From 4b39c760c08eed56d8a9e9802e1c9fc8629c366f Mon Sep 17 00:00:00 2001 From: ykiko Date: Sun, 16 Aug 2026 20:38:29 +0800 Subject: [PATCH 01/10] fix(index): reject duplicate symbol hashes, requeue orphan owners --- src/index/project_index.cpp | 14 +++++--- src/server/compiler/indexer.cpp | 29 +++++++++++++--- tests/unit/index/persisted_index_tests.cpp | 31 +++++++++++++++++ tests/unit/server/indexer_tests.cpp | 40 ++++++++++++++++++++++ 4 files changed, 105 insertions(+), 9 deletions(-) diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index 45cd010d3..0f4c2cb99 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -329,10 +329,12 @@ bool ProjectIndex::load_global(this ProjectIndex& self, } } - // Ids and (path, hash) pairs are both map keys in the writer, so a - // repeat of either marks a corrupt blob. A repeated id in particular - // would leave fv_ids interning the earlier pair to an id whose record - // names the later path, attributing contributions to the wrong file. + // Ids, (path, hash) pairs and symbol hashes are all map keys in the + // writer, so a repeat of any marks a corrupt blob. A repeated id in + // particular would leave fv_ids interning the earlier pair to an id + // whose record names the later path, attributing contributions to the + // wrong file; a repeated symbol hash would silently replace the earlier + // entry's identity and reference bitmap. llvm::DenseSet blob_fvs(blob.fv_ids.begin(), blob.fv_ids.end()); if(blob_fvs.size() != count) { return false; @@ -343,6 +345,10 @@ bool ProjectIndex::load_global(this ProjectIndex& self, return false; } } + llvm::DenseSet blob_syms(blob.sym_hashes.begin(), blob.sym_hashes.end()); + if(blob_syms.size() != sym_count) { + return false; + } // The writer only pins manifests whose tu_fv survived the same save's // garbage collection, so an unresolvable pin marks a corrupt blob. diff --git a/src/server/compiler/indexer.cpp b/src/server/compiler/indexer.cpp index 7fb9cf4f6..667868df4 100644 --- a/src/server/compiler/indexer.cpp +++ b/src/server/compiler/indexer.cpp @@ -112,7 +112,6 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { std::size_t hits = 0; std::size_t appended = 0; - std::size_t rebuilt = 0; auto lookup_symbol = [&](index::SymbolHash hash) { return view.find_symbol(hash); }; @@ -124,6 +123,7 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { // (TU-local path id, rows hash) per serving section; the FileVersions // these will reference are interned only at commit. llvm::SmallVector> section_contributions; + llvm::SmallVector rebuilt_ids; for(std::uint32_t section = 0; section < view.section_count(); section += 1) { auto local_id = view.section_path(section); auto rows_hash = view.section_rows_hash(section); @@ -201,7 +201,8 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { index::VariantInput fresh{rows_hash, &*rows, lookup_symbol}; std::string bytes; llvm::raw_string_ostream os(bytes); - if(shard && shard->loaded() && shard->content_hash() == generation) { + bool append = shard && shard->loaded() && shard->content_hash() == generation; + if(append) { // Same generation, new variant: merge it in, keeping every // stored variant — dead ones stay masked until the next save // compacts them. @@ -211,9 +212,9 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { // New content generation (or no blob at all): rows from other // generations must never share offset storage with these, so // the blob starts over. Stale contributions from other TUs - // simply stop matching any stored variant. + // stop matching any stored variant; the commit re-enqueues + // their owners. write_shard(index::Shard(), {}, fresh, content, generation, os); - rebuilt += 1; } auto replacement = index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); @@ -241,6 +242,9 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { return; } replacements.emplace_back(global_id, std::move(replacement)); + if(!append) { + rebuilt_ids.push_back(global_id); + } section_contributions.emplace_back(local_id, rows_hash); } @@ -313,6 +317,21 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { } it->second.set_live(project.live_variants(path_id)); } + + // A rebuild started its file's blob over, discarding the variants other + // TUs' contributions pin. Those owners are usually already pending from + // the same content event (enqueue dedupes); one whose attempt already + // failed — or that reads fresh by hash after a revert, which is why + // ContentChanged — has no in-process event left to rebuild its rows, + // only a restart reaching load()'s re-enqueue. + for(auto path_id: rebuilt_ids) { + auto& shard = workspace.shards.find(path_id)->second; + for(auto& [tu, hash]: project.contributions.find(path_id)->second) { + if(!shard.has_variant(hash)) { + enqueue(tu, ReindexReason::ContentChanged); + } + } + } dirty_manifests.insert(tu_path_id); global_dirty = true; @@ -323,7 +342,7 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { view.section_count(), hits, appended, - rebuilt, + rebuilt_ids.size(), workspace.shards.size()); } diff --git a/tests/unit/index/persisted_index_tests.cpp b/tests/unit/index/persisted_index_tests.cpp index 9790153ab..d57b71b76 100644 --- a/tests/unit/index/persisted_index_tests.cpp +++ b/tests/unit/index/persisted_index_tests.cpp @@ -327,6 +327,37 @@ TEST_CASE(GlobalDuplicateVersionsRejected) { ASSERT_EQ(loaded.file_versions.size(), std::size_t(2)); } +TEST_CASE(GlobalDuplicateSymbolRejected) { + // Symbol hashes are map keys in the writer; a structurally valid blob + // repeating one would silently replace the earlier entry's identity + // and reference bitmap while every manifest still loads as fresh. + GlobalBlobMirror mirror; + mirror.format_version = index::index_format_version; + mirror.sym_hashes = {42, 42}; + mirror.sym_names = {"sym", "impostor"}; + mirror.sym_kinds = {0, 0}; + clice::Bitmap bits; + bits.add(3); + mirror.sym_bitmaps = {index::write_bitmap(bits), index::write_bitmap(bits)}; + mirror.sym_paths = { + {3, "/proj/ref.h"} + }; + + clice::PathPool pool; + llvm::DenseMap pins; + auto dup = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(dup.has_value()); + index::ProjectIndex loaded; + ASSERT_FALSE(loaded.load_global(bytes_of(*dup), pool, pins)); + ASSERT_TRUE(loaded.symbols.empty()); + + mirror.sym_hashes = {42, 43}; + auto distinct = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(distinct.has_value()); + ASSERT_TRUE(loaded.load_global(bytes_of(*distinct), pool, pins)); + ASSERT_EQ(loaded.symbols.size(), std::size_t(2)); +} + TEST_CASE(UnknownFileVersionsDetected) { index::ProjectIndex project; auto known = project.intern_file_version(0, 0x1); diff --git a/tests/unit/server/indexer_tests.cpp b/tests/unit/server/indexer_tests.cpp index cbbecfbad..e7e3117d2 100644 --- a/tests/unit/server/indexer_tests.cpp +++ b/tests/unit/server/indexer_tests.cpp @@ -460,6 +460,10 @@ TEST_CASE(SaveRetiresPinnedShard) { indexer.merge(a2.data.data(), a2.data.size()); ASSERT_EQ(workspace.shards[header_id].variants().size(), std::size_t(1)); + // The rebuild re-enqueued pb; its pass then runs and fails, consuming + // the slot — the state the retirement below must repair on its own. + indexer.clear_pending(workspace.path_pool.intern(b.tu_path)); + // pa's index drops before pb reindexes: every stored variant is dead, // but pb's pinned hash keeps the live set nonempty. The save must // retire the shard rather than compact to an empty variant set. @@ -486,6 +490,42 @@ TEST_CASE(SaveRetiresPinnedShard) { ReindexReason::ContentChanged); } +TEST_CASE(RebuildRequeuesPinnedOwner) { + TempDir tmp; + tmp.touch("gen.h", + "#pragma once\n#ifdef MODE\nint gen_mode();\n#endif\n" + "inline int gen_fn() { return 1; }\n"); + tmp.touch("ga.cpp", "#include \"gen.h\"\nint ga() { return gen_fn(); }\n"); + tmp.touch("gb.cpp", "#include \"gen.h\"\nint gb() { return gen_fn(); }\n"); + + auto a = index_file(tmp, tmp.path("ga.cpp")); + auto b = index_file(tmp, tmp.path("gb.cpp"), {"-DMODE"}); + ASSERT_FALSE(a.data.empty()); + ASSERT_FALSE(b.data.empty()); + indexer.merge(a.data.data(), a.data.size()); + indexer.merge(b.data.data(), b.data.size()); + auto b_tu = workspace.path_pool.intern(b.tu_path); + ASSERT_FALSE(indexer.pending_reason(b_tu).has_value()); + + // The header moves to a new content generation and only ga catches up: + // the rebuilt blob discards gb's variant. With no pending slot left for + // gb, no in-process event would rebuild its rows — the rebuild itself + // must re-enqueue it, and as ContentChanged: a reverted header reads + // fresh by hash, which a deps-only slot would skip past. + tmp.touch("gen.h", + "#pragma once\n#ifdef MODE\nint gen_mode();\n#endif\n" + "inline int gen_fn() { return 2; }\n"); + auto a2 = index_file(tmp, tmp.path("ga.cpp")); + ASSERT_FALSE(a2.data.empty()); + indexer.merge(a2.data.data(), a2.data.size()); + auto header_id = workspace.path_pool.intern(tmp.path("gen.h")); + ASSERT_EQ(workspace.shards[header_id].variants().size(), std::size_t(1)); + + ASSERT_TRUE(indexer.pending_reason(b_tu) == ReindexReason::ContentChanged); + // ga's own fresh pin is stored: the rebuild must not re-enqueue it. + ASSERT_FALSE(indexer.pending_reason(workspace.path_pool.intern(a2.tu_path)).has_value()); +} + TEST_CASE(RejectsCorruptSection) { TempDir tmp; tmp.touch("cor.h", "#pragma once\ninline int cor() { return 1; }\n"); From a00945982fb7ec2171ba22306ec142c91d722bf6 Mon Sep 17 00:00:00 2001 From: ykiko Date: Mon, 17 Aug 2026 01:39:03 +0800 Subject: [PATCH 02/10] refactor(index): unified shard format core with byte-hash identity --- src/index/shard.cpp | 1129 +++++++++++++++++++----------- src/index/shard.h | 177 ++--- src/index/shard_layout.h | 138 ++++ tests/unit/index/shard_tests.cpp | 550 +++++++++++---- 4 files changed, 1298 insertions(+), 696 deletions(-) create mode 100644 src/index/shard_layout.h diff --git a/src/index/shard.cpp b/src/index/shard.cpp index ee4cbc06b..e68469d1a 100644 --- a/src/index/shard.cpp +++ b/src/index/shard.cpp @@ -7,8 +7,9 @@ #include #include "index/serialization.h" +#include "index/shard_layout.h" -#include "kota/ipc/lsp/position.h" +#include "kota/ipc/lsp/text.h" #include "llvm/ADT/DenseMap.h" #include "llvm/ADT/DenseSet.h" #include "llvm/ADT/STLExtras.h" @@ -18,11 +19,7 @@ namespace clice::index { namespace { -using ShardView = kota::codec::fbs::table_view; - -/// Sentinel in the length column: the real end lives in the sparse -/// (row, end) escape table. -constexpr std::uint8_t length_escape = 0xff; +using BlobView = kota::codec::fbs::table_view; enum class MaskTier : std::uint8_t { /// One variant: no mask columns at all. @@ -46,13 +43,27 @@ MaskTier tier_of(std::size_t variant_count) { } /// The blob was fully verified at load; per-query views skip that cost. -ShardView root_of(const llvm::MemoryBuffer& buffer) { - return ShardView::from_verified_bytes(blob_bytes(buffer.getBuffer())); +BlobView view_of(llvm::StringRef bytes) { + return BlobView::from_verified_bytes(blob_bytes(bytes)); +} + +BlobView root_of(const llvm::MemoryBuffer& buffer) { + return view_of(buffer.getBuffer()); +} + +/// The number of variants the blob's masks encode: the stored table's +/// size, or one for a worker-emitted blob (empty table, one anonymous +/// variant). +std::size_t variant_count_of(BlobView root) { + auto stored = to_array_ref(root[&ShardBlob::variants]); + return stored.empty() ? 1 : stored.size(); } -/// One side of the blob's row storage (occurrences or relations) as -/// contiguous column refs. -struct RowColumns { +/// One side of the blob's row storage as contiguous column refs, +/// tier-agnostic: `begin_of`/`end_of` read whichever range tier the blob +/// stores. +struct Ranges { + llvm::ArrayRef packed; llvm::ArrayRef begins; llvm::ArrayRef lengths; llvm::ArrayRef long_rows; @@ -62,65 +73,125 @@ struct RowColumns { llvm::ArrayRef roaring_offsets; llvm::ArrayRef roaring; + std::size_t size() const { + return packed.empty() ? begins.size() : packed.size(); + } + + std::uint32_t begin_of(std::uint32_t row) const { + if(packed.empty()) { + return begins[row]; + } + return packed[row] == packed_sentinel ? ~std::uint32_t(0) : packed[row] >> 8; + } + std::uint32_t end_of(std::uint32_t row) const { - auto length = lengths[row]; + if(!packed.empty() && packed[row] == packed_sentinel) { + return ~std::uint32_t(0); + } + auto length = + packed.empty() ? lengths[row] : static_cast(packed[row] & 0xff); if(length == length_escape) { // validate() proves every sentinel owns exactly one escape // entry, so the search always lands. auto it = std::ranges::lower_bound(long_rows, row); return long_ends[it - long_rows.begin()]; } - return begins[row] + length; + return begin_of(row) + length; } }; -RowColumns occ_columns(ShardView root) { +Ranges ranges_of(kota::codec::fbs::table_view view) { + if(!view.valid()) { + return {}; + } return { - to_array_ref(root[&ShardBlob::occ_begins]), - to_array_ref(root[&ShardBlob::occ_lengths]), - to_array_ref(root[&ShardBlob::occ_long_rows]), - to_array_ref(root[&ShardBlob::occ_long_ends]), - to_array_ref(root[&ShardBlob::occ_masks32]), - to_array_ref(root[&ShardBlob::occ_masks64]), - to_array_ref(root[&ShardBlob::occ_roaring_offsets]), - to_array_ref(root[&ShardBlob::occ_roaring]), + to_array_ref(view[&RowRanges::packed]), + to_array_ref(view[&RowRanges::begins]), + to_array_ref(view[&RowRanges::lengths]), + to_array_ref(view[&RowRanges::long_rows]), + to_array_ref(view[&RowRanges::long_ends]), + to_array_ref(view[&RowRanges::masks32]), + to_array_ref(view[&RowRanges::masks64]), + to_array_ref(view[&RowRanges::roaring_offsets]), + to_array_ref(view[&RowRanges::roaring]), }; } -RowColumns rel_columns(ShardView root) { - return { - to_array_ref(root[&ShardBlob::rel_begins]), - to_array_ref(root[&ShardBlob::rel_lengths]), - to_array_ref(root[&ShardBlob::rel_long_rows]), - to_array_ref(root[&ShardBlob::rel_long_ends]), - to_array_ref(root[&ShardBlob::rel_masks32]), - to_array_ref(root[&ShardBlob::rel_masks64]), - to_array_ref(root[&ShardBlob::rel_roaring_offsets]), - to_array_ref(root[&ShardBlob::rel_roaring]), - }; +Ranges occ_ranges(BlobView root) { + return ranges_of(root[&ShardBlob::occs]); +} + +Ranges rel_ranges(BlobView root) { + return ranges_of(root[&ShardBlob::rels]); } /// The slice bounds were validated monotonic and in-bounds at load, and /// every slice proven to decode, so this cannot fail. -Bitmap read_row_bitmap(const RowColumns& columns, std::uint32_t row) { +Bitmap read_row_bitmap(const Ranges& columns, std::uint32_t row) { auto begin = columns.roaring_offsets[row]; return *read_bitmap(columns.roaring.data() + begin, columns.roaring_offsets[row + 1] - begin); } +std::uint32_t occ_sym_id(BlobView root, std::uint32_t row) { + if(auto syms8 = to_array_ref(root[&ShardBlob::occ_syms8]); !syms8.empty()) { + return syms8[row]; + } + if(auto syms16 = to_array_ref(root[&ShardBlob::occ_syms16]); !syms16.empty()) { + return syms16[row]; + } + return to_array_ref(root[&ShardBlob::occ_syms32])[row]; +} + +/// The escape table must pair one-to-one, in row order, with the sentinel +/// lengths: readers trust the pairing, and a sentinel missing its entry +/// (or a stray entry masking one elsewhere) can pass every range bound +/// while serving a wrong value forever. +bool escapes_ok(llvm::ArrayRef lengths, + llvm::ArrayRef escape_rows, + std::size_t escape_values) { + if(escape_values != escape_rows.size()) { + return false; + } + std::size_t cursor = 0; + for(std::uint32_t row = 0; row < lengths.size(); row += 1) { + if(lengths[row] != length_escape) { + continue; + } + if(cursor == escape_rows.size() || escape_rows[cursor] != row) { + return false; + } + cursor += 1; + } + return cursor == escape_rows.size(); +} + +/// Sentinel lengths inside the packed column, extracted for escapes_ok. +/// The no-range sentinel word shares the escape byte pattern but owns no +/// escape entry — report it as a plain length. +llvm::SmallVector packed_lengths(llvm::ArrayRef packed) { + llvm::SmallVector lengths; + lengths.reserve(packed.size()); + for(auto value: packed) { + lengths.push_back(value == packed_sentinel ? 0 : static_cast(value & 0xff)); + } + return lengths; +} + /// Structural verification does not constrain field values; everything the /// readers dereference through raw column pointers or binary-search must be /// proven in-bounds and in order here, once, so queries stay check-free. -bool validate(ShardView root) { +/// The checks are also canonicality checks: every self-describing choice +/// (range tier, symbol-id width, content omission, mask tier) is a strict +/// function of the data, so one logical blob has exactly one encoding and +/// its byte hash is a usable identity. +bool validate(BlobView root) { auto variants = to_array_ref(root[&ShardBlob::variants]); auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); auto offsets = to_array_ref(root[&ShardBlob::sym_rel_offsets]); - if(variants.empty()) { - return false; - } - // A variant is identified by its rows hash everywhere (set_live, - // write_shard's keep filter), so hashes must be unique: rows owned only - // by a duplicated entry would serve and survive compaction with no + // A variant is identified by its hash everywhere (set_live, the merge + // keep filter), so stored hashes must be unique: rows owned only by a + // duplicated entry would serve and survive compaction with no // contribution owning them. llvm::SmallVector sorted_variants(variants.begin(), variants.end()); std::ranges::sort(sorted_variants); @@ -128,18 +199,78 @@ bool validate(ShardView root) { return false; } - // Every freshness decision compares the advertised content hash - // (manifest FileVersions, the merge's generation checks), so content - // bytes corrupted under an intact structure would keep loading as fresh - // while position mapping reads text the rows were not built from. - if(llvm::xxh3_64bits(to_ref(root[&ShardBlob::content])) != root[&ShardBlob::content_hash]) { + auto content = to_ref(root[&ShardBlob::content]); + auto content_size = root[&ShardBlob::content_size]; + if(!content.empty()) { + // Every freshness decision compares the advertised content hash + // (manifest FileVersions, the merge's generation checks), so + // content bytes corrupted under an intact structure would keep + // loading as fresh while position mapping reads text the rows were + // not built from. + if(content.size() != content_size || + llvm::xxh3_64bits(content) != root[&ShardBlob::content_hash]) { + return false; + } + // Pure-ASCII content must be omitted — the canonical form. + if(llvm::all_of(content, [](char c) { return static_cast(c) < 0x80; })) { + return false; + } + } + + // The line table reconstructs every line start by prefix sum, so it + // must both pair with its escape table and add up to exactly the + // content size — a drifted sum would shift every position mapping + // below the corruption. + auto line_lengths = to_array_ref(root[&ShardBlob::line_lengths]); + auto long_line_rows = to_array_ref(root[&ShardBlob::long_line_rows]); + auto long_line_lengths = to_array_ref(root[&ShardBlob::long_line_lengths]); + if(line_lengths.empty() || !escapes_ok(line_lengths, long_line_rows, long_line_lengths.size()) || + !std::ranges::is_sorted(long_line_rows, std::less_equal{})) { + return false; + } + std::uint64_t line_sum = 0; + std::size_t line_escape_cursor = 0; + for(auto length: line_lengths) { + if(length == length_escape) { + auto value = long_line_lengths[line_escape_cursor]; + line_escape_cursor += 1; + if(value < length_escape) { + return false; + } + line_sum += value; + } else { + line_sum += length; + } + } + if(line_sum != content_size) { return false; } - auto occ = occ_columns(root); - auto rel = rel_columns(root); - auto occ_count = occ.begins.size(); - auto rel_count = rel.begins.size(); + auto occ = occ_ranges(root); + auto rel = rel_ranges(root); + auto occ_count = occ.size(); + auto rel_count = rel.size(); + + // Exactly one range tier, chosen by the content size. + auto tier_ok = [&](const Ranges& columns) { + if(content_size <= packed_range_limit) { + return columns.begins.empty() && columns.lengths.empty(); + } + return columns.packed.empty() && columns.lengths.size() == columns.begins.size(); + }; + if(!tier_ok(occ) || !tier_ok(rel)) { + return false; + } + auto range_escapes_ok = [](const Ranges& columns) { + if(columns.packed.empty()) { + return escapes_ok(columns.lengths, columns.long_rows, columns.long_ends.size()); + } + return escapes_ok(packed_lengths(columns.packed), columns.long_rows, + columns.long_ends.size()); + }; + if(!range_escapes_ok(occ) || !range_escapes_ok(rel)) { + return false; + } if(offsets.size() != sym_hashes.size() + 1) { return false; @@ -157,64 +288,47 @@ bool validate(ShardView root) { auto sym_ids_ok = [&](auto ids) { return llvm::all_of(ids, [&](std::uint32_t id) { return id < sym_hashes.size(); }); }; + // Exactly one id width, chosen by the table size. + auto sym_width_ok = [&](std::size_t count, + llvm::ArrayRef ids8, + llvm::ArrayRef ids16, + llvm::ArrayRef ids32) { + if(ids8.size() + ids16.size() + ids32.size() != count) { + return false; + } + if(sym_hashes.size() <= 0x100) { + return ids16.empty() && ids32.empty(); + } + if(sym_hashes.size() <= 0x10000) { + return ids8.empty() && ids32.empty(); + } + return ids8.empty() && ids16.empty(); + }; + auto occ_syms8 = to_array_ref(root[&ShardBlob::occ_syms8]); auto occ_syms16 = to_array_ref(root[&ShardBlob::occ_syms16]); auto occ_syms32 = to_array_ref(root[&ShardBlob::occ_syms32]); - if(occ.lengths.size() != occ_count) { + if(!sym_width_ok(occ_count, occ_syms8, occ_syms16, occ_syms32)) { return false; } - if(occ_syms16.size() + occ_syms32.size() != occ_count || - (!occ_syms16.empty() && !occ_syms32.empty())) { - return false; - } - if(!sym_ids_ok(occ_syms16) || !sym_ids_ok(occ_syms32)) { + if(!sym_ids_ok(occ_syms8) || !sym_ids_ok(occ_syms16) || !sym_ids_ok(occ_syms32)) { return false; } auto rel_kinds = to_array_ref(root[&ShardBlob::rel_kinds]); - if(rel_kinds.size() != rel_count || rel.lengths.size() != rel_count) { + if(rel_kinds.size() != rel_count) { return false; } - auto sparse_ok = [](llvm::ArrayRef rows, std::size_t values, std::size_t count) { - return rows.size() == values && std::ranges::is_sorted(rows, std::less_equal{}) && - (rows.empty() || rows.back() < count); - }; - // The escape table must pair one-to-one, in row order, with the - // sentinel lengths: end_of trusts the pairing, and a sentinel missing - // its entry (or a stray entry masking one elsewhere) can pass every - // range bound below while serving a wrong end forever. - auto escapes_ok = [](llvm::ArrayRef lengths, - llvm::ArrayRef long_rows, - llvm::ArrayRef long_ends) { - if(long_ends.size() != long_rows.size()) { - return false; - } - std::size_t cursor = 0; - for(std::uint32_t row = 0; row < lengths.size(); row += 1) { - if(lengths[row] != length_escape) { - continue; - } - if(cursor == long_rows.size() || long_rows[cursor] != row) { - return false; - } - cursor += 1; - } - return cursor == long_rows.size(); - }; - if(!escapes_ok(occ.lengths, occ.long_rows, occ.long_ends)) { - return false; - } // lookup(offset) binary-searches the decoded end column and stops its // containment walk on begin order; rows out of either order (a corrupt // escaped end included) would silently miss or misresolve occurrences // on every query, forever — reject the blob so it is rebuilt instead. - // Ends are bounded by the stored content too: every decoded range is - // served as a source range into it. - auto content_size = root[&ShardBlob::content].size(); + // Ends are bounded by the content size too: every decoded range is + // served as a source range into the content. std::uint32_t prev_begin = 0; std::uint32_t prev_end = 0; for(std::uint32_t row = 0; row < occ_count; row += 1) { - auto begin = occ.begins[row]; + auto begin = occ.begin_of(row); auto end = occ.end_of(row); if(begin < prev_begin || end < prev_end || end < begin || end > content_size) { return false; @@ -222,9 +336,6 @@ bool validate(ShardView root) { prev_begin = begin; prev_end = end; } - if(!escapes_ok(rel.lengths, rel.long_rows, rel.long_ends)) { - return false; - } // Relation ranges carry no query order to enforce, but are served as // source ranges all the same — bound them like the occurrence ends. // The exception is the default LocalSourceRange, the writer's sentinel @@ -232,7 +343,7 @@ bool validate(ShardView root) { // kind is written with a real range, so a sentinel there is corruption // that would serve an invalid source range forever. for(std::uint32_t row = 0; row < rel_count; row += 1) { - auto begin = rel.begins[row]; + auto begin = rel.begin_of(row); auto end = rel.end_of(row); if((LocalSourceRange{begin, end}) == LocalSourceRange{}) { if(!RelationKind(static_cast(rel_kinds[row])).isBetweenSymbol()) { @@ -245,14 +356,21 @@ bool validate(ShardView root) { } } + auto sparse_ok = [](llvm::ArrayRef rows, std::size_t values, std::size_t count) { + return rows.size() == values && std::ranges::is_sorted(rows, std::less_equal{}) && + (rows.empty() || rows.back() < count); + }; auto rel_sym_rows = to_array_ref(root[&ShardBlob::rel_sym_rows]); + auto rel_sym8 = to_array_ref(root[&ShardBlob::rel_sym8]); auto rel_sym16 = to_array_ref(root[&ShardBlob::rel_sym16]); auto rel_sym32 = to_array_ref(root[&ShardBlob::rel_sym32]); - if(!sparse_ok(rel_sym_rows, rel_sym16.size() + rel_sym32.size(), rel_count) || - (!rel_sym16.empty() && !rel_sym32.empty())) { + if(!sparse_ok(rel_sym_rows, rel_sym8.size() + rel_sym16.size() + rel_sym32.size(), rel_count)) { + return false; + } + if(!sym_width_ok(rel_sym_rows.size(), rel_sym8, rel_sym16, rel_sym32)) { return false; } - if(!sym_ids_ok(rel_sym16) || !sym_ids_ok(rel_sym32)) { + if(!sym_ids_ok(rel_sym8) || !sym_ids_ok(rel_sym16) || !sym_ids_ok(rel_sym32)) { return false; } auto rel_def_rows = to_array_ref(root[&ShardBlob::rel_def_rows]); @@ -283,8 +401,9 @@ bool validate(ShardView root) { // live.all fast path skips the mask), vanishes once any variant dies, // and the next compaction erases it for real — every manifest still // fresh throughout. - auto masks_ok = [&](const RowColumns& columns, std::size_t count) { - switch(tier_of(variants.size())) { + auto variant_count = variant_count_of(root); + auto masks_ok = [&](const Ranges& columns, std::size_t count) { + switch(tier_of(variant_count)) { case MaskTier::Single: { return columns.masks32.empty() && columns.masks64.empty() && columns.roaring_offsets.empty() && columns.roaring.empty(); @@ -294,7 +413,7 @@ bool validate(ShardView root) { !columns.roaring_offsets.empty()) { return false; } - auto stray = variants.size() < 32 ? ~std::uint32_t(0) << variants.size() : 0; + auto stray = variant_count < 32 ? ~std::uint32_t(0) << variant_count : 0; return llvm::all_of(columns.masks32, [&](std::uint32_t mask) { return mask != 0 && (mask & stray) == 0; }); @@ -304,7 +423,7 @@ bool validate(ShardView root) { !columns.roaring_offsets.empty()) { return false; } - auto stray = variants.size() < 64 ? ~std::uint64_t(0) << variants.size() : 0; + auto stray = variant_count < 64 ? ~std::uint64_t(0) << variant_count : 0; return llvm::all_of(columns.masks64, [&](std::uint64_t mask) { return mask != 0 && (mask & stray) == 0; }); @@ -325,7 +444,7 @@ bool validate(ShardView root) { auto begin = columns.roaring_offsets[row]; auto mask = read_bitmap(columns.roaring.data() + begin, columns.roaring_offsets[row + 1] - begin); - if(!mask || mask->isEmpty() || mask->maximum() >= variants.size()) { + if(!mask || mask->isEmpty() || mask->maximum() >= variant_count) { return false; } } @@ -337,17 +456,11 @@ bool validate(ShardView root) { return masks_ok(occ, occ_count) && masks_ok(rel, rel_count); } -std::uint32_t occ_sym_id(ShardView root, std::uint32_t row) { - auto syms16 = to_array_ref(root[&ShardBlob::occ_syms16]); - if(!syms16.empty()) { - return syms16[row]; - } - return to_array_ref(root[&ShardBlob::occ_syms32])[row]; -} - } // namespace -Shard::Shard(std::unique_ptr buffer) : buffer(std::move(buffer)) {} +Shard::Shard(std::unique_ptr buffer) : buffer(std::move(buffer)) { + blob_hash = llvm::xxh3_64bits(this->buffer->getBuffer()); +} Shard Shard::from_bytes(llvm::StringRef data) { return from_buffer(llvm::MemoryBuffer::getMemBuffer(data, "", false)); @@ -360,10 +473,10 @@ Shard Shard::from_buffer(std::unique_ptr buffer) { // Stale or corrupt bytes (an older build's cache directory) must never // crash the server or be misread: deep structural verification first, - // then the format-version gate, then the cross-field size checks the - // raw column readers rely on. Anything failing loads as "not on disk" - // and the background indexer rebuilds it. - auto root = ShardView::from_bytes(blob_bytes(buffer->getBuffer())); + // then the format-version gate, then the cross-field checks the raw + // column readers rely on. Anything failing loads as "not on disk" and + // the background indexer rebuilds it. + auto root = BlobView::from_bytes(blob_bytes(buffer->getBuffer())); if(!root.valid() || root[&ShardBlob::format_version] != index_format_version || !validate(root)) { return {}; @@ -378,19 +491,37 @@ std::uint64_t Shard::content_hash() const { return root_of(*buffer)[&ShardBlob::content_hash]; } +std::uint32_t Shard::content_size() const { + if(!buffer) { + return 0; + } + return root_of(*buffer)[&ShardBlob::content_size]; +} + +llvm::StringRef Shard::content() const { + if(!buffer) { + return {}; + } + return to_ref(root_of(*buffer)[&ShardBlob::content]); +} + +bool Shard::ascii() const { + return loaded() && content().empty(); +} + std::vector Shard::variants() const { if(!buffer) { return {}; } auto stored = to_array_ref(root_of(*buffer)[&ShardBlob::variants]); + if(stored.empty()) { + return {blob_hash}; + } return {stored.begin(), stored.end()}; } bool Shard::has_variant(RowsHash hash) const { - if(!buffer) { - return false; - } - return llvm::is_contained(to_array_ref(root_of(*buffer)[&ShardBlob::variants]), hash); + return loaded() && llvm::is_contained(variants(), hash); } void Shard::set_live(llvm::ArrayRef live_hashes) { @@ -399,7 +530,7 @@ void Shard::set_live(llvm::ArrayRef live_hashes) { return; } - auto stored = to_array_ref(root_of(*buffer)[&ShardBlob::variants]); + auto stored = variants(); std::size_t matched = 0; Live next; for(std::uint32_t id = 0; id < stored.size(); id += 1) { @@ -425,8 +556,8 @@ bool Shard::row_live(bool occurrence, std::uint32_t row) const { return true; } auto root = root_of(*buffer); - auto columns = occurrence ? occ_columns(root) : rel_columns(root); - switch(tier_of(to_array_ref(root[&ShardBlob::variants]).size())) { + auto columns = occurrence ? occ_ranges(root) : rel_ranges(root); + switch(tier_of(variant_count_of(root))) { case MaskTier::Single: { return (live.bits & 1) != 0; } @@ -449,7 +580,7 @@ void Shard::lookup(std::uint32_t offset, return; } auto root = root_of(*buffer); - auto columns = occ_columns(root); + auto columns = occ_ranges(root); auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); // Binary search the first row whose end reaches the offset, then walk @@ -457,7 +588,7 @@ void Shard::lookup(std::uint32_t offset, // pairwise disjoint or identical, so under (begin, end) order the end // column is monotonic too. std::size_t lo = 0; - std::size_t hi = columns.begins.size(); + std::size_t hi = columns.size(); while(lo < hi) { auto mid = lo + (hi - lo) / 2; if(columns.end_of(mid) < offset) { @@ -467,9 +598,9 @@ void Shard::lookup(std::uint32_t offset, } } - for(; lo < columns.begins.size(); lo += 1) { + for(; lo < columns.size(); lo += 1) { auto row = static_cast(lo); - LocalSourceRange range{columns.begins[row], columns.end_of(row)}; + LocalSourceRange range{columns.begin_of(row), columns.end_of(row)}; if(!range.contains(offset)) { break; } @@ -501,9 +632,10 @@ void Shard::lookup(SymbolHash symbol, auto begin_row = offsets[id]; auto end_row = offsets[id + 1]; - auto columns = rel_columns(root); + auto columns = rel_ranges(root); auto kinds = to_array_ref(root[&ShardBlob::rel_kinds]); auto sym_rows = to_array_ref(root[&ShardBlob::rel_sym_rows]); + auto sym8 = to_array_ref(root[&ShardBlob::rel_sym8]); auto sym16 = to_array_ref(root[&ShardBlob::rel_sym16]); auto sym32 = to_array_ref(root[&ShardBlob::rel_sym32]); auto def_rows = to_array_ref(root[&ShardBlob::rel_def_rows]); @@ -533,7 +665,7 @@ void Shard::lookup(SymbolHash symbol, Relation relation{ .kind = row_kind, - .range = {columns.begins[row], columns.end_of(row)}, + .range = {columns.begin_of(row), columns.end_of(row)}, .target_symbol = 0, }; if(def_cursor < static_cast(def_rows.size()) && @@ -541,7 +673,9 @@ void Shard::lookup(SymbolHash symbol, relation.set_definition_range({def_begins[def_cursor], def_ends[def_cursor]}); } else if(sym_cursor < static_cast(sym_rows.size()) && sym_rows[sym_cursor] == row) { - auto payload = sym16.empty() ? sym32[sym_cursor] : sym16[sym_cursor]; + auto payload = !sym8.empty() ? sym8[sym_cursor] + : !sym16.empty() ? sym16[sym_cursor] + : sym32[sym_cursor]; relation.target_symbol = sym_hashes[payload]; } @@ -574,19 +708,26 @@ bool Shard::find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) co return true; } -llvm::StringRef Shard::content() const { - if(!buffer) { - return {}; - } - return to_ref(root_of(*buffer)[&ShardBlob::content]); -} - std::span Shard::line_starts() const { if(!buffer) { return {}; } if(line_starts_cache.empty()) { - line_starts_cache = kota::ipc::lsp::build_line_starts(content()); + auto root = root_of(*buffer); + auto lengths = to_array_ref(root[&ShardBlob::line_lengths]); + auto long_lengths = to_array_ref(root[&ShardBlob::long_line_lengths]); + line_starts_cache.reserve(lengths.size()); + std::uint32_t start = 0; + std::size_t escape_cursor = 0; + for(auto length: lengths) { + line_starts_cache.push_back(start); + if(length == length_escape) { + start += long_lengths[escape_cursor]; + escape_cursor += 1; + } else { + start += length; + } + } } return line_starts_cache; } @@ -644,8 +785,8 @@ void mask_or(MaskT& into, const MaskT& from) { /// dropped variant's bit vanishes; an all-dropped row reads as empty and /// is skipped by the caller). template -MaskT remap_mask(ShardView root, - const RowColumns& columns, +MaskT remap_mask(BlobView root, + const Ranges& columns, std::uint32_t row, llvm::ArrayRef id_map) { MaskT result{}; @@ -654,7 +795,7 @@ MaskT remap_mask(ShardView root, mask_or(result, single_bit(static_cast(id_map[old_id]))); } }; - switch(tier_of(to_array_ref(root[&ShardBlob::variants]).size())) { + switch(tier_of(variant_count_of(root))) { case MaskTier::Single: { apply(0); break; @@ -687,7 +828,7 @@ MaskT remap_mask(ShardView root, return result; } -/// Everything write_shard accumulates before choosing column tiers. +/// Everything a write accumulates before choosing column tiers. template struct MergedRows { std::vector> occurrences; @@ -722,59 +863,81 @@ void merge_sorted(std::vector old_rows, } } -template -void merge_occurrences(ShardView old_root, - llvm::ArrayRef id_map, - const VariantInput& fresh, - std::int64_t fresh_id, - std::vector>& out) { - std::vector> old_rows; - if(old_root.valid()) { - auto columns = occ_columns(old_root); - auto sym_hashes = to_array_ref(old_root[&ShardBlob::sym_hashes]); - old_rows.reserve(columns.begins.size()); - for(std::uint32_t row = 0; row < columns.begins.size(); row += 1) { - auto mask = remap_mask(old_root, columns, row, id_map); - if(mask_empty(mask)) { - continue; +/// Combine adjacent equal-key rows of a sorted run into one row with +/// OR-ed masks. +template +void combine_equal(std::vector& rows, Key key) { + std::size_t out = 0; + for(std::size_t i = 0; i < rows.size(); i += 1) { + if(out != 0 && key(rows[out - 1]) == key(rows[i])) { + mask_or(rows[out - 1].mask, rows[i].mask); + } else { + if(out != i) { + rows[out] = std::move(rows[i]); } - old_rows.push_back({columns.begins[row], - columns.end_of(row), - sym_hashes[occ_sym_id(old_root, row)], - std::move(mask)}); + out += 1; } } + rows.resize(out); +} - std::vector> fresh_rows; - if(fresh.rows) { - fresh_rows.reserve(fresh.rows->occurrences.size()); - auto bit = single_bit(static_cast(fresh_id)); - for(auto& occurrence: fresh.rows->occurrences) { - fresh_rows.push_back( - {occurrence.range.begin, occurrence.range.end, occurrence.target, bit}); +/// The symbol table a set of merged rows requires: occurrence targets, +/// relation group keys, and symbol payloads. +template +llvm::DenseSet referenced_symbols(const MergedRows& merged) { + llvm::DenseSet referenced; + for(auto& row: merged.occurrences) { + referenced.insert(row.sym); + } + for(auto& [hash, rows]: merged.relations) { + referenced.insert(hash); + for(auto& row: rows) { + if(row.payload != 0 && + !RelationKind(static_cast(row.kind)).isDeclOrDef()) { + referenced.insert(row.payload); + } } - std::ranges::sort(fresh_rows, [](const auto& lhs, const auto& rhs) { - return std::tuple(lhs.begin, lhs.end, lhs.sym) < - std::tuple(rhs.begin, rhs.end, rhs.sym); - }); } + return referenced; +} + +constexpr auto occ_key = [](const auto& row) { return std::tuple(row.begin, row.end, row.sym); }; +constexpr auto rel_key = [](const auto& row) { + return std::tuple(row.kind, row.begin, row.end, row.payload); +}; - merge_sorted( - std::move(old_rows), - std::move(fresh_rows), - [](const auto& row) { return std::tuple(row.begin, row.end, row.sym); }, - out); +/// Decode one blob's occurrence rows into working form; `mask_of` returns +/// the row's mask in the output id space (empty drops the row). +template +std::vector> decode_occurrences(BlobView root, MaskOf mask_of) { + auto columns = occ_ranges(root); + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + std::vector> rows; + rows.reserve(columns.size()); + for(std::uint32_t row = 0; row < columns.size(); row += 1) { + auto mask = mask_of(columns, row); + if(mask_empty(mask)) { + continue; + } + rows.push_back({columns.begin_of(row), + columns.end_of(row), + sym_hashes[occ_sym_id(root, row)], + std::move(mask)}); + } + return rows; } -template -std::vector> decode_relation_group(ShardView root, - const RowColumns& columns, +/// Decode one symbol's relation slice of a blob into working form. +template +std::vector> decode_relation_group(BlobView root, + const Ranges& columns, std::uint32_t begin_row, std::uint32_t end_row, - llvm::ArrayRef id_map) { + MaskOf mask_of) { auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); auto kinds = to_array_ref(root[&ShardBlob::rel_kinds]); auto sym_rows = to_array_ref(root[&ShardBlob::rel_sym_rows]); + auto sym8 = to_array_ref(root[&ShardBlob::rel_sym8]); auto sym16 = to_array_ref(root[&ShardBlob::rel_sym16]); auto sym32 = to_array_ref(root[&ShardBlob::rel_sym32]); auto def_rows = to_array_ref(root[&ShardBlob::rel_def_rows]); @@ -796,7 +959,7 @@ std::vector> decode_relation_group(ShardView root, def_cursor += 1; } - auto mask = remap_mask(root, columns, row, id_map); + auto mask = mask_of(columns, row); if(mask_empty(mask)) { continue; } @@ -808,197 +971,193 @@ std::vector> decode_relation_group(ShardView root, LocalSourceRange{def_begins[def_cursor], def_ends[def_cursor]}); } else if(sym_cursor < static_cast(sym_rows.size()) && sym_rows[sym_cursor] == row) { - auto id = sym16.empty() ? sym32[sym_cursor] : sym16[sym_cursor]; + auto id = !sym8.empty() ? sym8[sym_cursor] + : !sym16.empty() ? sym16[sym_cursor] + : sym32[sym_cursor]; payload = sym_hashes[id]; } rows.push_back( - {kinds[row], columns.begins[row], columns.end_of(row), payload, std::move(mask)}); + {kinds[row], columns.begin_of(row), columns.end_of(row), payload, std::move(mask)}); } return rows; } -template -void merge_relation_rows(std::vector> old_rows, - std::vector> fresh_rows, - std::vector>& out) { - merge_sorted( - std::move(old_rows), - std::move(fresh_rows), - [](const auto& row) { return std::tuple(row.kind, row.begin, row.end, row.payload); }, - out); -} +/// One blob's relation groups in symbol-hash order, decoded lazily by the +/// group merge. +struct GroupIndex { + std::uint64_t hash; + std::uint32_t begin_row; + std::uint32_t end_row; +}; -template -void merge_relations(ShardView old_root, - llvm::ArrayRef id_map, - const VariantInput& fresh, - std::int64_t fresh_id, - std::vector>>>& out) { - // Fresh groups, sorted by symbol hash; rows in a group follow the - // builder's canonical (kind, begin, end, payload) order already, but a - // hand-built FileIndex (tests) may not — sort defensively, it is cheap - // relative to the merge. - std::vector>>> fresh_groups; - if(fresh.rows) { - auto bit = single_bit(static_cast(fresh_id)); - fresh_groups.reserve(fresh.rows->relations.size()); - for(auto& [hash, relations]: fresh.rows->relations) { - std::vector> rows; - rows.reserve(relations.size()); - for(auto& relation: relations) { - rows.push_back({static_cast(relation.kind), - relation.range.begin, - relation.range.end, - relation.target_symbol, - bit}); - } - std::ranges::sort(rows, [](const auto& lhs, const auto& rhs) { - return std::tuple(lhs.kind, lhs.begin, lhs.end, lhs.payload) < - std::tuple(rhs.kind, rhs.begin, rhs.end, rhs.payload); - }); - fresh_groups.emplace_back(hash, std::move(rows)); +std::vector relation_groups(BlobView root) { + std::vector groups; + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + auto offsets = to_array_ref(root[&ShardBlob::sym_rel_offsets]); + for(std::uint32_t id = 0; id < sym_hashes.size(); id += 1) { + if(offsets[id] != offsets[id + 1]) { + groups.push_back({sym_hashes[id], offsets[id], offsets[id + 1]}); } - std::ranges::sort(fresh_groups, {}, [](const auto& group) { return group.first; }); } + return groups; +} - struct OldGroup { - std::uint64_t hash; - std::uint32_t begin_row; - std::uint32_t end_row; - }; +/// The content identity and line table one blob carries, copied verbatim +/// between blobs of the same generation. +struct ContentInfo { + std::uint64_t hash = 0; + std::uint32_t size = 0; + llvm::StringRef content; + std::vector line_lengths; + std::vector long_line_rows; + std::vector long_line_lengths; +}; - std::vector old_groups; - RowColumns old_columns; - if(old_root.valid()) { - old_columns = rel_columns(old_root); - auto sym_hashes = to_array_ref(old_root[&ShardBlob::sym_hashes]); - auto offsets = to_array_ref(old_root[&ShardBlob::sym_rel_offsets]); - for(std::uint32_t id = 0; id < sym_hashes.size(); id += 1) { - if(offsets[id] != offsets[id + 1]) { - old_groups.push_back({sym_hashes[id], offsets[id], offsets[id + 1]}); - } +ContentInfo content_info_of(BlobView root) { + ContentInfo info; + info.hash = root[&ShardBlob::content_hash]; + info.size = root[&ShardBlob::content_size]; + info.content = to_ref(root[&ShardBlob::content]); + auto lengths = to_array_ref(root[&ShardBlob::line_lengths]); + auto long_rows = to_array_ref(root[&ShardBlob::long_line_rows]); + auto long_lengths = to_array_ref(root[&ShardBlob::long_line_lengths]); + info.line_lengths.assign(lengths.begin(), lengths.end()); + info.long_line_rows.assign(long_rows.begin(), long_rows.end()); + info.long_line_lengths.assign(long_lengths.begin(), long_lengths.end()); + return info; +} + +ContentInfo content_info_of(llvm::StringRef content) { + ContentInfo info; + info.hash = llvm::xxh3_64bits(content); + info.size = static_cast(content.size()); + bool is_ascii = + llvm::all_of(content, [](char c) { return static_cast(c) < 0x80; }); + if(!is_ascii) { + info.content = content; + } + + auto starts = + kota::ipc::lsp::build_line_starts(std::string_view(content.data(), content.size())); + info.line_lengths.reserve(starts.size()); + for(std::size_t i = 0; i < starts.size(); i += 1) { + auto next = i + 1 < starts.size() ? starts[i + 1] : info.size; + auto length = next - starts[i]; + if(length >= length_escape) { + info.line_lengths.push_back(length_escape); + info.long_line_rows.push_back(static_cast(i)); + info.long_line_lengths.push_back(length); + } else { + info.line_lengths.push_back(static_cast(length)); } } + return info; +} - auto lhs = old_groups.begin(); - auto rhs = fresh_groups.begin(); - while(lhs != old_groups.end() || rhs != fresh_groups.end()) { - if(rhs == fresh_groups.end() || (lhs != old_groups.end() && lhs->hash < rhs->first)) { - auto rows = decode_relation_group(old_root, - old_columns, - lhs->begin_row, - lhs->end_row, - id_map); - if(!rows.empty()) { - out.emplace_back(lhs->hash, std::move(rows)); - } - lhs += 1; - } else if(lhs == old_groups.end() || rhs->first < lhs->hash) { - out.emplace_back(rhs->first, std::move(rhs->second)); - rhs += 1; +/// A local symbol's identity as stored in a blob's local-name table. +struct LocalInfo { + std::string name; + std::uint8_t kind; + std::uint8_t scope; +}; + +/// Collect one blob's local symbols; symbols the merged rows no longer +/// reference are filtered at emit time. +void collect_locals(BlobView root, llvm::DenseMap& locals) { + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + auto local_syms = to_array_ref(root[&ShardBlob::local_syms]); + auto kinds = to_array_ref(root[&ShardBlob::local_kinds]); + auto scopes = to_array_ref(root[&ShardBlob::local_scopes]); + auto names = root[&ShardBlob::local_names]; + for(std::uint32_t k = 0; k < local_syms.size(); k += 1) { + auto hash = sym_hashes[local_syms[k]]; + locals.try_emplace(hash, LocalInfo{std::string(names.at(k)), kinds[k], scopes[k]}); + } +} + +void emit_row_range(RowRanges& side, + bool narrow, + std::uint32_t row, + std::uint32_t begin, + std::uint32_t end) { + if((LocalSourceRange{begin, end}) == LocalSourceRange{}) { + // The no-range sentinel of pair relations; the wide columns hold + // it natively, the packed column spells it as the reserved word. + if(narrow) { + side.packed.push_back(packed_sentinel); } else { - auto old_rows = decode_relation_group(old_root, - old_columns, - lhs->begin_row, - lhs->end_row, - id_map); - std::vector> merged; - merge_relation_rows(std::move(old_rows), std::move(rhs->second), merged); - if(!merged.empty()) { - out.emplace_back(lhs->hash, std::move(merged)); - } - lhs += 1; - rhs += 1; + side.begins.push_back(begin); + side.lengths.push_back(0); } + return; + } + auto length = end - begin; + std::uint8_t stored = length >= length_escape ? length_escape + : static_cast(length); + if(stored == length_escape) { + side.long_rows.push_back(row); + side.long_ends.push_back(end); + } + if(narrow) { + side.packed.push_back(pack_range(begin, stored)); + } else { + side.begins.push_back(begin); + side.lengths.push_back(stored); } } template -void emit_mask(ShardBlob& blob, bool occurrence, MaskTier tier, const MaskT& mask) { - auto& masks32 = occurrence ? blob.occ_masks32 : blob.rel_masks32; - auto& masks64 = occurrence ? blob.occ_masks64 : blob.rel_masks64; - auto& roaring_offsets = occurrence ? blob.occ_roaring_offsets : blob.rel_roaring_offsets; - auto& roaring = occurrence ? blob.occ_roaring : blob.rel_roaring; - +void emit_mask(RowRanges& side, MaskTier tier, const MaskT& mask) { switch(tier) { case MaskTier::Single: { break; } case MaskTier::U32: { if constexpr(std::same_as) { - masks32.push_back(static_cast(mask)); + side.masks32.push_back(static_cast(mask)); } break; } case MaskTier::U64: { if constexpr(std::same_as) { - masks64.push_back(mask); + side.masks64.push_back(mask); } break; } case MaskTier::Roaring: { if constexpr(std::same_as) { auto size = mask.getSizeInBytes(true); - auto offset = roaring.size(); - roaring.resize(offset + size); - mask.write(reinterpret_cast(roaring.data() + offset), true); - roaring_offsets.push_back(static_cast(offset)); + auto offset = side.roaring.size(); + side.roaring.resize(offset + size); + mask.write(reinterpret_cast(side.roaring.data() + offset), true); + side.roaring_offsets.push_back(static_cast(offset)); } break; } } } -void emit_range(std::vector& lengths, - std::vector& long_rows, - std::vector& long_ends, - std::uint32_t row, - std::uint32_t begin, - std::uint32_t end) { - auto length = end - begin; - if(length >= length_escape) { - lengths.push_back(length_escape); - long_rows.push_back(row); - long_ends.push_back(end); - } else { - lengths.push_back(static_cast(length)); - } -} - +/// Encode merged rows, locals and content into canonical blob bytes. The +/// only entry point that writes a ShardBlob: every self-describing choice +/// (range tier, id width, mask tier, content omission) is made here, from +/// the data, so equal inputs produce equal bytes. template -void write_shard_impl(ShardView old_root, - llvm::ArrayRef id_map, - const VariantInput& fresh, - std::int64_t fresh_id, - std::vector variants, - llvm::StringRef content, - std::uint64_t content_hash, - llvm::raw_ostream& os) { - MergedRows merged; - merge_occurrences(old_root, id_map, fresh, fresh_id, merged.occurrences); - merge_relations(old_root, id_map, fresh, fresh_id, merged.relations); - - // The symbol table covers exactly what the merged rows reference: - // occurrence targets, relation group keys, and symbol payloads. - llvm::DenseSet referenced; - for(auto& row: merged.occurrences) { - referenced.insert(row.sym); - } - for(auto& [hash, rows]: merged.relations) { - referenced.insert(hash); - for(auto& row: rows) { - if(row.payload != 0 && - !RelationKind(static_cast(row.kind)).isDeclOrDef()) { - referenced.insert(row.payload); - } - } - } +void emit_blob(MergedRows& merged, + llvm::DenseMap& locals, + std::vector variants, + const ContentInfo& content, + llvm::raw_ostream& os) { + auto referenced = referenced_symbols(merged); ShardBlob blob; blob.format_version = index_format_version; - blob.content_hash = content_hash; - blob.content = content.str(); + blob.content_hash = content.hash; + blob.content_size = content.size; + blob.content = content.content.str(); + blob.line_lengths = content.line_lengths; + blob.long_line_rows = content.long_line_rows; + blob.long_line_lengths = content.long_line_lengths; blob.variants = std::move(variants); blob.sym_hashes.assign(referenced.begin(), referenced.end()); @@ -1008,48 +1167,12 @@ void write_shard_impl(ShardView old_root, blob.sym_hashes.begin()); }; - // Local symbol names: survivors from the old blob, plus the fresh - // variant's non-External symbols its rows referenced. External names - // live in the ProjectIndex and are never stored here. - struct LocalInfo { - std::string name; - std::uint8_t kind; - std::uint8_t scope; - }; - - llvm::DenseMap locals; - if(old_root.valid()) { - auto old_sym_hashes = to_array_ref(old_root[&ShardBlob::sym_hashes]); - auto old_local_syms = to_array_ref(old_root[&ShardBlob::local_syms]); - auto old_kinds = to_array_ref(old_root[&ShardBlob::local_kinds]); - auto old_scopes = to_array_ref(old_root[&ShardBlob::local_scopes]); - auto old_names = old_root[&ShardBlob::local_names]; - for(std::uint32_t k = 0; k < old_local_syms.size(); k += 1) { - auto hash = old_sym_hashes[old_local_syms[k]]; - if(referenced.contains(hash)) { - locals.try_emplace( - hash, - LocalInfo{std::string(old_names.at(k)), old_kinds[k], old_scopes[k]}); - } - } - } - if(fresh.symbols) { - for(auto hash: referenced) { - auto found = fresh.symbols(hash); - if(!found || found->scope == SymbolScope::External) { - continue; - } - locals.try_emplace(hash, - LocalInfo{std::string(found->name), - found->kind.value(), - static_cast(found->scope)}); - } - } - llvm::SmallVector> sorted_locals; sorted_locals.reserve(locals.size()); for(auto& [hash, info]: locals) { - sorted_locals.emplace_back(sym_id(hash), &info); + if(referenced.contains(hash)) { + sorted_locals.emplace_back(sym_id(hash), &info); + } } std::ranges::sort(sorted_locals, {}, [](const auto& entry) { return entry.first; }); for(auto& [id, info]: sorted_locals) { @@ -1059,28 +1182,31 @@ void write_shard_impl(ShardView old_root, blob.local_scopes.push_back(info->scope); } - auto tier = tier_of(blob.variants.size()); - bool wide_syms = blob.sym_hashes.size() > 0xffff; + auto tier = tier_of(blob.variants.empty() ? 1 : blob.variants.size()); + bool narrow = content.size <= packed_range_limit; + enum class SymWidth : std::uint8_t { U8, U16, U32 }; + auto width = blob.sym_hashes.size() <= 0x100 ? SymWidth::U8 + : blob.sym_hashes.size() <= 0x10000 ? SymWidth::U16 + : SymWidth::U32; + auto emit_sym_id = [&](std::uint32_t id, + std::vector& ids8, + std::vector& ids16, + std::vector& ids32) { + switch(width) { + case SymWidth::U8: ids8.push_back(static_cast(id)); break; + case SymWidth::U16: ids16.push_back(static_cast(id)); break; + case SymWidth::U32: ids32.push_back(id); break; + } + }; for(std::uint32_t row = 0; row < merged.occurrences.size(); row += 1) { auto& occurrence = merged.occurrences[row]; - blob.occ_begins.push_back(occurrence.begin); - emit_range(blob.occ_lengths, - blob.occ_long_rows, - blob.occ_long_ends, - row, - occurrence.begin, - occurrence.end); - auto id = sym_id(occurrence.sym); - if(wide_syms) { - blob.occ_syms32.push_back(id); - } else { - blob.occ_syms16.push_back(static_cast(id)); - } - emit_mask(blob, true, tier, occurrence.mask); + emit_row_range(blob.occs, narrow, row, occurrence.begin, occurrence.end); + emit_sym_id(sym_id(occurrence.sym), blob.occ_syms8, blob.occ_syms16, blob.occ_syms32); + emit_mask(blob.occs, tier, occurrence.mask); } if(tier == MaskTier::Roaring) { - blob.occ_roaring_offsets.push_back(static_cast(blob.occ_roaring.size())); + blob.occs.roaring_offsets.push_back(static_cast(blob.occs.roaring.size())); } // Relation groups follow symbol-table order; a symbol with occurrences @@ -1095,13 +1221,7 @@ void write_shard_impl(ShardView old_root, } for(auto& row: group->second) { blob.rel_kinds.push_back(row.kind); - blob.rel_begins.push_back(row.begin); - emit_range(blob.rel_lengths, - blob.rel_long_rows, - blob.rel_long_ends, - rel_row, - row.begin, - row.end); + emit_row_range(blob.rels, narrow, rel_row, row.begin, row.end); if(row.payload != 0) { if(RelationKind(static_cast(row.kind)).isDeclOrDef()) { auto range = std::bit_cast(row.payload); @@ -1109,38 +1229,201 @@ void write_shard_impl(ShardView old_root, blob.rel_def_begins.push_back(range.begin); blob.rel_def_ends.push_back(range.end); } else { - auto id = sym_id(row.payload); blob.rel_sym_rows.push_back(rel_row); - if(wide_syms) { - blob.rel_sym32.push_back(id); - } else { - blob.rel_sym16.push_back(static_cast(id)); - } + emit_sym_id(sym_id(row.payload), blob.rel_sym8, blob.rel_sym16, blob.rel_sym32); } } - emit_mask(blob, false, tier, row.mask); + emit_mask(blob.rels, tier, row.mask); rel_row += 1; } group += 1; } blob.sym_rel_offsets.push_back(rel_row); if(tier == MaskTier::Roaring) { - blob.rel_roaring_offsets.push_back(static_cast(blob.rel_roaring.size())); + blob.rels.roaring_offsets.push_back(static_cast(blob.rels.roaring.size())); } serialize_blob(blob, os); } +template +void merge_shards_impl(BlobView old_root, + llvm::ArrayRef id_map, + llvm::ArrayRef fresh, + std::uint32_t fresh_base, + std::vector variants, + const ContentInfo& content, + llvm::raw_ostream& os) { + auto old_mask = [&](const Ranges& columns, std::uint32_t row) { + return remap_mask(old_root, columns, row, id_map); + }; + + MergedRows merged; + + // Occurrences: concatenate every fresh blob's rows (each stamped with + // its new bit), sort, combine equal rows, then merge with the old + // rows. Fresh runs are individually sorted already; one sort over the + // concatenation keeps the merge two-way. + std::vector> fresh_occs; + for(std::uint32_t i = 0; i < fresh.size(); i += 1) { + auto root = view_of(fresh[i].bytes()); + auto bit = single_bit(fresh_base + i); + auto rows = decode_occurrences(root, [&](const Ranges&, std::uint32_t) { + return bit; + }); + fresh_occs.insert(fresh_occs.end(), + std::make_move_iterator(rows.begin()), + std::make_move_iterator(rows.end())); + } + std::ranges::sort(fresh_occs, {}, occ_key); + combine_equal(fresh_occs, occ_key); + + std::vector> old_occs; + if(old_root.valid()) { + old_occs = decode_occurrences(old_root, old_mask); + } + merge_sorted(std::move(old_occs), std::move(fresh_occs), occ_key, merged.occurrences); + + // Relations: gather fresh groups per symbol across all fresh blobs, + // then two-way merge with the old blob's groups in hash order. + llvm::DenseMap>> fresh_group_map; + for(std::uint32_t i = 0; i < fresh.size(); i += 1) { + auto root = view_of(fresh[i].bytes()); + auto bit = single_bit(fresh_base + i); + auto columns = rel_ranges(root); + for(auto& group: relation_groups(root)) { + auto rows = decode_relation_group(root, + columns, + group.begin_row, + group.end_row, + [&](const Ranges&, std::uint32_t) { + return bit; + }); + auto& into = fresh_group_map[group.hash]; + into.insert(into.end(), + std::make_move_iterator(rows.begin()), + std::make_move_iterator(rows.end())); + } + } + std::vector>>> fresh_groups; + fresh_groups.reserve(fresh_group_map.size()); + for(auto& [hash, rows]: fresh_group_map) { + std::ranges::sort(rows, {}, rel_key); + combine_equal(rows, rel_key); + fresh_groups.emplace_back(hash, std::move(rows)); + } + std::ranges::sort(fresh_groups, {}, [](const auto& group) { return group.first; }); + + std::vector old_groups; + Ranges old_columns; + if(old_root.valid()) { + old_columns = rel_ranges(old_root); + old_groups = relation_groups(old_root); + } + + auto lhs = old_groups.begin(); + auto rhs = fresh_groups.begin(); + while(lhs != old_groups.end() || rhs != fresh_groups.end()) { + if(rhs == fresh_groups.end() || (lhs != old_groups.end() && lhs->hash < rhs->first)) { + auto rows = decode_relation_group(old_root, + old_columns, + lhs->begin_row, + lhs->end_row, + old_mask); + if(!rows.empty()) { + merged.relations.emplace_back(lhs->hash, std::move(rows)); + } + lhs += 1; + } else if(lhs == old_groups.end() || rhs->first < lhs->hash) { + merged.relations.emplace_back(rhs->first, std::move(rhs->second)); + rhs += 1; + } else { + auto old_rows = decode_relation_group(old_root, + old_columns, + lhs->begin_row, + lhs->end_row, + old_mask); + std::vector> combined; + merge_sorted(std::move(old_rows), std::move(rhs->second), rel_key, combined); + if(!combined.empty()) { + merged.relations.emplace_back(lhs->hash, std::move(combined)); + } + lhs += 1; + rhs += 1; + } + } + + // Local names union: every input blob is self-contained, so the merge + // needs no external symbol resolver. First writer wins — identities of + // one symbol agree across blobs of one file. + llvm::DenseMap locals; + if(old_root.valid()) { + collect_locals(old_root, locals); + } + for(auto& shard: fresh) { + collect_locals(view_of(shard.bytes()), locals); + } + + emit_blob(merged, locals, std::move(variants), content, os); +} + } // namespace -void write_shard(const Shard& old, - llvm::ArrayRef keep, - const VariantInput& fresh, +void write_shard(const FileIndex& rows, + llvm::function_ref(SymbolHash)> symbols, llvm::StringRef content, - std::uint64_t content_hash, llvm::raw_ostream& os) { - ShardView old_root; - std::vector old_variants = old.variants(); + // Canonicalize straight from the in-memory rows: sorted, deduplicated. + // MaskT is irrelevant for a single variant (no mask columns) — use the + // cheap one. + MergedRows merged; + merged.occurrences.reserve(rows.occurrences.size()); + for(auto& occurrence: rows.occurrences) { + merged.occurrences.push_back( + {occurrence.range.begin, occurrence.range.end, occurrence.target, 1}); + } + std::ranges::sort(merged.occurrences, {}, occ_key); + combine_equal(merged.occurrences, occ_key); + + merged.relations.reserve(rows.relations.size()); + for(auto& [hash, relations]: rows.relations) { + std::vector> group; + group.reserve(relations.size()); + for(auto& relation: relations) { + group.push_back({static_cast(relation.kind), + relation.range.begin, + relation.range.end, + relation.target_symbol, + 1}); + } + std::ranges::sort(group, {}, rel_key); + combine_equal(group, rel_key); + merged.relations.emplace_back(hash, std::move(group)); + } + std::ranges::sort(merged.relations, {}, [](const auto& group) { return group.first; }); + + llvm::DenseMap locals; + if(symbols) { + for(auto hash: referenced_symbols(merged)) { + auto found = symbols(hash); + if(!found || found->scope == SymbolScope::External) { + continue; + } + locals.try_emplace(hash, + LocalInfo{std::string(found->name), + found->kind.value(), + static_cast(found->scope)}); + } + } + + emit_blob(merged, locals, {}, content_info_of(content), os); +} + +void merge_shards(const Shard& old, + llvm::ArrayRef keep, + llvm::ArrayRef fresh, + llvm::raw_ostream& os) { + auto old_variants = old.variants(); // old-id -> new-id; -1 drops the variant. llvm::SmallVector id_map(old_variants.size(), -1); @@ -1151,37 +1434,39 @@ void write_shard(const Shard& old, variants.push_back(old_variants[id]); } } + + BlobView old_root; if(!variants.empty()) { - old_root = root_of(*old.buffer); + old_root = view_of(old.bytes()); } - std::int64_t fresh_id = -1; - if(fresh.rows) { - assert(!llvm::is_contained(variants, fresh.hash) && + auto fresh_base = static_cast(variants.size()); + for(auto& shard: fresh) { + assert(shard.loaded() && "fresh shards must hold a blob"); + auto identity = shard.variants(); + assert(identity.size() == 1 && "fresh shards are single-variant worker blobs"); + assert(!llvm::is_contained(variants, identity.front()) && "a variant already stored must not be re-appended"); - fresh_id = static_cast(variants.size()); - variants.push_back(fresh.hash); + variants.push_back(identity.front()); } assert(!variants.empty() && "a shard blob holds at least one variant"); + // Content and line table travel verbatim from any input — all inputs + // share one content generation. Offsets from different generations + // must never share row storage, hence the assert. + ContentInfo content = old_root.valid() ? content_info_of(old_root) + : content_info_of(view_of(fresh.front().bytes())); + for([[maybe_unused]] auto& shard: fresh) { + assert(shard.content_hash() == content.hash && + "merge inputs must share one content generation"); + } + if(variants.size() <= 64) { - write_shard_impl(old_root, - id_map, - fresh, - fresh_id, - std::move(variants), - content, - content_hash, - os); + merge_shards_impl(old_root, id_map, fresh, fresh_base, + std::move(variants), content, os); } else { - write_shard_impl(old_root, - id_map, - fresh, - fresh_id, - std::move(variants), - content, - content_hash, - os); + merge_shards_impl(old_root, id_map, fresh, fresh_base, + std::move(variants), content, os); } } diff --git a/src/index/shard.h b/src/index/shard.h index e151c3392..749f5197c 100644 --- a/src/index/shard.h +++ b/src/index/shard.h @@ -16,109 +16,20 @@ namespace clice::index { -/// Identity of one preprocessing variant of a file: xxh3 over the file's -/// rows in canonical order (see FileIndex::rows_hash). Two compilation -/// contexts whose preprocessing of the file agrees produce the same value -/// and share one stored variant. +/// Identity of one variant of a file's rows: xxh3_64 over the encoded +/// single-variant blob bytes the worker produced. Encoding is canonical +/// and deterministic, so two compilation contexts whose preprocessing of +/// the file agrees produce byte-identical blobs and share one identity. using RowsHash = std::uint64_t; -/// A file's persisted index rows: every variant produced from one content -/// generation, merged so that a row shared by several variants is stored -/// once. The blob knows nothing about which TU contributed which variant — -/// that mapping is global state (ProjectIndex) — so re-indexing a TU whose -/// rows are unchanged never touches the blob. -/// -/// Columnar layout, chosen at serialize time from known bounds and -/// self-described by which columns are populated: -/// - symbol ids: u16 when the table has at most 65535 entries, else u32 -/// - row ranges: begin u32 + length u8, lengths >= 255 escape to a -/// sparse (row, end) table -/// - variant masks: absent when there is one variant, u32 up to 32, -/// u64 up to 64, concatenated roaring bitmaps beyond -/// -/// Relations are grouped by symbol: `sym_rel_offsets` slices the relation -/// columns per symbol-table entry, so a per-symbol lookup is a binary -/// search plus a slice walk. A relation's payload (`Relation::target_symbol`) -/// is sparse by class: a definition range for decl/def rows that recorded -/// one, a symbol reference for symbol-pair rows, nothing otherwise. -struct ShardBlob { - /// Persisted-blob schema version (index_format_version), stamped by the - /// writer and gated by Shard::from_bytes. - std::uint32_t format_version = 0; - - /// xxh3 of `content`: the content generation these rows were built - /// from. A variant produced from different bytes of the file starts a - /// new blob instead of merging in — offsets from two generations must - /// never share row storage. - std::uint64_t content_hash = 0; - - /// The file's text, for position mapping. Line starts are derived at - /// load time, never persisted. - std::string content; - - /// Variant id (mask bit position) -> rows hash. - std::vector variants; - - /// Referenced symbols, sorted by hash; the index into this table is the - /// symbol id the row columns use. - std::vector sym_hashes; - - /// Relation-column slice per symbol: entry i's relations occupy rows - /// [sym_rel_offsets[i], sym_rel_offsets[i + 1]). Size is table size + 1. - std::vector sym_rel_offsets; - - /// Symbols local to this file (FileLocal) or its TU (TULocal), whose - /// names live nowhere else: sparse over the symbol table, ascending. - std::vector local_syms; - std::vector local_names; - std::vector local_kinds; - std::vector local_scopes; - - /// Occurrences sorted by (begin, end, symbol hash). - std::vector occ_begins; - std::vector occ_lengths; - std::vector occ_long_rows; - std::vector occ_long_ends; - std::vector occ_syms16; - std::vector occ_syms32; - std::vector occ_masks32; - std::vector occ_masks64; - std::vector occ_roaring_offsets; - std::vector occ_roaring; - - /// Relations in symbol-table order, sorted by (kind, begin, end) within - /// each group. - std::vector rel_kinds; - std::vector rel_begins; - std::vector rel_lengths; - std::vector rel_long_rows; - std::vector rel_long_ends; - std::vector rel_sym_rows; - std::vector rel_sym16; - std::vector rel_sym32; - std::vector rel_def_rows; - std::vector rel_def_begins; - std::vector rel_def_ends; - std::vector rel_masks32; - std::vector rel_masks64; - std::vector rel_roaring_offsets; - std::vector rel_roaring; -}; - -/// A variant to write into a shard blob: the rows plus a resolver for the -/// symbols they reference. Only non-External entries land in the blob's -/// local-name table (External names live in the ProjectIndex); returned -/// name refs must stay valid for the duration of the write call. -struct VariantInput { - RowsHash hash = 0; - const FileIndex* rows = nullptr; - llvm::function_ref(SymbolHash)> symbols; -}; - /// Zero-copy reader over a shard blob, plus the live-variant mask the /// indexer maintains: a variant whose last contributing TU was removed or /// replaced stops serving immediately, and its rows are erased for real by -/// the next write_shard covering the blob. +/// the next merge_shards covering the blob. +/// +/// The same blob encoding serves every holder of a file's rows: the +/// worker's per-file product travelling in an indexing result, the disk +/// shard, a session's main-file rows and the preamble index entries. class Shard { public: Shard() = default; @@ -144,9 +55,26 @@ class Shard { return buffer ? buffer->getBuffer() : llvm::StringRef(); } + /// xxh3 of the content bytes the rows were built from (the content + /// generation), not of the blob. std::uint64_t content_hash() const; - /// All variants stored in the blob, in variant-id order. + /// Size of the content the rows were built from; the upper bound of + /// every stored range. + std::uint32_t content_size() const; + + /// The file's text. Empty for pure-ASCII content, which is not + /// stored: byte offsets are already UTF-16 column offsets, so + /// position mapping needs only line_starts(), and text previews + /// re-read the file from disk under a content_hash check. + llvm::StringRef content() const; + + /// Whether the content is pure ASCII (and therefore not stored). + bool ascii() const; + + /// All variant identities stored in the blob, in mask-bit order. A + /// worker-emitted blob holds one anonymous variant identified by its + /// own byte hash. std::vector variants() const; bool has_variant(RowsHash hash) const; @@ -168,22 +96,13 @@ class Shard { /// Look up a local symbol's name and kind. bool find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const; - llvm::StringRef content() const; - - /// Line start offsets for position mapping, derived from the content on - /// first use. + /// Line start offsets for position mapping, materialized from the + /// blob's line-length columns on first use. std::span line_starts() const; private: explicit Shard(std::unique_ptr buffer); - friend void write_shard(const Shard& old, - llvm::ArrayRef keep, - const VariantInput& fresh, - llvm::StringRef content, - std::uint64_t content_hash, - llvm::raw_ostream& os); - struct Live { /// Fast path: every stored variant is live, no per-row filtering. bool all = true; @@ -195,21 +114,37 @@ class Shard { std::unique_ptr buffer; Live live; - /// Lazily derived from the blob's content; content is immutable for the - /// shard's lifetime, so the cache never invalidates. + /// Byte hash of the blob — the identity of a worker-emitted blob's + /// single anonymous variant. Computed once at load. + std::uint64_t blob_hash = 0; + /// Lazily materialized from the line-length columns; the blob is + /// immutable for the shard's lifetime, so the cache never invalidates. mutable std::vector line_starts_cache; }; -/// Write a shard blob for one content generation: the variants of `old` -/// whose hash is in `keep` (in stored order), then `fresh` if its rows are -/// non-null. Rows shared between variants merge; masks are re-encoded for -/// the surviving variant set. `old` may be an empty shard (fresh build) and -/// `fresh.rows` may be null (pure compaction). -void write_shard(const Shard& old, - llvm::ArrayRef keep, - const VariantInput& fresh, +/// Encode one variant's rows as a self-contained single-variant blob — +/// the worker's per-file product. Rows are canonicalized (sorted, +/// deduplicated) and the encoding is deterministic: the blob's byte hash +/// is the variant's identity. `symbols` resolves referenced symbols so +/// non-External names land in the blob's local-name table (External names +/// live in the ProjectIndex); returned name refs must stay valid for the +/// duration of the call. `content` is the text the indexing compile +/// consumed; pure-ASCII content is hashed and measured but not stored. +void write_shard(const FileIndex& rows, + llvm::function_ref(SymbolHash)> symbols, llvm::StringRef content, - std::uint64_t content_hash, llvm::raw_ostream& os); +/// Merge blobs of one file and one content generation: the variants of +/// `old` whose identity is in `keep` (in stored order), then each blob of +/// `fresh` as one new variant. Rows shared between variants merge into +/// one row; masks are re-encoded for the surviving variant set. `old` may +/// be an empty shard (all-fresh merge) and `fresh` may be empty (pure +/// compaction); every fresh shard must be a loaded single-variant blob of +/// the same content generation as the other inputs. +void merge_shards(const Shard& old, + llvm::ArrayRef keep, + llvm::ArrayRef fresh, + llvm::raw_ostream& os); + } // namespace clice::index diff --git a/src/index/shard_layout.h b/src/index/shard_layout.h new file mode 100644 index 000000000..496860b7e --- /dev/null +++ b/src/index/shard_layout.h @@ -0,0 +1,138 @@ +#pragma once + +/// Internal: the persisted layout of a shard blob and the encoding rules +/// shared by its writer, reader and tests. Everything else consumes shards +/// through the `Shard` reader in shard.h — include this header only to +/// build blob bytes by hand (the writer, corruption tests). + +#include +#include +#include + +#include "index/shard.h" + +namespace clice::index { + +/// Sentinel in a length column (row ranges, line lengths): the real end +/// lives in the sparse escape table. +constexpr inline std::uint8_t length_escape = 0xff; + +/// Files whose content fits 24-bit offsets use the packed range column; +/// larger files fall back to the wide begin/length columns. +constexpr inline std::uint32_t packed_range_limit = 0xffffff; + +/// One row range in the packed column: begin in the high 24 bits, length +/// in the low 8. Raw u32 order equals (begin, length) lexicographic order. +constexpr inline std::uint32_t pack_range(std::uint32_t begin, std::uint8_t length) { + return (begin << 8) | length; +} + +/// The packed spelling of the no-range sentinel pair relations carry (the +/// default LocalSourceRange, ~0u/~0u). Unambiguous: a real packed row with +/// begin 0xffffff and an escaped length would need an end at least 255 +/// past a begin that already sits at the content limit. +constexpr inline std::uint32_t packed_sentinel = 0xffffffff; + +/// One side of the blob's row storage (occurrences or relations). +/// +/// Ranges use exactly one of two self-describing tiers: +/// - packed: `(begin << 8) | length` per row (content < 16MB) +/// - wide: begin u32 + length u8 columns +/// Lengths >= 255 escape to the sparse (row, end) table in either tier. +/// +/// Variant masks are absent for a single variant, then u32 / u64 / +/// concatenated roaring bitmaps by variant count. +struct RowRanges { + std::vector packed; + + std::vector begins; + std::vector lengths; + + std::vector long_rows; + std::vector long_ends; + + std::vector masks32; + std::vector masks64; + std::vector roaring_offsets; + std::vector roaring; +}; + +/// A file's index rows as one persisted blob. +/// +/// The worker encodes one blob per file per TU: single variant (an empty +/// `variants` table), self-contained (content, line table, local symbol +/// names), canonical byte-for-byte — its identity is the hash of its +/// bytes, computed by whoever holds them, never stored inside. The master +/// stores first variants verbatim and merges only when a second distinct +/// variant of the same content generation arrives; merged blobs list the +/// original single-variant identities in `variants` (mask bit position -> +/// identity) and deduplicate rows shared between variants via the masks. +/// +/// Symbol ids used by the row columns index `sym_hashes`; the id column +/// width is chosen from the table size (u8 / u16 / u32). +struct ShardBlob { + /// Persisted-blob schema version (index_format_version), stamped by + /// the writer and gated by Shard::from_bytes. + std::uint32_t format_version = 0; + + /// xxh3 of the content bytes the indexing compile consumed — the + /// content generation these rows were built from. Variants of + /// different generations never share a blob. + std::uint64_t content_hash = 0; + + /// Size of the consumed content; bounds every stored range. + std::uint32_t content_size = 0; + + /// The file's text, for UTF-16 position mapping and text previews. + /// Empty when the content is pure ASCII: byte offsets are already + /// UTF-16 column offsets, so the text itself is dead weight. The form + /// is canonical — a blob storing pure-ASCII content is invalid. + std::string content; + + /// Mask bit position -> variant identity. Empty for a worker-emitted + /// blob (one anonymous variant, identified by its own byte hash). + std::vector variants; + + /// Per-line byte lengths, up to and including the newline; the last + /// entry runs to end of file. Lengths >= 255 escape to the sparse + /// (line, length) table. Line starts are the prefix sums, materialized + /// once at load. + std::vector line_lengths; + std::vector long_line_rows; + std::vector long_line_lengths; + + /// Referenced symbols, sorted by hash; the index into this table is + /// the symbol id the row columns use. + std::vector sym_hashes; + + /// Relation slice per symbol: entry i's relations occupy rows + /// [sym_rel_offsets[i], sym_rel_offsets[i + 1]). Size is table size + 1. + std::vector sym_rel_offsets; + + /// Symbols local to this file (FileLocal) or its TU (TULocal), whose + /// names live nowhere else: sparse over the symbol table, ascending. + std::vector local_syms; + std::vector local_names; + std::vector local_kinds; + std::vector local_scopes; + + /// Occurrences sorted by (begin, end, symbol hash). + RowRanges occs; + std::vector occ_syms8; + std::vector occ_syms16; + std::vector occ_syms32; + + /// Relations in symbol-table order, sorted by (kind, begin, end, + /// payload) within each group. + RowRanges rels; + std::vector rel_kinds; + std::vector rel_sym_rows; + std::vector rel_sym8; + std::vector rel_sym16; + std::vector rel_sym32; + std::vector rel_def_rows; + std::vector rel_def_begins; + std::vector rel_def_ends; +}; + +} // namespace clice::index diff --git a/tests/unit/index/shard_tests.cpp b/tests/unit/index/shard_tests.cpp index 2efe49178..1bb403b96 100644 --- a/tests/unit/index/shard_tests.cpp +++ b/tests/unit/index/shard_tests.cpp @@ -7,8 +7,10 @@ #include "test/tester.h" #include "index/serialization.h" #include "index/shard.h" +#include "index/shard_layout.h" #include "index/tu_index.h" +#include "kota/ipc/lsp/text.h" #include "llvm/Support/MemoryBuffer.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Support/xxhash.h" @@ -36,33 +38,18 @@ std::optional lookup_symbol(index::SymbolHash hash) { } std::string write_fresh(const index::FileIndex& rows, - index::RowsHash hash, llvm::StringRef content, bool with_symbols = false) { - auto resolve = [this](index::SymbolHash symbol) { + llvm::function_ref(index::SymbolHash)> resolve; + auto resolver = [this](index::SymbolHash symbol) { return lookup_symbol(symbol); }; - index::VariantInput fresh{hash, &rows, {}}; if(with_symbols) { - fresh.symbols = resolve; + resolve = resolver; } std::string bytes; llvm::raw_string_ostream os(bytes); - index::write_shard(index::Shard(), {}, fresh, content, llvm::xxh3_64bits(content), os); - return bytes; -} - -std::string append_variant(const index::Shard& old, - const index::FileIndex& rows, - index::RowsHash hash) { - std::string bytes; - llvm::raw_string_ostream os(bytes); - index::write_shard(old, - old.variants(), - {hash, &rows, {}}, - old.content(), - old.content_hash(), - os); + index::write_shard(rows, resolve, content, os); return bytes; } @@ -72,6 +59,25 @@ index::Shard make_shard(llvm::StringRef bytes) { return index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); } +index::Shard merge(const index::Shard& old, + llvm::ArrayRef keep, + std::vector fresh) { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::merge_shards(old, keep, fresh, os); + return make_shard(bytes); +} + +/// `content` must repeat the text the shard was built from — ASCII blobs +/// do not store it, so it cannot be recovered from `old`. +index::Shard append_variant(const index::Shard& old, + const index::FileIndex& rows, + llvm::StringRef content) { + std::vector fresh; + fresh.push_back(make_shard(write_fresh(rows, content))); + return merge(old, old.variants(), std::move(fresh)); +} + index::SymbolHash hash_at(const index::Shard& shard, std::uint32_t offset) { index::SymbolHash result = 0; shard.lookup(offset, [&](const index::Occurrence& o) { @@ -81,6 +87,12 @@ index::SymbolHash hash_at(const index::Shard& shard, std::uint32_t offset) { return result; } +index::FileIndex simple_rows(std::initializer_list occurrences) { + index::FileIndex rows; + rows.occurrences = occurrences; + return rows; +} + TEST_CASE(RoundtripLookups) { build_index(R"( int §(def)⟦§(def)foo⟧() { return 42; } @@ -88,11 +100,10 @@ TEST_CASE(RoundtripLookups) { )"); auto content = sources.all_files.find("main.cpp")->second.content; - auto bytes = - write_fresh(tu_index.main_file_index, tu_index.main_file_index.rows_hash(), content); + auto bytes = write_fresh(tu_index.main_file_index, content); auto shard = make_shard(bytes); ASSERT_TRUE(shard.loaded()); - ASSERT_EQ(shard.content(), llvm::StringRef(content)); + ASSERT_EQ(shard.content_size(), static_cast(content.size())); ASSERT_FALSE(shard.line_starts().empty()); auto expected = range("ref"); @@ -117,10 +128,103 @@ TEST_CASE(RoundtripLookups) { ASSERT_TRUE(has_definition); } -index::FileIndex simple_rows(std::initializer_list occurrences) { +TEST_CASE(DeterministicEncoding) { + // The blob's byte hash is the variant's identity, so equal rows must + // encode to equal bytes regardless of in-memory insertion order. index::FileIndex rows; - rows.occurrences = occurrences; - return rows; + rows.occurrences = { + {{0, 3}, 111}, + {{10, 13}, 222}, + {{20, 23}, 111}, + }; + rows.relations[111] = { + {.kind = RelationKind::Definition, .range = {0, 3}, .target_symbol = 0}, + {.kind = RelationKind::Reference, .range = {20, 23}, .target_symbol = 0}, + }; + rows.relations[333] = { + {.kind = RelationKind::Base, .range = {10, 13}, .target_symbol = 444}, + }; + + index::FileIndex shuffled; + shuffled.occurrences = {rows.occurrences[2], rows.occurrences[0], rows.occurrences[1]}; + shuffled.relations[333] = rows.relations[333]; + shuffled.relations[111] = {rows.relations[111][1], rows.relations[111][0]}; + + auto content = "aaa bbb ccc ddd 111 222 333"; + ASSERT_EQ(write_fresh(rows, content), write_fresh(shuffled, content)); +} + +TEST_CASE(AnonymousVariantIdentity) { + auto rows = simple_rows({ + {{0, 3}, 111} + }); + auto bytes = write_fresh(rows, "aaa bbb"); + auto shard = make_shard(bytes); + + auto variants = shard.variants(); + ASSERT_EQ(variants.size(), std::size_t(1)); + ASSERT_EQ(variants.front(), llvm::xxh3_64bits(bytes)); + ASSERT_TRUE(shard.has_variant(variants.front())); +} + +TEST_CASE(AsciiContentOmitted) { + std::string content = "int x;\nint y;\n"; + auto rows = simple_rows({ + {{4, 5}, 111} + }); + auto shard = make_shard(write_fresh(rows, content)); + + ASSERT_TRUE(shard.ascii()); + ASSERT_TRUE(shard.content().empty()); + ASSERT_EQ(shard.content_size(), static_cast(content.size())); + ASSERT_EQ(shard.content_hash(), llvm::xxh3_64bits(content)); + ASSERT_EQ(hash_at(shard, 4), 111u); + + auto expected = kota::ipc::lsp::build_line_starts(content); + auto starts = shard.line_starts(); + ASSERT_EQ(std::vector(starts.begin(), starts.end()), expected); +} + +TEST_CASE(NonAsciiContentStored) { + std::string content = "int å;\nint y;\n"; + auto rows = simple_rows({ + {{4, 6}, 111} + }); + auto shard = make_shard(write_fresh(rows, content)); + + ASSERT_FALSE(shard.ascii()); + ASSERT_EQ(shard.content(), llvm::StringRef(content)); + + auto expected = kota::ipc::lsp::build_line_starts(content); + auto starts = shard.line_starts(); + ASSERT_EQ(std::vector(starts.begin(), starts.end()), expected); +} + +TEST_CASE(LongLineEscape) { + // A line past 255 bytes escapes to the sparse table; the materialized + // starts must match a direct scan of the content. + std::string content = "short\n" + std::string(300, 'a') + "\nshort again\n"; + auto rows = simple_rows({ + {{0, 5}, 111} + }); + auto shard = make_shard(write_fresh(rows, content)); + + auto expected = kota::ipc::lsp::build_line_starts(content); + auto starts = shard.line_starts(); + ASSERT_EQ(std::vector(starts.begin(), starts.end()), expected); +} + +TEST_CASE(WideRangeTier) { + // Past 16MB of content the packed range column cannot hold begins; + // the wide tier takes over transparently. + std::string content(index::packed_range_limit + 64, 'w'); + auto rows = simple_rows({ + {{0, 3}, 111}, + {{index::packed_range_limit + 8, index::packed_range_limit + 11}, 222}, + }); + auto shard = make_shard(write_fresh(rows, content)); + ASSERT_EQ(hash_at(shard, 1), 111u); + ASSERT_EQ(hash_at(shard, index::packed_range_limit + 9), 222u); } TEST_CASE(VariantMaskFiltering) { @@ -132,23 +236,23 @@ TEST_CASE(VariantMaskFiltering) { {{10, 13}, 222} }); - auto first = make_shard(write_fresh(a, 1, "aaa bbb ccc ddd")); - auto shard = make_shard(append_variant(first, b, 2)); - ASSERT_TRUE(shard.has_variant(1)); - ASSERT_TRUE(shard.has_variant(2)); + auto first = make_shard(write_fresh(a, "aaa bbb ccc ddd")); + auto shard = append_variant(first, b, "aaa bbb ccc ddd"); + auto variants = shard.variants(); + ASSERT_EQ(variants.size(), std::size_t(2)); // All variants live by default: both rows serve. ASSERT_EQ(hash_at(shard, 1), 111u); ASSERT_EQ(hash_at(shard, 11), 222u); - // Restricting to variant 1 hides the row only variant 2 holds, while - // the shared row keeps serving. - shard.set_live({1}); + // Restricting to the first variant hides the row only the second + // holds, while the shared row keeps serving. + shard.set_live({variants[0]}); ASSERT_TRUE(shard.has_dead_variants()); ASSERT_EQ(hash_at(shard, 1), 111u); ASSERT_EQ(hash_at(shard, 11), 0u); - shard.set_live({1, 2}); + shard.set_live(variants); ASSERT_FALSE(shard.has_dead_variants()); ASSERT_EQ(hash_at(shard, 11), 222u); @@ -164,19 +268,43 @@ TEST_CASE(CompactionDropsVariant) { {{0, 3}, 111}, {{10, 13}, 222} }); - auto first = make_shard(write_fresh(a, 1, "aaa bbb ccc ddd")); - auto both = make_shard(append_variant(first, b, 2)); + auto first = make_shard(write_fresh(a, "aaa bbb ccc ddd")); + auto both = append_variant(first, b, "aaa bbb ccc ddd"); + auto variants = both.variants(); - std::string bytes; - llvm::raw_string_ostream os(bytes); - index::write_shard(both, {1}, {}, both.content(), both.content_hash(), os); - auto compacted = make_shard(bytes); - ASSERT_TRUE(compacted.has_variant(1)); - ASSERT_FALSE(compacted.has_variant(2)); + auto compacted = merge(both, {variants[0]}, {}); + ASSERT_TRUE(compacted.has_variant(variants[0])); + ASSERT_FALSE(compacted.has_variant(variants[1])); ASSERT_EQ(hash_at(compacted, 1), 111u); ASSERT_EQ(hash_at(compacted, 11), 0u); } +TEST_CASE(KWayMerge) { + // Several fresh variants land in one write; shared rows collapse with + // OR-ed masks and each unique row stays filterable to its owner. + std::string content = "aaa bbb ccc ddd eee"; + std::vector fresh; + for(std::uint32_t i = 0; i < 3; i += 1) { + auto rows = simple_rows({ + {{0, 3}, 111 }, + {{4 * (i + 1), 4 * (i + 1) + 3}, 1000 + i}, + }); + fresh.push_back(make_shard(write_fresh(rows, content))); + } + auto shard = merge(index::Shard(), {}, std::move(fresh)); + + auto variants = shard.variants(); + ASSERT_EQ(variants.size(), std::size_t(3)); + for(std::uint32_t i = 0; i < 3; i += 1) { + ASSERT_EQ(hash_at(shard, 4 * (i + 1) + 1), 1000u + i); + } + + shard.set_live({variants[1]}); + ASSERT_EQ(hash_at(shard, 1), 111u); + ASSERT_EQ(hash_at(shard, 8 + 1), 1001u); + ASSERT_EQ(hash_at(shard, 4 + 1), 0u); +} + /// Grow a shard to `count` variants: variant i holds the shared occurrence /// and relation plus a unique one of each at offset i * 16. index::Shard grow_variants(std::uint32_t count) { @@ -191,8 +319,14 @@ index::Shard grow_variants(std::uint32_t count) { {.kind = RelationKind::Reference, .range = {0, 3}, .target_symbol = 0}, {.kind = RelationKind::Reference, .range = {i * 16, i * 16 + 3}, .target_symbol = 0}, }; - shard = make_shard(shard.loaded() ? append_variant(shard, rows, i) - : write_fresh(rows, i, content)); + auto fresh = make_shard(write_fresh(rows, content)); + if(!shard.loaded()) { + shard = std::move(fresh); + } else { + std::vector batch; + batch.push_back(std::move(fresh)); + shard = merge(shard, shard.variants(), std::move(batch)); + } } return shard; } @@ -208,7 +342,8 @@ std::size_t reference_count(const index::Shard& shard, index::SymbolHash symbol) void expect_tier_behavior(std::uint32_t count) { auto shard = grow_variants(count); - ASSERT_EQ(shard.variants().size(), std::size_t(count)); + auto variants = shard.variants(); + ASSERT_EQ(variants.size(), std::size_t(count)); // Every variant's unique row serves under the full live set, and the // shared relation collapsed to one row across all variants. @@ -219,7 +354,7 @@ void expect_tier_behavior(std::uint32_t count) { // One live variant: its unique rows and the shared rows serve, another // variant's do not — on the occurrence and the relation side alike. - shard.set_live({3}); + shard.set_live({variants[2]}); ASSERT_EQ(hash_at(shard, 1), 111u); ASSERT_EQ(hash_at(shard, 3 * 16 + 1), 1003u); ASSERT_EQ(hash_at(shard, 5 * 16 + 1), 0u); @@ -244,7 +379,7 @@ TEST_CASE(LongTokenEscape) { {{400, 404}, 222} }); std::string content(500, 'y'); - auto shard = make_shard(write_fresh(rows, 1, content)); + auto shard = make_shard(write_fresh(rows, content)); bool found = false; shard.lookup(299, [&](const index::Occurrence& o) { @@ -271,7 +406,7 @@ TEST_CASE(RelationPayloadRoundtrip) { {.kind = RelationKind::Base, .range = {20, 23}, .target_symbol = 444}, }; - auto shard = make_shard(write_fresh(rows, 1, std::string(60, 'z'))); + auto shard = make_shard(write_fresh(rows, std::string(60, 'z'))); bool checked_definition = false; shard.lookup(111, RelationKind::Definition, [&](const index::Relation& r) { @@ -307,10 +442,7 @@ TEST_CASE(LocalSymbolNames) { )"); auto content = sources.all_files.find("main.cpp")->second.content; - auto shard = make_shard(write_fresh(tu_index.main_file_index, - tu_index.main_file_index.rows_hash(), - content, - /*with_symbols=*/true)); + auto shard = make_shard(write_fresh(tu_index.main_file_index, content, /*with_symbols=*/true)); auto local = hash_at(shard, point("use")); ASSERT_TRUE(local != 0); @@ -332,8 +464,31 @@ TEST_CASE(LocalSymbolNames) { ASSERT_FALSE(shard.find_symbol(external, name, kind)); } +TEST_CASE(MergedLocalNames) { + // Merged blobs carry local names forward from their inputs without any + // external resolver — every input blob is self-contained. + build_index(R"( + static int §(local)⟦§(local)helper⟧() { return 1; } + int visible() { return helper(); } + )"); + auto content = sources.all_files.find("main.cpp")->second.content; + auto first = make_shard(write_fresh(tu_index.main_file_index, content, /*with_symbols=*/true)); + + auto extra = simple_rows({ + {{0, 3}, 424242} + }); + auto shard = append_variant(first, extra, content); + + auto local = hash_at(shard, point("local")); + ASSERT_TRUE(local != 0); + std::string name; + SymbolKind kind; + ASSERT_TRUE(shard.find_symbol(local, name, kind)); + ASSERT_EQ(name, "helper"); +} + TEST_CASE(WideSymbolIds) { - // Past 65535 distinct symbols the id columns must widen to u32; a + // Past 65536 distinct symbols the id columns must widen to u32; a // truncating writer corrupts resolution only on indexes this large. index::FileIndex rows; constexpr std::uint32_t count = 70000; @@ -345,7 +500,7 @@ TEST_CASE(WideSymbolIds) { }); } std::string content(count * 8 + 16, 'w'); - auto shard = make_shard(write_fresh(rows, 1, content)); + auto shard = make_shard(write_fresh(rows, content)); ASSERT_EQ(hash_at(shard, 69999 * 8 + 1), 0x100000u + 69999); ASSERT_EQ(hash_at(shard, 3 * 8 + 1), 0x100000u + 3); } @@ -361,6 +516,30 @@ TEST_CASE(UnloadedShardNoops) { ASSERT_TRUE(shard.line_starts().empty()); } +/// Fill the content identity and line table of a hand-built blob the way +/// the writer would (ASCII omitted, non-ASCII stored). +void fill_content(index::ShardBlob& blob, llvm::StringRef text) { + blob.content_hash = llvm::xxh3_64bits(text); + blob.content_size = static_cast(text.size()); + bool is_ascii = llvm::all_of(text, [](char c) { return static_cast(c) < 0x80; }); + blob.content = is_ascii ? std::string() : text.str(); + blob.line_lengths.clear(); + blob.long_line_rows.clear(); + blob.long_line_lengths.clear(); + auto starts = kota::ipc::lsp::build_line_starts(std::string_view(text.data(), text.size())); + for(std::size_t i = 0; i < starts.size(); i += 1) { + auto next = i + 1 < starts.size() ? starts[i + 1] : blob.content_size; + auto length = next - starts[i]; + if(length >= index::length_escape) { + blob.line_lengths.push_back(index::length_escape); + blob.long_line_rows.push_back(static_cast(i)); + blob.long_line_lengths.push_back(length); + } else { + blob.line_lengths.push_back(static_cast(length)); + } + } +} + TEST_CASE(CorruptBlobRejected) { ASSERT_FALSE(index::Shard::from_bytes("not a flatbuffer").loaded()); @@ -370,12 +549,12 @@ TEST_CASE(CorruptBlobRejected) { auto rows = simple_rows({ {{0, 3}, 111} }); - auto bytes = write_fresh(rows, 1, "aaaa"); + auto bytes = write_fresh(rows, "aaaa"); ASSERT_FALSE( index::Shard::from_bytes(llvm::StringRef(bytes).take_front(bytes.size() / 2)).loaded()); - // A structurally valid blob of the current version but with no variants - // is impossible output of the writer, and must not load either. + // A structurally valid table of the current version with no line table + // at all cannot be writer output. struct VersionOnly { std::uint32_t format_version = 0; }; @@ -392,8 +571,52 @@ TEST_CASE(ContentHashMismatchRejected) { // as fresh while position mapping reads the wrong text. index::ShardBlob blob; blob.format_version = index::index_format_version; + fill_content(blob, "aaåå"); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.content = "aaåb"; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(StoredAsciiContentRejected) { + // The encoding is canonical — one logical blob, one byte image — so + // pure-ASCII content stored in full is an invalid second spelling of + // the omitted form. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + fill_content(blob, "aaaa"); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + blob.content = "aaaa"; - blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(LineTableMismatchRejected) { + // Line starts are prefix sums of the length column; a sum drifting off + // the content size would shift every position mapping below the drift. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + fill_content(blob, "aaa\nbbb\n"); blob.variants = {1}; blob.sym_hashes = {111}; blob.sym_rel_offsets = {0, 0}; @@ -406,7 +629,10 @@ TEST_CASE(ContentHashMismatchRejected) { }; ASSERT_TRUE(make_shard(bytes_of()).loaded()); - blob.content = "aaab"; + blob.line_lengths = {4, 3}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + blob.line_lengths = {}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); } @@ -416,16 +642,14 @@ TEST_CASE(MisorderedRowsRejected) { // rebuilt, not keep misresolving queries on every restart. index::ShardBlob blob; blob.format_version = index::index_format_version; - blob.content = "aaaaaaaaaaaaaaaa"; - blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + fill_content(blob, "aaaaaaaaaaaaaaaa"); blob.variants = {1}; blob.sym_hashes = {111}; blob.sym_rel_offsets = {0, 0}; - blob.occ_begins = {0, 8}; - blob.occ_lengths = {3, 0xff}; // 0xff escapes to (row, end) - blob.occ_long_rows = {1}; - blob.occ_long_ends = {12}; - blob.occ_syms16 = {0, 0}; + blob.occs.packed = {index::pack_range(0, 3), index::pack_range(8, index::length_escape)}; + blob.occs.long_rows = {1}; + blob.occs.long_ends = {12}; + blob.occ_syms8 = {0, 0}; auto bytes_of = [&] { std::string bytes; @@ -436,20 +660,20 @@ TEST_CASE(MisorderedRowsRejected) { ASSERT_TRUE(make_shard(bytes_of()).loaded()); // Begins out of order. - blob.occ_begins = {8, 0}; + blob.occs.packed = {index::pack_range(8, index::length_escape), index::pack_range(0, 3)}; + blob.occs.long_rows = {0}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); // Begins sorted, but the escaped end regresses below the row before. - blob.occ_begins = {0, 8}; - blob.occ_lengths = {0xff, 3}; - blob.occ_long_rows = {0}; - blob.occ_long_ends = {14}; // ends decode to {14, 11} + blob.occs.packed = {index::pack_range(0, index::length_escape), index::pack_range(8, 3)}; + blob.occs.long_rows = {0}; + blob.occs.long_ends = {14}; // ends decode to {14, 11} ASSERT_FALSE(make_shard(bytes_of()).loaded()); // An escaped end before its own begin. - blob.occ_lengths = {3, 0xff}; - blob.occ_long_rows = {1}; - blob.occ_long_ends = {5}; // row 1: begin 8, end 5 + blob.occs.packed = {index::pack_range(0, 3), index::pack_range(8, index::length_escape)}; + blob.occs.long_rows = {1}; + blob.occs.long_ends = {5}; // row 1: begin 8, end 5 ASSERT_FALSE(make_shard(bytes_of()).loaded()); } @@ -460,16 +684,14 @@ TEST_CASE(EscapeTableMismatchRejected) { // wrong ranges forever, so only the pairing check can reject them. index::ShardBlob blob; blob.format_version = index::index_format_version; - blob.content = std::string(300, 'a'); - blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + fill_content(blob, std::string(300, 'a')); blob.variants = {1}; blob.sym_hashes = {111}; blob.sym_rel_offsets = {0, 0}; - blob.occ_begins = {0}; - blob.occ_lengths = {0xff}; - blob.occ_long_rows = {0}; - blob.occ_long_ends = {260}; - blob.occ_syms16 = {0}; + blob.occs.packed = {index::pack_range(0, index::length_escape)}; + blob.occs.long_rows = {0}; + blob.occs.long_ends = {260}; + blob.occ_syms8 = {0}; auto bytes_of = [&] { std::string bytes; @@ -480,43 +702,39 @@ TEST_CASE(EscapeTableMismatchRejected) { ASSERT_TRUE(make_shard(bytes_of()).loaded()); // A sentinel without its sparse entry. - blob.occ_long_rows = {}; - blob.occ_long_ends = {}; + blob.occs.long_rows = {}; + blob.occs.long_ends = {}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); // A sparse entry pointing at an unescaped row. - blob.occ_lengths = {3}; - blob.occ_long_rows = {0}; - blob.occ_long_ends = {260}; + blob.occs.packed = {index::pack_range(0, 3)}; + blob.occs.long_rows = {0}; + blob.occs.long_ends = {260}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); // The relation escape table is validated alike. - blob.occ_lengths = {0xff}; + blob.occs.packed = {index::pack_range(0, index::length_escape)}; blob.sym_rel_offsets = {0, 1}; blob.rel_kinds = {static_cast(RelationKind::Reference)}; - blob.rel_begins = {0}; - blob.rel_lengths = {0xff}; + blob.rels.packed = {index::pack_range(0, index::length_escape)}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); } TEST_CASE(RangesBeyondContentRejected) { - // Every decoded range is served as a source range into the stored - // content; an end past it would map positions through text that does - // not exist — forever, since the blob's content hash still matches the - // disk and nothing rebuilds it. + // Every decoded range is served as a source range into the content; an + // end past it would map positions through text that does not exist — + // forever, since the blob's content hash still matches the disk and + // nothing rebuilds it. index::ShardBlob blob; blob.format_version = index::index_format_version; - blob.content = "aaaaaaaaaaaaaaaa"; - blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + fill_content(blob, "aaaaaaaaaaaaaaaa"); blob.variants = {1}; blob.sym_hashes = {111}; blob.sym_rel_offsets = {0, 1}; - blob.occ_begins = {0}; - blob.occ_lengths = {3}; - blob.occ_syms16 = {0}; + blob.occs.packed = {index::pack_range(0, 3)}; + blob.occ_syms8 = {0}; blob.rel_kinds = {static_cast(RelationKind::Reference)}; - blob.rel_begins = {0}; - blob.rel_lengths = {3}; + blob.rels.packed = {index::pack_range(0, 3)}; auto bytes_of = [&] { std::string bytes; @@ -527,32 +745,30 @@ TEST_CASE(RangesBeyondContentRejected) { ASSERT_TRUE(make_shard(bytes_of()).loaded()); // A plain length overruns the 16-byte content. - blob.occ_lengths = {100}; + blob.occs.packed = {index::pack_range(0, 100)}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); // An escaped end does too. - blob.occ_lengths = {0xff}; - blob.occ_long_rows = {0}; - blob.occ_long_ends = {600}; + blob.occs.packed = {index::pack_range(0, index::length_escape)}; + blob.occs.long_rows = {0}; + blob.occs.long_ends = {600}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); - blob.occ_lengths = {3}; - blob.occ_long_rows = {}; - blob.occ_long_ends = {}; + blob.occs.packed = {index::pack_range(0, 3)}; + blob.occs.long_rows = {}; + blob.occs.long_ends = {}; // Relation ranges are bounded alike. - blob.rel_lengths = {100}; + blob.rels.packed = {index::pack_range(0, 100)}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); // Except the no-range sentinel a pair relation legitimately carries — // on a source-located kind the same sentinel is corruption. blob.rel_kinds = {static_cast(RelationKind::Base)}; - blob.rel_begins = {0xffffffff}; - blob.rel_lengths = {0}; + blob.rels.packed = {index::packed_sentinel}; ASSERT_TRUE(make_shard(bytes_of()).loaded()); blob.rel_kinds = {static_cast(RelationKind::Reference)}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); - blob.rel_begins = {0}; - blob.rel_lengths = {3}; + blob.rels.packed = {index::pack_range(0, 3)}; // And definition-range payloads. blob.rel_def_rows = {0}; @@ -561,14 +777,50 @@ TEST_CASE(RangesBeyondContentRejected) { ASSERT_FALSE(make_shard(bytes_of()).loaded()); } +TEST_CASE(WrongRangeTierRejected) { + // The range tier is a strict function of the content size — a second + // spelling of the same rows would fork the byte identity. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + fill_content(blob, "aaaaaaaaaaaaaaaa"); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + blob.occs.begins = {0}; + blob.occs.lengths = {3}; + blob.occ_syms8 = {0}; + + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + ASSERT_FALSE(make_shard(bytes).loaded()); +} + +TEST_CASE(WrongSymWidthRejected) { + // The symbol id width is a strict function of the table size, for the + // same canonicality reason. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + fill_content(blob, "aaaaaaaaaaaaaaaa"); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + blob.occs.packed = {index::pack_range(0, 3)}; + blob.occ_syms16 = {0}; + + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + ASSERT_FALSE(make_shard(bytes).loaded()); +} + TEST_CASE(DuplicateSymbolHashRejected) { // Symbol lookups lower-bound the hash column and read only the first // match's slices: a duplicated hash strands the later id's relations // unreachably while the blob keeps loading as fresh. index::ShardBlob blob; blob.format_version = index::index_format_version; - blob.content = "aaaa"; - blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + fill_content(blob, "aaaa"); blob.variants = {1}; blob.sym_hashes = {111, 222}; blob.sym_rel_offsets = {0, 0, 0}; @@ -586,13 +838,12 @@ TEST_CASE(DuplicateSymbolHashRejected) { } TEST_CASE(DuplicateVariantRejected) { - // Liveness and compaction select variants by rows hash; a duplicated + // Liveness and compaction select variants by identity; a duplicated // entry would make every copy live at once, and rows masked only to the // extra id would serve and survive with no contribution owning them. index::ShardBlob blob; blob.format_version = index::index_format_version; - blob.content = "aaaa"; - blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + fill_content(blob, "aaaa"); blob.variants = {1, 2}; blob.sym_hashes = {111}; blob.sym_rel_offsets = {0, 0}; @@ -615,19 +866,16 @@ TEST_CASE(StraySymbolIdRejected) { // or dropping the relation's target forever with no reindex triggered. index::ShardBlob blob; blob.format_version = index::index_format_version; - blob.content = "aaaaaaaaaaaaaaaa"; - blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + fill_content(blob, "aaaaaaaaaaaaaaaa"); blob.variants = {1}; blob.sym_hashes = {111}; blob.sym_rel_offsets = {0, 1}; - blob.occ_begins = {0}; - blob.occ_lengths = {3}; - blob.occ_syms16 = {0}; + blob.occs.packed = {index::pack_range(0, 3)}; + blob.occ_syms8 = {0}; blob.rel_kinds = {static_cast(RelationKind::Base)}; - blob.rel_begins = {4}; - blob.rel_lengths = {3}; + blob.rels.packed = {index::pack_range(4, 3)}; blob.rel_sym_rows = {0}; - blob.rel_sym16 = {0}; + blob.rel_sym8 = {0}; auto bytes_of = [&] { std::string bytes; @@ -637,11 +885,11 @@ TEST_CASE(StraySymbolIdRejected) { }; ASSERT_TRUE(make_shard(bytes_of()).loaded()); - blob.occ_syms16 = {5}; + blob.occ_syms8 = {5}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); - blob.occ_syms16 = {0}; + blob.occ_syms8 = {0}; - blob.rel_sym16 = {5}; + blob.rel_sym8 = {5}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); } @@ -652,15 +900,13 @@ TEST_CASE(OwnerlessMaskRejected) { // it for real — so it must reject the blob at load. index::ShardBlob blob; blob.format_version = index::index_format_version; - blob.content = "aaaa"; - blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + fill_content(blob, "aaaa"); blob.variants = {1, 2}; blob.sym_hashes = {111}; blob.sym_rel_offsets = {0, 0}; - blob.occ_begins = {0}; - blob.occ_lengths = {3}; - blob.occ_syms16 = {0}; - blob.occ_masks32 = {0b01}; + blob.occs.packed = {index::pack_range(0, 3)}; + blob.occ_syms8 = {0}; + blob.occs.masks32 = {0b01}; auto bytes_of = [&] { std::string bytes; @@ -671,33 +917,33 @@ TEST_CASE(OwnerlessMaskRejected) { ASSERT_TRUE(make_shard(bytes_of()).loaded()); // An empty mask, then one whose only bit lies past the variant table. - blob.occ_masks32 = {0}; + blob.occs.masks32 = {0}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); - blob.occ_masks32 = {0b100}; + blob.occs.masks32 = {0b100}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); // The u64 tier is bounded alike. for(std::uint32_t i = 3; i <= 40; i += 1) { blob.variants.push_back(i); } - blob.occ_masks32 = {}; - blob.occ_masks64 = {1}; + blob.occs.masks32 = {}; + blob.occs.masks64 = {1}; ASSERT_TRUE(make_shard(bytes_of()).loaded()); - blob.occ_masks64 = {std::uint64_t(1) << 45}; + blob.occs.masks64 = {std::uint64_t(1) << 45}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); // And roaring masks: decodable but empty, or holding only dropped ids. for(std::uint32_t i = 41; i <= 70; i += 1) { blob.variants.push_back(i); } - blob.occ_masks64 = {}; + blob.occs.masks64 = {}; auto set_mask = [&](const clice::Bitmap& mask) { - blob.occ_roaring.clear(); + blob.occs.roaring.clear(); for(auto byte: index::write_bitmap(mask)) { - blob.occ_roaring.push_back(static_cast(byte)); + blob.occs.roaring.push_back(static_cast(byte)); } - blob.occ_roaring_offsets = {0, static_cast(blob.occ_roaring.size())}; - blob.rel_roaring_offsets = {0}; + blob.occs.roaring_offsets = {0, static_cast(blob.occs.roaring.size())}; + blob.rels.roaring_offsets = {0}; }; clice::Bitmap in_range; in_range.add(69); @@ -718,24 +964,22 @@ TEST_CASE(CorruptRoaringMaskRejected) { // an undecodable slice must reject the blob at load. index::ShardBlob blob; blob.format_version = index::index_format_version; - blob.content = "aaaa"; - blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + fill_content(blob, "aaaa"); for(std::uint32_t i = 1; i <= 65; i += 1) { blob.variants.push_back(i); } blob.sym_hashes = {111}; blob.sym_rel_offsets = {0, 0}; - blob.occ_begins = {0}; - blob.occ_lengths = {3}; - blob.occ_syms16 = {0}; + blob.occs.packed = {index::pack_range(0, 3)}; + blob.occ_syms8 = {0}; clice::Bitmap mask; mask.add(2); for(auto byte: index::write_bitmap(mask)) { - blob.occ_roaring.push_back(static_cast(byte)); + blob.occs.roaring.push_back(static_cast(byte)); } - blob.occ_roaring_offsets = {0, static_cast(blob.occ_roaring.size())}; - blob.rel_roaring_offsets = {0}; + blob.occs.roaring_offsets = {0, static_cast(blob.occs.roaring.size())}; + blob.rels.roaring_offsets = {0}; auto bytes_of = [&] { std::string bytes; @@ -745,8 +989,8 @@ TEST_CASE(CorruptRoaringMaskRejected) { }; ASSERT_TRUE(make_shard(bytes_of()).loaded()); - blob.occ_roaring = {0xff, 0xff, 0xff}; - blob.occ_roaring_offsets = {0, 3}; + blob.occs.roaring = {0xff, 0xff, 0xff}; + blob.occs.roaring_offsets = {0, 3}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); } From 80e1c3fb2860b45a8616159d0e2c8ffb19fc1b04 Mon Sep 17 00:00:00 2001 From: ykiko Date: Mon, 17 Aug 2026 01:46:49 +0800 Subject: [PATCH 03/10] refactor(index): envelope carries shard blobs, PreambleIndex over unified readers --- src/index/preamble_index.cpp | 249 ++++++++++++++++ .../{preamble_state.h => preamble_index.h} | 76 +++-- src/index/preamble_state.cpp | 271 ------------------ src/index/serialization.h | 36 +-- src/index/tu_index.cpp | 240 ++++++---------- src/index/tu_index.h | 102 +++---- 6 files changed, 418 insertions(+), 556 deletions(-) create mode 100644 src/index/preamble_index.cpp rename src/index/{preamble_state.h => preamble_index.h} (63%) delete mode 100644 src/index/preamble_state.cpp diff --git a/src/index/preamble_index.cpp b/src/index/preamble_index.cpp new file mode 100644 index 000000000..aede500e5 --- /dev/null +++ b/src/index/preamble_index.cpp @@ -0,0 +1,249 @@ +#include "index/preamble_index.h" + +#include +#include + +#include "compile/compilation_unit.h" +#include "index/serialization.h" + +#include "llvm/Support/xxhash.h" + +namespace clice::index { + +namespace { + +/// One file covered by the preamble compilation: its shard blob, embedded +/// verbatim from the consumed TUIndex section. Entries are only ever +/// encoded — queries run on Shard readers wrapped at load(). +struct PreambleEntry { + std::uint32_t path_id = 0; + llvm::ArrayRef blob; +}; + +/// find_symbol serves only name and kind, so the blob stores this reduced +/// entry instead of the full Symbol — reflecting that would drag every +/// symbol's scope and reference bitmap into large SDK preamble blobs for +/// nothing. The name borrows the consumed TUIndex (encode-only, like +/// PreambleEntry). +struct PreambleSymbol { + llvm::StringRef name; + SymbolKind kind; +}; + +/// The persisted shape of a `.pch.idx` blob. +struct PreambleBlob { + std::uint32_t format_version = 0; + + /// Identity of the exact preamble text the PCH was built from: + /// xxh3 and byte size. Consumers serve preamble-derived state only + /// while the live buffer's prefix still matches (matches_prefix) — + /// independent of whether the region produced any rows. + std::uint64_t preamble_hash = 0; + std::uint32_t preamble_size = 0; + + std::vector paths; + std::vector files; + PreambleEntry preamble; + llvm::DenseMap symbols; + llvm::ArrayRef links; + llvm::ArrayRef inactive_regions; + llvm::ArrayRef open_conditionals; +}; + +using BlobView = kota::codec::fbs::table_view; + +/// The blob was fully verified at load(); per-query views skip that cost. +BlobView root_of(const llvm::MemoryBuffer& buffer) { + return BlobView::from_verified_bytes(blob_bytes(buffer.getBuffer())); +} + +llvm::StringRef entry_bytes(kota::codec::fbs::table_view entry) { + auto blob = to_array_ref(entry[&PreambleEntry::blob]); + return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); +} + +} // namespace + +void PreambleIndex::serialize(CompilationUnitRef unit, + TUIndex index, + llvm::ArrayRef links, + llvm::ArrayRef inactive_regions, + llvm::ArrayRef open_conditionals, + llvm::raw_ostream& os) { + PreambleBlob blob; + blob.format_version = preamble_format_version; + + // The preamble compile remaps the buffer truncated at the bound, so + // interested_content() is exactly the preamble text the PCH was built + // from. + auto preamble_text = unit.interested_content(); + blob.preamble_hash = llvm::xxh3_64bits(preamble_text); + blob.preamble_size = static_cast(preamble_text.size()); + + // The source file is the last path in graph.paths (convention from + // IncludeGraph); its section holds the preamble region's own rows. + auto main_id = static_cast(index.graph.paths.size()) - 1; + blob.files.reserve(index.sections.size()); + for(auto& section: index.sections) { + if(section.path_id == main_id) { + blob.preamble = {section.path_id, section.blob}; + } else { + blob.files.push_back({section.path_id, section.blob}); + } + } + + blob.symbols.reserve(index.symbols.size()); + for(const auto& [hash, symbol]: index.symbols) { + blob.symbols.try_emplace(hash, PreambleSymbol{.name = symbol.name, .kind = symbol.kind}); + } + blob.paths = std::move(index.graph.paths); + blob.links = links; + blob.inactive_regions = inactive_regions; + blob.open_conditionals = open_conditionals; + + serialize_blob(blob, os); +} + +std::shared_ptr PreambleIndex::load(llvm::StringRef path) { + auto buffer = llvm::MemoryBuffer::getFile(path); + if(!buffer) { + return nullptr; + } + + // A stale or truncated blob must never crash the server. from_bytes + // deep-verifies every offset, string, vector and table the views can + // reach, and each embedded shard blob is verified once by the Shard + // wrap below — queries then run unchecked. Anything failing loads as + // "missing" and the PCH pair is rebuilt. + auto root = BlobView::from_bytes(blob_bytes((*buffer)->getBuffer())); + if(!root.valid() || root[&PreambleBlob::format_version] != preamble_format_version) { + return nullptr; + } + + std::shared_ptr state(new PreambleIndex()); + state->buffer = std::move(*buffer); + + auto verified = root_of(*state->buffer); + auto paths = verified[&PreambleBlob::paths]; + auto files = verified[&PreambleBlob::files]; + state->file_shards.reserve(files.size()); + state->file_paths.reserve(files.size()); + for(std::size_t i = 0; i < files.size(); i += 1) { + auto entry = files[i]; + if(entry[&PreambleEntry::path_id] >= paths.size()) { + return nullptr; + } + auto shard = Shard::from_bytes(entry_bytes(entry)); + if(!shard.loaded()) { + return nullptr; + } + state->file_paths.push_back(to_ref(paths[entry[&PreambleEntry::path_id]])); + state->file_shards.push_back(std::move(shard)); + } + + // The preamble region may legitimately have no rows — an absent blob + // stays an empty shard; corrupt bytes still reject the pair. + auto preamble = entry_bytes(verified[&PreambleBlob::preamble]); + if(!preamble.empty()) { + state->preamble_shard = Shard::from_bytes(preamble); + if(!state->preamble_shard.loaded()) { + return nullptr; + } + } + + return state; +} + +void PreambleIndex::lookup(SymbolHash symbol, + RelationKind kind, + llvm::function_ref callback) const { + for(std::size_t i = 0; i < file_shards.size(); i += 1) { + auto& shard = file_shards[i]; + File file{ + .path = file_paths[i], + .content = shard.content(), + .content_size = shard.content_size(), + .line_starts = shard.line_starts(), + }; + bool stopped = false; + shard.lookup(symbol, kind, [&](const Relation& relation) { + if(!callback(file, relation)) { + stopped = true; + return false; + } + return true; + }); + if(stopped) { + return; + } + } +} + +llvm::StringRef PreambleIndex::source_path() const { + auto root = root_of(*buffer); + auto paths = root[&PreambleBlob::paths]; + if(paths.empty()) { + return {}; + } + // The source file is the last path, by IncludeGraph convention. + return to_ref(paths[paths.size() - 1]); +} + +bool PreambleIndex::matches_prefix(llvm::StringRef text) const { + auto root = root_of(*buffer); + auto size = root[&PreambleBlob::preamble_size]; + return text.size() >= size && + llvm::xxh3_64bits(text.take_front(size)) == root[&PreambleBlob::preamble_hash]; +} + +void PreambleIndex::lookup_preamble(std::uint32_t offset, + llvm::function_ref callback) const { + preamble_shard.lookup(offset, callback); +} + +void PreambleIndex::lookup_preamble(SymbolHash symbol, + RelationKind kind, + llvm::function_ref callback) const { + preamble_shard.lookup(symbol, kind, callback); +} + +bool PreambleIndex::find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const { + auto root = root_of(*buffer); + auto found = root[&PreambleBlob::symbols].find(hash); + if(!found) { + return false; + } + + auto symbol = found->get<1>(); + name = std::string(symbol[&PreambleSymbol::name]); + kind = SymbolKind(symbol[&PreambleSymbol::kind]); + return true; +} + +std::vector PreambleIndex::links() const { + auto root = root_of(*buffer); + auto entries = root[&PreambleBlob::links]; + + std::vector links; + links.reserve(entries.size()); + for(std::size_t i = 0; i < entries.size(); i += 1) { + auto entry = entries[i]; + links.push_back(feature::DocumentLink{ + .range = entry[&feature::DocumentLink::range], + .target = std::string(entry[&feature::DocumentLink::target]), + }); + } + return links; +} + +llvm::ArrayRef PreambleIndex::inactive_regions() const { + auto root = root_of(*buffer); + return to_array_ref(root[&PreambleBlob::inactive_regions]); +} + +llvm::ArrayRef PreambleIndex::open_conditionals() const { + auto root = root_of(*buffer); + return to_array_ref(root[&PreambleBlob::open_conditionals]); +} + +} // namespace clice::index diff --git a/src/index/preamble_state.h b/src/index/preamble_index.h similarity index 63% rename from src/index/preamble_state.h rename to src/index/preamble_index.h index c04ce72a2..6969ae742 100644 --- a/src/index/preamble_state.h +++ b/src/index/preamble_index.h @@ -6,6 +6,7 @@ #include #include "feature/feature.h" +#include "index/shard.h" #include "index/tu_index.h" #include "llvm/ADT/ArrayRef.h" @@ -14,24 +15,25 @@ namespace clice::index { -/// On-disk PreambleState blob schema version (the PCH's `.pch.idx` pair). -/// Bump whenever the persisted PreambleState layout (its reflected repr) -/// changes; a blob carrying a different value loads as "missing" and the -/// PCH pair is rebuilt. cache.json records it so a version change is -/// caught at load time instead of on the first overlay query. -constexpr inline std::uint32_t preamble_format_version = 5; +/// On-disk PreambleIndex blob schema version (the PCH's `.pch.idx` pair). +/// Bump whenever the persisted layout (its reflected repr or the nested +/// shard blob format) changes; a blob carrying a different value loads as +/// "missing" and the PCH pair is rebuilt. cache.json records it so a +/// version change is caught at load time instead of on the first overlay +/// query. +constexpr inline std::uint32_t preamble_format_version = 6; /// All master-visible state derived from one PCH build. /// /// The stateless worker serializes it next to the PCH blob (the store's -/// `.pch.idx` pair) and the master opens it as a memory-mapped FlatBuffer: +/// `.pch.idx` pair) and the master opens it as a memory-mapped blob: /// queries run directly on the serialized data, nothing is deserialized up -/// front. It carries the preamble's full symbol index — every header the -/// PCH covers plus the main file's preamble region — with per-file content -/// and line starts for position mapping (mirroring the disk shards), -/// and the PCH-derived feature state that is spliced into main-file -/// results: document links, inactive regions and the open conditional -/// stack at the preamble bound. +/// front. It carries the preamble's full symbol index — one shard blob per +/// header the PCH covers plus the main file's preamble region, the same +/// encoding every other holder of a file's rows uses — and the +/// PCH-derived feature state that is spliced into main-file results: +/// document links, inactive regions and the open conditional stack at the +/// preamble bound. /// /// Lifecycle equals the PCH's: the pair is committed, hit and evicted /// together, so no separate invalidation is needed — a preamble change or @@ -42,23 +44,28 @@ constexpr inline std::uint32_t preamble_format_version = 5; /// and replayed on every request. New preamble-region features add /// precomputed rows to this blob and define a per-feature merge with the /// live results of the rest of the file. -class PreambleState { +class PreambleIndex { public: /// A file entry handed to lookup callbacks: everything needed to turn /// a byte-offset hit into an LSP location. Views borrow the mapped - /// blob; keep the PreambleState alive while using them. + /// blob; keep the PreambleIndex alive while using them. struct File { llvm::StringRef path; + + /// Empty for pure-ASCII content, which the blob does not store — + /// byte offsets are already UTF-16 column offsets there. llvm::StringRef content; + + std::uint32_t content_size = 0; + std::span line_starts; }; /// Serialize a preamble compilation's state. `index` must be built - /// over the preamble unit with interested_only=false and its - /// main_file_index intact (it holds the preamble region's own - /// occurrences — macro definitions and references before the bound). - /// Taken by value and consumed: the blob is assembled by moving the - /// index's rows, never copying them. + /// over the preamble unit with interested_only=false; its sections + /// (one shard blob per covered file) are embedded verbatim. Taken by + /// value and consumed: the blob borrows the index's bytes and names + /// for the duration of the write. static void serialize(CompilationUnitRef unit, TUIndex index, llvm::ArrayRef links, @@ -67,9 +74,10 @@ class PreambleState { llvm::raw_ostream& os); /// Open a blob from disk (memory-mapped). Returns nullptr when the - /// file is unreadable, structurally invalid or written by a different - /// format version — callers treat all of these as a PCH cache miss. - static std::shared_ptr load(llvm::StringRef path); + /// file is unreadable, structurally invalid, written by a different + /// format version, or any embedded shard blob fails verification — + /// callers treat all of these as a PCH cache miss. + static std::shared_ptr load(llvm::StringRef path); /// Iterate relations of `symbol` matching `kind` across all header /// entries. Return false from the callback to stop. This is the only @@ -86,11 +94,11 @@ class PreambleState { /// to this file. Borrows the mapped blob. llvm::StringRef source_path() const; - /// The exact preamble text this blob was built from. Consumers serve - /// preamble-entry rows only while the live buffer still starts with - /// it — the rows are buffer offsets into this prefix. Borrows the - /// mapped blob. - llvm::StringRef preamble_content() const; + /// Whether `text` still begins with the exact preamble this blob was + /// built from — the gate for serving preamble-derived state against a + /// live buffer (the rows are offsets into that prefix). Compared by + /// hash: the text itself is not stored. + bool matches_prefix(llvm::StringRef text) const; /// Occurrence lookup in the source file's preamble region (buffer /// offsets below the preamble bound). @@ -122,9 +130,19 @@ class PreambleState { } private: - explicit PreambleState(std::unique_ptr buffer); + PreambleIndex() = default; std::unique_ptr buffer; + + /// One reader per header entry, wrapping the mapped blob's bytes; + /// line-start caches accumulate here across queries. Paths borrow the + /// mapped blob, parallel to the shards. + std::vector file_shards; + std::vector file_paths; + + /// The source file's preamble-region rows; an empty shard when the + /// region had none. + Shard preamble_shard; }; } // namespace clice::index diff --git a/src/index/preamble_state.cpp b/src/index/preamble_state.cpp deleted file mode 100644 index 0724d3b94..000000000 --- a/src/index/preamble_state.cpp +++ /dev/null @@ -1,271 +0,0 @@ -#include "index/preamble_state.h" - -#include -#include -#include -#include - -#include "compile/compilation_unit.h" -#include "index/serialization.h" - -#include "kota/ipc/lsp/text.h" - -namespace clice::index { - -namespace { - -/// One file covered by the preamble compilation: its rows plus content and -/// line starts for position mapping. The rows are moved out of the TUIndex -/// and the content borrows the compilation's buffers — entries are only -/// ever encoded, never decoded (queries run on the zero-copy view). -struct PreambleFileEntry { - std::uint32_t path_id = 0; - FileIndex index; - llvm::StringRef content; - std::vector line_starts; -}; - -/// find_symbol serves only name and kind, so the blob stores this reduced -/// entry instead of the full Symbol — reflecting that would drag every -/// symbol's scope and reference bitmap into large SDK preamble blobs for -/// nothing. The name borrows the consumed TUIndex (encode-only, like -/// PreambleFileEntry). -struct PreambleSymbol { - llvm::StringRef name; - SymbolKind kind; -}; - -/// The persisted shape of a `.pch.idx` blob. Queries run on a zero-copy -/// view of this layout; nothing is deserialized up front. -struct PreambleBlob { - std::uint32_t format_version = 0; - std::vector paths; - std::vector files; - PreambleFileEntry preamble; - llvm::DenseMap symbols; - llvm::ArrayRef links; - llvm::ArrayRef inactive_regions; - llvm::ArrayRef open_conditionals; -}; - -using StateView = kota::codec::fbs::table_view; -using FileEntryView = kota::codec::fbs::table_view; - -/// The blob was fully verified at load(); per-query views skip that cost. -StateView root_of(const llvm::MemoryBuffer& buffer) { - return StateView::from_verified_bytes(blob_bytes(buffer.getBuffer())); -} - -PreambleState::File file_of(kota::codec::fbs::array_view paths, FileEntryView entry) { - auto line_starts = to_array_ref(entry[&PreambleFileEntry::line_starts]); - return PreambleState::File{ - .path = to_ref(paths[entry[&PreambleFileEntry::path_id]]), - .content = to_ref(entry[&PreambleFileEntry::content]), - .line_starts = std::span(line_starts.data(), line_starts.size()), - }; -} - -} // namespace - -void PreambleState::serialize(CompilationUnitRef unit, - TUIndex index, - llvm::ArrayRef links, - llvm::ArrayRef inactive_regions, - llvm::ArrayRef open_conditionals, - llvm::raw_ostream& os) { - PreambleBlob blob; - blob.format_version = preamble_format_version; - - blob.files.reserve(index.file_indices.size()); - for(auto& [fid, file_index]: index.file_indices) { - // A file with no include edge is a synthetic buffer (predefines, - // ): it has no real path to attribute rows to, and - // path_id() would misfile them under the source file. Real files - // forced in via -include are not affected — clang records their - // include edge in the predefines buffer, which is a valid - // location (covered by ForcedIncludeServed). - if(index.graph.include_location_id(fid) == static_cast(-1)) { - continue; - } - auto content = unit.file_content(fid); - blob.files.push_back({ - .path_id = index.graph.path_id(fid), - .index = std::move(file_index), - .content = content, - .line_starts = - kota::ipc::lsp::build_line_starts(std::string_view(content.data(), content.size())), - }); - } - - // The source file is the last path in graph.paths (convention from - // IncludeGraph). The preamble compile remaps the buffer truncated at - // the bound, so interested_content() is exactly the preamble text the - // PCH was built from — stored so consumers can compare it against the - // live buffer's prefix before serving these rows. - auto preamble_text = unit.interested_content(); - blob.preamble = { - .path_id = static_cast(index.graph.paths.size() - 1), - .index = std::move(index.main_file_index), - .content = preamble_text, - .line_starts = kota::ipc::lsp::build_line_starts( - std::string_view(preamble_text.data(), preamble_text.size())), - }; - - blob.symbols.reserve(index.symbols.size()); - for(const auto& [hash, symbol]: index.symbols) { - blob.symbols.try_emplace(hash, PreambleSymbol{.name = symbol.name, .kind = symbol.kind}); - } - blob.paths = std::move(index.graph.paths); - blob.links = links; - blob.inactive_regions = inactive_regions; - blob.open_conditionals = open_conditionals; - - serialize_blob(blob, os); -} - -PreambleState::PreambleState(std::unique_ptr buffer) : - buffer(std::move(buffer)) {} - -std::shared_ptr PreambleState::load(llvm::StringRef path) { - auto buffer = llvm::MemoryBuffer::getFile(path); - if(!buffer) { - return nullptr; - } - - // A stale or truncated blob must never crash the server. from_bytes - // deep-verifies every offset, string, vector and table the views can - // reach — queries then run unchecked (root_of). Blobs of a different - // format version load as "missing" (version-less blobs read back 0) - // and the PCH pair is rebuilt. - auto root = StateView::from_bytes(blob_bytes((*buffer)->getBuffer())); - if(!root.valid() || root[&PreambleBlob::format_version] != preamble_format_version) { - return nullptr; - } - - return std::shared_ptr(new PreambleState(std::move(*buffer))); -} - -void PreambleState::lookup(SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) const { - auto root = root_of(*buffer); - auto paths = root[&PreambleBlob::paths]; - auto files = root[&PreambleBlob::files]; - - for(std::size_t i = 0; i < files.size(); ++i) { - auto entry = files[i]; - auto relations = entry[&PreambleFileEntry::index][&FileIndex::relations]; - auto found = relations.find(symbol); - if(!found) { - continue; - } - - // The verifier checks structure, not cross-references: a corrupt - // path_id must not attribute rows to an arbitrary path. - if(entry[&PreambleFileEntry::path_id] >= paths.size()) { - continue; - } - auto file = file_of(paths, entry); - - auto rels = found->get<1>(); - for(std::size_t j = 0; j < rels.size(); ++j) { - Relation relation = rels[j]; - if(RelationKind(relation.kind) & kind) { - if(!callback(file, relation)) { - return; - } - } - } - } -} - -llvm::StringRef PreambleState::source_path() const { - auto root = root_of(*buffer); - auto paths = root[&PreambleBlob::paths]; - if(paths.empty()) { - return {}; - } - // The source file is the last path, by IncludeGraph convention. - return to_ref(paths[paths.size() - 1]); -} - -llvm::StringRef PreambleState::preamble_content() const { - auto root = root_of(*buffer); - return to_ref(root[&PreambleBlob::preamble][&PreambleFileEntry::content]); -} - -void PreambleState::lookup_preamble(std::uint32_t offset, - llvm::function_ref callback) const { - auto root = root_of(*buffer); - auto occurrences = - root[&PreambleBlob::preamble][&PreambleFileEntry::index][&FileIndex::occurrences]; - - scan_occurrences_at( - occurrences.size(), - offset, - [&](std::size_t i) { return occurrences[i]; }, - callback); -} - -void PreambleState::lookup_preamble(SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) const { - auto root = root_of(*buffer); - auto relations = - root[&PreambleBlob::preamble][&PreambleFileEntry::index][&FileIndex::relations]; - auto found = relations.find(symbol); - if(!found) { - return; - } - - auto rels = found->get<1>(); - for(std::size_t i = 0; i < rels.size(); ++i) { - Relation relation = rels[i]; - if(RelationKind(relation.kind) & kind) { - if(!callback(relation)) { - return; - } - } - } -} - -bool PreambleState::find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const { - auto root = root_of(*buffer); - auto found = root[&PreambleBlob::symbols].find(hash); - if(!found) { - return false; - } - - auto symbol = found->get<1>(); - name = std::string(symbol[&PreambleSymbol::name]); - kind = SymbolKind(symbol[&PreambleSymbol::kind]); - return true; -} - -std::vector PreambleState::links() const { - auto root = root_of(*buffer); - auto entries = root[&PreambleBlob::links]; - - std::vector links; - links.reserve(entries.size()); - for(std::size_t i = 0; i < entries.size(); ++i) { - auto entry = entries[i]; - links.push_back(feature::DocumentLink{ - .range = entry[&feature::DocumentLink::range], - .target = std::string(entry[&feature::DocumentLink::target]), - }); - } - return links; -} - -llvm::ArrayRef PreambleState::inactive_regions() const { - auto root = root_of(*buffer); - return to_array_ref(root[&PreambleBlob::inactive_regions]); -} - -llvm::ArrayRef PreambleState::open_conditionals() const { - auto root = root_of(*buffer); - return to_array_ref(root[&PreambleBlob::open_conditionals]); -} - -} // namespace clice::index diff --git a/src/index/serialization.h b/src/index/serialization.h index c890bae60..1090b441a 100644 --- a/src/index/serialization.h +++ b/src/index/serialization.h @@ -113,7 +113,7 @@ namespace clice::index { /// regular field and every loader discards blobs with a different value — /// including version-less blobs from older builds, which read back as 0. /// Bump it whenever a persisted type's reflected layout changes. -constexpr inline std::uint32_t index_format_version = 5; +constexpr inline std::uint32_t index_format_version = 6; /// Serialize a reflected index blob to `os` as a verified-readable /// flatbuffer. Encoding only fails on structural impossibilities (e.g. more @@ -139,40 +139,6 @@ bool deserialize_blob(llvm::StringRef data, T& out) { return kota::codec::fbs::from_bytes(blob_bytes(data), out).has_value(); } -/// Scan the occurrences containing `offset` in a sequence sorted by -/// (range.begin, range.end, target): binary-search the first entry whose -/// range ends at or past the offset, then walk while ranges contain it. -/// Binary-searching on range.end is sound because occurrence ranges are -/// name-token spans, pairwise disjoint or identical — never partially -/// overlapping or nested — so under this order range.end is monotonic too. -/// `get(i)` yields the i-th Occurrence. -template -void scan_occurrences_at(std::size_t size, - std::uint32_t offset, - GetOccurrence&& get, - llvm::function_ref callback) { - std::size_t lo = 0; - std::size_t hi = size; - while(lo < hi) { - auto mid = lo + (hi - lo) / 2; - if(get(mid).range.end < offset) { - lo = mid + 1; - } else { - hi = mid; - } - } - - for(; lo < size; ++lo) { - Occurrence occurrence = get(lo); - if(!occurrence.range.contains(offset)) { - break; - } - if(!callback(occurrence)) { - break; - } - } -} - inline llvm::StringRef to_ref(std::string_view text) { return {text.data(), text.size()}; } diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index 387d3647e..53d8a830e 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -5,6 +5,7 @@ #include "compile/compilation_unit.h" #include "index/serialization.h" +#include "index/shard.h" #include "semantic/decls.h" #include "semantic/display.h" #include "semantic/semantics.h" @@ -45,7 +46,7 @@ class Projector { if(interested_only && fid != unit.interested_file()) { return nullptr; } - return &result.file_indices[fid]; + return &file_indices[fid]; } void add_occurrence(const clang::NamedDecl* decl, @@ -527,126 +528,105 @@ class Projector { // every fid keying `file_indices` gets its include chain resolved // through the SourceManager, so the lookups below cannot miss. llvm::SmallVector indexed_fids; - indexed_fids.reserve(result.file_indices.size()); - for(auto& [fid, index]: result.file_indices) { + indexed_fids.reserve(file_indices.size()); + for(auto& [fid, index]: file_indices) { indexed_fids.push_back(fid); } result.graph = IncludeGraph::from(unit, indexed_fids); - for(auto& [fid, index]: result.file_indices) { - for(auto& [symbol_id, relations]: index.relations) { - std::ranges::sort(relations, [](const Relation& lhs, const Relation& rhs) { - return std::tuple(lhs.kind, lhs.range.begin, lhs.range.end, lhs.target_symbol) < - std::tuple(rhs.kind, rhs.range.begin, rhs.range.end, rhs.target_symbol); - }); - auto range = - std::ranges::unique(relations, [](const Relation& lhs, const Relation& rhs) { - return lhs.kind == rhs.kind && lhs.range == rhs.range && - lhs.target_symbol == rhs.target_symbol; - }); - relations.erase(range.begin(), range.end()); + for(auto& [fid, index]: file_indices) { + for(auto symbol_id: llvm::make_first_range(index.relations)) { result.symbols[symbol_id].reference_files.add(result.graph.path_id(fid)); } - - std::ranges::sort(index.occurrences, [](const Occurrence& lhs, const Occurrence& rhs) { - return std::tuple(lhs.range.begin, lhs.range.end, lhs.target) < - std::tuple(rhs.range.begin, rhs.range.end, rhs.target); - }); - auto range = - std::ranges::unique(index.occurrences, - [](const Occurrence& lhs, const Occurrence& rhs) { - return lhs.range == rhs.range && lhs.target == rhs.target; - }); - index.occurrences.erase(range.begin(), range.end()); - - if(fid == unit.interested_file()) { - result.main_file_index = std::move(index); + } + auto finish_ms = finish_timer.ms_f(); + + // Encode one blob per path. A header entered several times (its + // FileIDs differ, its path id does not) contributes the union of + // its entries' rows: write_shard canonicalizes — sorts and + // deduplicates — so concatenation is union. + ScopedTimer encode_timer; + llvm::DenseMap by_path; + llvm::DenseMap path_fids; + for(auto& [fid, index]: file_indices) { + // A file with no include edge is a synthetic buffer (predefines, + // ): it has no real path to attribute rows to, and + // path_id() would misfile them under the source file. Real files + // forced in via -include are not affected — clang records their + // include edge in the predefines buffer, which is a valid + // location. The interested file legitimately has no edge. + if(fid != unit.interested_file() && + result.graph.include_location_id(fid) == static_cast(-1)) { + continue; + } + auto path_id = result.graph.path_id(fid); + path_fids.try_emplace(path_id, fid); + auto& into = by_path[path_id]; + if(into.empty()) { + into = std::move(index); + continue; + } + into.occurrences.insert(into.occurrences.end(), + index.occurrences.begin(), + index.occurrences.end()); + for(auto& [hash, relations]: index.relations) { + auto& group = into.relations[hash]; + group.insert(group.end(), relations.begin(), relations.end()); } } - result.file_indices.erase(unit.interested_file()); + auto resolve = [&](SymbolHash hash) -> std::optional { + auto it = result.symbols.find(hash); + if(it == result.symbols.end()) { + return std::nullopt; + } + return SymbolIdentity{it->second.name, it->second.kind, it->second.scope}; + }; + + llvm::SmallVector path_ids; + path_ids.reserve(by_path.size()); + for(auto path_id: llvm::make_first_range(by_path)) { + path_ids.push_back(path_id); + } + // Ascending path ids put the interested file (the last path id) + // last — the position main_section() relies on. + llvm::sort(path_ids); + for(auto path_id: path_ids) { + auto& rows = by_path[path_id]; + if(rows.empty()) { + continue; + } + std::string bytes; + llvm::raw_string_ostream os(bytes); + write_shard(rows, resolve, unit.file_content(path_fids[path_id]), os); + auto hash = llvm::xxh3_64bits(bytes); + result.sections.push_back( + {path_id, hash, std::vector(bytes.begin(), bytes.end())}); + } LOG_PERF("index_detail", - "op=build scope={} semantics_ms={:.2f} project_ms={:.2f} finish_ms={:.2f}", + "op=build scope={} semantics_ms={:.2f} project_ms={:.2f} finish_ms={:.2f} " + "encode_ms={:.2f}", interested_only ? "interested" : "full", semantics_ms, project_ms, - finish_timer.ms_f()); + finish_ms, + encode_timer.ms_f()); } private: TUIndex& result; CompilationUnitRef unit; bool interested_only; + /// Build-time working state keyed by FileID — clang::FileID means + /// nothing outside the compilation, so it never leaves the builder; + /// the encode step above converts it through graph.path_id. + llvm::DenseMap file_indices; llvm::DenseMap enclosing_cache; }; } // namespace -void FileIndex::lookup(std::uint32_t offset, - llvm::function_ref callback) const { - auto it = std::ranges::lower_bound(occurrences, offset, {}, [](const Occurrence& o) { - return o.range.end; - }); - while(it != occurrences.end() && it->range.contains(offset)) { - if(!callback(*it)) - return; - ++it; - } -} - -void FileIndex::lookup(SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) const { - auto it = relations.find(symbol); - if(it == relations.end()) - return; - for(auto& r: it->second) { - if(RelationKind(r.kind) & kind) { - if(!callback(r)) - return; - } - } -} - -std::uint64_t FileIndex::rows_hash() const { - static_assert(sizeof(Occurrence) == sizeof(Range) + sizeof(SymbolHash)); - static_assert(sizeof(Relation) == - sizeof(RelationKind) + 4 + sizeof(Range) + sizeof(SymbolHash)); - - // One flat buffer in a deterministic order: the sorted occurrences, - // then each relation group in ascending symbol order (DenseMap - // iteration order must never leak into the hash). - std::vector buffer; - std::size_t size = occurrences.size() * sizeof(Occurrence); - for(auto& [symbol, group]: relations) { - size += sizeof(symbol) + group.size() * sizeof(Relation); - } - buffer.reserve(size); - - auto append = [&](const void* data, std::size_t bytes) { - auto* raw = static_cast(data); - buffer.insert(buffer.end(), raw, raw + bytes); - }; - - append(occurrences.data(), occurrences.size() * sizeof(Occurrence)); - - llvm::SmallVector keys; - keys.reserve(relations.size()); - for(auto symbol: llvm::make_first_range(relations)) { - keys.push_back(symbol); - } - llvm::sort(keys); - for(auto symbol: keys) { - append(&symbol, sizeof(symbol)); - auto& group = relations.find(symbol)->second; - append(group.data(), group.size() * sizeof(Relation)); - } - - return llvm::xxh3_64bits( - llvm::StringRef(reinterpret_cast(buffer.data()), buffer.size())); -} - TUIndex TUIndex::build(CompilationUnitRef unit, bool interested_only) { TUIndex index; index.built_at = unit.build_at(); @@ -660,50 +640,9 @@ TUIndex TUIndex::build(CompilationUnitRef unit, bool interested_only) { void TUIndex::serialize(llvm::raw_ostream& os) { format_version = index_format_version; - /// Convert the FileID-keyed working state into wire sections; multiple - /// FileIDs can share a path id (repeated header contexts), last-wins. - /// A deserialized index has no FileID-keyed state at all — its sections - /// already are the persisted form, so re-serializing must not wipe them. - ScopedTimer copy_timer; - if(!file_indices.empty() || !main_file_index.empty()) { - sections.clear(); - llvm::DenseMap positions; - auto add = [&](std::uint32_t path_id, const FileIndex& index) { - if(index.empty()) { - return; - } - auto encoded = kota::codec::fbs::to_bytes(index); - assert(encoded.has_value()); - FileSection section{path_id, index.rows_hash(), std::move(*encoded)}; - auto [it, inserted] = positions.try_emplace(path_id, sections.size()); - if(inserted) { - sections.push_back(std::move(section)); - } else { - sections[it->second] = std::move(section); - } - }; - for(auto& [fid, file_index]: file_indices) { - add(graph.path_id(fid), file_index); - } - // size() - 1 would wrap on an empty path table and name the main - // file with an id every reader rejects, losing the rows silently. - assert(!graph.paths.empty() && "rows cannot exist without a path table naming their file"); - add(static_cast(graph.paths.size() - 1), main_file_index); - } - auto copy_ms = copy_timer.ms_f(); - - // The interested file's rows travel only as their section; the - // reflected field is written empty and restored after the pack. - auto main_rows = std::move(main_file_index); - main_file_index = FileIndex(); - ScopedTimer pack_timer; serialize_blob(*this, os); - main_file_index = std::move(main_rows); - LOG_PERF("index_detail", - "op=serialize copy_ms={:.2f} pack_ms={:.2f}", - copy_ms, - pack_timer.ms_f()); + LOG_PERF("index_detail", "op=serialize pack_ms={:.2f}", pack_timer.ms_f()); } std::optional TUIndex::from(llvm::StringRef data) { @@ -756,16 +695,6 @@ const FileSection* TUIndex::main_section() const { return nullptr; } -std::optional TUIndex::decode_rows(const FileSection& section) { - std::optional rows{std::in_place}; - auto data = - llvm::StringRef(reinterpret_cast(section.rows.data()), section.rows.size()); - if(!deserialize_blob(data, *rows)) { - return std::nullopt; - } - return rows; -} - namespace { using WireView = kota::codec::fbs::table_view; @@ -859,18 +788,13 @@ std::uint32_t TUIndexView::section_path(std::uint32_t i) const { return wire_root(data)[&TUIndex::sections].at(i)[&FileSection::path_id]; } -std::uint64_t TUIndexView::section_rows_hash(std::uint32_t i) const { - return wire_root(data)[&TUIndex::sections].at(i)[&FileSection::rows_hash]; +std::uint64_t TUIndexView::section_hash(std::uint32_t i) const { + return wire_root(data)[&TUIndex::sections].at(i)[&FileSection::hash]; } -std::optional TUIndexView::decode_section_rows(std::uint32_t i) const { - auto rows = to_array_ref(wire_root(data)[&TUIndex::sections].at(i)[&FileSection::rows]); - std::optional decoded{std::in_place}; - auto bytes = llvm::StringRef(reinterpret_cast(rows.data()), rows.size()); - if(!deserialize_blob(bytes, *decoded)) { - return std::nullopt; - } - return decoded; +llvm::StringRef TUIndexView::section_blob(std::uint32_t i) const { + auto blob = to_array_ref(wire_root(data)[&TUIndex::sections].at(i)[&FileSection::blob]); + return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); } std::optional TUIndexView::main_section_index() const { diff --git a/src/index/tu_index.h b/src/index/tu_index.h index 7a397b14e..eba6ee916 100644 --- a/src/index/tu_index.h +++ b/src/index/tu_index.h @@ -11,8 +11,6 @@ #include "semantic/symbol.h" #include "support/bitmap.h" -#include "kota/codec/macro.h" -#include "kota/meta/annotation.h" #include "llvm/ADT/STLFunctionalExtras.h" #include "llvm/Support/raw_ostream.h" @@ -66,6 +64,8 @@ struct Occurrence { friend bool operator==(const Occurrence&, const Occurrence&) = default; }; +/// One file's rows while a build accumulates them; encoded into a shard +/// blob (index/shard.h) at build end and consumed as bytes from then on. struct FileIndex { /// The braces matter: fbs decode value-constructs map entries with /// `FileIndex{}`, and without an initializer this member would be @@ -75,22 +75,9 @@ struct FileIndex { std::vector occurrences; - void lookup(std::uint32_t offset, llvm::function_ref callback) const; - - void lookup(SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) const; - bool empty() const { return occurrences.empty() && relations.empty(); } - - /// Content identity of the rows: xxh3 over the occurrences and the - /// relation groups in ascending symbol order. Requires the canonical - /// row order build() establishes (sorted, deduplicated); two files - /// preprocessed identically hash equal, and that equality is what - /// deduplicates variants across compilation contexts. - std::uint64_t rows_hash() const; }; struct Symbol { @@ -108,25 +95,31 @@ struct Symbol { using SymbolTable = llvm::DenseMap; -/// One file's rows on the wire: the hash first, so the master can skip the -/// nested decode for rows it already stores, and the rows themselves as a -/// self-contained nested blob decoded per miss. +/// One file's rows on the wire: a self-contained single-variant shard +/// blob (index/shard.h), stored verbatim by the master when the variant +/// is new and merged byte-for-byte otherwise. `hash` is xxh3 of `blob` — +/// the variant's identity — so the master can skip blobs it already +/// stores without touching their bytes. struct FileSection { std::uint32_t path_id = 0; - /// FileIndex::rows_hash of the nested rows. - std::uint64_t rows_hash = 0; + std::uint64_t hash = 0; - /// Nested fbs FileIndex blob (TUIndex::decode_rows). - std::vector rows; + std::vector blob; }; +/// What indexing one TU produced, in transit from worker to master: the +/// include graph (interned into a manifest), the TU's symbol table with +/// per-symbol reference files (merged into the project table), and one +/// shard blob per file that received rows (stored or merged into the +/// file's disk shard). Never persisted itself; it travels worker→server +/// over IPC and is dismantled into the three persistent layers on +/// arrival. struct TUIndex { - /// Persisted-blob schema version (index_format_version), stamped by - /// serialize() and gated by from(). These blobs never touch disk — they - /// travel worker→server over IPC — but a worker respawned after the - /// binary on disk changed can be one build ahead of the server, and a - /// layout change need not be structurally detectable. + /// Wire schema version (index_format_version), stamped by serialize() + /// and gated by from(). A worker respawned after the binary on disk + /// changed can be one build ahead of the server, and a layout change + /// need not be structurally detectable. std::uint32_t format_version = 0; /// The building timestamp of this file. @@ -137,48 +130,28 @@ struct TUIndex { SymbolTable symbols; - /// Build-time working state keyed by FileID — clang::FileID means nothing - /// outside the compilation, so it never persists; serialize() converts it - /// through graph.path_id. - KOTATSU_ANNOTATE(skip = true) - > file_indices; - - /// The interested file's rows, used in memory (sessions, preamble - /// state). serialize() moves it into its wire section for the duration - /// of the write, so the reflected field always travels empty. - FileIndex main_file_index; - - /// The wire form of the per-file rows: populated by serialize() from - /// file_indices and main_file_index (the interested file's section is - /// last), kept raw by from(). Files whose rows are empty get no - /// section — no rows means no contribution. + /// One entry per file with rows, ascending by path id; the interested + /// file (the last path id) comes last. Rows of a header entered + /// several times are one union blob. Files whose rows are empty get + /// no section — no rows means no contribution. std::vector sections; /// Build the index for `unit`. With interested_only, only rows in - /// the interested file are kept. Note that a full build over a unit - /// compiled with a preamble PCH is not a production combination - /// (background indexing compiles without PCH): rows landing in the - /// PCH's loaded copy of the main file would serialize a second entry - /// under the main path id. + /// the interested file are kept. static TUIndex build(CompilationUnitRef unit, bool interested_only = false); - /// Serialization reflects this object directly (sections are populated - /// from the row state first — hence non-const). + /// Serialization reflects this object directly. void serialize(llvm::raw_ostream& os); /// Verify and deserialize a buffer; nullopt when structural /// verification fails, the format version differs, or a decoded path id - /// falls outside the blob's own path table. Section rows stay raw — - /// decode them per file with decode_rows. + /// falls outside the blob's own path table. Section blobs stay raw — + /// wrap them in a Shard per file to query them. static std::optional from(llvm::StringRef data); /// The interested file's wire section, or nullptr when its rows were /// empty. const FileSection* main_section() const; - - /// Verify and decode one section's rows; nullopt for a corrupt nested - /// blob. - static std::optional decode_rows(const FileSection& section); }; /// A symbol's identity as a merge consumer needs it; the name borrows the @@ -190,10 +163,12 @@ struct SymbolIdentity { }; /// Zero-copy reader over a serialized TUIndex, for the master's merge path: -/// the graph and the per-file rows hashes are read straight off the wire, -/// section rows are decoded only for actual misses, and symbol names are -/// touched only when a symbol is genuinely new to the global table. The -/// view borrows the wire bytes; keep them alive while using it. +/// the graph, the per-file blob hashes and the blob bytes themselves are +/// read straight off the wire — a new variant's bytes are sliced out and +/// written or merged without ever decoding the envelope around them — and +/// symbol names are touched only when a symbol is genuinely new to the +/// global table. The view borrows the wire bytes; keep them alive while +/// using it. /// /// TUIndex::from stays the full-decode entry for consumers that need the /// whole object (sessions, tests). @@ -203,7 +178,8 @@ class TUIndexView { /// the graph and sections carry. Symbol reference-file ids are NOT /// validated here — iterate_symbols hands them out raw and the consumer /// bounds them (decoding every bitmap twice just to validate would - /// defeat the view). + /// defeat the view). Section blob bytes are not verified either; + /// Shard::from_bytes verifies each blob the consumer actually uses. static std::optional from(llvm::StringRef data); std::int64_t built_at() const; @@ -222,10 +198,10 @@ class TUIndexView { std::uint32_t section_path(std::uint32_t i) const; - std::uint64_t section_rows_hash(std::uint32_t i) const; + std::uint64_t section_hash(std::uint32_t i) const; - /// Verify and decode one section's rows straight from the wire buffer. - std::optional decode_section_rows(std::uint32_t i) const; + /// One section's shard blob bytes, borrowing the wire buffer. + llvm::StringRef section_blob(std::uint32_t i) const; /// The section index of the interested file (path_count() - 1), or /// nullopt when its rows were empty. From 9de3ba0710b71ee20cf2b2c66cb13fad91425a72 Mon Sep 17 00:00:00 2001 From: ykiko Date: Mon, 17 Aug 2026 01:58:10 +0800 Subject: [PATCH 04/10] refactor(server): byte-passthrough merge, Shard sessions, IndexedLineMap --- src/feature/feature.h | 2 +- src/semantic/semantics.h | 2 +- src/server/compiler/compiler.cpp | 34 ++-- src/server/compiler/compiler.h | 2 +- src/server/compiler/indexer.cpp | 190 ++++-------------- src/server/protocol/extension.h | 2 +- src/server/protocol/position.h | 80 ++++++++ src/server/protocol/worker.h | 4 +- src/server/service/feature_router.cpp | 2 +- src/server/service/query.cpp | 147 ++++++++------ src/server/service/query.h | 8 +- src/server/state/session.h | 13 +- src/server/state/workspace.cpp | 14 +- src/server/state/workspace.h | 20 +- src/server/worker/stateless_worker.cpp | 10 +- src/support/logging.h | 2 +- ...ate_tests.cpp => preamble_index_tests.cpp} | 0 17 files changed, 265 insertions(+), 267 deletions(-) rename tests/unit/index/{preamble_state_tests.cpp => preamble_index_tests.cpp} (100%) diff --git a/src/feature/feature.h b/src/feature/feature.h index 52ba57eae..b8b2b7146 100644 --- a/src/feature/feature.h +++ b/src/feature/feature.h @@ -275,7 +275,7 @@ struct FoldingRange { /// A resolved document link: the argument range of an include-like /// directive (byte offsets in the containing file) and the absolute path /// of the target file. Plain data — it serializes over the worker RPC and -/// the PCH's PreambleState blob as-is and becomes an LSP DocumentLink only +/// the PCH's PreambleIndex blob as-is and becomes an LSP DocumentLink only /// at the reply edge, where the session's line map does the conversion. struct DocumentLink { LocalSourceRange range; diff --git a/src/semantic/semantics.h b/src/semantic/semantics.h index 14b92847e..286324256 100644 --- a/src/semantic/semantics.h +++ b/src/semantic/semantics.h @@ -320,7 +320,7 @@ class Semantics { /// The spelled tokens of the interested file (a view into the unit's /// TokenBuffer, not a copy). They cover the whole file even under a /// preamble PCH — what the PCH consumes is the preamble's AST and - /// directives (those travel through PreambleState instead), not its + /// directives (those travel through PreambleIndex instead), not its /// spelling. llvm::ArrayRef spelled_tokens() const { return tokens; diff --git a/src/server/compiler/compiler.cpp b/src/server/compiler/compiler.cpp index 07a04f785..60d68bea9 100644 --- a/src/server/compiler/compiler.cpp +++ b/src/server/compiler/compiler.cpp @@ -8,7 +8,7 @@ #include #include "command/argument_parser.h" -#include "index/preamble_state.h" +#include "index/preamble_index.h" #include "index/tu_index.h" #include "server/compiler/context_resolver.h" #include "server/protocol/extension.h" @@ -493,7 +493,7 @@ kota::task Compiler::ensure_pch(Session& session, if(auto it = workspace.pch_cache.find(pch_key); it != workspace.pch_cache.end()) { auto& st = it->second; // Both halves of the pair must be present: a PCH whose - // PreambleState blob is gone (crash between commits, failed aux + // PreambleIndex blob is gone (crash between commits, failed aux // commit) rebuilds whole. bool in_store = workspace.store && workspace.store->lookup("pch", pch_key) && workspace.store->lookup_aux("pch", pch_key); @@ -576,7 +576,7 @@ kota::task Compiler::ensure_pch(Session& session, } // Build a new PCH pair via stateless worker: it writes the PCH and its - // PreambleState blob to the tmp paths allocated here; the store + // PreambleIndex blob to the tmp paths allocated here; the store // commits (fsync + rename) both on success, primary first. auto pending = workspace.store->begin_store("pch", pch_key); auto pending_idx = workspace.store->begin_store_aux("pch", pch_key); @@ -628,7 +628,7 @@ kota::task Compiler::ensure_pch(Session& session, struct PairCommit { std::optional pch_path; std::optional index_path; - std::shared_ptr state; + std::shared_ptr state; }; auto committed = co_await kota::queue([&]() -> PairCommit { @@ -648,7 +648,7 @@ kota::task Compiler::ensure_pch(Session& session, return outcome; } outcome.index_path = std::move(*index_path); - outcome.state = index::PreambleState::load(*outcome.index_path); + outcome.state = index::PreambleIndex::load(*outcome.index_path); return outcome; }); if(!committed.has_value() || !committed.value().pch_path.has_value()) { @@ -656,7 +656,7 @@ kota::task Compiler::ensure_pch(Session& session, co_return false; } if(!committed.value().index_path.has_value()) { - LOG_WARN("Failed to commit PreambleState blob for {}", path); + LOG_WARN("Failed to commit PreambleIndex blob for {}", path); // A rebuild of an existing key just had its blobs retracted from // the store; the entry's paths now dangle and waiters checking // `!path.empty()` would hand the compile a deleted PCH. Drop it — @@ -974,7 +974,7 @@ kota::task<> Compiler::run_compile(std::shared_ptr session) { // out: concurrent compiles can insert into pch_cache across the // await below and rehash the map from under a held pointer. std::vector pch_inactive; - std::shared_ptr preamble_state; + std::shared_ptr preamble_state; if(session->pch_key.has_value()) { preamble_state = workspace.preamble_state(*session->pch_key); } @@ -1193,12 +1193,18 @@ kota::task<> Compiler::run_compile(std::shared_ptr session) { ? std::nullopt : index::TUIndex::from(result.value().tu_index_data); if(tu_index) { - // The interested file's rows travel as a wire section; a file - // with no rows at all has no section and gets an empty index. - auto* section = tu_index->main_section(); - auto rows = section ? index::TUIndex::decode_rows(*section) - : std::optional{std::in_place}; - session->file_index = rows ? std::move(*rows) : index::FileIndex(); + // The interested file's rows travel as its shard blob; a file + // with no rows at all has no section and gets an empty-rows + // blob, so a settled empty index still outranks disk fallbacks. + std::string blob; + if(auto* section = tu_index->main_section()) { + blob.assign(section->blob.begin(), section->blob.end()); + } else { + llvm::raw_string_ostream os(blob); + index::write_shard(index::FileIndex(), {}, llvm::StringRef(), os); + } + session->file_index = + index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(blob)); session->symbols = std::move(tu_index->symbols); } else { // The AST and the file index settle together — that pairing is @@ -1206,7 +1212,7 @@ kota::task<> Compiler::run_compile(std::shared_ptr session) { // compile that produced no index data (fatal error, no AST) must // therefore drop the previous buffer's index rather than leave // it posing as current: an honest gap over yesterday's offsets. - session->file_index.reset(); + session->file_index = index::Shard(); session->symbols.reset(); } diff --git a/src/server/compiler/compiler.h b/src/server/compiler/compiler.h index 5a28d6c55..1d8d2b354 100644 --- a/src/server/compiler/compiler.h +++ b/src/server/compiler/compiler.h @@ -113,7 +113,7 @@ class Compiler { /// Forward a document-link query to the stateful worker holding this /// file's AST. Covers the main-file region only: the preamble's links - /// live in the PCH's PreambleState blob (see PCHState::load_state). + /// live in the PCH's PreambleIndex blob (see PCHState::load_state). /// `token`: see forward_query. kota::task, kota::ipc::Error> forward_document_links(std::shared_ptr session, diff --git a/src/server/compiler/indexer.cpp b/src/server/compiler/indexer.cpp index 667868df4..e3777a0ce 100644 --- a/src/server/compiler/indexer.cpp +++ b/src/server/compiler/indexer.cpp @@ -35,8 +35,9 @@ static std::string blob_key(llvm::StringRef path) { } void Indexer::merge(const void* tu_index_data, std::size_t size) { - // Zero-copy consumption: the wire stays serialized; only miss sections - // and genuinely new symbol names are ever materialized. + // Zero-copy consumption: the wire stays serialized; a new variant's + // blob bytes are sliced out and installed or merged without decoding + // the envelope, and only genuinely new symbol names are materialized. auto loaded = index::TUIndexView::from(llvm::StringRef(static_cast(tu_index_data), size)); if(!loaded) { @@ -51,34 +52,6 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { auto main_local_id = view.path_count() - 1; llvm::StringRef main_tu_path = view.path(main_local_id); - // Shards pair the worker's rows with content read from disk here; if the - // disk moved on since the worker read it, the rows' offsets describe - // bytes that no longer exist and merging would misplace every position - // until the next reindex. The worker's consumed-content hash arbitrates. - // A missing hash (0) proceeds as before. - auto content_matches = [&](std::uint32_t local_id, llvm::StringRef disk_content) { - auto consumed = view.path_hash(local_id); - return consumed == 0 || llvm::xxh3_64bits(disk_content) == consumed; - }; - - // The main file's verdict gates the WHOLE result: every section and the - // manifest describe this one compile, so applying any of it against a - // moved-on main file would mix two generations. Skipping everything - // keeps the last-known state consistent; the changed file fails the - // next staleness check (or is already pending), and a follow-up pass - // redoes the merge against settled content. - auto main_buf = llvm::MemoryBuffer::getFile(main_tu_path); - if(!main_buf) { - LOG_WARN("Skip merge for {}: cannot read content: {}", - main_tu_path, - main_buf.getError().message()); - return; - } - if(!content_matches(main_local_id, (*main_buf)->getBuffer())) { - LOG_INFO("Skip merge for {}: disk moved on since it was indexed", main_tu_path); - return; - } - // Interning paths only names them — pool ids left behind by a rejected // result are inert. Everything that is index STATE (symbols, // FileVersions, the manifest, shards) commits only below the section @@ -94,158 +67,79 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { index::TUManifest manifest; manifest.built_at = static_cast(view.built_at()); - // The previous contribution entry for a file this pass must skip (disk - // moved on under a section): its rows stay consistent with the shard - // they live in, so it keeps serving; a later pass redoes the file. - auto carry_old_contribution = [&](std::uint32_t global_id) { - auto manifest_it = project.manifests.find(tu_path_id); - if(manifest_it == project.manifests.end()) { - return; - } - for(auto& [fv, hash]: manifest_it->second.contributions) { - if(project.file_versions.find(fv)->second.path_id == global_id) { - manifest.contributions.emplace_back(fv, hash); - return; - } - } - }; - std::size_t hits = 0; std::size_t appended = 0; - auto lookup_symbol = [&](index::SymbolHash hash) { - return view.find_symbol(hash); - }; - // Staged, not committed: a section that fails to decode rejects the + // Staged, not committed: a section that fails verification rejects the // whole result mid-loop, and shards installed before that point would // leave the surviving manifest referencing variants the new blobs no // longer store. llvm::SmallVector> replacements; - // (TU-local path id, rows hash) per serving section; the FileVersions - // these will reference are interned only at commit. + // (TU-local path id, variant identity) per serving section; the + // FileVersions these will reference are interned only at commit. llvm::SmallVector> section_contributions; llvm::SmallVector rebuilt_ids; for(std::uint32_t section = 0; section < view.section_count(); section += 1) { auto local_id = view.section_path(section); - auto rows_hash = view.section_rows_hash(section); + auto blob_hash = view.section_hash(section); auto global_id = file_ids_map[local_id]; - auto consumed = view.path_hash(local_id); - bool is_main = local_id == main_local_id; auto shard_it = workspace.shards.find(global_id); auto* shard = shard_it != workspace.shards.end() ? &shard_it->second : nullptr; - // Fast path: this content generation's blob already stores these - // rows — recording the contribution is the only work, no IO at all. - if(consumed != 0 && shard && shard->loaded() && shard->content_hash() == consumed && - shard->has_variant(rows_hash)) { - section_contributions.emplace_back(local_id, rows_hash); - hits += 1; - continue; - } - - // Read and arbitrate the content the blob will pair the rows with. - std::string header_content; - llvm::StringRef content; - if(is_main) { - content = (*main_buf)->getBuffer(); - } else { - auto path = workspace.path_pool.resolve(global_id); - auto buf = llvm::MemoryBuffer::getFile(path); - if(buf) { - header_content = (*buf)->getBuffer().str(); - content = header_content; - } - // Unconditional, unlike the main-file read: an unreadable or - // truncated-to-empty header must not slip past the arbitration - // and pair the rows with content they were not built from. A - // failed read is checked on its own — with no worker hash the - // arbitration would otherwise wave the empty content through. - if(!buf || !content_matches(local_id, content)) { - LOG_INFO("Skip merge for {}: disk moved on since it was indexed", path); - carry_old_contribution(global_id); - continue; - } - } - auto generation = consumed != 0 ? consumed : llvm::xxh3_64bits(content); - - // The same hit as the fast path above, verifiable for a hash-less - // section only now that the disk read pinned the generation: - // re-appending an already-stored variant would duplicate it in the - // blob's variant table (write_shard asserts against exactly that). - if(consumed == 0 && shard && shard->loaded() && shard->content_hash() == generation && - shard->has_variant(rows_hash)) { - section_contributions.emplace_back(local_id, rows_hash); + // Fast path: the blob already stores this variant. The identity + // hashes the blob bytes, which embed the content generation, so + // one membership test is the whole check — no IO, no bytes read. + if(shard && shard->loaded() && shard->has_variant(blob_hash)) { + section_contributions.emplace_back(local_id, blob_hash); hits += 1; continue; } // The recomputed hash guards the variant identity alongside the - // structural decode: rows installed under a hash they do not - // reproduce would satisfy every later hit-path check for that hash - // while the shard stores different rows. - auto rows = view.decode_section_rows(section); - if(!rows || rows->rows_hash() != rows_hash) { - // Not a carry-and-skip like the moved-on case above: that one - // self-corrects because the skipped file's recorded version is - // stale by hash. A decode failure leaves every recorded version - // matching the disk, so an installed manifest would be judged - // fresh forever with this file's rows missing or stale — even - // across restarts. Reject the whole result like the main-file - // gate does; nothing is committed yet. + // structural verification below: bytes installed under a hash they + // do not reproduce would satisfy every later hit-path check for + // that hash while the shard stores different rows. A mismatch (or + // an invalid blob) leaves every recorded version matching the + // disk, so an installed manifest would be judged fresh forever + // with this file's rows missing or stale — reject the whole + // result; nothing is committed yet. + auto bytes = view.section_blob(section); + if(llvm::xxh3_64bits(bytes) != blob_hash) { LOG_WARN("Reject merge for {}: rows section for {} failed verification", main_tu_path, workspace.path_pool.resolve(global_id)); return; } + auto fresh = index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); + if(!fresh.loaded()) { + LOG_WARN("Reject merge for {}: rows for {} do not form a valid shard", + main_tu_path, + workspace.path_pool.resolve(global_id)); + return; + } - index::VariantInput fresh{rows_hash, &*rows, lookup_symbol}; - std::string bytes; - llvm::raw_string_ostream os(bytes); - bool append = shard && shard->loaded() && shard->content_hash() == generation; - if(append) { + index::Shard replacement; + if(shard && shard->loaded() && shard->content_hash() == fresh.content_hash()) { // Same generation, new variant: merge it in, keeping every // stored variant — dead ones stay masked until the next save // compacts them. - write_shard(*shard, shard->variants(), fresh, content, generation, os); + std::string merged; + llvm::raw_string_ostream os(merged); + index::merge_shards(*shard, shard->variants(), llvm::ArrayRef(fresh), os); + replacement = index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(merged)); + assert(replacement.loaded() && "a freshly merged shard blob must verify"); appended += 1; } else { // New content generation (or no blob at all): rows from other // generations must never share offset storage with these, so - // the blob starts over. Stale contributions from other TUs - // stop matching any stored variant; the commit re-enqueues - // their owners. - write_shard(index::Shard(), {}, fresh, content, generation, os); - } - - auto replacement = index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); - if(!replacement.loaded()) { - if(consumed == 0) { - // Unverified pairing: the arbitration above cannot see the - // disk shrinking under rows built from longer content, and - // the blob's own range bounds catch it here instead. A - // carry self-corrects — the recorded version is stale by - // hash. - LOG_INFO("Skip merge for {}: disk moved on since it was indexed", - workspace.path_pool.resolve(global_id)); - carry_old_contribution(global_id); - continue; - } - // Hash-verified content always fits rows built from it: this - // blob failed on the rows themselves (inverted or out-of-range - // spans that kept wire structure and hash). Carrying would - // install a manifest whose versions all match the disk, pinning - // the stale rows as fresh forever — reject like the decode - // failure above. - LOG_WARN("Reject merge for {}: rows for {} do not form a valid shard", - main_tu_path, - workspace.path_pool.resolve(global_id)); - return; - } - replacements.emplace_back(global_id, std::move(replacement)); - if(!append) { + // the worker's bytes become the blob verbatim. Stale + // contributions from other TUs stop matching any stored + // variant; the commit re-enqueues their owners. + replacement = std::move(fresh); rebuilt_ids.push_back(global_id); } - section_contributions.emplace_back(local_id, rows_hash); + replacements.emplace_back(global_id, std::move(replacement)); + section_contributions.emplace_back(local_id, blob_hash); } // The last gate and the first commit. A malformed reference bitmap (or @@ -394,7 +288,7 @@ kota::task<> Indexer::save() { } std::string bytes; llvm::raw_string_ostream os(bytes); - write_shard(shard, live, {}, shard.content(), shard.content_hash(), os); + index::merge_shards(shard, live, {}, os); auto replacement = index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); assert(replacement.loaded() && "a freshly written shard blob must verify"); shard = std::move(replacement); diff --git a/src/server/protocol/extension.h b/src/server/protocol/extension.h index 2b4937fde..3cf77a319 100644 --- a/src/server/protocol/extension.h +++ b/src/server/protocol/extension.h @@ -118,7 +118,7 @@ struct LogFloodResult { struct StatsParams {}; struct StatsResult { - /// pch_cache entries whose PreambleState blob is currently open, and + /// pch_cache entries whose PreambleIndex blob is currently open, and /// their mapped bytes. Steady state after closing documents: bounded /// by the loaded-state budget, not by every key ever touched. std::uint32_t pch_loaded_states = 0; diff --git a/src/server/protocol/position.h b/src/server/protocol/position.h index 8ee74b650..3887d4f68 100644 --- a/src/server/protocol/position.h +++ b/src/server/protocol/position.h @@ -2,7 +2,12 @@ /// Shared LSP position clamping for master-side buffer access. +#include +#include +#include + #include "kota/ipc/lsp/position.h" +#include "llvm/ADT/StringRef.h" namespace clice { @@ -24,4 +29,79 @@ inline kota::ipc::lsp::LineMap::Offset clamped_offset(const kota::ipc::lsp::Line return map.line_bounds(starts[position.line]).end; } +/// Position mapping in an indexed file's coordinates. Index blobs omit +/// pure-ASCII text — byte offsets are UTF-16 offsets there, so the +/// mapping is line-table arithmetic over the blob's content size; +/// non-ASCII content delegates to LineMap over the stored text. +class IndexedLineMap { +public: + IndexedLineMap(llvm::StringRef content, + std::uint32_t content_size, + std::span line_starts) : + content(content), content_size(content_size), starts(line_starts) {} + + std::optional to_position(std::uint32_t offset) const { + if(!content.empty()) { + return text_map().to_position(offset); + } + if(offset > content_size || starts.empty()) { + return std::nullopt; + } + auto line = line_of(offset); + if(offset > line_end(line)) { + return std::nullopt; + } + return protocol::Position{.line = line, .character = offset - starts[line]}; + } + + std::optional to_offset(const protocol::Position& position) const { + if(!content.empty()) { + return text_map().to_offset(position); + } + if(position.line >= starts.size()) { + return std::nullopt; + } + auto offset = starts[position.line] + position.character; + if(offset > line_end(position.line)) { + return std::nullopt; + } + return offset; + } + + std::optional to_range(std::uint32_t begin, std::uint32_t end) const { + if(begin > end) { + return std::nullopt; + } + auto start = to_position(begin); + if(!start) { + return std::nullopt; + } + auto stop = to_position(end); + if(!stop) { + return std::nullopt; + } + return protocol::Range{.start = *start, .end = *stop}; + } + +private: + kota::ipc::lsp::LineMap text_map() const { + return kota::ipc::lsp::LineMap(std::string_view(content.data(), content.size()), starts); + } + + std::uint32_t line_of(std::uint32_t offset) const { + auto it = std::upper_bound(starts.begin(), starts.end(), offset); + return it == starts.begin() ? 0 : static_cast(it - starts.begin()) - 1; + } + + /// Byte offset of the line's end, before its newline (mirroring + /// LineMap::line_bounds). + std::uint32_t line_end(std::uint32_t line) const { + return line + 1 < starts.size() ? starts[line + 1] - 1 : content_size; + } + + llvm::StringRef content; + std::uint32_t content_size; + std::span starts; +}; + } // namespace clice diff --git a/src/server/protocol/worker.h b/src/server/protocol/worker.h index 321774a07..2745aaaa2 100644 --- a/src/server/protocol/worker.h +++ b/src/server/protocol/worker.h @@ -210,7 +210,7 @@ struct BuildParams { std::string output_path; ///< BuildPCH, BuildPCM - /// BuildPCH: tmp path for the PreambleState blob (the PCH's paired + /// BuildPCH: tmp path for the PreambleIndex blob (the PCH's paired /// `.pch.idx`), allocated by the master's store alongside output_path. /// The worker serializes the preamble's index and feature state into /// it; the master commits both blobs together. @@ -248,7 +248,7 @@ struct BuildResult { /// Request the document links of an open file's AST. Only the main-file /// region is covered: the preamble is compiled into the PCH, and its links -/// live in the PCH's PreambleState blob (spliced in by the master). +/// live in the PCH's PreambleIndex blob (spliced in by the master). struct DocumentLinkParams { std::string path; }; diff --git a/src/server/service/feature_router.cpp b/src/server/service/feature_router.cpp index b658a842c..8cd878c1d 100644 --- a/src/server/service/feature_router.cpp +++ b/src/server/service/feature_router.cpp @@ -45,7 +45,7 @@ std::vector FeatureRouter::find_preamble_links(const Sess // Link offsets are buffer coordinates as of the PCH build; serve them // only while the buffer still starts with that exact preamble text // (a deferred rebuild mid-edit keeps an old blob for a moved buffer). - if(!llvm::StringRef(session.text).starts_with(state->preamble_content())) + if(!state->matches_prefix(session.text)) return {}; return state->links(); } diff --git a/src/server/service/query.cpp b/src/server/service/query.cpp index 9c0c55410..3cc0f02ad 100644 --- a/src/server/service/query.cpp +++ b/src/server/service/query.cpp @@ -1,5 +1,10 @@ #include "server/service/query.h" +#include "server/protocol/position.h" + +#include "llvm/Support/MemoryBuffer.h" +#include "llvm/Support/xxhash.h" + #include #include #include @@ -9,7 +14,7 @@ #include #include "feature/feature.h" -#include "index/preamble_state.h" +#include "index/preamble_index.h" #include "index/tu_index.h" #include "server/compiler/compiler.h" #include "server/compiler/indexer.h" @@ -38,7 +43,7 @@ void IndexQuery::visit_sessions(SessionVisitor visitor) const { sessions.for_each([&](std::uint32_t path_id, const Session& session) -> bool { // Freshness contract, clause 3: a dirty session's file index may // describe a buffer that no longer exists — skip it. - if(session.file_index && session.symbols && !session.ast_dirty) { + if(session.file_index.loaded() && session.symbols && !session.ast_dirty) { return visitor(path_id, session); } return true; @@ -49,7 +54,7 @@ bool IndexQuery::is_path_open(std::uint32_t path_id) const { return sessions.find(path_id) != nullptr; } -std::shared_ptr IndexQuery::overlay_of(const Session& session) const { +std::shared_ptr IndexQuery::overlay_of(const Session& session) const { if(!session.pch_key) { return nullptr; } @@ -59,7 +64,7 @@ std::shared_ptr IndexQuery::overlay_of(const Session& sess } void IndexQuery::visit_overlays( - llvm::function_ref visitor) const { + llvm::function_ref visitor) const { if(options.disk_only) { return; } @@ -75,7 +80,7 @@ void IndexQuery::visit_overlays( } void IndexQuery::visit_preambles( - llvm::function_ref visitor) + llvm::function_ref visitor) const { if(options.disk_only) { return; @@ -89,7 +94,7 @@ void IndexQuery::visit_preambles( }); } -bool IndexQuery::serves_preamble(const Session& session, const index::PreambleState& state) const { +bool IndexQuery::serves_preamble(const Session& session, const index::PreambleIndex& state) const { // The preamble entry's rows are buffer offsets of the file that built // the blob: serve them only for that very file (identical preambles // share a PCH, but macro USRs embed the source path) and only while @@ -100,7 +105,7 @@ bool IndexQuery::serves_preamble(const Session& session, const index::PreambleSt // (backslashes on Windows) while the pool normalizes separators, so // compare through the pool's lookup, not raw strings. return workspace.path_pool.find(state.source_path()) == session.path_id && - llvm::StringRef(session.text).starts_with(state.preamble_content()); + state.matches_prefix(session.text); } bool IndexQuery::should_serve_overlay_file(llvm::StringRef path) const { @@ -181,7 +186,7 @@ bool IndexQuery::find_symbol_info(index::SymbolHash hash, // Check PCH overlays: a symbol that exists only under an open buffer's // context (or in headers no disk TU has been indexed with) is in no // disk table. - visit_overlays([&](const index::PreambleState& state) { + visit_overlays([&](const index::PreambleIndex& state) { found = state.find_symbol(hash, name, kind); return !found; }); @@ -209,16 +214,16 @@ IndexQuery::CursorHit IndexQuery::resolve_cursor(llvm::StringRef path, // cross-file visit already skips shards of open files for the same // reason). A dirty-after-await session (failed or superseded compile) // or an index-less one therefore reports no hit. - if(session && (!session->file_index || session->ast_dirty)) { + if(session && (!session->file_index.loaded() || session->ast_dirty)) { return {}; } - if(session && session->file_index && !session->ast_dirty) { + if(session && session->file_index.loaded() && !session->ast_dirty) { auto map = session->line_map(); auto offset = map.to_offset(position); if(!offset) return {}; CursorHit hit; - session->file_index->lookup(*offset, [&](const index::Occurrence& occ) { + session->file_index.lookup(*offset, [&](const index::Occurrence& occ) { auto range = map.to_range(occ.range.begin, occ.range.end); if(range) { hit = {occ.target, *range}; @@ -260,10 +265,9 @@ IndexQuery::CursorHit IndexQuery::resolve_cursor(llvm::StringRef path, return {}; auto& merged_index = shard_it->second; - auto ls = merged_index.line_starts(); - if(ls.empty()) - return {}; - lsp::LineMap map(merged_index.content(), ls); + IndexedLineMap map(merged_index.content(), + merged_index.content_size(), + merged_index.line_starts()); auto offset = session ? session->line_map().to_offset(position) : map.to_offset(position); if(!offset) @@ -309,10 +313,9 @@ std::vector IndexQuery::query_relations(llvm::StringRef path if(!uri) continue; auto& merged_index = shard_it->second; - auto ls = merged_index.line_starts(); - if(ls.empty()) - continue; - lsp::LineMap map(merged_index.content(), ls); + IndexedLineMap map(merged_index.content(), + merged_index.content_size(), + merged_index.line_starts()); merged_index.lookup(hit.hash, kind, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) locations.push_back({uri->str(), *range}); @@ -326,7 +329,7 @@ std::vector IndexQuery::query_relations(llvm::StringRef path if(!uri) return true; auto map = session.line_map(); - session.file_index->lookup(hit.hash, kind, [&](const index::Relation& r) { + session.file_index.lookup(hit.hash, kind, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) locations.push_back({uri->str(), *range}); return true; @@ -337,16 +340,16 @@ std::vector IndexQuery::query_relations(llvm::StringRef path // PCH overlays: header rows under each open buffer's live context. // Rows a disk shard also holds come out identical and collapse in the // dedup below. - visit_overlays([&](const index::PreambleState& state) { + visit_overlays([&](const index::PreambleIndex& state) { state.lookup(hit.hash, kind, - [&](const index::PreambleState::File& file, const index::Relation& r) { - if(!should_serve_overlay_file(file.path) || file.line_starts.empty()) + [&](const index::PreambleIndex::File& file, const index::Relation& r) { + if(!should_serve_overlay_file(file.path)) return true; // to_uri canonicalizes clang's raw spelling (drive // case) before emitting. auto uri = feature::to_uri(file.path); - lsp::LineMap map(file.content, file.line_starts); + IndexedLineMap map(file.content, file.content_size, file.line_starts); if(auto range = map.to_range(r.range.begin, r.range.end)) locations.push_back({uri, *range}); return true; @@ -356,7 +359,7 @@ std::vector IndexQuery::query_relations(llvm::StringRef path // Preamble entries: the buffers' own preamble regions. visit_preambles( - [&](std::uint32_t id, const Session& session, const index::PreambleState& state) { + [&](std::uint32_t id, const Session& session, const index::PreambleIndex& state) { auto uri = lsp::URI::from_file_path(std::string(workspace.path_pool.resolve(id))); if(!uri) return true; @@ -422,7 +425,7 @@ std::optional IndexQuery::find_definition_location(index::Sy if(!uri) return true; auto map = session.line_map(); - session.file_index->lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { + session.file_index.lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) { session_result = protocol::Location{uri->str(), *range}; return false; @@ -440,7 +443,7 @@ std::optional IndexQuery::find_definition_location(index::Sy // First the buffers' own preamble regions, then the header entries. std::optional overlay_result; visit_preambles( - [&](std::uint32_t id, const Session& session, const index::PreambleState& state) { + [&](std::uint32_t id, const Session& session, const index::PreambleIndex& state) { auto uri = lsp::URI::from_file_path(std::string(workspace.path_pool.resolve(id))); if(!uri) return true; @@ -457,14 +460,14 @@ std::optional IndexQuery::find_definition_location(index::Sy if(overlay_result) return overlay_result; - visit_overlays([&](const index::PreambleState& state) { + visit_overlays([&](const index::PreambleIndex& state) { state.lookup(hash, RelationKind::Definition, - [&](const index::PreambleState::File& file, const index::Relation& r) { - if(!should_serve_overlay_file(file.path) || file.line_starts.empty()) + [&](const index::PreambleIndex::File& file, const index::Relation& r) { + if(!should_serve_overlay_file(file.path)) return true; auto uri = feature::to_uri(file.path); - lsp::LineMap map(file.content, file.line_starts); + IndexedLineMap map(file.content, file.content_size, file.line_starts); if(auto range = map.to_range(r.range.begin, r.range.end)) { overlay_result = protocol::Location{uri, *range}; return false; @@ -542,10 +545,9 @@ void IndexQuery::collect_grouped_relations( if(shard_it == workspace.shards.end()) continue; auto& merged_index = shard_it->second; - auto ls = merged_index.line_starts(); - if(ls.empty()) - continue; - lsp::LineMap map(merged_index.content(), ls); + IndexedLineMap map(merged_index.content(), + merged_index.content_size(), + merged_index.line_starts()); merged_index.lookup(hash, kind, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) target_ranges[r.target_symbol].push_back(*range); @@ -555,7 +557,7 @@ void IndexQuery::collect_grouped_relations( } visit_sessions([&](std::uint32_t, const Session& session) -> bool { auto map = session.line_map(); - session.file_index->lookup(hash, kind, [&](const index::Relation& r) { + session.file_index.lookup(hash, kind, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) target_ranges[r.target_symbol].push_back(*range); return true; @@ -566,13 +568,13 @@ void IndexQuery::collect_grouped_relations( // PCH overlays: call/type relations inside headers under an open // buffer's context. The main-file entry cannot contribute — the // preamble region holds only preprocessor directives. - visit_overlays([&](const index::PreambleState& state) { + visit_overlays([&](const index::PreambleIndex& state) { state.lookup(hash, kind, - [&](const index::PreambleState::File& file, const index::Relation& r) { - if(!should_serve_overlay_file(file.path) || file.line_starts.empty()) + [&](const index::PreambleIndex::File& file, const index::Relation& r) { + if(!should_serve_overlay_file(file.path)) return true; - lsp::LineMap map(file.content, file.line_starts); + IndexedLineMap map(file.content, file.content_size, file.line_starts); if(auto range = map.to_range(r.range.begin, r.range.end)) target_ranges[r.target_symbol].push_back(*range); return true; @@ -619,16 +621,12 @@ void IndexQuery::collect_unique_targets(index::SymbolHash hash, } } visit_sessions([&](std::uint32_t, const Session& session) -> bool { - auto rel_it = session.file_index->relations.find(hash); - if(rel_it == session.file_index->relations.end()) - return true; - for(auto& r: rel_it->second) { - if(RelationKind(r.kind) & kind) { - if(seen.insert(r.target_symbol).second) { - targets.push_back(r.target_symbol); - } + session.file_index.lookup(hash, kind, [&](const index::Relation& r) { + if(seen.insert(r.target_symbol).second) { + targets.push_back(r.target_symbol); } - } + return true; + }); return true; }); @@ -636,10 +634,10 @@ void IndexQuery::collect_unique_targets(index::SymbolHash hash, // open header's session is authoritative for its relations (an edited // `struct D : NewBase` must not resurface the disk snapshot's OldBase // through another file's overlay). - visit_overlays([&](const index::PreambleState& state) { + visit_overlays([&](const index::PreambleIndex& state) { state.lookup(hash, kind, - [&](const index::PreambleState::File& file, const index::Relation& r) { + [&](const index::PreambleIndex::File& file, const index::Relation& r) { if(!should_serve_overlay_file(file.path)) return true; if(seen.insert(r.target_symbol).second) { @@ -664,6 +662,25 @@ std::optional IndexQuery::resolve_symbol(index::SymbolHash hash) { return SymbolInfo{hash, std::move(name), kind, def_loc->uri, def_loc->range}; } +/// The stored text of an indexed file, for preview slicing. ASCII blobs +/// do not store it: re-read the disk and serve it only while its hash +/// still matches what the rows were built from — a moved-on file +/// degrades to no preview rather than slicing mismatched text. +static std::optional indexed_text(llvm::StringRef path, const index::Shard& shard) { + if(!shard.ascii()) { + return shard.content().str(); + } + auto buffer = llvm::MemoryBuffer::getFile(path); + if(!buffer) { + return std::nullopt; + } + auto text = (*buffer)->getBuffer(); + if(llvm::xxh3_64bits(text) != shard.content_hash()) { + return std::nullopt; + } + return text.str(); +} + static std::string extract_line(llvm::StringRef content, std::uint32_t offset) { if(content.empty() || offset >= content.size()) return {}; @@ -691,11 +708,12 @@ std::optional IndexQuery::get_definition_text(index: if(shard_it == workspace.shards.end()) continue; auto& merged_index = shard_it->second; - auto ls = merged_index.line_starts(); - if(ls.empty()) + auto file_path = workspace.path_pool.resolve(file_id); + auto text = indexed_text(file_path, merged_index); + if(!text) continue; - auto content = merged_index.content(); - lsp::LineMap map(content, ls); + llvm::StringRef content = *text; + lsp::LineMap map(content, merged_index.line_starts()); std::optional result; merged_index.lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { @@ -706,7 +724,7 @@ std::optional IndexQuery::get_definition_text(index: if(!range) return true; result = DefinitionText{ - .file = workspace.path_pool.resolve(file_id).str(), + .file = file_path.str(), .start_line = static_cast(range->start.line) + 1, .end_line = static_cast(range->end.line) + 1, .text = @@ -734,12 +752,12 @@ std::vector IndexQuery::collect_references(ind if(shard_it == workspace.shards.end()) continue; auto& merged_index = shard_it->second; - auto ls = merged_index.line_starts(); - if(ls.empty()) - continue; - auto content = merged_index.content(); - lsp::LineMap map(content, ls); auto file_path = workspace.path_pool.resolve(file_id); + auto text = indexed_text(file_path, merged_index); + if(!text) + continue; + llvm::StringRef content = *text; + lsp::LineMap map(content, merged_index.line_starts()); merged_index.lookup(hash, kind, [&](const index::Relation& r) { auto pos = map.to_position(r.range.begin); @@ -980,10 +998,9 @@ std::vector IndexQuery::locate_symbols(const agentic::ReadSymbol return {}; auto& merged_index = shard_it->second; - auto ls = merged_index.line_starts(); - if(ls.empty()) - return {}; - lsp::LineMap map(merged_index.content(), ls); + IndexedLineMap map(merged_index.content(), + merged_index.content_size(), + merged_index.line_starts()); for(auto& [hash, symbol]: workspace.project_index.symbols) { if(!symbol.reference_files.contains(*path_id)) diff --git a/src/server/service/query.h b/src/server/service/query.h index 24e6cb800..36bffb4e1 100644 --- a/src/server/service/query.h +++ b/src/server/service/query.h @@ -231,23 +231,23 @@ class IndexQuery { /// staleness follows the PCH's dependency discipline. Identical rows /// also present in disk shards are collapsed by per-location dedup at /// result assembly. Return false from the visitor to stop. - void visit_overlays(llvm::function_ref visitor) const; + void visit_overlays(llvm::function_ref visitor) const; /// Visit each open session whose overlay preamble entry may serve /// (see serves_preamble), paired with that blob. void visit_preambles(llvm::function_ref visitor) const; + const index::PreambleIndex& state)> visitor) const; /// The PCH overlay of a session, or nullptr when it has no PCH or the /// blob is unreadable. - std::shared_ptr overlay_of(const Session& session) const; + std::shared_ptr overlay_of(const Session& session) const; /// Whether a session's overlay preamble entry may serve: the blob was /// built from this very file (identical preambles share one PCH, but /// macro USRs embed the source path) and the buffer still starts with /// the blob's stored preamble text. - bool serves_preamble(const Session& session, const index::PreambleState& state) const; + bool serves_preamble(const Session& session, const index::PreambleIndex& state) const; /// Whether an overlay file entry may contribute results. Filters /// synthesized context artifacts (their positions live in diff --git a/src/server/state/session.h b/src/server/state/session.h index 4895c62aa..f35649b6a 100644 --- a/src/server/state/session.h +++ b/src/server/state/session.h @@ -6,7 +6,7 @@ #include #include -#include "index/tu_index.h" +#include "index/shard.h" #include "server/state/quarantine.h" #include "server/state/workspace.h" @@ -149,11 +149,12 @@ struct Session { /// errors never trigger a pointless prefix synthesis. bool trial_done = false; - /// Symbol index built from the latest compilation of this file's buffer. - /// Used for queries (hover, goto, references) on this file. - /// NOT merged into Workspace.project_index — that only gets disk-derived - /// data from background indexing. - std::optional file_index; + /// Symbol index built from the latest compilation of this file's buffer, + /// held as the worker's shard blob and queried through the unified + /// reader (empty = no index). Used for queries (hover, goto, + /// references) on this file. NOT merged into Workspace.project_index — + /// that only gets disk-derived data from background indexing. + index::Shard file_index; /// Symbol table from the latest compilation, mapping symbol hashes to /// names and kinds. diff --git a/src/server/state/workspace.cpp b/src/server/state/workspace.cpp index 1705ba0cc..35f1559dd 100644 --- a/src/server/state/workspace.cpp +++ b/src/server/state/workspace.cpp @@ -374,21 +374,21 @@ struct CacheData { } // namespace -const std::shared_ptr& PCHState::load_state() { +const std::shared_ptr& PCHState::load_state() { if(!state && !index_path.empty()) { - state = index::PreambleState::load(index_path); + state = index::PreambleIndex::load(index_path); if(!state) { // Unreadable blob: clear the path so queries don't retry the // mmap + verification on every call. The pair now looks // incomplete and ensure_pch rebuilds it on the next compile. - LOG_WARN("Failed to open PreambleState blob {}", index_path); + LOG_WARN("Failed to open PreambleIndex blob {}", index_path); index_path.clear(); } } return state; } -std::shared_ptr Workspace::preamble_state(llvm::StringRef pch_key) { +std::shared_ptr Workspace::preamble_state(llvm::StringRef pch_key) { auto it = pch_cache.find(pch_key); if(it == pch_cache.end()) { return nullptr; @@ -403,7 +403,7 @@ std::shared_ptr Workspace::preamble_state(llvm::StringRef // would be served to every session for the rest of the store's // life. Retract it now; the entry itself stays until ensure_pch // re-checks the store and rebuilds the pair. - LOG_WARN("Retracting PCH pair {} with unreadable PreambleState blob", pch_key); + LOG_WARN("Retracting PCH pair {} with unreadable PreambleIndex blob", pch_key); store->invalidate("pch", pch_key); } if(state) { @@ -448,7 +448,7 @@ void Workspace::enforce_loaded_budget() { i += 1; continue; } - LOG_DEBUG("Unloading PreambleState of {} (budget {})", loaded_state_lru[i], budget); + LOG_DEBUG("Unloading PreambleIndex of {} (budget {})", loaded_state_lru[i], budget); it->second.state.reset(); loaded_state_lru.erase(loaded_state_lru.begin() + i); } @@ -501,7 +501,7 @@ void Workspace::load_cache(ContextResolver& contexts) { if(!pch_path) continue; - // A PCH without its PreambleState blob is an incomplete pair + // A PCH without its PreambleIndex blob is an incomplete pair // (crash between the two commits): treat it as absent so the next // compile rebuilds both. auto index_path = store->lookup_aux("pch", entry.key); diff --git a/src/server/state/workspace.h b/src/server/state/workspace.h index e9ee46cd0..cb32a4184 100644 --- a/src/server/state/workspace.h +++ b/src/server/state/workspace.h @@ -12,7 +12,7 @@ #include "command/command.h" #include "command/toolchain.h" #include "compile/dep_file.h" -#include "index/preamble_state.h" +#include "index/preamble_index.h" #include "index/project_index.h" #include "index/shard.h" #include "index/storage.h" @@ -146,25 +146,25 @@ struct SavedContext { /// /// Everything derived from the PCH build beyond validity metadata — the /// preamble's symbol index, document links, inactive regions, the open -/// conditional stack — lives in the paired PreambleState blob (the store's +/// conditional stack — lives in the paired PreambleIndex blob (the store's /// `.pch.idx` aux file), committed and evicted together with the PCH. struct PCHState { std::string path; std::uint32_t bound = 0; DepsSnapshot deps; - /// Path of the paired PreambleState blob. + /// Path of the paired PreambleIndex blob. std::string index_path; /// Lazily opened blob; shared so a consumer holding it across an await /// survives concurrent entry replacement or eviction. - std::shared_ptr state; + std::shared_ptr state; /// Open the blob on first use (memory-mapped, no deserialization). /// Returns nullptr when the blob is missing or unreadable — consumers /// degrade (no overlay, no preamble links) and the next ensure_pch /// treats the incomplete pair as a cache miss. - const std::shared_ptr& load_state(); + const std::shared_ptr& load_state(); std::shared_ptr building; }; @@ -233,7 +233,7 @@ struct Workspace { /// of CacheStore state; blob paths come from the store. llvm::StringMap pch_cache; - /// Keys of pch_cache entries whose PreambleState is currently loaded, + /// Keys of pch_cache entries whose PreambleIndex is currently loaded, /// most recently used first (see enforce_loaded_budget). llvm::SmallVector loaded_state_lru; @@ -311,20 +311,20 @@ struct Workspace { /// is a module unit so dependents can be re-evaluated on next compile. void on_file_closed(std::uint32_t path_id); - /// Open the PreambleState blob of a cached PCH. The single consumption + /// Open the PreambleIndex blob of a cached PCH. The single consumption /// gate for `.pch.idx` blobs: when the blob turns out unreadable, the /// on-disk pair is retracted from the store as well — otherwise every /// later session re-adopts the corrupt pair from cache.json and /// silently degrades again. With the pair gone the next ensure_pch is /// a miss and rebuilds both halves. Loads count against the /// loaded-state budget (see enforce_loaded_budget). - std::shared_ptr preamble_state(llvm::StringRef pch_key); + std::shared_ptr preamble_state(llvm::StringRef pch_key); /// Move a pch key to the front of the loaded-state LRU. Called - /// whenever an entry's PreambleState is opened or replaced. + /// whenever an entry's PreambleIndex is opened or replaced. void touch_loaded_state(llvm::StringRef pch_key); - /// Unload PreambleState blobs beyond the budget (open documents + 2), + /// Unload PreambleIndex blobs beyond the budget (open documents + 2), /// least recently used first. Without this every preamble key ever /// touched keeps its blob mapped for the server's lifetime — tens of /// MB per key on real projects, released by neither didClose nor diff --git a/src/server/worker/stateless_worker.cpp b/src/server/worker/stateless_worker.cpp index e75f07434..6c863c658 100644 --- a/src/server/worker/stateless_worker.cpp +++ b/src/server/worker/stateless_worker.cpp @@ -7,7 +7,7 @@ #include "compile/compilation.h" #include "feature/feature.h" -#include "index/preamble_state.h" +#include "index/preamble_index.h" #include "index/tu_index.h" #include "server/protocol/worker.h" #include "server/worker/worker_common.h" @@ -41,7 +41,7 @@ struct ScopedNice { using kota::ipc::RequestResult; using RequestContext = kota::ipc::BincodePeer::RequestContext; -/// Serialize the preamble's PreambleState blob (full index + document +/// Serialize the preamble's PreambleIndex blob (full index + document /// links + inactive regions) into a string. Runs while the freshly /// parsed AST is still in memory — the only moment the preamble's index /// is obtainable without deserializing the whole PCH. The file write @@ -57,7 +57,7 @@ static std::string serialize_preamble_state(CompilationUnit& unit, std::uint32_t ScopedTimer blob_timer; std::string blob; llvm::raw_string_ostream os(blob); - index::PreambleState::serialize(unit, + index::PreambleIndex::serialize(unit, std::move(tu_index), links, inactive.regions, @@ -79,14 +79,14 @@ static std::optional write_preamble_state(llvm::StringRef blob, llvm::raw_fd_ostream os(output_path, ec); if(ec) { auto message = - std::format("cannot open PreambleState blob {}: {}", output_path, ec.message()); + std::format("cannot open PreambleIndex blob {}: {}", output_path, ec.message()); LOG_ERROR("BuildPCH: {}", message); return message; } os << blob; os.flush(); if(os.has_error()) { - auto message = std::format("failed writing PreambleState blob {}: {}", + auto message = std::format("failed writing PreambleIndex blob {}: {}", output_path, os.error().message()); os.clear_error(); diff --git a/src/support/logging.h b/src/support/logging.h index f0cb38ba9..9e27602e4 100644 --- a/src/support/logging.h +++ b/src/support/logging.h @@ -75,7 +75,7 @@ /// "index_detail" (inside one index pass: op=build splits the semantics /// table from projection and finishing, op=serialize splits the path-id /// rekeying copy from the flatbuffers pack, op=preamble the document -/// links and the PreambleState blob). Use stable key=value pairs and +/// links and the PreambleIndex blob). Use stable key=value pairs and /// `_ms` suffixes for durations — scripts aggregate these lines /// (tools/bench/perf_report.ts). /// diff --git a/tests/unit/index/preamble_state_tests.cpp b/tests/unit/index/preamble_index_tests.cpp similarity index 100% rename from tests/unit/index/preamble_state_tests.cpp rename to tests/unit/index/preamble_index_tests.cpp From 20663221f3d0aac3cb963d33f21c0d00589712f5 Mon Sep 17 00:00:00 2001 From: ykiko Date: Mon, 17 Aug 2026 02:33:16 +0800 Subject: [PATCH 05/10] refactor(index): single envelope format, worker emits bytes, readers everywhere --- src/driver/inspect.cc | 25 +- src/feature/feature.h | 2 +- src/index/preamble_index.cpp | 249 ---------------- src/index/preamble_index.h | 148 --------- src/index/project_index.cpp | 2 +- src/index/project_index.h | 2 +- src/index/serialization.h | 142 ++++++++- src/index/shard.cpp | 114 +++++-- src/index/shard.h | 11 +- src/index/shard_layout.h | 138 --------- src/index/tu_index.cpp | 397 ++++++++++++++++--------- src/index/tu_index.h | 277 +++++++---------- src/index/types.h | 107 +++++++ src/semantic/semantics.h | 2 +- src/server/compiler/compiler.cpp | 48 +-- src/server/compiler/compiler.h | 2 +- src/server/compiler/indexer.cpp | 7 +- src/server/protocol/extension.h | 2 +- src/server/protocol/worker.h | 4 +- src/server/service/query.cpp | 152 +++++++--- src/server/service/query.h | 8 +- src/server/state/session.h | 24 +- src/server/state/workspace.cpp | 35 ++- src/server/state/workspace.h | 26 +- src/server/transport/lsp_client.cpp | 2 +- src/server/worker/stateful_worker.cpp | 4 +- src/server/worker/stateless_worker.cpp | 43 +-- src/support/logging.h | 2 +- 28 files changed, 923 insertions(+), 1052 deletions(-) delete mode 100644 src/index/preamble_index.cpp delete mode 100644 src/index/preamble_index.h delete mode 100644 src/index/shard_layout.h create mode 100644 src/index/types.h diff --git a/src/driver/inspect.cc b/src/driver/inspect.cc index f09c0ea8f..ebd3de65d 100644 --- a/src/driver/inspect.cc +++ b/src/driver/inspect.cc @@ -9,6 +9,7 @@ #include "compile/compilation.h" #include "driver/driver.h" #include "feature/feature.h" +#include "index/shard.h" #include "index/tu_index.h" #include "support/filesystem.h" #include "syntax/annotation.h" @@ -235,21 +236,28 @@ struct RawOccurrence { std::optional run_tu_index(CompilationUnitRef unit, [[maybe_unused]] llvm::StringRef config) { auto index = index::TUIndex::build(unit); - auto sorted = index.main_file_index.occurrences; - std::ranges::sort(sorted, {}, [](const index::Occurrence& occurrence) { - return std::tuple(occurrence.range.begin, occurrence.range.end, occurrence.target); + index::Shard rows; + if(auto* section = index.main_section()) { + rows = index::Shard::from_bytes( + llvm::StringRef(reinterpret_cast(section->blob.data()), + section->blob.size())); + } + + llvm::DenseMap> relations; + rows.for_each_relation([&](index::SymbolHash hash, const index::Relation& relation) { + relations[hash].push_back(relation); + return true; }); std::vector out; - for(const auto& occurrence: sorted) { + rows.for_each_occurrence([&](const index::Occurrence& occurrence) { RawOccurrence raw; raw.range = LocalSourceRange(occurrence.range.begin, occurrence.range.end); auto symbol = index.symbols.find(occurrence.target); raw.kind = symbol != index.symbols.end() ? symbol->second.kind : SymbolKind(SymbolKind::Invalid); - if(auto relations = index.main_file_index.relations.find(occurrence.target); - relations != index.main_file_index.relations.end()) { - for(const auto& relation: relations->second) { + if(auto found = relations.find(occurrence.target); found != relations.end()) { + for(const auto& relation: found->second) { if(relation.range == occurrence.range) { raw.relations.emplace_back( kota::meta::enum_name(static_cast(relation.kind), @@ -258,7 +266,8 @@ std::optional run_tu_index(CompilationUnitRef unit, } } out.push_back(std::move(raw)); - } + return true; + }); return to_raw_json(out); } diff --git a/src/feature/feature.h b/src/feature/feature.h index b8b2b7146..14b490ce7 100644 --- a/src/feature/feature.h +++ b/src/feature/feature.h @@ -275,7 +275,7 @@ struct FoldingRange { /// A resolved document link: the argument range of an include-like /// directive (byte offsets in the containing file) and the absolute path /// of the target file. Plain data — it serializes over the worker RPC and -/// the PCH's PreambleIndex blob as-is and becomes an LSP DocumentLink only +/// the PCH's pch.idx envelope as-is and becomes an LSP DocumentLink only /// at the reply edge, where the session's line map does the conversion. struct DocumentLink { LocalSourceRange range; diff --git a/src/index/preamble_index.cpp b/src/index/preamble_index.cpp deleted file mode 100644 index aede500e5..000000000 --- a/src/index/preamble_index.cpp +++ /dev/null @@ -1,249 +0,0 @@ -#include "index/preamble_index.h" - -#include -#include - -#include "compile/compilation_unit.h" -#include "index/serialization.h" - -#include "llvm/Support/xxhash.h" - -namespace clice::index { - -namespace { - -/// One file covered by the preamble compilation: its shard blob, embedded -/// verbatim from the consumed TUIndex section. Entries are only ever -/// encoded — queries run on Shard readers wrapped at load(). -struct PreambleEntry { - std::uint32_t path_id = 0; - llvm::ArrayRef blob; -}; - -/// find_symbol serves only name and kind, so the blob stores this reduced -/// entry instead of the full Symbol — reflecting that would drag every -/// symbol's scope and reference bitmap into large SDK preamble blobs for -/// nothing. The name borrows the consumed TUIndex (encode-only, like -/// PreambleEntry). -struct PreambleSymbol { - llvm::StringRef name; - SymbolKind kind; -}; - -/// The persisted shape of a `.pch.idx` blob. -struct PreambleBlob { - std::uint32_t format_version = 0; - - /// Identity of the exact preamble text the PCH was built from: - /// xxh3 and byte size. Consumers serve preamble-derived state only - /// while the live buffer's prefix still matches (matches_prefix) — - /// independent of whether the region produced any rows. - std::uint64_t preamble_hash = 0; - std::uint32_t preamble_size = 0; - - std::vector paths; - std::vector files; - PreambleEntry preamble; - llvm::DenseMap symbols; - llvm::ArrayRef links; - llvm::ArrayRef inactive_regions; - llvm::ArrayRef open_conditionals; -}; - -using BlobView = kota::codec::fbs::table_view; - -/// The blob was fully verified at load(); per-query views skip that cost. -BlobView root_of(const llvm::MemoryBuffer& buffer) { - return BlobView::from_verified_bytes(blob_bytes(buffer.getBuffer())); -} - -llvm::StringRef entry_bytes(kota::codec::fbs::table_view entry) { - auto blob = to_array_ref(entry[&PreambleEntry::blob]); - return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); -} - -} // namespace - -void PreambleIndex::serialize(CompilationUnitRef unit, - TUIndex index, - llvm::ArrayRef links, - llvm::ArrayRef inactive_regions, - llvm::ArrayRef open_conditionals, - llvm::raw_ostream& os) { - PreambleBlob blob; - blob.format_version = preamble_format_version; - - // The preamble compile remaps the buffer truncated at the bound, so - // interested_content() is exactly the preamble text the PCH was built - // from. - auto preamble_text = unit.interested_content(); - blob.preamble_hash = llvm::xxh3_64bits(preamble_text); - blob.preamble_size = static_cast(preamble_text.size()); - - // The source file is the last path in graph.paths (convention from - // IncludeGraph); its section holds the preamble region's own rows. - auto main_id = static_cast(index.graph.paths.size()) - 1; - blob.files.reserve(index.sections.size()); - for(auto& section: index.sections) { - if(section.path_id == main_id) { - blob.preamble = {section.path_id, section.blob}; - } else { - blob.files.push_back({section.path_id, section.blob}); - } - } - - blob.symbols.reserve(index.symbols.size()); - for(const auto& [hash, symbol]: index.symbols) { - blob.symbols.try_emplace(hash, PreambleSymbol{.name = symbol.name, .kind = symbol.kind}); - } - blob.paths = std::move(index.graph.paths); - blob.links = links; - blob.inactive_regions = inactive_regions; - blob.open_conditionals = open_conditionals; - - serialize_blob(blob, os); -} - -std::shared_ptr PreambleIndex::load(llvm::StringRef path) { - auto buffer = llvm::MemoryBuffer::getFile(path); - if(!buffer) { - return nullptr; - } - - // A stale or truncated blob must never crash the server. from_bytes - // deep-verifies every offset, string, vector and table the views can - // reach, and each embedded shard blob is verified once by the Shard - // wrap below — queries then run unchecked. Anything failing loads as - // "missing" and the PCH pair is rebuilt. - auto root = BlobView::from_bytes(blob_bytes((*buffer)->getBuffer())); - if(!root.valid() || root[&PreambleBlob::format_version] != preamble_format_version) { - return nullptr; - } - - std::shared_ptr state(new PreambleIndex()); - state->buffer = std::move(*buffer); - - auto verified = root_of(*state->buffer); - auto paths = verified[&PreambleBlob::paths]; - auto files = verified[&PreambleBlob::files]; - state->file_shards.reserve(files.size()); - state->file_paths.reserve(files.size()); - for(std::size_t i = 0; i < files.size(); i += 1) { - auto entry = files[i]; - if(entry[&PreambleEntry::path_id] >= paths.size()) { - return nullptr; - } - auto shard = Shard::from_bytes(entry_bytes(entry)); - if(!shard.loaded()) { - return nullptr; - } - state->file_paths.push_back(to_ref(paths[entry[&PreambleEntry::path_id]])); - state->file_shards.push_back(std::move(shard)); - } - - // The preamble region may legitimately have no rows — an absent blob - // stays an empty shard; corrupt bytes still reject the pair. - auto preamble = entry_bytes(verified[&PreambleBlob::preamble]); - if(!preamble.empty()) { - state->preamble_shard = Shard::from_bytes(preamble); - if(!state->preamble_shard.loaded()) { - return nullptr; - } - } - - return state; -} - -void PreambleIndex::lookup(SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) const { - for(std::size_t i = 0; i < file_shards.size(); i += 1) { - auto& shard = file_shards[i]; - File file{ - .path = file_paths[i], - .content = shard.content(), - .content_size = shard.content_size(), - .line_starts = shard.line_starts(), - }; - bool stopped = false; - shard.lookup(symbol, kind, [&](const Relation& relation) { - if(!callback(file, relation)) { - stopped = true; - return false; - } - return true; - }); - if(stopped) { - return; - } - } -} - -llvm::StringRef PreambleIndex::source_path() const { - auto root = root_of(*buffer); - auto paths = root[&PreambleBlob::paths]; - if(paths.empty()) { - return {}; - } - // The source file is the last path, by IncludeGraph convention. - return to_ref(paths[paths.size() - 1]); -} - -bool PreambleIndex::matches_prefix(llvm::StringRef text) const { - auto root = root_of(*buffer); - auto size = root[&PreambleBlob::preamble_size]; - return text.size() >= size && - llvm::xxh3_64bits(text.take_front(size)) == root[&PreambleBlob::preamble_hash]; -} - -void PreambleIndex::lookup_preamble(std::uint32_t offset, - llvm::function_ref callback) const { - preamble_shard.lookup(offset, callback); -} - -void PreambleIndex::lookup_preamble(SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) const { - preamble_shard.lookup(symbol, kind, callback); -} - -bool PreambleIndex::find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const { - auto root = root_of(*buffer); - auto found = root[&PreambleBlob::symbols].find(hash); - if(!found) { - return false; - } - - auto symbol = found->get<1>(); - name = std::string(symbol[&PreambleSymbol::name]); - kind = SymbolKind(symbol[&PreambleSymbol::kind]); - return true; -} - -std::vector PreambleIndex::links() const { - auto root = root_of(*buffer); - auto entries = root[&PreambleBlob::links]; - - std::vector links; - links.reserve(entries.size()); - for(std::size_t i = 0; i < entries.size(); i += 1) { - auto entry = entries[i]; - links.push_back(feature::DocumentLink{ - .range = entry[&feature::DocumentLink::range], - .target = std::string(entry[&feature::DocumentLink::target]), - }); - } - return links; -} - -llvm::ArrayRef PreambleIndex::inactive_regions() const { - auto root = root_of(*buffer); - return to_array_ref(root[&PreambleBlob::inactive_regions]); -} - -llvm::ArrayRef PreambleIndex::open_conditionals() const { - auto root = root_of(*buffer); - return to_array_ref(root[&PreambleBlob::open_conditionals]); -} - -} // namespace clice::index diff --git a/src/index/preamble_index.h b/src/index/preamble_index.h deleted file mode 100644 index 6969ae742..000000000 --- a/src/index/preamble_index.h +++ /dev/null @@ -1,148 +0,0 @@ -#pragma once - -#include -#include -#include -#include - -#include "feature/feature.h" -#include "index/shard.h" -#include "index/tu_index.h" - -#include "llvm/ADT/ArrayRef.h" -#include "llvm/Support/MemoryBuffer.h" -#include "llvm/Support/raw_ostream.h" - -namespace clice::index { - -/// On-disk PreambleIndex blob schema version (the PCH's `.pch.idx` pair). -/// Bump whenever the persisted layout (its reflected repr or the nested -/// shard blob format) changes; a blob carrying a different value loads as -/// "missing" and the PCH pair is rebuilt. cache.json records it so a -/// version change is caught at load time instead of on the first overlay -/// query. -constexpr inline std::uint32_t preamble_format_version = 6; - -/// All master-visible state derived from one PCH build. -/// -/// The stateless worker serializes it next to the PCH blob (the store's -/// `.pch.idx` pair) and the master opens it as a memory-mapped blob: -/// queries run directly on the serialized data, nothing is deserialized up -/// front. It carries the preamble's full symbol index — one shard blob per -/// header the PCH covers plus the main file's preamble region, the same -/// encoding every other holder of a file's rows uses — and the -/// PCH-derived feature state that is spliced into main-file results: -/// document links, inactive regions and the open conditional stack at the -/// preamble bound. -/// -/// Lifecycle equals the PCH's: the pair is committed, hit and evicted -/// together, so no separate invalidation is needed — a preamble change or -/// a stale dependency rebuilds both. -/// -/// The preamble cannot change while its PCH lives, so feature results are -/// computed once here — by the only live AST that ever sees the preamble — -/// and replayed on every request. New preamble-region features add -/// precomputed rows to this blob and define a per-feature merge with the -/// live results of the rest of the file. -class PreambleIndex { -public: - /// A file entry handed to lookup callbacks: everything needed to turn - /// a byte-offset hit into an LSP location. Views borrow the mapped - /// blob; keep the PreambleIndex alive while using them. - struct File { - llvm::StringRef path; - - /// Empty for pure-ASCII content, which the blob does not store — - /// byte offsets are already UTF-16 column offsets there. - llvm::StringRef content; - - std::uint32_t content_size = 0; - - std::span line_starts; - }; - - /// Serialize a preamble compilation's state. `index` must be built - /// over the preamble unit with interested_only=false; its sections - /// (one shard blob per covered file) are embedded verbatim. Taken by - /// value and consumed: the blob borrows the index's bytes and names - /// for the duration of the write. - static void serialize(CompilationUnitRef unit, - TUIndex index, - llvm::ArrayRef links, - llvm::ArrayRef inactive_regions, - llvm::ArrayRef open_conditionals, - llvm::raw_ostream& os); - - /// Open a blob from disk (memory-mapped). Returns nullptr when the - /// file is unreadable, structurally invalid, written by a different - /// format version, or any embedded shard blob fails verification — - /// callers treat all of these as a PCH cache miss. - static std::shared_ptr load(llvm::StringRef path); - - /// Iterate relations of `symbol` matching `kind` across all header - /// entries. Return false from the callback to stop. This is the only - /// query shape overlays serve: hash-anchored answering. Discovery - /// inputs (by name, by path and line) are the disk index's job. - void lookup(SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) const; - - /// Path of the file whose preamble built this blob. Files with - /// identical preambles share one PCH (the key excludes the source - /// path), but the preamble entry carries file-local symbol identities - /// — macro USRs embed the source path — so its lookups must be scoped - /// to this file. Borrows the mapped blob. - llvm::StringRef source_path() const; - - /// Whether `text` still begins with the exact preamble this blob was - /// built from — the gate for serving preamble-derived state against a - /// live buffer (the rows are offsets into that prefix). Compared by - /// hash: the text itself is not stored. - bool matches_prefix(llvm::StringRef text) const; - - /// Occurrence lookup in the source file's preamble region (buffer - /// offsets below the preamble bound). - void lookup_preamble(std::uint32_t offset, - llvm::function_ref callback) const; - - /// Relations of `symbol` in the source file's preamble region. - void lookup_preamble(SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) const; - - /// Look up a symbol's name and kind in the blob's symbol table. - bool find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const; - - /// Document links of the preamble region, materialized from the blob. - std::vector links() const; - - /// Inactive regions within the preamble (flat begin/end offset pairs). - /// Borrows the mapped blob. - llvm::ArrayRef inactive_regions() const; - - /// Conditional stack still open at the preamble bound. Borrows the - /// mapped blob. - llvm::ArrayRef open_conditionals() const; - - /// Size of the mapped blob in bytes (memory accounting gauge). - std::size_t size() const { - return buffer ? buffer->getBufferSize() : 0; - } - -private: - PreambleIndex() = default; - - std::unique_ptr buffer; - - /// One reader per header entry, wrapping the mapped blob's bytes; - /// line-start caches accumulate here across queries. Paths borrow the - /// mapped blob, parallel to the shards. - std::vector file_shards; - std::vector file_paths; - - /// The source file's preamble-region rows; an empty shard when the - /// region had none. - Shard preamble_shard; -}; - -} // namespace clice::index diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index 0f4c2cb99..d55a456db 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -57,7 +57,7 @@ struct GlobalBlob { } // namespace bool ProjectIndex::merge(this ProjectIndex& self, - const TUIndexView& view, + const TUIndex& view, llvm::ArrayRef file_ids_map) { // Decode and bound every reference bitmap before touching the table: // merged bits persist in the global blob while the result's recorded diff --git a/src/index/project_index.h b/src/index/project_index.h index b74c5bec8..024029726 100644 --- a/src/index/project_index.h +++ b/src/index/project_index.h @@ -86,7 +86,7 @@ struct ProjectIndex { /// the result's recorded versions match the disk, so lost bits would /// never be rebuilt. bool merge(this ProjectIndex& self, - const TUIndexView& view, + const TUIndex& view, llvm::ArrayRef file_ids_map); /// The FileVersion id for (path, content hash), interning a new record diff --git a/src/index/serialization.h b/src/index/serialization.h index 1090b441a..c400db162 100644 --- a/src/index/serialization.h +++ b/src/index/serialization.h @@ -11,7 +11,8 @@ #include #include -#include "index/tu_index.h" +#include "index/shard.h" +#include "index/types.h" #include "semantic/symbol.h" #include "support/bitmap.h" @@ -92,19 +93,6 @@ struct repr { } }; -template <> -struct repr { - using type = std::int64_t; - - static type to(std::chrono::milliseconds ms) { - return ms.count(); - } - - static std::chrono::milliseconds from(type count) { - return std::chrono::milliseconds(count); - } -}; - } // namespace kota::meta namespace clice::index { @@ -125,6 +113,132 @@ void serialize_blob(const T& value, llvm::raw_ostream& os) { os.write(reinterpret_cast(encoded->data()), encoded->size()); } +/// --------------------------------------------------------------------- +/// Persisted blob layouts. Shared by writers, load-time validation and the +/// hand-built blobs of corruption tests; everything else consumes blobs +/// through their readers. +/// Sentinel in a length column (row ranges, line lengths): the real end +/// lives in the sparse escape table. +constexpr inline std::uint8_t length_escape = 0xff; + +/// Files whose content fits 24-bit offsets use the packed range column; +/// larger files fall back to the wide begin/length columns. +constexpr inline std::uint32_t packed_range_limit = 0xffffff; + +/// One row range in the packed column: begin in the high 24 bits, length +/// in the low 8. Raw u32 order equals (begin, length) lexicographic order. +constexpr inline std::uint32_t pack_range(std::uint32_t begin, std::uint8_t length) { + return (begin << 8) | length; +} + +/// The packed spelling of the no-range sentinel pair relations carry (the +/// default LocalSourceRange, ~0u/~0u). Unambiguous: a real packed row with +/// begin 0xffffff and an escaped length would need an end at least 255 +/// past a begin that already sits at the content limit. +constexpr inline std::uint32_t packed_sentinel = 0xffffffff; + +/// One side of the blob's row storage (occurrences or relations). +/// +/// Ranges use exactly one of two self-describing tiers: +/// - packed: `(begin << 8) | length` per row (content < 16MB) +/// - wide: begin u32 + length u8 columns +/// Lengths >= 255 escape to the sparse (row, end) table in either tier. +/// +/// Variant masks are absent for a single variant, then u32 / u64 / +/// concatenated roaring bitmaps by variant count. +struct RowRanges { + std::vector packed; + + std::vector begins; + std::vector lengths; + + std::vector long_rows; + std::vector long_ends; + + std::vector masks32; + std::vector masks64; + std::vector roaring_offsets; + std::vector roaring; +}; + +/// A file's index rows as one persisted blob. +/// +/// The worker encodes one blob per file per TU: single variant (an empty +/// `variants` table), self-contained (content, line table, local symbol +/// names), canonical byte-for-byte — its identity is the hash of its +/// bytes, computed by whoever holds them, never stored inside. The master +/// stores first variants verbatim and merges only when a second distinct +/// variant of the same content generation arrives; merged blobs list the +/// original single-variant identities in `variants` (mask bit position -> +/// identity) and deduplicate rows shared between variants via the masks. +/// +/// Symbol ids used by the row columns index `sym_hashes`; the id column +/// width is chosen from the table size (u8 / u16 / u32). +struct ShardBlob { + /// Persisted-blob schema version (index_format_version), stamped by + /// the writer and gated by Shard::from_bytes. + std::uint32_t format_version = 0; + + /// xxh3 of the content bytes the indexing compile consumed — the + /// content generation these rows were built from. Variants of + /// different generations never share a blob. + std::uint64_t content_hash = 0; + + /// Size of the consumed content; bounds every stored range. + std::uint32_t content_size = 0; + + /// The file's text, for UTF-16 position mapping and text previews. + /// Empty when the content is pure ASCII: byte offsets are already + /// UTF-16 column offsets, so the text itself is dead weight. The form + /// is canonical — a blob storing pure-ASCII content is invalid. + std::string content; + + /// Mask bit position -> variant identity. Empty for a worker-emitted + /// blob (one anonymous variant, identified by its own byte hash). + std::vector variants; + + /// Per-line byte lengths, up to and including the newline; the last + /// entry runs to end of file. Lengths >= 255 escape to the sparse + /// (line, length) table. Line starts are the prefix sums, materialized + /// once at load. + std::vector line_lengths; + std::vector long_line_rows; + std::vector long_line_lengths; + + /// Referenced symbols, sorted by hash; the index into this table is + /// the symbol id the row columns use. + std::vector sym_hashes; + + /// Relation slice per symbol: entry i's relations occupy rows + /// [sym_rel_offsets[i], sym_rel_offsets[i + 1]). Size is table size + 1. + std::vector sym_rel_offsets; + + /// Symbols local to this file (FileLocal) or its TU (TULocal), whose + /// names live nowhere else: sparse over the symbol table, ascending. + std::vector local_syms; + std::vector local_names; + std::vector local_kinds; + std::vector local_scopes; + + /// Occurrences sorted by (begin, end, symbol hash). + RowRanges occs; + std::vector occ_syms8; + std::vector occ_syms16; + std::vector occ_syms32; + + /// Relations in symbol-table order, sorted by (kind, begin, end, + /// payload) within each group. + RowRanges rels; + std::vector rel_kinds; + std::vector rel_sym_rows; + std::vector rel_sym8; + std::vector rel_sym16; + std::vector rel_sym32; + std::vector rel_def_rows; + std::vector rel_def_begins; + std::vector rel_def_ends; +}; + /// The bytes of `data` as the span every kota fbs entry point takes. inline std::span blob_bytes(llvm::StringRef data) { return {reinterpret_cast(data.data()), data.size()}; diff --git a/src/index/shard.cpp b/src/index/shard.cpp index e68469d1a..08030011b 100644 --- a/src/index/shard.cpp +++ b/src/index/shard.cpp @@ -7,7 +7,6 @@ #include #include "index/serialization.h" -#include "index/shard_layout.h" #include "kota/ipc/lsp/text.h" #include "llvm/ADT/DenseMap.h" @@ -614,25 +613,17 @@ void Shard::lookup(std::uint32_t offset, } } -void Shard::lookup(SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) const { - if(!buffer) { - return; - } - auto root = root_of(*buffer); - auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); - auto it = std::ranges::lower_bound(sym_hashes, symbol); - if(it == sym_hashes.end() || *it != symbol) [[unlikely]] { - return; - } - auto id = static_cast(it - sym_hashes.begin()); - - auto offsets = to_array_ref(root[&ShardBlob::sym_rel_offsets]); - auto begin_row = offsets[id]; - auto end_row = offsets[id + 1]; +namespace { +/// Reconstruct and visit the relation rows [begin_row, end_row); `live` +/// filters dead rows, the callback's false stops the walk. +void visit_relation_rows(BlobView root, + std::uint32_t begin_row, + std::uint32_t end_row, + llvm::function_ref live, + llvm::function_ref callback) { auto columns = rel_ranges(root); + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); auto kinds = to_array_ref(root[&ShardBlob::rel_kinds]); auto sym_rows = to_array_ref(root[&ShardBlob::rel_sym_rows]); auto sym8 = to_array_ref(root[&ShardBlob::rel_sym8]); @@ -655,16 +646,12 @@ void Shard::lookup(SymbolHash symbol, def_cursor += 1; } - auto row_kind = static_cast(kinds[row]); - if(!(RelationKind(row_kind) & kind)) { - continue; - } - if(!row_live(false, row)) { + if(!live(row)) { continue; } Relation relation{ - .kind = row_kind, + .kind = static_cast(kinds[row]), .range = {columns.begin_of(row), columns.end_of(row)}, .target_symbol = 0, }; @@ -680,7 +667,84 @@ void Shard::lookup(SymbolHash symbol, } if(!callback(relation)) { - break; + return; + } + } +} + +} // namespace + +void Shard::lookup(SymbolHash symbol, + RelationKind kind, + llvm::function_ref callback) const { + if(!buffer) { + return; + } + auto root = root_of(*buffer); + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + auto it = std::ranges::lower_bound(sym_hashes, symbol); + if(it == sym_hashes.end() || *it != symbol) [[unlikely]] { + return; + } + auto id = static_cast(it - sym_hashes.begin()); + + auto offsets = to_array_ref(root[&ShardBlob::sym_rel_offsets]); + visit_relation_rows( + root, + offsets[id], + offsets[id + 1], + [&](std::uint32_t row) { return row_live(false, row); }, + [&](const Relation& relation) { + if(!(RelationKind(relation.kind) & kind)) { + return true; + } + return callback(relation); + }); +} + +void Shard::for_each_occurrence(llvm::function_ref callback) const { + if(!buffer) { + return; + } + auto root = root_of(*buffer); + auto columns = occ_ranges(root); + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + for(std::uint32_t row = 0; row < columns.size(); row += 1) { + if(!row_live(true, row)) { + continue; + } + Occurrence occurrence{{columns.begin_of(row), columns.end_of(row)}, + sym_hashes[occ_sym_id(root, row)]}; + if(!callback(occurrence)) { + return; + } + } +} + +void Shard::for_each_relation( + llvm::function_ref callback) const { + if(!buffer) { + return; + } + auto root = root_of(*buffer); + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + auto offsets = to_array_ref(root[&ShardBlob::sym_rel_offsets]); + for(std::uint32_t id = 0; id < sym_hashes.size(); id += 1) { + bool stopped = false; + visit_relation_rows( + root, + offsets[id], + offsets[id + 1], + [&](std::uint32_t row) { return row_live(false, row); }, + [&](const Relation& relation) { + if(!callback(sym_hashes[id], relation)) { + stopped = true; + return false; + } + return true; + }); + if(stopped) { + return; } } } diff --git a/src/index/shard.h b/src/index/shard.h index 749f5197c..e46bbfcee 100644 --- a/src/index/shard.h +++ b/src/index/shard.h @@ -7,7 +7,7 @@ #include #include -#include "index/tu_index.h" +#include "index/types.h" #include "support/bitmap.h" #include "llvm/ADT/ArrayRef.h" @@ -93,6 +93,15 @@ class Shard { RelationKind kind, llvm::function_ref callback) const; + /// Visit every live occurrence in row order (sorted by range, then + /// target hash). + void for_each_occurrence(llvm::function_ref callback) const; + + /// Visit every live relation, grouped by symbol in ascending hash + /// order, rows in (kind, range, payload) order within each group. + void for_each_relation( + llvm::function_ref callback) const; + /// Look up a local symbol's name and kind. bool find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const; diff --git a/src/index/shard_layout.h b/src/index/shard_layout.h deleted file mode 100644 index 496860b7e..000000000 --- a/src/index/shard_layout.h +++ /dev/null @@ -1,138 +0,0 @@ -#pragma once - -/// Internal: the persisted layout of a shard blob and the encoding rules -/// shared by its writer, reader and tests. Everything else consumes shards -/// through the `Shard` reader in shard.h — include this header only to -/// build blob bytes by hand (the writer, corruption tests). - -#include -#include -#include - -#include "index/shard.h" - -namespace clice::index { - -/// Sentinel in a length column (row ranges, line lengths): the real end -/// lives in the sparse escape table. -constexpr inline std::uint8_t length_escape = 0xff; - -/// Files whose content fits 24-bit offsets use the packed range column; -/// larger files fall back to the wide begin/length columns. -constexpr inline std::uint32_t packed_range_limit = 0xffffff; - -/// One row range in the packed column: begin in the high 24 bits, length -/// in the low 8. Raw u32 order equals (begin, length) lexicographic order. -constexpr inline std::uint32_t pack_range(std::uint32_t begin, std::uint8_t length) { - return (begin << 8) | length; -} - -/// The packed spelling of the no-range sentinel pair relations carry (the -/// default LocalSourceRange, ~0u/~0u). Unambiguous: a real packed row with -/// begin 0xffffff and an escaped length would need an end at least 255 -/// past a begin that already sits at the content limit. -constexpr inline std::uint32_t packed_sentinel = 0xffffffff; - -/// One side of the blob's row storage (occurrences or relations). -/// -/// Ranges use exactly one of two self-describing tiers: -/// - packed: `(begin << 8) | length` per row (content < 16MB) -/// - wide: begin u32 + length u8 columns -/// Lengths >= 255 escape to the sparse (row, end) table in either tier. -/// -/// Variant masks are absent for a single variant, then u32 / u64 / -/// concatenated roaring bitmaps by variant count. -struct RowRanges { - std::vector packed; - - std::vector begins; - std::vector lengths; - - std::vector long_rows; - std::vector long_ends; - - std::vector masks32; - std::vector masks64; - std::vector roaring_offsets; - std::vector roaring; -}; - -/// A file's index rows as one persisted blob. -/// -/// The worker encodes one blob per file per TU: single variant (an empty -/// `variants` table), self-contained (content, line table, local symbol -/// names), canonical byte-for-byte — its identity is the hash of its -/// bytes, computed by whoever holds them, never stored inside. The master -/// stores first variants verbatim and merges only when a second distinct -/// variant of the same content generation arrives; merged blobs list the -/// original single-variant identities in `variants` (mask bit position -> -/// identity) and deduplicate rows shared between variants via the masks. -/// -/// Symbol ids used by the row columns index `sym_hashes`; the id column -/// width is chosen from the table size (u8 / u16 / u32). -struct ShardBlob { - /// Persisted-blob schema version (index_format_version), stamped by - /// the writer and gated by Shard::from_bytes. - std::uint32_t format_version = 0; - - /// xxh3 of the content bytes the indexing compile consumed — the - /// content generation these rows were built from. Variants of - /// different generations never share a blob. - std::uint64_t content_hash = 0; - - /// Size of the consumed content; bounds every stored range. - std::uint32_t content_size = 0; - - /// The file's text, for UTF-16 position mapping and text previews. - /// Empty when the content is pure ASCII: byte offsets are already - /// UTF-16 column offsets, so the text itself is dead weight. The form - /// is canonical — a blob storing pure-ASCII content is invalid. - std::string content; - - /// Mask bit position -> variant identity. Empty for a worker-emitted - /// blob (one anonymous variant, identified by its own byte hash). - std::vector variants; - - /// Per-line byte lengths, up to and including the newline; the last - /// entry runs to end of file. Lengths >= 255 escape to the sparse - /// (line, length) table. Line starts are the prefix sums, materialized - /// once at load. - std::vector line_lengths; - std::vector long_line_rows; - std::vector long_line_lengths; - - /// Referenced symbols, sorted by hash; the index into this table is - /// the symbol id the row columns use. - std::vector sym_hashes; - - /// Relation slice per symbol: entry i's relations occupy rows - /// [sym_rel_offsets[i], sym_rel_offsets[i + 1]). Size is table size + 1. - std::vector sym_rel_offsets; - - /// Symbols local to this file (FileLocal) or its TU (TULocal), whose - /// names live nowhere else: sparse over the symbol table, ascending. - std::vector local_syms; - std::vector local_names; - std::vector local_kinds; - std::vector local_scopes; - - /// Occurrences sorted by (begin, end, symbol hash). - RowRanges occs; - std::vector occ_syms8; - std::vector occ_syms16; - std::vector occ_syms32; - - /// Relations in symbol-table order, sorted by (kind, begin, end, - /// payload) within each group. - RowRanges rels; - std::vector rel_kinds; - std::vector rel_sym_rows; - std::vector rel_sym8; - std::vector rel_sym16; - std::vector rel_sym32; - std::vector rel_def_rows; - std::vector rel_def_begins; - std::vector rel_def_ends; -}; - -} // namespace clice::index diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index 53d8a830e..ab7e031ca 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -20,6 +20,63 @@ namespace clice::index { namespace { +/// One file's rows on the wire: a self-contained single-variant shard +/// blob (index/shard.h). `hash` is xxh3 of `blob` — the variant's +/// identity — so the master can skip blobs it already stores without +/// touching their bytes. +struct FileSection { + std::uint32_t path_id = 0; + + std::uint64_t hash = 0; + + std::vector blob; +}; + +/// The envelope's wire layout. Only the builder below ever materializes +/// it; every consumer reads the bytes through the TUIndex reader. +struct EnvelopeBlob { + /// Wire schema version (index_format_version), gated by TUIndex::from. + /// A worker respawned after the binary on disk changed can be one + /// build ahead of the server, and a layout change need not be + /// structurally detectable. + std::uint32_t format_version = 0; + + /// Milliseconds since epoch, sampled before the build started. + std::int64_t built_at = 0; + + /// The include graph (IncludeGraph's persisted vectors): the path + /// table, the consumed-content hash per path, and every include edge + /// of the parse. + std::vector paths; + std::vector path_hashes; + std::vector locations; + + SymbolTable symbols; + + /// One entry per file with rows, ascending by path id. + std::vector sections; + + /// Preamble ride-alongs, empty on ordinary envelopes: identity of the + /// exact preamble text the PCH was built from (matches_prefix), and + /// the PCH-derived feature state spliced into main-file results. The + /// refs borrow the builder's inputs — encode-only, like the section + /// blobs are for readers. + std::uint64_t preamble_hash = 0; + std::uint32_t preamble_size = 0; + llvm::ArrayRef links; + llvm::ArrayRef inactive_regions; + llvm::ArrayRef open_conditionals; +}; + +/// What build_preamble_index adds on top of an ordinary build. +struct PreambleExtras { + std::uint64_t hash = 0; + std::uint32_t size = 0; + llvm::ArrayRef links; + llvm::ArrayRef inactive_regions; + llvm::ArrayRef open_conditionals; +}; + SymbolScope classify_scope(const clang::NamedDecl* decl) { auto linkage = decl->getFormalLinkage(); if(linkage == clang::Linkage::None) @@ -33,8 +90,8 @@ SymbolScope classify_scope(const clang::NamedDecl* decl) { /// relations from the resolve facts, macros from the preprocessor directives. class Projector { public: - Projector(TUIndex& result, CompilationUnitRef unit, bool interested_only) : - result(result), unit(unit), interested_only(interested_only) {} + Projector(CompilationUnitRef unit, bool interested_only) : + unit(unit), interested_only(interested_only) {} /// The only gate through which rows enter `file_indices`. With /// interested_only, the index covers just the interested file — yet @@ -75,7 +132,7 @@ class Projector { } auto symbol_id = unit.getSymbolID(decl); - auto [it, success] = result.symbols.try_emplace(symbol_id.hash); + auto [it, success] = symbols.try_emplace(symbol_id.hash); if(success) { auto& symbol = it->second; symbol.name = display::name_of(decl); @@ -102,7 +159,7 @@ class Projector { // build() would default-construct a nameless entry when recording // reference files, and every name lookup for the macro would come // back empty. - auto [it, success] = result.symbols.try_emplace(symbol_id.hash); + auto [it, success] = symbols.try_emplace(symbol_id.hash); if(success) { auto& symbol = it->second; symbol.name = unit.token_spelling(location).str(); @@ -234,7 +291,7 @@ class Projector { } index->relations[hash].emplace_back(relation); - auto& symbol = result.symbols[hash]; + auto& symbol = symbols[hash]; if(symbol.name.empty()) { symbol.name = name.str(); symbol.kind = SymbolKind::Module; @@ -505,7 +562,7 @@ class Projector { } } - void build() { + std::string build(const PreambleExtras* extras) { ScopedTimer semantics_timer; /// The interested-only shape is the one features share, cached on the /// unit; the whole-TU shape is transient — projected and dropped. @@ -532,11 +589,11 @@ class Projector { for(auto& [fid, index]: file_indices) { indexed_fids.push_back(fid); } - result.graph = IncludeGraph::from(unit, indexed_fids); + graph = IncludeGraph::from(unit, indexed_fids); for(auto& [fid, index]: file_indices) { for(auto symbol_id: llvm::make_first_range(index.relations)) { - result.symbols[symbol_id].reference_files.add(result.graph.path_id(fid)); + symbols[symbol_id].reference_files.add(graph.path_id(fid)); } } auto finish_ms = finish_timer.ms_f(); @@ -556,10 +613,10 @@ class Projector { // include edge in the predefines buffer, which is a valid // location. The interested file legitimately has no edge. if(fid != unit.interested_file() && - result.graph.include_location_id(fid) == static_cast(-1)) { + graph.include_location_id(fid) == static_cast(-1)) { continue; } - auto path_id = result.graph.path_id(fid); + auto path_id = graph.path_id(fid); path_fids.try_emplace(path_id, fid); auto& into = by_path[path_id]; if(into.empty()) { @@ -576,8 +633,8 @@ class Projector { } auto resolve = [&](SymbolHash hash) -> std::optional { - auto it = result.symbols.find(hash); - if(it == result.symbols.end()) { + auto it = symbols.find(hash); + if(it == symbols.end()) { return std::nullopt; } return SymbolIdentity{it->second.name, it->second.kind, it->second.scope}; @@ -588,9 +645,8 @@ class Projector { for(auto path_id: llvm::make_first_range(by_path)) { path_ids.push_back(path_id); } - // Ascending path ids put the interested file (the last path id) - // last — the position main_section() relies on. llvm::sort(path_ids); + std::vector sections; for(auto path_id: path_ids) { auto& rows = by_path[path_id]; if(rows.empty()) { @@ -600,107 +656,88 @@ class Projector { llvm::raw_string_ostream os(bytes); write_shard(rows, resolve, unit.file_content(path_fids[path_id]), os); auto hash = llvm::xxh3_64bits(bytes); - result.sections.push_back( + sections.push_back( {path_id, hash, std::vector(bytes.begin(), bytes.end())}); } + auto encode_ms = encode_timer.ms_f(); + + EnvelopeBlob blob; + blob.format_version = index_format_version; + blob.built_at = unit.build_at().count(); + blob.paths = std::move(graph.paths); + blob.path_hashes = std::move(graph.path_hashes); + blob.locations = std::move(graph.locations); + blob.symbols = std::move(symbols); + blob.sections = std::move(sections); + if(extras) { + blob.preamble_hash = extras->hash; + blob.preamble_size = extras->size; + blob.links = extras->links; + blob.inactive_regions = extras->inactive_regions; + blob.open_conditionals = extras->open_conditionals; + } + + ScopedTimer pack_timer; + std::string envelope; + llvm::raw_string_ostream os(envelope); + serialize_blob(blob, os); LOG_PERF("index_detail", "op=build scope={} semantics_ms={:.2f} project_ms={:.2f} finish_ms={:.2f} " - "encode_ms={:.2f}", + "encode_ms={:.2f} pack_ms={:.2f}", interested_only ? "interested" : "full", semantics_ms, project_ms, finish_ms, - encode_timer.ms_f()); + encode_ms, + pack_timer.ms_f()); + return envelope; } private: - TUIndex& result; CompilationUnitRef unit; bool interested_only; + IncludeGraph graph; + SymbolTable symbols; /// Build-time working state keyed by FileID — clang::FileID means /// nothing outside the compilation, so it never leaves the builder; - /// the encode step above converts it through graph.path_id. + /// the encode step converts it through graph.path_id. llvm::DenseMap file_indices; llvm::DenseMap enclosing_cache; }; } // namespace -TUIndex TUIndex::build(CompilationUnitRef unit, bool interested_only) { - TUIndex index; - index.built_at = unit.build_at(); - - Projector projector(index, unit, interested_only); - projector.build(); - - return index; -} - -void TUIndex::serialize(llvm::raw_ostream& os) { - format_version = index_format_version; - - ScopedTimer pack_timer; - serialize_blob(*this, os); - LOG_PERF("index_detail", "op=serialize pack_ms={:.2f}", pack_timer.ms_f()); +std::string build_tu_index(CompilationUnitRef unit, bool interested_only) { + Projector projector(unit, interested_only); + return projector.build(nullptr); } -std::optional TUIndex::from(llvm::StringRef data) { - std::optional index{std::in_place}; - if(!deserialize_blob(data, *index) || index->format_version != index_format_version) { - return std::nullopt; - } - // The verifier checks structure, not cross-field consistency: consumers - // index path_hashes by path id, so normalize its length to the path - // table's (absent hashes read as 0 = "unavailable"). - index->graph.path_hashes.resize(index->graph.paths.size(), 0); - - // Nor does it constrain field values, and every decoded path id is - // dereferenced against the path table without further checks — graph - // locations and wire sections in Indexer::merge, reference_files - // through ProjectIndex::merge's file_ids_map. A blob carrying an - // out-of-range one is rejected as a whole. - auto in_range = [count = index->graph.paths.size()](std::uint32_t path_id) { - return path_id < count; +std::string build_preamble_index(CompilationUnitRef unit, + llvm::ArrayRef links, + llvm::ArrayRef inactive_regions, + llvm::ArrayRef open_conditionals) { + // The preamble compile remaps the buffer truncated at the bound, so + // interested_content() is exactly the preamble text the PCH was built + // from. + auto preamble_text = unit.interested_content(); + PreambleExtras extras{ + .hash = llvm::xxh3_64bits(preamble_text), + .size = static_cast(preamble_text.size()), + .links = links, + .inactive_regions = inactive_regions, + .open_conditionals = open_conditionals, }; - for(auto& location: index->graph.locations) { - if(!in_range(location.path_id)) { - return std::nullopt; - } - } - for(auto& section: index->sections) { - if(!in_range(section.path_id)) { - return std::nullopt; - } - } - for(auto& [_, symbol]: index->symbols) { - if(!symbol.reference_files.isEmpty() && !in_range(symbol.reference_files.maximum())) { - return std::nullopt; - } - } - return index; -} - -const FileSection* TUIndex::main_section() const { - if(graph.paths.empty()) { - return nullptr; - } - auto main_id = static_cast(graph.paths.size() - 1); - // The interested file's section is appended last by serialize(). - for(auto& section: std::ranges::reverse_view(sections)) { - if(section.path_id == main_id) { - return §ion; - } - } - return nullptr; + Projector projector(unit, false); + return projector.build(&extras); } namespace { -using WireView = kota::codec::fbs::table_view; +using WireView = kota::codec::fbs::table_view; -/// The buffer was fully verified at TUIndexView::from; per-accessor views -/// skip that cost. +/// The buffer was fully verified at TUIndex::from_bytes; per-accessor +/// views skip that cost. WireView wire_root(llvm::StringRef data) { return WireView::from_verified_bytes(blob_bytes(data)); } @@ -723,110 +760,204 @@ llvm::StringRef bitmap_bytes(kota::codec::fbs::table_view symbol) { } // namespace -std::optional TUIndexView::from(llvm::StringRef data) { +TUIndex TUIndex::from_bytes(llvm::StringRef data) { auto root = WireView::from_bytes(blob_bytes(data)); - if(!root.valid() || root[&TUIndex::format_version] != index_format_version) { - return std::nullopt; + if(!root.valid() || root[&EnvelopeBlob::format_version] != index_format_version) { + return {}; } // Structural verification does not constrain field values; every path // id the merge dereferences against the path table is bounded here so // the accessors stay check-free. - auto graph = root[&TUIndex::graph]; - auto count = graph[&IncludeGraph::paths].size(); - auto locations = graph[&IncludeGraph::locations]; + auto count = root[&EnvelopeBlob::paths].size(); + auto locations = root[&EnvelopeBlob::locations]; for(std::size_t i = 0; i < locations.size(); i += 1) { IncludeLocation location = locations.at(i); if(location.path_id >= count) { - return std::nullopt; + return {}; } } - auto sections = root[&TUIndex::sections]; + auto sections = root[&EnvelopeBlob::sections]; for(std::size_t i = 0; i < sections.size(); i += 1) { if(sections.at(i)[&FileSection::path_id] >= count) { - return std::nullopt; + return {}; } } - return TUIndexView(data); + + TUIndex result; + result.data = data; + return result; +} + +TUIndex TUIndex::from_buffer(std::unique_ptr buffer) { + if(!buffer) { + return {}; + } + auto result = from_bytes(buffer->getBuffer()); + if(result.loaded()) { + result.owned = std::move(buffer); + } + return result; } -std::int64_t TUIndexView::built_at() const { - return wire_root(data)[&TUIndex::built_at]; +std::int64_t TUIndex::built_at() const { + return loaded() ? wire_root(data)[&EnvelopeBlob::built_at] : 0; } -std::uint32_t TUIndexView::path_count() const { - return static_cast( - wire_root(data)[&TUIndex::graph][&IncludeGraph::paths].size()); +std::uint32_t TUIndex::path_count() const { + return loaded() ? static_cast(wire_root(data)[&EnvelopeBlob::paths].size()) : 0; } -llvm::StringRef TUIndexView::path(std::uint32_t id) const { - return to_ref(wire_root(data)[&TUIndex::graph][&IncludeGraph::paths].at(id)); +llvm::StringRef TUIndex::path(std::uint32_t id) const { + return to_ref(wire_root(data)[&EnvelopeBlob::paths].at(id)); } -std::uint64_t TUIndexView::path_hash(std::uint32_t id) const { +std::uint64_t TUIndex::path_hash(std::uint32_t id) const { // The hash column may be shorter than the path table on a foreign - // blob; an absent hash reads as 0, "unavailable" — the same - // normalization TUIndex::from applies. - auto hashes = wire_root(data)[&TUIndex::graph][&IncludeGraph::path_hashes]; + // blob; an absent hash reads as 0, "unavailable". + auto hashes = wire_root(data)[&EnvelopeBlob::path_hashes]; return id < hashes.size() ? hashes.at(id) : 0; } -std::uint32_t TUIndexView::location_count() const { - return static_cast( - wire_root(data)[&TUIndex::graph][&IncludeGraph::locations].size()); +std::uint32_t TUIndex::location_count() const { + return loaded() + ? static_cast(wire_root(data)[&EnvelopeBlob::locations].size()) + : 0; } -IncludeLocation TUIndexView::location(std::uint32_t i) const { - return wire_root(data)[&TUIndex::graph][&IncludeGraph::locations].at(i); +IncludeLocation TUIndex::location(std::uint32_t i) const { + return wire_root(data)[&EnvelopeBlob::locations].at(i); } -std::uint32_t TUIndexView::section_count() const { - return static_cast(wire_root(data)[&TUIndex::sections].size()); +std::uint32_t TUIndex::section_count() const { + return loaded() ? static_cast(wire_root(data)[&EnvelopeBlob::sections].size()) + : 0; } -std::uint32_t TUIndexView::section_path(std::uint32_t i) const { - return wire_root(data)[&TUIndex::sections].at(i)[&FileSection::path_id]; +std::uint32_t TUIndex::section_path(std::uint32_t i) const { + return wire_root(data)[&EnvelopeBlob::sections].at(i)[&FileSection::path_id]; } -std::uint64_t TUIndexView::section_hash(std::uint32_t i) const { - return wire_root(data)[&TUIndex::sections].at(i)[&FileSection::hash]; +std::uint64_t TUIndex::section_hash(std::uint32_t i) const { + return wire_root(data)[&EnvelopeBlob::sections].at(i)[&FileSection::hash]; } -llvm::StringRef TUIndexView::section_blob(std::uint32_t i) const { - auto blob = to_array_ref(wire_root(data)[&TUIndex::sections].at(i)[&FileSection::blob]); +llvm::StringRef TUIndex::section_blob(std::uint32_t i) const { + auto blob = to_array_ref(wire_root(data)[&EnvelopeBlob::sections].at(i)[&FileSection::blob]); return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); } -std::optional TUIndexView::main_section_index() const { - auto count = path_count(); - if(count == 0) { - return std::nullopt; - } - auto main_id = count - 1; - // The interested file's section is appended last by serialize(). - for(auto i = section_count(); i > 0; i -= 1) { - if(section_path(i - 1) == main_id) { - return i - 1; +std::optional TUIndex::section_of(std::uint32_t path_id) const { + std::uint32_t lo = 0; + std::uint32_t hi = section_count(); + while(lo < hi) { + auto mid = lo + (hi - lo) / 2; + if(section_path(mid) < path_id) { + lo = mid + 1; + } else { + hi = mid; } } + if(lo < section_count() && section_path(lo) == path_id) { + return lo; + } return std::nullopt; } -void TUIndexView::iterate_symbols( +const Shard& TUIndex::shard_of(std::uint32_t path_id) const { + static const Shard missing; + auto section = section_of(path_id); + if(!section) { + return missing; + } + if(shards.empty()) { + shards.resize(section_count()); + } + auto& slot = shards[*section]; + if(!slot.loaded()) { + slot = Shard::from_bytes(section_blob(*section)); + } + return slot; +} + +bool TUIndex::shards_verify() const { + if(shards.empty()) { + shards.resize(section_count()); + } + for(std::uint32_t i = 0; i < section_count(); i += 1) { + if(!shards[i].loaded()) { + shards[i] = Shard::from_bytes(section_blob(i)); + if(!shards[i].loaded()) { + return false; + } + } + } + return true; +} + +void TUIndex::iterate_symbols( llvm::function_ref callback) const { - auto symbols = wire_root(data)[&TUIndex::symbols]; + if(!loaded()) { + return; + } + auto symbols = wire_root(data)[&EnvelopeBlob::symbols]; for(std::size_t i = 0; i < symbols.size(); i += 1) { auto entry = symbols.at(i); callback(entry.get<0>(), identity_of(entry.get<1>()), bitmap_bytes(entry.get<1>())); } } -std::optional TUIndexView::find_symbol(SymbolHash hash) const { - auto found = wire_root(data)[&TUIndex::symbols].find(hash); +std::optional TUIndex::find_symbol(SymbolHash hash) const { + if(!loaded()) { + return std::nullopt; + } + auto found = wire_root(data)[&EnvelopeBlob::symbols].find(hash); if(!found) { return std::nullopt; } return identity_of(found->get<1>()); } +bool TUIndex::matches_prefix(llvm::StringRef text) const { + if(!loaded()) { + return false; + } + auto root = wire_root(data); + auto size = root[&EnvelopeBlob::preamble_size]; + return text.size() >= size && + llvm::xxh3_64bits(text.take_front(size)) == root[&EnvelopeBlob::preamble_hash]; +} + +std::vector TUIndex::links() const { + if(!loaded()) { + return {}; + } + auto entries = wire_root(data)[&EnvelopeBlob::links]; + + std::vector links; + links.reserve(entries.size()); + for(std::size_t i = 0; i < entries.size(); i += 1) { + auto entry = entries[i]; + links.push_back(feature::DocumentLink{ + .range = entry[&feature::DocumentLink::range], + .target = std::string(entry[&feature::DocumentLink::target]), + }); + } + return links; +} + +llvm::ArrayRef TUIndex::inactive_regions() const { + if(!loaded()) { + return {}; + } + return to_array_ref(wire_root(data)[&EnvelopeBlob::inactive_regions]); +} + +llvm::ArrayRef TUIndex::open_conditionals() const { + if(!loaded()) { + return {}; + } + return to_array_ref(wire_root(data)[&EnvelopeBlob::open_conditionals]); +} + } // namespace clice::index diff --git a/src/index/tu_index.h b/src/index/tu_index.h index eba6ee916..198231ed1 100644 --- a/src/index/tu_index.h +++ b/src/index/tu_index.h @@ -1,189 +1,87 @@ #pragma once -#include -#include #include +#include #include #include #include +#include "feature/feature.h" #include "index/include_graph.h" -#include "semantic/symbol.h" -#include "support/bitmap.h" +#include "index/shard.h" +#include "index/types.h" #include "llvm/ADT/STLFunctionalExtras.h" -#include "llvm/Support/raw_ostream.h" +#include "llvm/ADT/StringRef.h" +#include "llvm/Support/MemoryBuffer.h" -namespace clice::index { - -using Range = LocalSourceRange; -using SymbolHash = std::uint64_t; - -/// Visibility scope of a symbol, determining which level of the multi-level -/// symbol table stores it. -enum class SymbolScope : std::uint8_t { - /// Can be referenced from any TU (external linkage). Stored in ProjectIndex. - External = 0, - /// Can be referenced across files within one TU but not across TUs - /// (internal linkage: static, anonymous namespace). Stored in the main - /// file's Shard blob. - TULocal = 1, - /// Cannot be referenced from any other file (local variables, parameters, - /// labels). Stored in the defining file's Shard blob. - FileLocal = 2, -}; +namespace clice { -struct Relation { - /// The raw enum rather than the RelationKind wrapper: the wrapper's - /// constructors hide it from reflection, and reflection is what lets a - /// relation vector persist as one contiguous struct vector. - RelationKind::Kind kind = RelationKind::Invalid; +class CompilationUnitRef; - std::uint32_t padding = 0; +} - LocalSourceRange range; - - SymbolHash target_symbol; - - constexpr void set_definition_range(LocalSourceRange range) { - target_symbol = std::bit_cast(range); - } +namespace clice::index { - constexpr auto definition_range() { - return std::bit_cast(target_symbol); +/// Index one TU and encode the result as its envelope bytes: the include +/// graph (interned into a manifest), the TU's symbol table with +/// per-symbol reference files (merged into the project table), and one +/// self-contained shard blob per file that received rows (stored or +/// merged into the file's disk shard). Rows of a header entered several +/// times are one union blob. The envelope travels worker→server over IPC +/// and is dismantled into the three persistent layers on arrival. With +/// interested_only, only rows in the interested file are kept. +std::string build_tu_index(CompilationUnitRef unit, bool interested_only = false); + +/// The preamble variant: a preamble is a TU cut off at the preamble +/// bound, and its index is the same envelope — persisted verbatim as the +/// PCH's `.pch.idx` pair — plus the preamble-specific fields ordinary +/// envelopes leave empty: the identity of the exact preamble text, and +/// the PCH-derived feature state spliced into main-file results +/// (document links, inactive regions, the open conditional stack at the +/// bound). +std::string build_preamble_index(CompilationUnitRef unit, + llvm::ArrayRef links, + llvm::ArrayRef inactive_regions, + llvm::ArrayRef open_conditionals); + +/// Zero-copy reader over an envelope: the graph, the per-file blob hashes +/// and the blob bytes themselves are read straight off the wire — a new +/// variant's bytes are sliced out and written or merged without ever +/// decoding the envelope around them — and symbol names are touched only +/// when a consumer genuinely needs them. Accessors on an empty reader +/// answer empty/zero. +class TUIndex { +public: + TUIndex() = default; + + /// Wrap verified envelope bytes without owning them (the caller keeps + /// the bytes alive). Verification gates the format version and bounds + /// every path id the graph and sections carry; corrupt bytes load as + /// an empty reader. Symbol reference-file ids are NOT validated — + /// iterate_symbols hands them out raw and the consumer bounds them. + /// Section blob bytes are verified per section, by shard_of on first + /// use or by shards_verify in one pass. + static TUIndex from_bytes(llvm::StringRef data); + + /// Adopt an owning buffer of envelope bytes (a mapped `.pch.idx`, a + /// session's IPC result). Same verification as from_bytes. + static TUIndex from_buffer(std::unique_ptr buffer); + + /// Whether this reader holds an envelope. + bool loaded() const { + return !data.empty(); } -}; - -struct Occurrence { - /// range of this occurrence. - Range range; - /// - SymbolHash target; - - friend bool operator==(const Occurrence&, const Occurrence&) = default; -}; - -/// One file's rows while a build accumulates them; encoded into a shard -/// blob (index/shard.h) at build end and consumed as bytes from then on. -struct FileIndex { - /// The braces matter: fbs decode value-constructs map entries with - /// `FileIndex{}`, and without an initializer this member would be - /// copy-initialized from an empty list, which DenseMap's explicit - /// default constructor rejects. - llvm::DenseMap> relations{}; - - std::vector occurrences; - - bool empty() const { - return occurrences.empty() && relations.empty(); + /// The envelope bytes backing this reader. + llvm::StringRef bytes() const { + return data; } -}; - -struct Symbol { - std::string name; - - SymbolKind kind; - - SymbolScope scope = SymbolScope::External; - - /// All files that referenced this symbol. - Bitmap reference_files; - - friend bool operator==(const Symbol&, const Symbol&) = default; -}; - -using SymbolTable = llvm::DenseMap; - -/// One file's rows on the wire: a self-contained single-variant shard -/// blob (index/shard.h), stored verbatim by the master when the variant -/// is new and merged byte-for-byte otherwise. `hash` is xxh3 of `blob` — -/// the variant's identity — so the master can skip blobs it already -/// stores without touching their bytes. -struct FileSection { - std::uint32_t path_id = 0; - - std::uint64_t hash = 0; - - std::vector blob; -}; - -/// What indexing one TU produced, in transit from worker to master: the -/// include graph (interned into a manifest), the TU's symbol table with -/// per-symbol reference files (merged into the project table), and one -/// shard blob per file that received rows (stored or merged into the -/// file's disk shard). Never persisted itself; it travels worker→server -/// over IPC and is dismantled into the three persistent layers on -/// arrival. -struct TUIndex { - /// Wire schema version (index_format_version), stamped by serialize() - /// and gated by from(). A worker respawned after the binary on disk - /// changed can be one build ahead of the server, and a layout change - /// need not be structurally detectable. - std::uint32_t format_version = 0; - - /// The building timestamp of this file. - std::chrono::milliseconds built_at; - - /// The include information of this file. - IncludeGraph graph; - - SymbolTable symbols; - - /// One entry per file with rows, ascending by path id; the interested - /// file (the last path id) comes last. Rows of a header entered - /// several times are one union blob. Files whose rows are empty get - /// no section — no rows means no contribution. - std::vector sections; - - /// Build the index for `unit`. With interested_only, only rows in - /// the interested file are kept. - static TUIndex build(CompilationUnitRef unit, bool interested_only = false); - - /// Serialization reflects this object directly. - void serialize(llvm::raw_ostream& os); - - /// Verify and deserialize a buffer; nullopt when structural - /// verification fails, the format version differs, or a decoded path id - /// falls outside the blob's own path table. Section blobs stay raw — - /// wrap them in a Shard per file to query them. - static std::optional from(llvm::StringRef data); - - /// The interested file's wire section, or nullptr when its rows were - /// empty. - const FileSection* main_section() const; -}; - -/// A symbol's identity as a merge consumer needs it; the name borrows the -/// wire buffer. -struct SymbolIdentity { - llvm::StringRef name; - SymbolKind kind; - SymbolScope scope; -}; - -/// Zero-copy reader over a serialized TUIndex, for the master's merge path: -/// the graph, the per-file blob hashes and the blob bytes themselves are -/// read straight off the wire — a new variant's bytes are sliced out and -/// written or merged without ever decoding the envelope around them — and -/// symbol names are touched only when a symbol is genuinely new to the -/// global table. The view borrows the wire bytes; keep them alive while -/// using it. -/// -/// TUIndex::from stays the full-decode entry for consumers that need the -/// whole object (sessions, tests). -class TUIndexView { -public: - /// Verify the buffer, gate the format version, and bound every path id - /// the graph and sections carry. Symbol reference-file ids are NOT - /// validated here — iterate_symbols hands them out raw and the consumer - /// bounds them (decoding every bitmap twice just to validate would - /// defeat the view). Section blob bytes are not verified either; - /// Shard::from_bytes verifies each blob the consumer actually uses. - static std::optional from(llvm::StringRef data); std::int64_t built_at() const; + /// The interested file's path is always the last id, by IncludeGraph + /// convention. std::uint32_t path_count() const; llvm::StringRef path(std::uint32_t id) const; @@ -200,12 +98,23 @@ class TUIndexView { std::uint64_t section_hash(std::uint32_t i) const; - /// One section's shard blob bytes, borrowing the wire buffer. + /// One section's shard blob bytes, borrowing the envelope. llvm::StringRef section_blob(std::uint32_t i) const; - /// The section index of the interested file (path_count() - 1), or - /// nullopt when its rows were empty. - std::optional main_section_index() const; + /// The section holding `path_id`'s rows, or nullopt when the file had + /// none (no rows means no contribution). Sections ascend by path id. + std::optional section_of(std::uint32_t path_id) const; + + /// A reader over `path_id`'s rows, wrapped on first use and cached + /// for the envelope's lifetime (the wrap verifies the blob and later + /// materializes its line table). An empty shard when the file has no + /// section or its blob fails verification. + const Shard& shard_of(std::uint32_t path_id) const; + + /// Wrap and verify every section's blob in one pass — the load gate + /// for persisted envelopes, where a corrupt blob must read as "pair + /// missing" and rebuild instead of silently serving nothing. + bool shards_verify() const; /// Visit every symbol: hash, identity, and the raw serialized /// reference-files bitmap (a read_bitmap'able portable image). @@ -216,12 +125,34 @@ class TUIndexView { /// Look up one symbol's identity by hash. std::optional find_symbol(SymbolHash hash) const; -private: - explicit TUIndexView(llvm::StringRef data) : data(data) {} + /// Whether `text` still begins with the exact preamble this envelope + /// was built from — the gate for serving preamble-derived state + /// against a live buffer (the rows are offsets into that prefix). + /// Compared by hash: the text itself is not stored. Always false for + /// an ordinary envelope. + bool matches_prefix(llvm::StringRef text) const; - /// The verified wire bytes; accessors rebuild the (pointer-sized) fbs - /// view from them on demand. + /// Document links of the preamble region, materialized from the + /// envelope; empty for an ordinary one. + std::vector links() const; + + /// Inactive regions within the preamble (flat begin/end offset + /// pairs); empty for an ordinary envelope. Borrows the envelope. + llvm::ArrayRef inactive_regions() const; + + /// Conditional stack still open at the preamble bound; empty for an + /// ordinary envelope. Borrows the envelope. + llvm::ArrayRef open_conditionals() const; + +private: + /// The verified envelope bytes (owned iff `owned` is set); accessors + /// rebuild the (pointer-sized) fbs view from them on demand. + std::unique_ptr owned; llvm::StringRef data; + + /// Lazily wrapped per-section readers; the envelope is immutable for + /// the reader's lifetime, so the cache never invalidates. + mutable std::vector shards; }; } // namespace clice::index diff --git a/src/index/types.h b/src/index/types.h new file mode 100644 index 000000000..bf613c287 --- /dev/null +++ b/src/index/types.h @@ -0,0 +1,107 @@ +#pragma once + +/// The index vocabulary: row and symbol types shared by every layer — +/// builders accumulate them, blob readers hand them out, the project +/// table stores them. + +#include +#include +#include +#include + +#include "semantic/symbol.h" +#include "syntax/token.h" +#include "support/bitmap.h" + +#include "llvm/ADT/DenseMap.h" + +namespace clice::index { + +using Range = LocalSourceRange; +using SymbolHash = std::uint64_t; + +/// Visibility scope of a symbol, determining which level of the multi-level +/// symbol table stores it. +enum class SymbolScope : std::uint8_t { + /// Can be referenced from any TU (external linkage). Stored in ProjectIndex. + External = 0, + /// Can be referenced across files within one TU but not across TUs + /// (internal linkage: static, anonymous namespace). Stored in the main + /// file's Shard blob. + TULocal = 1, + /// Cannot be referenced from any other file (local variables, parameters, + /// labels). Stored in the defining file's Shard blob. + FileLocal = 2, +}; + +struct Relation { + /// The raw enum rather than the RelationKind wrapper: the wrapper's + /// constructors hide it from reflection, and reflection is what lets a + /// relation vector persist as one contiguous struct vector. + RelationKind::Kind kind = RelationKind::Invalid; + + std::uint32_t padding = 0; + + LocalSourceRange range; + + SymbolHash target_symbol; + + constexpr void set_definition_range(LocalSourceRange range) { + target_symbol = std::bit_cast(range); + } + + constexpr auto definition_range() { + return std::bit_cast(target_symbol); + } +}; + +struct Occurrence { + /// range of this occurrence. + Range range; + + /// + SymbolHash target; + + friend bool operator==(const Occurrence&, const Occurrence&) = default; +}; + +/// One file's rows while a build accumulates them; encoded into a shard +/// blob (index/shard.h) at build end and consumed as bytes from then on. +struct FileIndex { + /// The braces matter: fbs decode value-constructs map entries with + /// `FileIndex{}`, and without an initializer this member would be + /// copy-initialized from an empty list, which DenseMap's explicit + /// default constructor rejects. + llvm::DenseMap> relations{}; + + std::vector occurrences; + + bool empty() const { + return occurrences.empty() && relations.empty(); + } +}; + +struct Symbol { + std::string name; + + SymbolKind kind; + + SymbolScope scope = SymbolScope::External; + + /// All files that referenced this symbol. + Bitmap reference_files; + + friend bool operator==(const Symbol&, const Symbol&) = default; +}; + +using SymbolTable = llvm::DenseMap; + +/// A symbol's identity as a blob reader hands it out; the name borrows +/// the blob's bytes. +struct SymbolIdentity { + llvm::StringRef name; + SymbolKind kind; + SymbolScope scope; +}; + +} // namespace clice::index diff --git a/src/semantic/semantics.h b/src/semantic/semantics.h index 286324256..82c3b1852 100644 --- a/src/semantic/semantics.h +++ b/src/semantic/semantics.h @@ -320,7 +320,7 @@ class Semantics { /// The spelled tokens of the interested file (a view into the unit's /// TokenBuffer, not a copy). They cover the whole file even under a /// preamble PCH — what the PCH consumes is the preamble's AST and - /// directives (those travel through PreambleIndex instead), not its + /// directives (those travel through the pch.idx envelope instead), not its /// spelling. llvm::ArrayRef spelled_tokens() const { return tokens; diff --git a/src/server/compiler/compiler.cpp b/src/server/compiler/compiler.cpp index 60d68bea9..f286b2c5b 100644 --- a/src/server/compiler/compiler.cpp +++ b/src/server/compiler/compiler.cpp @@ -8,7 +8,6 @@ #include #include "command/argument_parser.h" -#include "index/preamble_index.h" #include "index/tu_index.h" #include "server/compiler/context_resolver.h" #include "server/protocol/extension.h" @@ -493,7 +492,7 @@ kota::task Compiler::ensure_pch(Session& session, if(auto it = workspace.pch_cache.find(pch_key); it != workspace.pch_cache.end()) { auto& st = it->second; // Both halves of the pair must be present: a PCH whose - // PreambleIndex blob is gone (crash between commits, failed aux + // pch.idx envelope is gone (crash between commits, failed aux // commit) rebuilds whole. bool in_store = workspace.store && workspace.store->lookup("pch", pch_key) && workspace.store->lookup_aux("pch", pch_key); @@ -576,7 +575,7 @@ kota::task Compiler::ensure_pch(Session& session, } // Build a new PCH pair via stateless worker: it writes the PCH and its - // PreambleIndex blob to the tmp paths allocated here; the store + // pch.idx envelope to the tmp paths allocated here; the store // commits (fsync + rename) both on success, primary first. auto pending = workspace.store->begin_store("pch", pch_key); auto pending_idx = workspace.store->begin_store_aux("pch", pch_key); @@ -628,7 +627,7 @@ kota::task Compiler::ensure_pch(Session& session, struct PairCommit { std::optional pch_path; std::optional index_path; - std::shared_ptr state; + std::shared_ptr state; }; auto committed = co_await kota::queue([&]() -> PairCommit { @@ -648,7 +647,7 @@ kota::task Compiler::ensure_pch(Session& session, return outcome; } outcome.index_path = std::move(*index_path); - outcome.state = index::PreambleIndex::load(*outcome.index_path); + outcome.state = load_pch_envelope(*outcome.index_path); return outcome; }); if(!committed.has_value() || !committed.value().pch_path.has_value()) { @@ -656,7 +655,7 @@ kota::task Compiler::ensure_pch(Session& session, co_return false; } if(!committed.value().index_path.has_value()) { - LOG_WARN("Failed to commit PreambleIndex blob for {}", path); + LOG_WARN("Failed to commit pch.idx envelope for {}", path); // A rebuild of an existing key just had its blobs retracted from // the store; the entry's paths now dangle and waiters checking // `!path.empty()` would hand the compile a deleted PCH. Drop it — @@ -974,7 +973,7 @@ kota::task<> Compiler::run_compile(std::shared_ptr session) { // out: concurrent compiles can insert into pch_cache across the // await below and rehash the map from under a held pointer. std::vector pch_inactive; - std::shared_ptr preamble_state; + std::shared_ptr preamble_state; if(session->pch_key.has_value()) { preamble_state = workspace.preamble_state(*session->pch_key); } @@ -1189,32 +1188,15 @@ kota::task<> Compiler::run_compile(std::shared_ptr session) { pc->succeeded = true; record_deps(*session, result.value().deps, result.value().build_at); - auto tu_index = result.value().tu_index_data.empty() - ? std::nullopt - : index::TUIndex::from(result.value().tu_index_data); - if(tu_index) { - // The interested file's rows travel as its shard blob; a file - // with no rows at all has no section and gets an empty-rows - // blob, so a settled empty index still outranks disk fallbacks. - std::string blob; - if(auto* section = tu_index->main_section()) { - blob.assign(section->blob.begin(), section->blob.end()); - } else { - llvm::raw_string_ostream os(blob); - index::write_shard(index::FileIndex(), {}, llvm::StringRef(), os); - } - session->file_index = - index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(blob)); - session->symbols = std::move(tu_index->symbols); - } else { - // The AST and the file index settle together — that pairing is - // what lets navigation trust the index after ensure_compiled. A - // compile that produced no index data (fatal error, no AST) must - // therefore drop the previous buffer's index rather than leave - // it posing as current: an honest gap over yesterday's offsets. - session->file_index = index::Shard(); - session->symbols.reset(); - } + // The AST and the file index settle together — that pairing is + // what lets navigation trust the index after ensure_compiled. A + // compile that produced no index data (fatal error, no AST) must + // therefore drop the previous buffer's index rather than leave it + // posing as current: an honest gap over yesterday's offsets. + auto& index_data = result.value().tu_index_data; + session->index = index_data.empty() ? index::TUIndex() + : index::TUIndex::from_buffer( + llvm::MemoryBuffer::getMemBufferCopy(index_data)); auto version = session->version; diff --git a/src/server/compiler/compiler.h b/src/server/compiler/compiler.h index 1d8d2b354..0c0274129 100644 --- a/src/server/compiler/compiler.h +++ b/src/server/compiler/compiler.h @@ -113,7 +113,7 @@ class Compiler { /// Forward a document-link query to the stateful worker holding this /// file's AST. Covers the main-file region only: the preamble's links - /// live in the PCH's PreambleIndex blob (see PCHState::load_state). + /// live in the PCH's pch.idx envelope (see PCHState::load_state). /// `token`: see forward_query. kota::task, kota::ipc::Error> forward_document_links(std::shared_ptr session, diff --git a/src/server/compiler/indexer.cpp b/src/server/compiler/indexer.cpp index e3777a0ce..742d789f2 100644 --- a/src/server/compiler/indexer.cpp +++ b/src/server/compiler/indexer.cpp @@ -38,13 +38,12 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { // Zero-copy consumption: the wire stays serialized; a new variant's // blob bytes are sliced out and installed or merged without decoding // the envelope, and only genuinely new symbol names are materialized. - auto loaded = - index::TUIndexView::from(llvm::StringRef(static_cast(tu_index_data), size)); - if(!loaded) { + auto view = + index::TUIndex::from_bytes(llvm::StringRef(static_cast(tu_index_data), size)); + if(!view.loaded()) { LOG_WARN("Ignoring TUIndex that failed verification"); return; } - auto& view = *loaded; if(view.path_count() == 0) { LOG_WARN("Ignoring TUIndex with empty path graph"); return; diff --git a/src/server/protocol/extension.h b/src/server/protocol/extension.h index 3cf77a319..598f78e47 100644 --- a/src/server/protocol/extension.h +++ b/src/server/protocol/extension.h @@ -118,7 +118,7 @@ struct LogFloodResult { struct StatsParams {}; struct StatsResult { - /// pch_cache entries whose PreambleIndex blob is currently open, and + /// pch_cache entries whose pch.idx envelope is currently open, and /// their mapped bytes. Steady state after closing documents: bounded /// by the loaded-state budget, not by every key ever touched. std::uint32_t pch_loaded_states = 0; diff --git a/src/server/protocol/worker.h b/src/server/protocol/worker.h index 2745aaaa2..b9ad29a00 100644 --- a/src/server/protocol/worker.h +++ b/src/server/protocol/worker.h @@ -210,7 +210,7 @@ struct BuildParams { std::string output_path; ///< BuildPCH, BuildPCM - /// BuildPCH: tmp path for the PreambleIndex blob (the PCH's paired + /// BuildPCH: tmp path for the pch.idx envelope (the PCH's paired /// `.pch.idx`), allocated by the master's store alongside output_path. /// The worker serializes the preamble's index and feature state into /// it; the master commits both blobs together. @@ -248,7 +248,7 @@ struct BuildResult { /// Request the document links of an open file's AST. Only the main-file /// region is covered: the preamble is compiled into the PCH, and its links -/// live in the PCH's PreambleIndex blob (spliced in by the master). +/// live in the PCH's pch.idx envelope (spliced in by the master). struct DocumentLinkParams { std::string path; }; diff --git a/src/server/service/query.cpp b/src/server/service/query.cpp index 3cc0f02ad..586c4de05 100644 --- a/src/server/service/query.cpp +++ b/src/server/service/query.cpp @@ -14,7 +14,6 @@ #include #include "feature/feature.h" -#include "index/preamble_index.h" #include "index/tu_index.h" #include "server/compiler/compiler.h" #include "server/compiler/indexer.h" @@ -43,7 +42,7 @@ void IndexQuery::visit_sessions(SessionVisitor visitor) const { sessions.for_each([&](std::uint32_t path_id, const Session& session) -> bool { // Freshness contract, clause 3: a dirty session's file index may // describe a buffer that no longer exists — skip it. - if(session.file_index.loaded() && session.symbols && !session.ast_dirty) { + if(session.index.loaded() && !session.ast_dirty) { return visitor(path_id, session); } return true; @@ -54,7 +53,7 @@ bool IndexQuery::is_path_open(std::uint32_t path_id) const { return sessions.find(path_id) != nullptr; } -std::shared_ptr IndexQuery::overlay_of(const Session& session) const { +std::shared_ptr IndexQuery::overlay_of(const Session& session) const { if(!session.pch_key) { return nullptr; } @@ -64,7 +63,7 @@ std::shared_ptr IndexQuery::overlay_of(const Session& sess } void IndexQuery::visit_overlays( - llvm::function_ref visitor) const { + llvm::function_ref visitor) const { if(options.disk_only) { return; } @@ -80,7 +79,7 @@ void IndexQuery::visit_overlays( } void IndexQuery::visit_preambles( - llvm::function_ref visitor) + llvm::function_ref visitor) const { if(options.disk_only) { return; @@ -94,7 +93,7 @@ void IndexQuery::visit_preambles( }); } -bool IndexQuery::serves_preamble(const Session& session, const index::PreambleIndex& state) const { +bool IndexQuery::serves_preamble(const Session& session, const index::TUIndex& state) const { // The preamble entry's rows are buffer offsets of the file that built // the blob: serve them only for that very file (identical preambles // share a PCH, but macro USRs embed the source path) and only while @@ -104,7 +103,7 @@ bool IndexQuery::serves_preamble(const Session& session, const index::PreambleIn // gating is needed on top. The blob stores clang's native path // (backslashes on Windows) while the pool normalizes separators, so // compare through the pool's lookup, not raw strings. - return workspace.path_pool.find(state.source_path()) == session.path_id && + return workspace.path_pool.find(state.path(state.path_count() - 1)) == session.path_id && state.matches_prefix(session.text); } @@ -124,6 +123,62 @@ bool IndexQuery::should_serve_overlay_file(llvm::StringRef path) const { return !workspace.is_synthesized_artifact(path); } +/// A header entry of an overlay envelope, as lookup callbacks consume it. +/// Views borrow the envelope; keep the overlay alive while using them. +struct OverlayFile { + llvm::StringRef path; + + /// Empty for pure-ASCII content, which blobs do not store. + llvm::StringRef content; + + std::uint32_t content_size = 0; + + std::span line_starts; +}; + +/// Iterate relations of `symbol` matching `kind` across an overlay +/// envelope's header entries (every section except the source file's +/// own, which has its own serving gate). This is the only query shape +/// overlays serve: hash-anchored answering; discovery inputs (by name, +/// by path and line) are the disk index's job. +static void overlay_lookup( + const index::TUIndex& state, + index::SymbolHash symbol, + RelationKind kind, + llvm::function_ref callback) { + auto main_id = state.path_count() - 1; + for(std::uint32_t i = 0; i < state.section_count(); i += 1) { + auto path_id = state.section_path(i); + if(path_id == main_id) { + continue; + } + auto& shard = state.shard_of(path_id); + OverlayFile file{ + .path = state.path(path_id), + .content = shard.content(), + .content_size = shard.content_size(), + .line_starts = shard.line_starts(), + }; + bool stopped = false; + shard.lookup(symbol, kind, [&](const index::Relation& relation) { + if(!callback(file, relation)) { + stopped = true; + return false; + } + return true; + }); + if(stopped) { + return; + } + } +} + +/// The source file's preamble-region rows of an overlay envelope (buffer +/// offsets below the preamble bound). +static const index::Shard& preamble_rows(const index::TUIndex& state) { + return state.shard_of(state.path_count() - 1); +} + /// Cross-source dedup: a row present in both a disk shard and a PCH /// overlay (or in two overlays sharing a preamble) comes out identical. static void dedup_locations(std::vector& locations) { @@ -161,12 +216,9 @@ bool IndexQuery::find_symbol_info(index::SymbolHash hash, // Check open sessions first (has all symbols for unsaved buffers). bool found = false; visit_sessions([&](std::uint32_t, const Session& session) -> bool { - if(!session.symbols) - return true; - auto it = session.symbols->find(hash); - if(it != session.symbols->end()) { - name = it->second.name; - kind = it->second.kind; + if(auto identity = session.index.find_symbol(hash)) { + name = std::string(identity->name); + kind = identity->kind; found = true; return false; } @@ -186,8 +238,12 @@ bool IndexQuery::find_symbol_info(index::SymbolHash hash, // Check PCH overlays: a symbol that exists only under an open buffer's // context (or in headers no disk TU has been indexed with) is in no // disk table. - visit_overlays([&](const index::PreambleIndex& state) { - found = state.find_symbol(hash, name, kind); + visit_overlays([&](const index::TUIndex& state) { + if(auto identity = state.find_symbol(hash)) { + name = std::string(identity->name); + kind = identity->kind; + found = true; + } return !found; }); if(found) @@ -214,16 +270,16 @@ IndexQuery::CursorHit IndexQuery::resolve_cursor(llvm::StringRef path, // cross-file visit already skips shards of open files for the same // reason). A dirty-after-await session (failed or superseded compile) // or an index-less one therefore reports no hit. - if(session && (!session->file_index.loaded() || session->ast_dirty)) { + if(session && (!session->index.loaded() || session->ast_dirty)) { return {}; } - if(session && session->file_index.loaded() && !session->ast_dirty) { + if(session && session->index.loaded() && !session->ast_dirty) { auto map = session->line_map(); auto offset = map.to_offset(position); if(!offset) return {}; CursorHit hit; - session->file_index.lookup(*offset, [&](const index::Occurrence& occ) { + session->file_rows().lookup(*offset, [&](const index::Occurrence& occ) { auto range = map.to_range(occ.range.begin, occ.range.end); if(range) { hit = {occ.target, *range}; @@ -238,7 +294,7 @@ IndexQuery::CursorHit IndexQuery::resolve_cursor(llvm::StringRef path, // (preamble drift, shared-PCH identity). auto overlay = hit.hash == 0 ? overlay_of(*session) : nullptr; if(overlay && serves_preamble(*session, *overlay)) { - overlay->lookup_preamble(*offset, [&](const index::Occurrence& occ) { + preamble_rows(*overlay).lookup(*offset, [&](const index::Occurrence& occ) { auto range = map.to_range(occ.range.begin, occ.range.end); if(range) { hit = {occ.target, *range}; @@ -329,7 +385,7 @@ std::vector IndexQuery::query_relations(llvm::StringRef path if(!uri) return true; auto map = session.line_map(); - session.file_index.lookup(hit.hash, kind, [&](const index::Relation& r) { + session.file_rows().lookup(hit.hash, kind, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) locations.push_back({uri->str(), *range}); return true; @@ -340,10 +396,10 @@ std::vector IndexQuery::query_relations(llvm::StringRef path // PCH overlays: header rows under each open buffer's live context. // Rows a disk shard also holds come out identical and collapse in the // dedup below. - visit_overlays([&](const index::PreambleIndex& state) { - state.lookup(hit.hash, + visit_overlays([&](const index::TUIndex& state) { + overlay_lookup(state, hit.hash, kind, - [&](const index::PreambleIndex::File& file, const index::Relation& r) { + [&](const OverlayFile& file, const index::Relation& r) { if(!should_serve_overlay_file(file.path)) return true; // to_uri canonicalizes clang's raw spelling (drive @@ -359,12 +415,12 @@ std::vector IndexQuery::query_relations(llvm::StringRef path // Preamble entries: the buffers' own preamble regions. visit_preambles( - [&](std::uint32_t id, const Session& session, const index::PreambleIndex& state) { + [&](std::uint32_t id, const Session& session, const index::TUIndex& state) { auto uri = lsp::URI::from_file_path(std::string(workspace.path_pool.resolve(id))); if(!uri) return true; auto map = session.line_map(); - state.lookup_preamble(hit.hash, kind, [&](const index::Relation& r) { + preamble_rows(state).lookup(hit.hash, kind, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) locations.push_back({uri->str(), *range}); return true; @@ -425,7 +481,7 @@ std::optional IndexQuery::find_definition_location(index::Sy if(!uri) return true; auto map = session.line_map(); - session.file_index.lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { + session.file_rows().lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) { session_result = protocol::Location{uri->str(), *range}; return false; @@ -443,12 +499,12 @@ std::optional IndexQuery::find_definition_location(index::Sy // First the buffers' own preamble regions, then the header entries. std::optional overlay_result; visit_preambles( - [&](std::uint32_t id, const Session& session, const index::PreambleIndex& state) { + [&](std::uint32_t id, const Session& session, const index::TUIndex& state) { auto uri = lsp::URI::from_file_path(std::string(workspace.path_pool.resolve(id))); if(!uri) return true; auto map = session.line_map(); - state.lookup_preamble(hash, RelationKind::Definition, [&](const index::Relation& r) { + preamble_rows(state).lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) { overlay_result = protocol::Location{uri->str(), *range}; return false; @@ -460,10 +516,10 @@ std::optional IndexQuery::find_definition_location(index::Sy if(overlay_result) return overlay_result; - visit_overlays([&](const index::PreambleIndex& state) { - state.lookup(hash, + visit_overlays([&](const index::TUIndex& state) { + overlay_lookup(state, hash, RelationKind::Definition, - [&](const index::PreambleIndex::File& file, const index::Relation& r) { + [&](const OverlayFile& file, const index::Relation& r) { if(!should_serve_overlay_file(file.path)) return true; auto uri = feature::to_uri(file.path); @@ -557,7 +613,7 @@ void IndexQuery::collect_grouped_relations( } visit_sessions([&](std::uint32_t, const Session& session) -> bool { auto map = session.line_map(); - session.file_index.lookup(hash, kind, [&](const index::Relation& r) { + session.file_rows().lookup(hash, kind, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) target_ranges[r.target_symbol].push_back(*range); return true; @@ -568,10 +624,10 @@ void IndexQuery::collect_grouped_relations( // PCH overlays: call/type relations inside headers under an open // buffer's context. The main-file entry cannot contribute — the // preamble region holds only preprocessor directives. - visit_overlays([&](const index::PreambleIndex& state) { - state.lookup(hash, + visit_overlays([&](const index::TUIndex& state) { + overlay_lookup(state, hash, kind, - [&](const index::PreambleIndex::File& file, const index::Relation& r) { + [&](const OverlayFile& file, const index::Relation& r) { if(!should_serve_overlay_file(file.path)) return true; IndexedLineMap map(file.content, file.content_size, file.line_starts); @@ -621,7 +677,7 @@ void IndexQuery::collect_unique_targets(index::SymbolHash hash, } } visit_sessions([&](std::uint32_t, const Session& session) -> bool { - session.file_index.lookup(hash, kind, [&](const index::Relation& r) { + session.file_rows().lookup(hash, kind, [&](const index::Relation& r) { if(seen.insert(r.target_symbol).second) { targets.push_back(r.target_symbol); } @@ -634,10 +690,10 @@ void IndexQuery::collect_unique_targets(index::SymbolHash hash, // open header's session is authoritative for its relations (an edited // `struct D : NewBase` must not resurface the disk snapshot's OldBase // through another file's overlay). - visit_overlays([&](const index::PreambleIndex& state) { - state.lookup(hash, + visit_overlays([&](const index::TUIndex& state) { + overlay_lookup(state, hash, kind, - [&](const index::PreambleIndex::File& file, const index::Relation& r) { + [&](const OverlayFile& file, const index::Relation& r) { if(!should_serve_overlay_file(file.path)) return true; if(seen.insert(r.target_symbol).second) { @@ -881,26 +937,28 @@ std::vector IndexQuery::search_symbols(llvm::String visit_sessions([&](std::uint32_t, const Session& session) -> bool { if(results.size() >= max_results) return false; - for(auto& [hash, symbol]: *session.symbols) { + session.index.iterate_symbols([&](index::SymbolHash hash, + const index::SymbolIdentity& symbol, + llvm::StringRef) { if(results.size() >= max_results) - return false; + return; if(seen.contains(hash)) - continue; + return; if(!is_indexable_kind(symbol.kind) || symbol.name.empty()) - continue; + return; if(!matches_query(symbol.name)) - continue; + return; auto def_loc = find_definition_location(hash); if(!def_loc) - continue; + return; protocol::SymbolInformation info; - info.name = symbol.name; + info.name = std::string(symbol.name); info.kind = to_lsp_symbol_kind(symbol.kind); info.location = std::move(*def_loc); results.push_back(std::move(info)); seen.insert(hash); - } + }); return true; }); // The query is arbitrary LSP input; its length is logged instead of its diff --git a/src/server/service/query.h b/src/server/service/query.h index 36bffb4e1..000ad3ce1 100644 --- a/src/server/service/query.h +++ b/src/server/service/query.h @@ -231,23 +231,23 @@ class IndexQuery { /// staleness follows the PCH's dependency discipline. Identical rows /// also present in disk shards are collapsed by per-location dedup at /// result assembly. Return false from the visitor to stop. - void visit_overlays(llvm::function_ref visitor) const; + void visit_overlays(llvm::function_ref visitor) const; /// Visit each open session whose overlay preamble entry may serve /// (see serves_preamble), paired with that blob. void visit_preambles(llvm::function_ref visitor) const; + const index::TUIndex& state)> visitor) const; /// The PCH overlay of a session, or nullptr when it has no PCH or the /// blob is unreadable. - std::shared_ptr overlay_of(const Session& session) const; + std::shared_ptr overlay_of(const Session& session) const; /// Whether a session's overlay preamble entry may serve: the blob was /// built from this very file (identical preambles share one PCH, but /// macro USRs embed the source path) and the buffer still starts with /// the blob's stored preamble text. - bool serves_preamble(const Session& session, const index::PreambleIndex& state) const; + bool serves_preamble(const Session& session, const index::TUIndex& state) const; /// Whether an overlay file entry may contribute results. Filters /// synthesized context artifacts (their positions live in diff --git a/src/server/state/session.h b/src/server/state/session.h index f35649b6a..8ab871960 100644 --- a/src/server/state/session.h +++ b/src/server/state/session.h @@ -6,7 +6,7 @@ #include #include -#include "index/shard.h" +#include "index/tu_index.h" #include "server/state/quarantine.h" #include "server/state/workspace.h" @@ -149,16 +149,18 @@ struct Session { /// errors never trigger a pointless prefix synthesis. bool trial_done = false; - /// Symbol index built from the latest compilation of this file's buffer, - /// held as the worker's shard blob and queried through the unified - /// reader (empty = no index). Used for queries (hover, goto, - /// references) on this file. NOT merged into Workspace.project_index — - /// that only gets disk-derived data from background indexing. - index::Shard file_index; - - /// Symbol table from the latest compilation, mapping symbol hashes to - /// names and kinds. - std::optional symbols; + /// The latest compilation's index envelope, owned; the readers it + /// hands out (main-file rows, symbol identities) borrow it. Empty + /// until a compile lands index data. NOT merged into + /// Workspace.project_index — that only gets disk-derived data from + /// background indexing. + index::TUIndex index; + + /// The interested file's rows within `index` (an empty shard when the + /// compile produced none). + const index::Shard& file_rows() const { + return index.shard_of(index.path_count() - 1); + } /// Publishable products of the latest compilation, kept for the /// transport push path (see CompileOutput). diff --git a/src/server/state/workspace.cpp b/src/server/state/workspace.cpp index 35f1559dd..64e6aa0f0 100644 --- a/src/server/state/workspace.cpp +++ b/src/server/state/workspace.cpp @@ -6,6 +6,7 @@ #include #include "command/search_config.h" +#include "index/serialization.h" #include "server/compiler/context_resolver.h" #include "support/filesystem.h" #include "support/logging.h" @@ -358,7 +359,7 @@ struct CachePCMEntry { struct CacheData { std::vector paths; - // preamble_format_version the .pch.idx blobs were written with (one + // index_format_version the .pch.idx envelopes were written with (one // binary writes them all). A mismatch drops every PCH entry at load so // the pairs rebuild immediately, instead of the mismatch surfacing // lazily on the first overlay query — which cannot trigger a rebuild. @@ -374,21 +375,37 @@ struct CacheData { } // namespace -const std::shared_ptr& PCHState::load_state() { +std::shared_ptr load_pch_envelope(llvm::StringRef path) { + auto buffer = llvm::MemoryBuffer::getFile(path); + if(!buffer) { + return nullptr; + } + // A stale or truncated pair must never crash the server: the envelope + // is deep-verified, and every embedded shard blob once — queries then + // run unchecked. Anything failing reads as "pair missing" and the PCH + // is rebuilt. + auto envelope = index::TUIndex::from_buffer(std::move(*buffer)); + if(!envelope.loaded() || !envelope.shards_verify()) { + return nullptr; + } + return std::make_shared(std::move(envelope)); +} + +const std::shared_ptr& PCHState::load_state() { if(!state && !index_path.empty()) { - state = index::PreambleIndex::load(index_path); + state = load_pch_envelope(index_path); if(!state) { // Unreadable blob: clear the path so queries don't retry the // mmap + verification on every call. The pair now looks // incomplete and ensure_pch rebuilds it on the next compile. - LOG_WARN("Failed to open PreambleIndex blob {}", index_path); + LOG_WARN("Failed to open pch.idx envelope {}", index_path); index_path.clear(); } } return state; } -std::shared_ptr Workspace::preamble_state(llvm::StringRef pch_key) { +std::shared_ptr Workspace::preamble_state(llvm::StringRef pch_key) { auto it = pch_cache.find(pch_key); if(it == pch_cache.end()) { return nullptr; @@ -403,7 +420,7 @@ std::shared_ptr Workspace::preamble_state(llvm::StringRef // would be served to every session for the rest of the store's // life. Retract it now; the entry itself stays until ensure_pch // re-checks the store and rebuilds the pair. - LOG_WARN("Retracting PCH pair {} with unreadable PreambleIndex blob", pch_key); + LOG_WARN("Retracting PCH pair {} with unreadable pch.idx envelope", pch_key); store->invalidate("pch", pch_key); } if(state) { @@ -491,7 +508,7 @@ void Workspace::load_cache(ContextResolver& contexts) { return deps; }; - bool pch_format_ok = data.pch_index_format == index::preamble_format_version; + bool pch_format_ok = data.pch_index_format == index::index_format_version; for(auto& entry: data.pch) { if(!pch_format_ok) { break; @@ -501,7 +518,7 @@ void Workspace::load_cache(ContextResolver& contexts) { if(!pch_path) continue; - // A PCH without its PreambleIndex blob is an incomplete pair + // A PCH without its pch.idx envelope is an incomplete pair // (crash between the two commits): treat it as absent so the next // compile rebuilds both. auto index_path = store->lookup_aux("pch", entry.key); @@ -553,7 +570,7 @@ void Workspace::save_cache(const ContextResolver& contexts) { return; CacheData data; - data.pch_index_format = index::preamble_format_version; + data.pch_index_format = index::index_format_version; std::unordered_map index_map; auto intern = [&](std::uint32_t runtime_path_id) -> std::uint32_t { diff --git a/src/server/state/workspace.h b/src/server/state/workspace.h index cb32a4184..4181e044b 100644 --- a/src/server/state/workspace.h +++ b/src/server/state/workspace.h @@ -12,7 +12,7 @@ #include "command/command.h" #include "command/toolchain.h" #include "compile/dep_file.h" -#include "index/preamble_index.h" +#include "index/tu_index.h" #include "index/project_index.h" #include "index/shard.h" #include "index/storage.h" @@ -146,25 +146,31 @@ struct SavedContext { /// /// Everything derived from the PCH build beyond validity metadata — the /// preamble's symbol index, document links, inactive regions, the open -/// conditional stack — lives in the paired PreambleIndex blob (the store's +/// conditional stack — lives in the paired pch.idx envelope (the store's /// `.pch.idx` aux file), committed and evicted together with the PCH. +/// Open a PCH's `.pch.idx` envelope (memory-mapped). Returns nullptr when +/// the file is unreadable, structurally invalid, of a different format +/// version, or any embedded shard blob fails verification — callers treat +/// all of these as a PCH cache miss. +std::shared_ptr load_pch_envelope(llvm::StringRef path); + struct PCHState { std::string path; std::uint32_t bound = 0; DepsSnapshot deps; - /// Path of the paired PreambleIndex blob. + /// Path of the paired pch.idx envelope. std::string index_path; /// Lazily opened blob; shared so a consumer holding it across an await /// survives concurrent entry replacement or eviction. - std::shared_ptr state; + std::shared_ptr state; /// Open the blob on first use (memory-mapped, no deserialization). /// Returns nullptr when the blob is missing or unreadable — consumers /// degrade (no overlay, no preamble links) and the next ensure_pch /// treats the incomplete pair as a cache miss. - const std::shared_ptr& load_state(); + const std::shared_ptr& load_state(); std::shared_ptr building; }; @@ -233,7 +239,7 @@ struct Workspace { /// of CacheStore state; blob paths come from the store. llvm::StringMap pch_cache; - /// Keys of pch_cache entries whose PreambleIndex is currently loaded, + /// Keys of pch_cache entries whose envelope is currently loaded, /// most recently used first (see enforce_loaded_budget). llvm::SmallVector loaded_state_lru; @@ -311,20 +317,20 @@ struct Workspace { /// is a module unit so dependents can be re-evaluated on next compile. void on_file_closed(std::uint32_t path_id); - /// Open the PreambleIndex blob of a cached PCH. The single consumption + /// Open the pch.idx envelope of a cached PCH. The single consumption /// gate for `.pch.idx` blobs: when the blob turns out unreadable, the /// on-disk pair is retracted from the store as well — otherwise every /// later session re-adopts the corrupt pair from cache.json and /// silently degrades again. With the pair gone the next ensure_pch is /// a miss and rebuilds both halves. Loads count against the /// loaded-state budget (see enforce_loaded_budget). - std::shared_ptr preamble_state(llvm::StringRef pch_key); + std::shared_ptr preamble_state(llvm::StringRef pch_key); /// Move a pch key to the front of the loaded-state LRU. Called - /// whenever an entry's PreambleIndex is opened or replaced. + /// whenever an entry's envelope is opened or replaced. void touch_loaded_state(llvm::StringRef pch_key); - /// Unload PreambleIndex blobs beyond the budget (open documents + 2), + /// Unload pch.idx envelopes beyond the budget (open documents + 2), /// least recently used first. Without this every preamble key ever /// touched keeps its blob mapped for the server's lifetime — tens of /// MB per key on real projects, released by neither didClose nor diff --git a/src/server/transport/lsp_client.cpp b/src/server/transport/lsp_client.cpp index e0358b7e5..ecd6347c0 100644 --- a/src/server/transport/lsp_client.cpp +++ b/src/server/transport/lsp_client.cpp @@ -640,7 +640,7 @@ void LSPClient::register_extensions() { auto& st = entry.second; if(st.state) { stats.pch_loaded_states += 1; - stats.pch_state_bytes += st.state->size(); + stats.pch_state_bytes += st.state->bytes().size(); } } stats.pch_cache_entries = static_cast(srv.workspace.pch_cache.size()); diff --git a/src/server/worker/stateful_worker.cpp b/src/server/worker/stateful_worker.cpp index 3720b939a..cd0b39cf4 100644 --- a/src/server/worker/stateful_worker.cpp +++ b/src/server/worker/stateful_worker.cpp @@ -321,9 +321,7 @@ void StatefulWorker::register_handlers() { result.deps = doc->unit.deps(); // Build index for main file only (interested_only=true). - auto tu_index = index::TUIndex::build(doc->unit, true); - llvm::raw_string_ostream os(result.tu_index_data); - tu_index.serialize(os); + result.tu_index_data = index::build_tu_index(doc->unit, true); } // A unit that is neither complete nor a fatal-error result diff --git a/src/server/worker/stateless_worker.cpp b/src/server/worker/stateless_worker.cpp index 6c863c658..4249814a3 100644 --- a/src/server/worker/stateless_worker.cpp +++ b/src/server/worker/stateless_worker.cpp @@ -7,7 +7,6 @@ #include "compile/compilation.h" #include "feature/feature.h" -#include "index/preamble_index.h" #include "index/tu_index.h" #include "server/protocol/worker.h" #include "server/worker/worker_common.h" @@ -41,28 +40,19 @@ struct ScopedNice { using kota::ipc::RequestResult; using RequestContext = kota::ipc::BincodePeer::RequestContext; -/// Serialize the preamble's PreambleIndex blob (full index + document -/// links + inactive regions) into a string. Runs while the freshly -/// parsed AST is still in memory — the only moment the preamble's index -/// is obtainable without deserializing the whole PCH. The file write +/// Serialize the preamble's index envelope (full index + document links +/// + inactive regions) into a string. Runs while the freshly parsed AST +/// is still in memory — the only moment the preamble's index is +/// obtainable without deserializing the whole PCH. The file write /// happens separately, after the PCH itself is flushed. static std::string serialize_preamble_state(CompilationUnit& unit, std::uint32_t preamble_bound) { - auto tu_index = index::TUIndex::build(unit); - ScopedTimer links_timer; auto links = feature::document_links(unit); auto inactive = feature::inactive_regions(unit, {}, 0, preamble_bound); auto links_ms = links_timer.ms_f(); ScopedTimer blob_timer; - std::string blob; - llvm::raw_string_ostream os(blob); - index::PreambleIndex::serialize(unit, - std::move(tu_index), - links, - inactive.regions, - inactive.open_stack, - os); + auto blob = index::build_preamble_index(unit, links, inactive.regions, inactive.open_stack); LOG_PERF("index_detail", "op=preamble links_ms={:.2f} blob_ms={:.2f} bytes={}", links_ms, @@ -79,14 +69,14 @@ static std::optional write_preamble_state(llvm::StringRef blob, llvm::raw_fd_ostream os(output_path, ec); if(ec) { auto message = - std::format("cannot open PreambleIndex blob {}: {}", output_path, ec.message()); + std::format("cannot open pch.idx envelope {}: {}", output_path, ec.message()); LOG_ERROR("BuildPCH: {}", message); return message; } os << blob; os.flush(); if(os.has_error()) { - auto message = std::format("failed writing PreambleIndex blob {}: {}", + auto message = std::format("failed writing pch.idx envelope {}: {}", output_path, os.error().message()); os.clear_error(); @@ -292,33 +282,22 @@ static worker::BuildResult handle_index(const worker::BuildParams& params, return {false, "Index cancelled"}; } ScopedTimer index_timer; - auto tu_index = index::TUIndex::build(unit); + auto serialized = index::build_tu_index(unit); auto index_ms = index_timer.ms(); - ScopedTimer serialize_timer; - std::string serialized; - llvm::raw_string_ostream os(serialized); - tu_index.serialize(os); - auto serialize_ms = serialize_timer.ms(); - // AST teardown for a large TU is material work that belongs to this - // task: sample the total only after the unit and index are gone, so - // the logged span covers everything that blocks the worker. - auto symbol_count = tu_index.symbols.size(); + // task: sample the total only after the unit is gone, so the logged + // span covers everything that blocks the worker. ScopedTimer teardown_timer; - tu_index = index::TUIndex(); unit = CompilationUnit(nullptr); auto teardown_ms = teardown_timer.ms(); LOG_PERF("build", - "kind=index file={} symbols={} bytes={} compile_ms={} index_ms={} serialize_ms={} " - "teardown_ms={} total_ms={}", + "kind=index file={} bytes={} compile_ms={} index_ms={} teardown_ms={} total_ms={}", params.file, - symbol_count, serialized.size(), compile_ms, index_ms, - serialize_ms, teardown_ms, timer.ms()); worker::BuildResult result; diff --git a/src/support/logging.h b/src/support/logging.h index 9e27602e4..ba53022a1 100644 --- a/src/support/logging.h +++ b/src/support/logging.h @@ -75,7 +75,7 @@ /// "index_detail" (inside one index pass: op=build splits the semantics /// table from projection and finishing, op=serialize splits the path-id /// rekeying copy from the flatbuffers pack, op=preamble the document -/// links and the PreambleIndex blob). Use stable key=value pairs and +/// links and the pch.idx envelope). Use stable key=value pairs and /// `_ms` suffixes for durations — scripts aggregate these lines /// (tools/bench/perf_report.ts). /// From ac5eddef0512f62a8dc9d3a4f7619bb3f1986d20 Mon Sep 17 00:00:00 2001 From: ykiko Date: Mon, 17 Aug 2026 04:21:51 +0800 Subject: [PATCH 06/10] test(index): migrate tests and benchmarks to envelope readers --- benchmarks/README.md | 8 +- benchmarks/index_stats_benchmark.cpp | 96 ++--- benchmarks/pipeline_benchmark.cpp | 60 +-- src/driver/inspect.cc | 21 +- src/index/shard.cpp | 63 +-- src/index/shard.h | 3 +- src/index/tu_index.cpp | 7 +- src/index/types.h | 2 +- src/server/compiler/compiler.cpp | 7 +- src/server/service/query.cpp | 215 +++++----- src/server/state/workspace.h | 2 +- src/server/transport/agent_client.cpp | 3 +- tests/unit/index/index_query_tests.cpp | 53 +-- tests/unit/index/preamble_index_tests.cpp | 264 ++++++------ tests/unit/index/project_index_tests.cpp | 70 ++-- tests/unit/index/shard_tests.cpp | 61 ++- tests/unit/index/tu_index_tests.cpp | 421 ++++++++++---------- tests/unit/server/indexer_tests.cpp | 235 ++++++----- tests/unit/server/pch_worker_tests.cpp | 6 +- tests/unit/server/query_freshness_tests.cpp | 49 +-- tests/unit/server/query_overlay_tests.cpp | 159 +++++--- 21 files changed, 918 insertions(+), 887 deletions(-) diff --git a/benchmarks/README.md b/benchmarks/README.md index 91b42a42d..8851c2d99 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -70,10 +70,10 @@ frontend-internal breakdown (preprocessing, parsing, Sema, PCH deserialization) that wall-clock stage timing cannot separate. `--log-level info` additionally surfaces the `[perf:index_detail]` lines -from inside the index stages: semantics-table build vs projection vs -finishing within `TUIndex::build`, and the path-rekeying copy vs the -flatbuffers pack within `serialize`. The same lines appear in worker logs -of a real session, so production runs decompose identically. +from inside the index stage: semantics-table build vs projection vs +finishing vs per-file blob encoding vs the envelope pack within +`build_tu_index`. The same lines appear in worker logs of a real session, +so production runs decompose identically. E2E scenarios, clice vs clangd: diff --git a/benchmarks/index_stats_benchmark.cpp b/benchmarks/index_stats_benchmark.cpp index 49e53b768..271476d0b 100644 --- a/benchmarks/index_stats_benchmark.cpp +++ b/benchmarks/index_stats_benchmark.cpp @@ -1,6 +1,6 @@ /// In-process measurement probe for the on-disk index redesign. Compiles /// every TU in a compilation database exactly like the background-index -/// worker does (full parse without PCH + TUIndex::build), serializes the +/// worker does (full parse without PCH + envelope build), measures the /// index the way production consumes it, then walks the resulting structures /// and accumulates the distributions the new "merged blob" format needs to /// pick its column tiers. @@ -8,7 +8,7 @@ /// Compiles run on a few worker threads, each accumulating into its own /// Stats; the only shared state is a per-(path, variant-hash) registry /// deciding which thread walks a distinct variant's rows — touched once per -/// FileIndex, never per row — so "per distinct variant" populations are not +/// file section, never per row — so "per distinct variant" populations are not /// double-counted across threads. Accumulators merge after join. /// /// This is measurement scratch code: it favours being obvious over being @@ -189,10 +189,10 @@ struct Hist { /// Per source-file aggregates, keyed by path string and folded across every /// TU that touched the file. struct PathAgg { - /// Distinct FileIndex rows hashes this thread claimed for this path; + /// Distinct section blob hashes this thread claimed for this path; /// claims are globally unique, so the merged union is the file's M. std::set variants; - /// Number of TU contributions (one FileIndex per TU per path) — N. + /// Number of TU contributions (one section per TU per path) — N. std::uint32_t contributions = 0; bool size_probed = false; @@ -220,7 +220,7 @@ struct DirAgg { }; /// Arbitrates which thread accumulates a distinct (path, variant) exactly -/// once. Touched once per FileIndex per TU — a coarse coordination point, +/// once. Touched once per section per TU — a coarse coordination point, /// not the per-row hot path. struct VariantRegistry { std::mutex mutex; @@ -256,7 +256,14 @@ struct Stats { std::uint64_t indexed = 0; std::uint64_t had_diagnostics = 0; - void add_file_index(index::FileIndex& fi, llvm::StringRef path) { + /// One file's rows decoded back out of its envelope section. + struct DecodedRows { + std::vector occurrences; + llvm::DenseMap> relations{}; + }; + + /// `hash` is the section's blob hash — the variant's byte identity. + void add_file_index(const DecodedRows& fi, llvm::StringRef path, std::uint64_t hash) { if(fi.occurrences.empty() && fi.relations.empty()) { return; } @@ -272,7 +279,6 @@ struct Stats { } agg.contributions += 1; - auto hash = fi.rows_hash(); bool inserted = registry->try_claim(path, hash); if(inserted) { agg.variants.insert(hash); @@ -341,16 +347,16 @@ struct Stats { } } - void add_directives(index::TUIndex& tu) { - auto& graph = tu.graph; - if(graph.paths.empty()) { + void add_directives(const index::TUIndex& tu) { + if(tu.path_count() == 0) { return; } - auto root_path_id = static_cast(graph.paths.size() - 1); + auto root_path_id = tu.path_count() - 1; // Outgoing edges keyed by parent location index (-1 = TU root). llvm::DenseMap>> outgoing; - for(auto& loc: graph.locations) { + for(std::uint32_t i = 0; i < tu.location_count(); i += 1) { + auto loc = tu.location(i); std::int64_t parent = loc.include == static_cast(-1) ? -1 : static_cast(loc.include); @@ -360,30 +366,28 @@ struct Stats { llvm::StringMap> local; for(auto& [parent, list]: outgoing) { std::uint32_t parent_path_id = - parent < 0 ? root_path_id : graph.locations[parent].path_id; - if(parent_path_id >= graph.paths.size()) { + parent < 0 ? root_path_id : tu.location(static_cast(parent)).path_id; + if(parent_path_id >= tu.path_count()) { continue; } - std::uint64_t parent_hash = - parent_path_id < graph.path_hashes.size() ? graph.path_hashes[parent_path_id] : 0; + std::uint64_t parent_hash = tu.path_hash(parent_path_id); std::ranges::sort(list, [&](auto& a, auto& b) { if(a.first != b.first) { return a.first < b.first; } - return graph.paths[a.second] < graph.paths[b.second]; + return tu.path(a.second) < tu.path(b.second); }); std::string shape; for(auto& [line, child_path_id]: list) { shape += std::format("{},{}\n", line, - child_path_id < graph.paths.size() - ? llvm::StringRef(graph.paths[child_path_id]) - : llvm::StringRef()); + child_path_id < tu.path_count() ? tu.path(child_path_id) + : llvm::StringRef()); } - auto key = std::format("{}#{:016x}", graph.paths[parent_path_id], parent_hash); + auto key = std::format("{}#{:016x}", tu.path(parent_path_id), parent_hash); local[key].insert(llvm::xxh3_64bits(shape)); } @@ -396,28 +400,30 @@ struct Stats { } } - void add_tu(index::TUIndex& tu) { + void add_tu(const index::TUIndex& tu) { indexed += 1; - for(auto& [hash, symbol]: tu.symbols) { - auto scope = static_cast(symbol.scope); - if(scope < scope_n.size()) { - scope_n[scope] += 1; - } - scope_map.try_emplace(hash, symbol.scope); - } + tu.iterate_symbols( + [&](index::SymbolHash hash, const index::SymbolIdentity& symbol, llvm::StringRef) { + auto scope = static_cast(symbol.scope); + if(scope < scope_n.size()) { + scope_n[scope] += 1; + } + scope_map.try_emplace(hash, symbol.scope); + }); - if(!tu.graph.paths.empty()) { - add_file_index(tu.main_file_index, tu.graph.paths.back()); - } - // Multiple FileIDs can share a path id (repeated header contexts); - // last-wins, like the wire sections. - llvm::DenseMap by_path; - for(auto& [fid, fi]: tu.file_indices) { - by_path[tu.graph.path_id(fid)] = &fi; - } - for(auto& [path_id, fi]: by_path) { - add_file_index(*fi, tu.graph.paths[path_id]); + for(std::uint32_t i = 0; i < tu.section_count(); i += 1) { + const auto& shard = tu.shard_of(tu.section_path(i)); + DecodedRows rows; + shard.for_each_occurrence([&](const index::Occurrence& occurrence) { + rows.occurrences.push_back(occurrence); + return true; + }); + shard.for_each_relation([&](index::SymbolHash hash, const index::Relation& relation) { + rows.relations[hash].push_back(relation); + return true; + }); + add_file_index(rows, tu.path(tu.section_path(i)), tu.section_hash(i)); } add_directives(tu); @@ -1178,14 +1184,10 @@ int main(int argc, const char** argv) { stats.had_diagnostics += 1; } - auto tu_index = index::TUIndex::build(unit); - - llvm::SmallString<0> buffer; - llvm::raw_svector_ostream os(buffer); - tu_index.serialize(os); - stats.wire_sizes.add(buffer.size()); + auto envelope = index::build_tu_index(unit); + stats.wire_sizes.add(envelope.size()); - stats.add_tu(tu_index); + stats.add_tu(index::TUIndex::from_bytes(envelope)); finish(""); } }; diff --git a/benchmarks/pipeline_benchmark.cpp b/benchmarks/pipeline_benchmark.cpp index 6fae25b5f..c10463761 100644 --- a/benchmarks/pipeline_benchmark.cpp +++ b/benchmarks/pipeline_benchmark.cpp @@ -6,12 +6,12 @@ /// read source file I/O /// preprocess PreprocessOnlyAction, TokenBuffer off /// preprocess_tokens PreprocessOnlyAction, TokenBuffer on (delta = TokenBuffer cost) -/// parse full parse without PCH + TUIndex build + serialize -/// (the background-index worker shape) -/// pch_build preamble PCH build + preamble index/state blob incl. -/// disk writes (first didOpen shape) -/// parse_pch full parse over the PCH + interactive index build + -/// serialize (the didChange shape) +/// parse full parse without PCH + envelope build (the +/// background-index worker shape) +/// pch_build preamble PCH build + preamble envelope incl. disk +/// writes (first didOpen shape) +/// parse_pch full parse over the PCH + interactive envelope +/// build (the didChange shape) /// /// Usage: /// pipeline_benchmark [OPTIONS] @@ -33,7 +33,6 @@ #include "command/toolchain.h" #include "compile/compilation.h" #include "feature/feature.h" -#include "index/preamble_state.h" #include "index/tu_index.h" #include "support/filesystem.h" #include "support/logging.h" @@ -95,7 +94,6 @@ struct FileResult { double preprocess_tokens_ms = -1; double parse_ms = -1; double index_ms = -1; - double index_serialize_ms = -1; double pch_build_ms = -1; double parse_pch_ms = -1; @@ -216,16 +214,14 @@ FileResult profile_file(llvm::StringRef file, }; ScopedTimer index_timer; - auto tu_index = index::TUIndex::build(unit); + auto serialized = index::build_tu_index(unit); keep_min(result.index_ms, index_timer.ms_f()); - result.symbols = tu_index.symbols.size(); - - ScopedTimer serialize_timer; - std::string serialized; - llvm::raw_string_ostream os(serialized); - tu_index.serialize(os); - keep_min(result.index_serialize_ms, serialize_timer.ms_f()); result.index_bytes = serialized.size(); + + auto view = index::TUIndex::from_bytes(serialized); + std::uint64_t symbols = 0; + view.iterate_symbols([&](auto, auto&, auto) { symbols += 1; }); + result.symbols = symbols; return true; }); if(tracing) { @@ -281,21 +277,13 @@ FileResult profile_file(llvm::StringRef file, } // The production PCH pass (stateless worker) also builds the - // preamble's full index, document links and inactive regions and - // writes the PreambleState blob next to the PCH before reporting - // success; the stage must carry that cost to match a didOpen. - auto tu_index = index::TUIndex::build(unit); + // preamble's envelope, document links and inactive regions and + // writes the blob next to the PCH before reporting success; the + // stage must carry that cost to match a didOpen. auto links = feature::document_links(unit); auto inactive = feature::inactive_regions(unit, {}, 0, result.preamble_bound); open_conditionals = std::move(inactive.open_stack); - std::string blob; - llvm::raw_string_ostream os(blob); - index::PreambleState::serialize(unit, - std::move(tu_index), - links, - inactive.regions, - open_conditionals, - os); + auto blob = index::build_preamble_index(unit, links, inactive.regions, open_conditionals); // The PCH is flushed to disk by the unit's destructor; the blob // write follows it, like the worker's on-disk ordering contract. @@ -325,14 +313,10 @@ FileResult profile_file(llvm::StringRef file, } // The didChange pass (stateful worker) also computes inactive - // regions and builds and serializes the interested-only index - // before replying; include them so parse and parse_pch bound the - // same work. + // regions and builds the interested-only envelope before replying; + // include them so parse and parse_pch bound the same work. feature::inactive_regions(unit, open_conditionals, result.preamble_bound); - auto tu_index = index::TUIndex::build(unit, /*interested_only=*/true); - std::string serialized; - llvm::raw_string_ostream os(serialized); - tu_index.serialize(os); + index::build_tu_index(unit, /*interested_only=*/true); return true; }); @@ -373,7 +357,6 @@ void print_summary(std::vector& results) { {"preprocess_tokens"}, {"parse"}, {"index"}, - {"index_serialize"}, {"pch_build"}, {"parse_pch"}, }; @@ -383,9 +366,8 @@ void print_summary(std::vector& results) { stats[2].add(result.preprocess_tokens_ms); stats[3].add(result.parse_ms); stats[4].add(result.index_ms); - stats[5].add(result.index_serialize_ms); - stats[6].add(result.pch_build_ms); - stats[7].add(result.parse_pch_ms); + stats[5].add(result.pch_build_ms); + stats[6].add(result.parse_pch_ms); } std::println(""); diff --git a/src/driver/inspect.cc b/src/driver/inspect.cc index ebd3de65d..c6bbb0dcc 100644 --- a/src/driver/inspect.cc +++ b/src/driver/inspect.cc @@ -235,13 +235,9 @@ struct RawOccurrence { std::optional run_tu_index(CompilationUnitRef unit, [[maybe_unused]] llvm::StringRef config) { - auto index = index::TUIndex::build(unit); - index::Shard rows; - if(auto* section = index.main_section()) { - rows = index::Shard::from_bytes( - llvm::StringRef(reinterpret_cast(section->blob.data()), - section->blob.size())); - } + auto envelope = index::build_tu_index(unit); + auto index = index::TUIndex::from_bytes(envelope); + const index::Shard& rows = index.shard_of(index.path_count() - 1); llvm::DenseMap> relations; rows.for_each_relation([&](index::SymbolHash hash, const index::Relation& relation) { @@ -252,16 +248,13 @@ std::optional run_tu_index(CompilationUnitRef unit, std::vector out; rows.for_each_occurrence([&](const index::Occurrence& occurrence) { RawOccurrence raw; - raw.range = LocalSourceRange(occurrence.range.begin, occurrence.range.end); - auto symbol = index.symbols.find(occurrence.target); - raw.kind = - symbol != index.symbols.end() ? symbol->second.kind : SymbolKind(SymbolKind::Invalid); + raw.range = occurrence.range; + auto symbol = index.find_symbol(occurrence.target); + raw.kind = symbol ? symbol->kind : SymbolKind(SymbolKind::Invalid); if(auto found = relations.find(occurrence.target); found != relations.end()) { for(const auto& relation: found->second) { if(relation.range == occurrence.range) { - raw.relations.emplace_back( - kota::meta::enum_name(static_cast(relation.kind), - "Invalid")); + raw.relations.emplace_back(kota::meta::enum_name(relation.kind, "Invalid")); } } } diff --git a/src/index/shard.cpp b/src/index/shard.cpp index 08030011b..667543318 100644 --- a/src/index/shard.cpp +++ b/src/index/shard.cpp @@ -87,8 +87,7 @@ struct Ranges { if(!packed.empty() && packed[row] == packed_sentinel) { return ~std::uint32_t(0); } - auto length = - packed.empty() ? lengths[row] : static_cast(packed[row] & 0xff); + auto length = packed.empty() ? lengths[row] : static_cast(packed[row] & 0xff); if(length == length_escape) { // validate() proves every sentinel owns exactly one escape // entry, so the search always lands. @@ -223,7 +222,8 @@ bool validate(BlobView root) { auto line_lengths = to_array_ref(root[&ShardBlob::line_lengths]); auto long_line_rows = to_array_ref(root[&ShardBlob::long_line_rows]); auto long_line_lengths = to_array_ref(root[&ShardBlob::long_line_lengths]); - if(line_lengths.empty() || !escapes_ok(line_lengths, long_line_rows, long_line_lengths.size()) || + if(line_lengths.empty() || + !escapes_ok(line_lengths, long_line_rows, long_line_lengths.size()) || !std::ranges::is_sorted(long_line_rows, std::less_equal{})) { return false; } @@ -264,7 +264,8 @@ bool validate(BlobView root) { if(columns.packed.empty()) { return escapes_ok(columns.lengths, columns.long_rows, columns.long_ends.size()); } - return escapes_ok(packed_lengths(columns.packed), columns.long_rows, + return escapes_ok(packed_lengths(columns.packed), + columns.long_rows, columns.long_ends.size()); }; if(!range_escapes_ok(occ) || !range_escapes_ok(rel)) { @@ -713,16 +714,18 @@ void Shard::for_each_occurrence(llvm::function_ref call if(!row_live(true, row)) { continue; } - Occurrence occurrence{{columns.begin_of(row), columns.end_of(row)}, - sym_hashes[occ_sym_id(root, row)]}; + Occurrence occurrence{ + {columns.begin_of(row), columns.end_of(row)}, + sym_hashes[occ_sym_id(root, row)] + }; if(!callback(occurrence)) { return; } } } -void Shard::for_each_relation( - llvm::function_ref callback) const { +void + Shard::for_each_relation(llvm::function_ref callback) const { if(!buffer) { return; } @@ -965,7 +968,9 @@ llvm::DenseSet referenced_symbols(const MergedRows& merged return referenced; } -constexpr auto occ_key = [](const auto& row) { return std::tuple(row.begin, row.end, row.sym); }; +constexpr auto occ_key = [](const auto& row) { + return std::tuple(row.begin, row.end, row.sym); +}; constexpr auto rel_key = [](const auto& row) { return std::tuple(row.kind, row.begin, row.end, row.payload); }; @@ -1157,8 +1162,8 @@ void emit_row_range(RowRanges& side, return; } auto length = end - begin; - std::uint8_t stored = length >= length_escape ? length_escape - : static_cast(length); + std::uint8_t stored = + length >= length_escape ? length_escape : static_cast(length); if(stored == length_escape) { side.long_rows.push_back(row); side.long_ends.push_back(end); @@ -1332,9 +1337,8 @@ void merge_shards_impl(BlobView old_root, for(std::uint32_t i = 0; i < fresh.size(); i += 1) { auto root = view_of(fresh[i].bytes()); auto bit = single_bit(fresh_base + i); - auto rows = decode_occurrences(root, [&](const Ranges&, std::uint32_t) { - return bit; - }); + auto rows = + decode_occurrences(root, [&](const Ranges&, std::uint32_t) { return bit; }); fresh_occs.insert(fresh_occs.end(), std::make_move_iterator(rows.begin()), std::make_move_iterator(rows.end())); @@ -1356,13 +1360,12 @@ void merge_shards_impl(BlobView old_root, auto bit = single_bit(fresh_base + i); auto columns = rel_ranges(root); for(auto& group: relation_groups(root)) { - auto rows = decode_relation_group(root, - columns, - group.begin_row, - group.end_row, - [&](const Ranges&, std::uint32_t) { - return bit; - }); + auto rows = + decode_relation_group(root, + columns, + group.begin_row, + group.end_row, + [&](const Ranges&, std::uint32_t) { return bit; }); auto& into = fresh_group_map[group.hash]; into.insert(into.end(), std::make_move_iterator(rows.begin()), @@ -1526,11 +1529,21 @@ void merge_shards(const Shard& old, } if(variants.size() <= 64) { - merge_shards_impl(old_root, id_map, fresh, fresh_base, - std::move(variants), content, os); + merge_shards_impl(old_root, + id_map, + fresh, + fresh_base, + std::move(variants), + content, + os); } else { - merge_shards_impl(old_root, id_map, fresh, fresh_base, - std::move(variants), content, os); + merge_shards_impl(old_root, + id_map, + fresh, + fresh_base, + std::move(variants), + content, + os); } } diff --git a/src/index/shard.h b/src/index/shard.h index e46bbfcee..24dc8f285 100644 --- a/src/index/shard.h +++ b/src/index/shard.h @@ -99,8 +99,7 @@ class Shard { /// Visit every live relation, grouped by symbol in ascending hash /// order, rows in (kind, range, payload) order within each group. - void for_each_relation( - llvm::function_ref callback) const; + void for_each_relation(llvm::function_ref callback) const; /// Look up a local symbol's name and kind. bool find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const; diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index ab7e031ca..75d4d0f3b 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -820,9 +820,8 @@ std::uint64_t TUIndex::path_hash(std::uint32_t id) const { } std::uint32_t TUIndex::location_count() const { - return loaded() - ? static_cast(wire_root(data)[&EnvelopeBlob::locations].size()) - : 0; + return loaded() ? static_cast(wire_root(data)[&EnvelopeBlob::locations].size()) + : 0; } IncludeLocation TUIndex::location(std::uint32_t i) const { @@ -865,7 +864,7 @@ std::optional TUIndex::section_of(std::uint32_t path_id) const { } const Shard& TUIndex::shard_of(std::uint32_t path_id) const { - static const Shard missing; + const static Shard missing; auto section = section_of(path_id); if(!section) { return missing; diff --git a/src/index/types.h b/src/index/types.h index bf613c287..d094ce0fe 100644 --- a/src/index/types.h +++ b/src/index/types.h @@ -10,8 +10,8 @@ #include #include "semantic/symbol.h" -#include "syntax/token.h" #include "support/bitmap.h" +#include "syntax/token.h" #include "llvm/ADT/DenseMap.h" diff --git a/src/server/compiler/compiler.cpp b/src/server/compiler/compiler.cpp index f286b2c5b..602f46dc4 100644 --- a/src/server/compiler/compiler.cpp +++ b/src/server/compiler/compiler.cpp @@ -1194,9 +1194,10 @@ kota::task<> Compiler::run_compile(std::shared_ptr session) { // therefore drop the previous buffer's index rather than leave it // posing as current: an honest gap over yesterday's offsets. auto& index_data = result.value().tu_index_data; - session->index = index_data.empty() ? index::TUIndex() - : index::TUIndex::from_buffer( - llvm::MemoryBuffer::getMemBufferCopy(index_data)); + session->index = + index_data.empty() + ? index::TUIndex() + : index::TUIndex::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(index_data)); auto version = session->version; diff --git a/src/server/service/query.cpp b/src/server/service/query.cpp index 586c4de05..e6d7fd0bc 100644 --- a/src/server/service/query.cpp +++ b/src/server/service/query.cpp @@ -1,10 +1,5 @@ #include "server/service/query.h" -#include "server/protocol/position.h" - -#include "llvm/Support/MemoryBuffer.h" -#include "llvm/Support/xxhash.h" - #include #include #include @@ -17,6 +12,7 @@ #include "index/tu_index.h" #include "server/compiler/compiler.h" #include "server/compiler/indexer.h" +#include "server/protocol/position.h" #include "server/state/session.h" #include "server/state/session_store.h" #include "support/filesystem.h" @@ -29,7 +25,9 @@ #include "kota/meta/enum.h" #include "llvm/ADT/DenseSet.h" #include "llvm/ADT/StringSet.h" +#include "llvm/Support/MemoryBuffer.h" #include "llvm/Support/Path.h" +#include "llvm/Support/xxhash.h" namespace clice { @@ -62,8 +60,7 @@ std::shared_ptr IndexQuery::overlay_of(const Session& session) c return workspace.preamble_state(*session.pch_key); } -void IndexQuery::visit_overlays( - llvm::function_ref visitor) const { +void IndexQuery::visit_overlays(llvm::function_ref visitor) const { if(options.disk_only) { return; } @@ -79,8 +76,7 @@ void IndexQuery::visit_overlays( } void IndexQuery::visit_preambles( - llvm::function_ref visitor) - const { + llvm::function_ref visitor) const { if(options.disk_only) { return; } @@ -141,11 +137,11 @@ struct OverlayFile { /// own, which has its own serving gate). This is the only query shape /// overlays serve: hash-anchored answering; discovery inputs (by name, /// by path and line) are the disk index's job. -static void overlay_lookup( - const index::TUIndex& state, - index::SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) { +static void + overlay_lookup(const index::TUIndex& state, + index::SymbolHash symbol, + RelationKind kind, + llvm::function_ref callback) { auto main_id = state.path_count() - 1; for(std::uint32_t i = 0; i < state.section_count(); i += 1) { auto path_id = state.section_path(i); @@ -175,7 +171,7 @@ static void overlay_lookup( /// The source file's preamble-region rows of an overlay envelope (buffer /// offsets below the preamble bound). -static const index::Shard& preamble_rows(const index::TUIndex& state) { +const static index::Shard& preamble_rows(const index::TUIndex& state) { return state.shard_of(state.path_count() - 1); } @@ -397,36 +393,36 @@ std::vector IndexQuery::query_relations(llvm::StringRef path // Rows a disk shard also holds come out identical and collapse in the // dedup below. visit_overlays([&](const index::TUIndex& state) { - overlay_lookup(state, hit.hash, - kind, - [&](const OverlayFile& file, const index::Relation& r) { - if(!should_serve_overlay_file(file.path)) - return true; - // to_uri canonicalizes clang's raw spelling (drive - // case) before emitting. - auto uri = feature::to_uri(file.path); - IndexedLineMap map(file.content, file.content_size, file.line_starts); - if(auto range = map.to_range(r.range.begin, r.range.end)) - locations.push_back({uri, *range}); - return true; - }); + overlay_lookup(state, + hit.hash, + kind, + [&](const OverlayFile& file, const index::Relation& r) { + if(!should_serve_overlay_file(file.path)) + return true; + // to_uri canonicalizes clang's raw spelling (drive + // case) before emitting. + auto uri = feature::to_uri(file.path); + IndexedLineMap map(file.content, file.content_size, file.line_starts); + if(auto range = map.to_range(r.range.begin, r.range.end)) + locations.push_back({uri, *range}); + return true; + }); return true; }); // Preamble entries: the buffers' own preamble regions. - visit_preambles( - [&](std::uint32_t id, const Session& session, const index::TUIndex& state) { - auto uri = lsp::URI::from_file_path(std::string(workspace.path_pool.resolve(id))); - if(!uri) - return true; - auto map = session.line_map(); - preamble_rows(state).lookup(hit.hash, kind, [&](const index::Relation& r) { - if(auto range = map.to_range(r.range.begin, r.range.end)) - locations.push_back({uri->str(), *range}); - return true; - }); + visit_preambles([&](std::uint32_t id, const Session& session, const index::TUIndex& state) { + auto uri = lsp::URI::from_file_path(std::string(workspace.path_pool.resolve(id))); + if(!uri) + return true; + auto map = session.line_map(); + preamble_rows(state).lookup(hit.hash, kind, [&](const index::Relation& r) { + if(auto range = map.to_range(r.range.begin, r.range.end)) + locations.push_back({uri->str(), *range}); return true; }); + return true; + }); dedup_locations(locations); LOG_PERF("index_query", @@ -498,38 +494,38 @@ std::optional IndexQuery::find_definition_location(index::Sy // been indexed — the in-memory-file case behind empty go-to-definition. // First the buffers' own preamble regions, then the header entries. std::optional overlay_result; - visit_preambles( - [&](std::uint32_t id, const Session& session, const index::TUIndex& state) { - auto uri = lsp::URI::from_file_path(std::string(workspace.path_pool.resolve(id))); - if(!uri) - return true; - auto map = session.line_map(); - preamble_rows(state).lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { - if(auto range = map.to_range(r.range.begin, r.range.end)) { - overlay_result = protocol::Location{uri->str(), *range}; - return false; - } - return true; - }); - return !overlay_result.has_value(); + visit_preambles([&](std::uint32_t id, const Session& session, const index::TUIndex& state) { + auto uri = lsp::URI::from_file_path(std::string(workspace.path_pool.resolve(id))); + if(!uri) + return true; + auto map = session.line_map(); + preamble_rows(state).lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { + if(auto range = map.to_range(r.range.begin, r.range.end)) { + overlay_result = protocol::Location{uri->str(), *range}; + return false; + } + return true; }); + return !overlay_result.has_value(); + }); if(overlay_result) return overlay_result; visit_overlays([&](const index::TUIndex& state) { - overlay_lookup(state, hash, - RelationKind::Definition, - [&](const OverlayFile& file, const index::Relation& r) { - if(!should_serve_overlay_file(file.path)) - return true; - auto uri = feature::to_uri(file.path); - IndexedLineMap map(file.content, file.content_size, file.line_starts); - if(auto range = map.to_range(r.range.begin, r.range.end)) { - overlay_result = protocol::Location{uri, *range}; - return false; - } - return true; - }); + overlay_lookup(state, + hash, + RelationKind::Definition, + [&](const OverlayFile& file, const index::Relation& r) { + if(!should_serve_overlay_file(file.path)) + return true; + auto uri = feature::to_uri(file.path); + IndexedLineMap map(file.content, file.content_size, file.line_starts); + if(auto range = map.to_range(r.range.begin, r.range.end)) { + overlay_result = protocol::Location{uri, *range}; + return false; + } + return true; + }); return !overlay_result.has_value(); }); if(overlay_result) @@ -553,7 +549,7 @@ std::optional IndexQuery::find_definition_location(index::Sy auto ls = merged_index.line_starts(); if(ls.empty()) continue; - lsp::LineMap map(merged_index.content(), ls); + IndexedLineMap map(merged_index.content(), merged_index.content_size(), ls); std::optional result; merged_index.lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) { @@ -625,16 +621,14 @@ void IndexQuery::collect_grouped_relations( // buffer's context. The main-file entry cannot contribute — the // preamble region holds only preprocessor directives. visit_overlays([&](const index::TUIndex& state) { - overlay_lookup(state, hash, - kind, - [&](const OverlayFile& file, const index::Relation& r) { - if(!should_serve_overlay_file(file.path)) - return true; - IndexedLineMap map(file.content, file.content_size, file.line_starts); - if(auto range = map.to_range(r.range.begin, r.range.end)) - target_ranges[r.target_symbol].push_back(*range); - return true; - }); + overlay_lookup(state, hash, kind, [&](const OverlayFile& file, const index::Relation& r) { + if(!should_serve_overlay_file(file.path)) + return true; + IndexedLineMap map(file.content, file.content_size, file.line_starts); + if(auto range = map.to_range(r.range.begin, r.range.end)) + target_ranges[r.target_symbol].push_back(*range); + return true; + }); return true; }); @@ -691,16 +685,14 @@ void IndexQuery::collect_unique_targets(index::SymbolHash hash, // `struct D : NewBase` must not resurface the disk snapshot's OldBase // through another file's overlay). visit_overlays([&](const index::TUIndex& state) { - overlay_lookup(state, hash, - kind, - [&](const OverlayFile& file, const index::Relation& r) { - if(!should_serve_overlay_file(file.path)) - return true; - if(seen.insert(r.target_symbol).second) { - targets.push_back(r.target_symbol); - } - return true; - }); + overlay_lookup(state, hash, kind, [&](const OverlayFile& file, const index::Relation& r) { + if(!should_serve_overlay_file(file.path)) + return true; + if(seen.insert(r.target_symbol).second) { + targets.push_back(r.target_symbol); + } + return true; + }); return true; }); } @@ -809,11 +801,11 @@ std::vector IndexQuery::collect_references(ind continue; auto& merged_index = shard_it->second; auto file_path = workspace.path_pool.resolve(file_id); + // A moved-on ASCII file yields no text: positions still map + // through the line table, only the context line degrades. auto text = indexed_text(file_path, merged_index); - if(!text) - continue; - llvm::StringRef content = *text; - lsp::LineMap map(content, merged_index.line_starts()); + llvm::StringRef content = text ? llvm::StringRef(*text) : llvm::StringRef(); + IndexedLineMap map(content, merged_index.content_size(), merged_index.line_starts()); merged_index.lookup(hash, kind, [&](const index::Relation& r) { auto pos = map.to_position(r.range.begin); @@ -937,28 +929,27 @@ std::vector IndexQuery::search_symbols(llvm::String visit_sessions([&](std::uint32_t, const Session& session) -> bool { if(results.size() >= max_results) return false; - session.index.iterate_symbols([&](index::SymbolHash hash, - const index::SymbolIdentity& symbol, - llvm::StringRef) { - if(results.size() >= max_results) - return; - if(seen.contains(hash)) - return; - if(!is_indexable_kind(symbol.kind) || symbol.name.empty()) - return; - if(!matches_query(symbol.name)) - return; - auto def_loc = find_definition_location(hash); - if(!def_loc) - return; + session.index.iterate_symbols( + [&](index::SymbolHash hash, const index::SymbolIdentity& symbol, llvm::StringRef) { + if(results.size() >= max_results) + return; + if(seen.contains(hash)) + return; + if(!is_indexable_kind(symbol.kind) || symbol.name.empty()) + return; + if(!matches_query(symbol.name)) + return; + auto def_loc = find_definition_location(hash); + if(!def_loc) + return; - protocol::SymbolInformation info; - info.name = std::string(symbol.name); - info.kind = to_lsp_symbol_kind(symbol.kind); - info.location = std::move(*def_loc); - results.push_back(std::move(info)); - seen.insert(hash); - }); + protocol::SymbolInformation info; + info.name = std::string(symbol.name); + info.kind = to_lsp_symbol_kind(symbol.kind); + info.location = std::move(*def_loc); + results.push_back(std::move(info)); + seen.insert(hash); + }); return true; }); // The query is arbitrary LSP input; its length is logged instead of its diff --git a/src/server/state/workspace.h b/src/server/state/workspace.h index 4181e044b..edb58b6b1 100644 --- a/src/server/state/workspace.h +++ b/src/server/state/workspace.h @@ -12,10 +12,10 @@ #include "command/command.h" #include "command/toolchain.h" #include "compile/dep_file.h" -#include "index/tu_index.h" #include "index/project_index.h" #include "index/shard.h" #include "index/storage.h" +#include "index/tu_index.h" #include "semantic/symbol.h" #include "server/compiler/compile_graph.h" #include "server/state/config.h" diff --git a/src/server/transport/agent_client.cpp b/src/server/transport/agent_client.cpp index 43c204813..545d790a2 100644 --- a/src/server/transport/agent_client.cpp +++ b/src/server/transport/agent_client.cpp @@ -8,6 +8,7 @@ #include #include "server/protocol/agentic.h" +#include "server/protocol/position.h" #include "server/transport/master_server.h" #include "support/filesystem.h" #include "support/logging.h" @@ -363,7 +364,7 @@ AgentClient::AgentClient(MasterServer& server, kota::ipc::JsonPeer& peer) : auto ls = merged_index.line_starts(); if(ls.empty()) co_return result; - lsp::LineMap map(merged_index.content(), ls); + IndexedLineMap map(merged_index.content(), merged_index.content_size(), ls); for(auto& [hash, symbol]: srv.workspace.project_index.symbols) { if(symbol.name.empty()) diff --git a/tests/unit/index/index_query_tests.cpp b/tests/unit/index/index_query_tests.cpp index 0025d67f7..495de78f5 100644 --- a/tests/unit/index/index_query_tests.cpp +++ b/tests/unit/index/index_query_tests.cpp @@ -36,49 +36,32 @@ std::uint32_t header_id = 0; /// per-section shard blobs, and the TU manifest with its contributions — /// so live-variant masks and staleness gates behave as in production. void merge_into_workspace() { - auto tu_index = index::TUIndex::build(*unit); - std::string wire; - llvm::raw_string_ostream wos(wire); - tu_index.serialize(wos); - auto view = index::TUIndexView::from(wire); - ASSERT_TRUE(view.has_value()); + auto wire = index::build_tu_index(*unit); + auto view = index::TUIndex::from_bytes(wire); + ASSERT_TRUE(view.loaded()); auto& project = workspace.project_index; llvm::SmallVector file_ids_map; - for(std::uint32_t i = 0; i < view->path_count(); i += 1) { - file_ids_map.push_back(workspace.path_pool.intern(view->path(i))); + for(std::uint32_t i = 0; i < view.path_count(); i += 1) { + file_ids_map.push_back(workspace.path_pool.intern(view.path(i))); } - ASSERT_TRUE(project.merge(*view, file_ids_map)); - main_id = file_ids_map[view->path_count() - 1]; - - auto content_of = [&](llvm::StringRef path) -> llvm::StringRef { - auto it = sources.all_files.find(llvm::sys::path::filename(path)); - return it != sources.all_files.end() ? llvm::StringRef(it->second.content) - : llvm::StringRef(); - }; - auto lookup_symbol = [&](index::SymbolHash hash) { - return view->find_symbol(hash); - }; + ASSERT_TRUE(project.merge(view, file_ids_map)); + main_id = file_ids_map[view.path_count() - 1]; index::TUManifest manifest; - manifest.tu_fv = project.intern_file_version(main_id, view->path_hash(view->path_count() - 1)); + manifest.tu_fv = project.intern_file_version(main_id, view.path_hash(view.path_count() - 1)); - for(std::uint32_t section = 0; section < view->section_count(); section += 1) { - auto local_id = view->section_path(section); + for(std::uint32_t section = 0; section < view.section_count(); section += 1) { + auto local_id = view.section_path(section); auto global_id = file_ids_map[local_id]; - auto rows = view->decode_section_rows(section); - ASSERT_TRUE(rows.has_value()); - auto content = content_of(view->path(local_id)); - index::VariantInput fresh{view->section_rows_hash(section), &*rows, lookup_symbol}; - std::string bytes; - llvm::raw_string_ostream os(bytes); - index::write_shard(index::Shard(), {}, fresh, content, llvm::xxh3_64bits(content), os); - workspace.shards[global_id] = - index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); - - auto fv = project.intern_file_version(global_id, view->path_hash(local_id)); - manifest.contributions.emplace_back(fv, view->section_rows_hash(section)); - if(llvm::sys::path::filename(view->path(local_id)) == "header.h") { + // A section blob is already the final shard encoding: install the + // bytes verbatim, as the indexer's first-variant path does. + workspace.shards[global_id] = index::Shard::from_buffer( + llvm::MemoryBuffer::getMemBufferCopy(view.section_blob(section))); + + auto fv = project.intern_file_version(global_id, view.path_hash(local_id)); + manifest.contributions.emplace_back(fv, view.section_hash(section)); + if(llvm::sys::path::filename(view.path(local_id)) == "header.h") { header_id = global_id; } } diff --git a/tests/unit/index/preamble_index_tests.cpp b/tests/unit/index/preamble_index_tests.cpp index df0004675..81a59f197 100644 --- a/tests/unit/index/preamble_index_tests.cpp +++ b/tests/unit/index/preamble_index_tests.cpp @@ -1,8 +1,10 @@ #include "test/temp_dir.h" #include "test/test.h" #include "test/tester.h" -#include "index/preamble_state.h" #include "index/serialization.h" +#include "index/shard.h" +#include "index/tu_index.h" +#include "server/state/workspace.h" #include "llvm/Support/raw_ostream.h" @@ -10,21 +12,19 @@ namespace clice::testing { namespace { -TEST_SUITE(PreambleState, Tester) { +TEST_SUITE(PreambleIndex, Tester) { -index::TUIndex tu_index; TempDir dir; -std::shared_ptr state; +std::shared_ptr state; std::vector links; std::vector inactive; std::vector conditionals; -/// Compile, build a full TUIndex, serialize a PreambleState blob to disk -/// and load it back. +/// Compile, build a preamble envelope, persist it as the `.pch.idx` pair +/// and load it back through the production gate. void build_state(std::source_location location = std::source_location::current()) { ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); links.resize(1); links[0].range = {12, 20}; @@ -32,14 +32,8 @@ void build_state(std::source_location location = std::source_location::current() inactive = {4, 9, 30, 42}; conditionals = {1, 0, 2}; - auto blob_path = dir.path("state.pch.idx"); - std::error_code ec; - llvm::raw_fd_ostream os(blob_path, ec); - ASSERT_FALSE(bool(ec)); - index::PreambleState::serialize(*unit, tu_index, links, inactive, conditionals, os); - os.close(); - - state = index::PreambleState::load(blob_path); + dir.touch("state.pch.idx", index::build_preamble_index(*unit, links, inactive, conditionals)); + state = load_pch_envelope(dir.path("state.pch.idx")); ASSERT_TRUE(state != nullptr); } @@ -47,24 +41,46 @@ index::SymbolHash hash_of(llvm::StringRef name, std::source_location location = std::source_location::current()) { index::SymbolHash hash = 0; std::uint32_t count = 0; - for(auto& [symbol_id, symbol]: tu_index.symbols) { - if(symbol.name == name) { - hash = symbol_id; - count += 1; - } - } + state->iterate_symbols( + [&](index::SymbolHash symbol_id, const index::SymbolIdentity& symbol, llvm::StringRef) { + if(symbol.name == name) { + hash = symbol_id; + count += 1; + } + }); EXPECT_EQ(count, 1); return hash; } +/// Walk a symbol's relation rows in every header section (all but the +/// main file's), the way the query layer's overlay lookup serves them. +void lookup_headers(index::SymbolHash hash, + RelationKind kind, + llvm::function_ref callback) { + for(std::uint32_t i = 0; i < state->section_count(); i += 1) { + auto path_id = state->section_path(i); + if(path_id == state->path_count() - 1) { + continue; + } + bool keep = true; + state->shard_of(path_id).lookup(hash, kind, [&](const index::Relation& r) { + keep = callback(state->path(path_id), r); + return keep; + }); + if(!keep) { + return; + } + } +} + TEST_CASE(ForcedIncludeServed) { add_file("forced.h", R"(int §(def)⟦forced_value⟧ = 1;)"); add_main("main.cpp", R"(int x = forced_value;)"); // A compile-command forced include: clang records its include edge in // the predefines buffer, which is a valid location — so unlike the - // synthetic buffers themselves, the file must stay in the blob under - // its own path. + // synthetic buffers themselves, the file must stay in the envelope + // under its own path. prepare(); owned_args.insert(owned_args.end() - 1, "-include"); owned_args.insert(owned_args.end() - 1, TestVFS::path("forced.h")); @@ -73,27 +89,20 @@ TEST_CASE(ForcedIncludeServed) { params.arguments.push_back(arg.c_str()); } ASSERT_TRUE(try_compile()); - tu_index = index::TUIndex::build(*unit); - - auto blob_path = dir.path("state.pch.idx"); - std::error_code ec; - llvm::raw_fd_ostream os(blob_path, ec); - ASSERT_FALSE(bool(ec)); - index::PreambleState::serialize(*unit, tu_index, {}, {}, {}, os); - os.close(); - state = index::PreambleState::load(blob_path); + dir.touch("state.pch.idx", index::build_preamble_index(*unit, {}, {}, {})); + state = load_pch_envelope(dir.path("state.pch.idx")); ASSERT_TRUE(state != nullptr); bool found = false; - state->lookup(hash_of("forced_value"), - RelationKind::Definition, - [&](const index::PreambleState::File& file, const index::Relation& r) { - EXPECT_TRUE(file.path.ends_with("forced.h")); - EXPECT_EQ(dump(r.range), dump(range("def", "forced.h"))); - found = true; - return false; - }); + lookup_headers(hash_of("forced_value"), + RelationKind::Definition, + [&](llvm::StringRef path, const index::Relation& r) { + EXPECT_TRUE(path.ends_with("forced.h")); + EXPECT_EQ(dump(r.range), dump(range("def", "forced.h"))); + found = true; + return false; + }); EXPECT_TRUE(found); } @@ -110,33 +119,38 @@ int main() { §(ref)⟦foo⟧(); return 0; } auto foo = hash_of("foo"); - // The definition inside the header is served from the blob, together - // with everything needed to map it to an LSP location. + // The definition inside the header is served from its section. bool found_def = false; - state->lookup(foo, - RelationKind::Definition, - [&](const index::PreambleState::File& file, const index::Relation& r) { - EXPECT_TRUE(file.path.ends_with("foo.h")); - EXPECT_FALSE(file.content.empty()); - EXPECT_FALSE(file.line_starts.empty()); - EXPECT_EQ(dump(r.range), dump(range("def", "foo.h"))); - found_def = true; - return false; - }); + lookup_headers(foo, + RelationKind::Definition, + [&](llvm::StringRef path, const index::Relation& r) { + EXPECT_TRUE(path.ends_with("foo.h")); + EXPECT_EQ(dump(r.range), dump(range("def", "foo.h"))); + found_def = true; + return false; + }); EXPECT_TRUE(found_def); - // Header-internal references are in the blob too. + // Header-internal references are in the envelope too. bool found_ref = false; - state->lookup(foo, - RelationKind::Reference, - [&](const index::PreambleState::File& file, const index::Relation& r) { - if(r.range == range("href", "foo.h")) { - found_ref = true; - return false; - } - return true; - }); + lookup_headers(foo, RelationKind::Reference, [&](llvm::StringRef, const index::Relation& r) { + if(r.range == range("href", "foo.h")) { + found_ref = true; + return false; + } + return true; + }); EXPECT_TRUE(found_ref); + + // Everything needed to map rows to LSP positions rides in each shard; + // pure-ASCII content itself is omitted. + for(std::uint32_t i = 0; i < state->section_count(); i += 1) { + auto& shard = state->shard_of(state->section_path(i)); + EXPECT_TRUE(shard.content_size() > 0); + EXPECT_FALSE(shard.line_starts().empty()); + EXPECT_TRUE(shard.ascii()); + EXPECT_TRUE(shard.content().empty()); + } } TEST_CASE(PreambleLookup) { @@ -150,10 +164,12 @@ int main() { §(ref)⟦§(ref)foo⟧(); return 0; } build_state(); auto foo = hash_of("foo"); + const index::Shard& preamble = state->shard_of(state->path_count() - 1); + ASSERT_TRUE(preamble.loaded()); // Occurrence lookup by offset in the preamble entry. bool found_occurrence = false; - state->lookup_preamble(point("ref"), [&](const index::Occurrence& occurrence) { + preamble.lookup(point("ref"), [&](const index::Occurrence& occurrence) { EXPECT_EQ(occurrence.target, foo); EXPECT_EQ(dump(occurrence.range), dump(range("ref"))); found_occurrence = true; @@ -163,7 +179,7 @@ int main() { §(ref)⟦§(ref)foo⟧(); return 0; } // Relation lookup by symbol in the preamble entry. bool found_relation = false; - state->lookup_preamble(foo, RelationKind::Reference, [&](const index::Relation& r) { + preamble.lookup(foo, RelationKind::Reference, [&](const index::Relation& r) { EXPECT_EQ(dump(r.range), dump(range("ref"))); found_relation = true; return false; @@ -171,48 +187,6 @@ int main() { §(ref)⟦§(ref)foo⟧(); return 0; } EXPECT_TRUE(found_relation); } -TEST_CASE(MoveConsumedIndex) { - // The production path (stateless worker) moves the TUIndex into - // serialize; the blob must be complete even though the index is - // consumed rather than copied. - add_file("foo.h", R"( -inline void §(def)⟦foo⟧() {} -)"); - add_main("main.cpp", R"( -#include "foo.h" -int main() { §(ref)⟦foo⟧(); return 0; } -)"); - ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); - auto foo = hash_of("foo"); - - auto blob_path = dir.path("moved.pch.idx"); - std::error_code ec; - llvm::raw_fd_ostream os(blob_path, ec); - ASSERT_FALSE(bool(ec)); - index::PreambleState::serialize(*unit, std::move(tu_index), {}, {}, {}, os); - os.close(); - - state = index::PreambleState::load(blob_path); - ASSERT_TRUE(state != nullptr); - - bool found = false; - state->lookup(foo, - RelationKind::Definition, - [&](const index::PreambleState::File& file, const index::Relation& r) { - EXPECT_TRUE(file.path.ends_with("foo.h")); - EXPECT_EQ(dump(r.range), dump(range("def", "foo.h"))); - found = true; - return false; - }); - EXPECT_TRUE(found); - - std::string name; - SymbolKind kind; - EXPECT_TRUE(state->find_symbol(foo, name, kind)); - EXPECT_EQ(name, "foo"); -} - TEST_CASE(SymbolTableLookup) { add_file("foo.h", R"( inline void §(def)⟦foo⟧() {} @@ -225,13 +199,12 @@ int main() { §(ref)⟦foo⟧(); return 0; } auto foo = hash_of("foo"); - std::string name; - SymbolKind kind; - ASSERT_TRUE(state->find_symbol(foo, name, kind)); - EXPECT_EQ(name, "foo"); - EXPECT_EQ(kind.value(), SymbolKind(SymbolKind::Function).value()); + auto identity = state->find_symbol(foo); + ASSERT_TRUE(identity.has_value()); + EXPECT_EQ(identity->name, "foo"); + EXPECT_EQ(identity->kind.value(), SymbolKind(SymbolKind::Function).value()); - EXPECT_FALSE(state->find_symbol(foo + 1, name, kind)); + EXPECT_FALSE(state->find_symbol(foo + 1).has_value()); } TEST_CASE(FeatureStateRoundtrip) { @@ -248,9 +221,10 @@ int main() { return 0; } EXPECT_EQ(state->inactive_regions(), llvm::ArrayRef(inactive)); EXPECT_EQ(state->open_conditionals(), llvm::ArrayRef(conditionals)); - // A blob with no header entries answers lookups with silence, not UB. + // An envelope with no header sections answers lookups with silence, + // not UB. bool visited = false; - state->lookup(42, RelationKind::Reference, [&](auto&, auto&) { + lookup_headers(42, RelationKind::Reference, [&](llvm::StringRef, const index::Relation&) { visited = true; return true; }); @@ -258,10 +232,10 @@ int main() { return 0; } } TEST_CASE(RejectBadBlob) { - EXPECT_TRUE(index::PreambleState::load(dir.path("missing.pch.idx")) == nullptr); + EXPECT_TRUE(load_pch_envelope(dir.path("missing.pch.idx")) == nullptr); dir.touch("garbage.pch.idx", "not a flatbuffer at all"); - EXPECT_TRUE(index::PreambleState::load(dir.path("garbage.pch.idx")) == nullptr); + EXPECT_TRUE(load_pch_envelope(dir.path("garbage.pch.idx")) == nullptr); } TEST_CASE(RejectVersionMismatch) { @@ -280,7 +254,7 @@ TEST_CASE(RejectVersionMismatch) { auto blob_path = dir.path("stale.pch.idx"); dir.touch("stale.pch.idx", llvm::StringRef(reinterpret_cast(blob->data()), blob->size())); - EXPECT_TRUE(index::PreambleState::load(blob_path) == nullptr); + EXPECT_TRUE(load_pch_envelope(blob_path) == nullptr); } TEST_CASE(AcceptCurrentVersionBlob) { @@ -291,12 +265,12 @@ TEST_CASE(AcceptCurrentVersionBlob) { std::uint32_t format_version = 0; }; - auto blob = kota::codec::fbs::to_bytes(VersionOnly{index::preamble_format_version}); + auto blob = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); ASSERT_TRUE(blob.has_value()); dir.touch("current.pch.idx", llvm::StringRef(reinterpret_cast(blob->data()), blob->size())); - EXPECT_TRUE(index::PreambleState::load(dir.path("current.pch.idx")) != nullptr); + EXPECT_TRUE(load_pch_envelope(dir.path("current.pch.idx")) != nullptr); } TEST_CASE(RejectCorruptBlob) { @@ -311,30 +285,68 @@ int main() { return 0; } ASSERT_TRUE(bytes.size() > 8); dir.touch("truncated.pch.idx", bytes.take_front(bytes.size() / 2)); - EXPECT_TRUE(index::PreambleState::load(dir.path("truncated.pch.idx")) == nullptr); + EXPECT_TRUE(load_pch_envelope(dir.path("truncated.pch.idx")) == nullptr); // Bytes 4-7 carry the buffer identifier; a blob from another format // must be rejected up front. std::string clobbered = bytes.str(); - for(std::size_t i = 4; i < 8; ++i) { + for(std::size_t i = 4; i < 8; i += 1) { clobbered[i] = 'X'; } dir.touch("clobbered.pch.idx", clobbered); - EXPECT_TRUE(index::PreambleState::load(dir.path("clobbered.pch.idx")) == nullptr); + EXPECT_TRUE(load_pch_envelope(dir.path("clobbered.pch.idx")) == nullptr); +} + +TEST_CASE(RejectCorruptSectionBlob) { + add_main("main.cpp", R"( +int main() { return 0; } +)"); + build_state(); + + // Overwrite one section's blob bytes in place: the envelope stays + // structurally valid, but the load gate verifies every blob and must + // read the pair as missing instead of silently serving nothing. + auto buffer = llvm::MemoryBuffer::getFile(dir.path("state.pch.idx")); + ASSERT_TRUE(bool(buffer)); + std::string bytes = (*buffer)->getBuffer().str(); + + auto view = index::TUIndex::from_bytes(bytes); + ASSERT_TRUE(view.loaded()); + ASSERT_TRUE(view.section_count() > 0); + auto blob = view.section_blob(0); + auto pos = llvm::StringRef(bytes).find(blob); + ASSERT_TRUE(pos != llvm::StringRef::npos); + for(std::size_t i = 0; i < blob.size(); i += 1) { + bytes[pos + i] = 'X'; + } + + dir.touch("bad_section.pch.idx", bytes); + EXPECT_TRUE(load_pch_envelope(dir.path("bad_section.pch.idx")) == nullptr); } -TEST_CASE(SourcePathAndContent) { +TEST_CASE(SourcePathAndPrefix) { add_main("main.cpp", R"( int value = 42; int other = 1; )"); build_state(); - EXPECT_TRUE(state->source_path().ends_with("main.cpp")); - EXPECT_EQ(state->preamble_content(), unit->interested_content()); + EXPECT_TRUE(state->path(state->path_count() - 1).ends_with("main.cpp")); + + // The preamble text itself is not stored; the envelope keeps only the + // identity of the exact prefix it was built from. + auto content = unit->interested_content(); + EXPECT_TRUE(state->matches_prefix(content)); + EXPECT_TRUE(state->matches_prefix(content.str() + "\nint more = 2;")); + EXPECT_FALSE(state->matches_prefix(content.drop_back(1))); + EXPECT_FALSE(state->matches_prefix("int changed = 0;")); + + // An ordinary envelope never serves preamble state. + auto ordinary = index::build_tu_index(*unit); + EXPECT_FALSE(index::TUIndex::from_bytes(ordinary).matches_prefix(content)); } -}; // TEST_SUITE(PreambleState) +}; // TEST_SUITE(PreambleIndex) } // namespace diff --git a/tests/unit/index/project_index_tests.cpp b/tests/unit/index/project_index_tests.cpp index 9a5253141..d07e6e906 100644 --- a/tests/unit/index/project_index_tests.cpp +++ b/tests/unit/index/project_index_tests.cpp @@ -8,6 +8,7 @@ #include "test/tester.h" #include "index/project_index.h" #include "index/serialization.h" +#include "index/tu_index.h" #include "llvm/Support/raw_ostream.h" @@ -18,14 +19,11 @@ TEST_SUITE(ProjectIndex, Tester) { std::string wire; -/// Build the current unit's TUIndex and return the zero-copy view the +/// Build the current unit's envelope and return the zero-copy reader the /// merge path consumes; `wire` keeps the bytes alive. -std::optional build_view() { - auto tu_index = index::TUIndex::build(*unit); - wire.clear(); - llvm::raw_string_ostream os(wire); - tu_index.serialize(os); - return index::TUIndexView::from(wire); +index::TUIndex build_view() { + wire = index::build_tu_index(*unit); + return index::TUIndex::from_bytes(wire); } index::SymbolHash find_symbol(const index::ProjectIndex& project, llvm::StringRef name) { @@ -39,8 +37,7 @@ index::SymbolHash find_symbol(const index::ProjectIndex& project, llvm::StringRe /// The TU-local id -> pool id mapping merge() consumes, as Indexer::merge /// computes it. -llvm::SmallVector intern_paths(const index::TUIndexView& view, - clice::PathPool& pool) { +llvm::SmallVector intern_paths(const index::TUIndex& view, clice::PathPool& pool) { llvm::SmallVector ids; for(std::uint32_t i = 0; i < view.path_count(); i += 1) { ids.push_back(pool.intern(view.path(i))); @@ -66,8 +63,8 @@ TEST_CASE(MergeCollectsExternalSymbols) { clice::PathPool pool; index::ProjectIndex project; auto view = build_view(); - ASSERT_TRUE(view.has_value()); - ASSERT_TRUE(project.merge(*view, intern_paths(*view, pool))); + ASSERT_TRUE(view.loaded()); + ASSERT_TRUE(project.merge(view, intern_paths(view, pool))); auto external = find_symbol(project, "external_fn"); ASSERT_TRUE(external != 0); @@ -79,9 +76,9 @@ TEST_CASE(MergeCollectsExternalSymbols) { } TEST_CASE(MergeRejectsBadBitmap) { - // Field order MUST mirror TUIndex up to `symbols` (the skip-annotated - // file_indices holds no slot): serialize() always writes valid bitmap - // images, so a malformed one has to be planted by hand. + // Field order MUST mirror the envelope layout (tu_index.cpp) up to + // `symbols`: the builder always writes valid bitmap images, so a + // malformed one has to be planted by hand. struct SymbolMirror { std::string name; std::uint8_t kind = 0; @@ -89,16 +86,18 @@ TEST_CASE(MergeRejectsBadBitmap) { std::vector reference_files; }; - struct TUIndexPrefixMirror { + struct EnvelopePrefixMirror { std::uint32_t format_version = 0; std::int64_t built_at = 0; - index::IncludeGraph graph; + std::vector paths; + std::vector path_hashes; + std::vector locations; llvm::DenseMap symbols{}; }; - TUIndexPrefixMirror mirror; + EnvelopePrefixMirror mirror; mirror.format_version = index::index_format_version; - mirror.graph.paths = {"/proj/main.cpp"}; + mirror.paths = {"/proj/main.cpp"}; clice::Bitmap bits; bits.add(0); mirror.symbols[42] = {.name = "good_sym", .reference_files = index::write_bitmap(bits)}; @@ -107,11 +106,11 @@ TEST_CASE(MergeRejectsBadBitmap) { // valid image merges. auto valid = kota::codec::fbs::to_bytes(mirror); ASSERT_TRUE(valid.has_value()); - auto valid_view = index::TUIndexView::from(bytes_of(*valid)); - ASSERT_TRUE(valid_view.has_value()); + auto valid_view = index::TUIndex::from_bytes(bytes_of(*valid)); + ASSERT_TRUE(valid_view.loaded()); clice::PathPool pool; index::ProjectIndex accepting; - ASSERT_TRUE(accepting.merge(*valid_view, intern_paths(*valid_view, pool))); + ASSERT_TRUE(accepting.merge(valid_view, intern_paths(valid_view, pool))); ASSERT_EQ(find_symbol(accepting, "good_sym"), 42u); // One malformed image rejects the whole result: merged bits would @@ -123,24 +122,25 @@ TEST_CASE(MergeRejectsBadBitmap) { }; auto corrupt = kota::codec::fbs::to_bytes(mirror); ASSERT_TRUE(corrupt.has_value()); - auto corrupt_view = index::TUIndexView::from(bytes_of(*corrupt)); - ASSERT_TRUE(corrupt_view.has_value()); + auto corrupt_view = index::TUIndex::from_bytes(bytes_of(*corrupt)); + ASSERT_TRUE(corrupt_view.loaded()); index::ProjectIndex rejecting; - ASSERT_FALSE(rejecting.merge(*corrupt_view, intern_paths(*corrupt_view, pool))); + ASSERT_FALSE(rejecting.merge(corrupt_view, intern_paths(corrupt_view, pool))); ASSERT_TRUE(rejecting.symbols.empty()); // An id past the path table is the same corruption in a decodable // coat: silently dropped, the symbol's relations would sit in a shard - // its fan-out never visits — reject like the full TUIndex::from does. + // its fan-out never visits. The reader hands reference-file ids out + // raw, so this merge is the only gate. clice::Bitmap stray; stray.add(7); mirror.symbols[43] = {.name = "bad_sym", .reference_files = index::write_bitmap(stray)}; auto out_of_range = kota::codec::fbs::to_bytes(mirror); ASSERT_TRUE(out_of_range.has_value()); - auto stray_view = index::TUIndexView::from(bytes_of(*out_of_range)); - ASSERT_TRUE(stray_view.has_value()); + auto stray_view = index::TUIndex::from_bytes(bytes_of(*out_of_range)); + ASSERT_TRUE(stray_view.loaded()); index::ProjectIndex bounding; - ASSERT_FALSE(bounding.merge(*stray_view, intern_paths(*stray_view, pool))); + ASSERT_FALSE(bounding.merge(stray_view, intern_paths(stray_view, pool))); ASSERT_TRUE(bounding.symbols.empty()); } @@ -219,17 +219,17 @@ TEST_CASE(GlobalRoundTripWithRealMerge) { clice::PathPool pool; index::ProjectIndex project; auto view = build_view(); - ASSERT_TRUE(view.has_value()); - auto file_ids_map = intern_paths(*view, pool); - ASSERT_TRUE(project.merge(*view, file_ids_map)); + ASSERT_TRUE(view.loaded()); + auto file_ids_map = intern_paths(view, pool); + ASSERT_TRUE(project.merge(view, file_ids_map)); // A manifest referencing the main file keeps its FileVersion alive // through the write's garbage collection. - auto main_fv = project.intern_file_version(file_ids_map[view->path_count() - 1], - view->path_hash(view->path_count() - 1)); + auto main_fv = project.intern_file_version(file_ids_map[view.path_count() - 1], + view.path_hash(view.path_count() - 1)); index::TUManifest manifest; manifest.tu_fv = main_fv; - project.apply_manifest(file_ids_map[view->path_count() - 1], std::move(manifest)); + project.apply_manifest(file_ids_map[view.path_count() - 1], std::move(manifest)); llvm::SmallString<4096> buf; llvm::raw_svector_ostream os(buf); @@ -242,7 +242,7 @@ TEST_CASE(GlobalRoundTripWithRealMerge) { auto symbol = find_symbol(loaded, "global_value"); ASSERT_TRUE(symbol != 0); - auto main_path = pool.resolve(file_ids_map[view->path_count() - 1]); + auto main_path = pool.resolve(file_ids_map[view.path_count() - 1]); auto fresh_id = fresh.find(main_path); ASSERT_TRUE(fresh_id.has_value()); ASSERT_TRUE(loaded.symbols[symbol].reference_files.contains(*fresh_id)); diff --git a/tests/unit/index/shard_tests.cpp b/tests/unit/index/shard_tests.cpp index 1bb403b96..3d793044f 100644 --- a/tests/unit/index/shard_tests.cpp +++ b/tests/unit/index/shard_tests.cpp @@ -7,7 +7,6 @@ #include "test/tester.h" #include "index/serialization.h" #include "index/shard.h" -#include "index/shard_layout.h" #include "index/tu_index.h" #include "kota/ipc/lsp/text.h" @@ -26,33 +25,27 @@ void build_index(llvm::StringRef code, std::source_location location = std::source_location::current()) { add_main("main.cpp", code); ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); + tu_index = index::TUIndex::from_buffer( + llvm::MemoryBuffer::getMemBufferCopy(index::build_tu_index(*unit))); + ASSERT_TRUE(tu_index.loaded()); } -std::optional lookup_symbol(index::SymbolHash hash) { - auto it = tu_index.symbols.find(hash); - if(it == tu_index.symbols.end()) { - return std::nullopt; - } - return index::SymbolIdentity{it->second.name, it->second.kind, it->second.scope}; -} - -std::string write_fresh(const index::FileIndex& rows, - llvm::StringRef content, - bool with_symbols = false) { - llvm::function_ref(index::SymbolHash)> resolve; - auto resolver = [this](index::SymbolHash symbol) { - return lookup_symbol(symbol); - }; - if(with_symbols) { - resolve = resolver; - } +std::string write_fresh(const index::FileIndex& rows, llvm::StringRef content) { std::string bytes; llvm::raw_string_ostream os(bytes); - index::write_shard(rows, resolve, content, os); + index::write_shard(rows, {}, content, os); return bytes; } +/// The interested file's worker-encoded blob, straight from the envelope. +std::string main_blob() { + auto section = tu_index.section_of(tu_index.path_count() - 1); + if(!section) { + return {}; + } + return tu_index.section_blob(*section).str(); +} + /// Owning wrap: from_bytes borrows, and every builder here returns a /// temporary string. index::Shard make_shard(llvm::StringRef bytes) { @@ -100,8 +93,7 @@ TEST_CASE(RoundtripLookups) { )"); auto content = sources.all_files.find("main.cpp")->second.content; - auto bytes = write_fresh(tu_index.main_file_index, content); - auto shard = make_shard(bytes); + auto shard = make_shard(main_blob()); ASSERT_TRUE(shard.loaded()); ASSERT_EQ(shard.content_size(), static_cast(content.size())); ASSERT_FALSE(shard.line_starts().empty()); @@ -219,7 +211,7 @@ TEST_CASE(WideRangeTier) { // the wide tier takes over transparently. std::string content(index::packed_range_limit + 64, 'w'); auto rows = simple_rows({ - {{0, 3}, 111}, + {{0, 3}, 111}, {{index::packed_range_limit + 8, index::packed_range_limit + 11}, 222}, }); auto shard = make_shard(write_fresh(rows, content)); @@ -286,7 +278,7 @@ TEST_CASE(KWayMerge) { std::vector fresh; for(std::uint32_t i = 0; i < 3; i += 1) { auto rows = simple_rows({ - {{0, 3}, 111 }, + {{0, 3}, 111 }, {{4 * (i + 1), 4 * (i + 1) + 3}, 1000 + i}, }); fresh.push_back(make_shard(write_fresh(rows, content))); @@ -441,8 +433,7 @@ TEST_CASE(LocalSymbolNames) { int visible() { return §(use)⟦§(use)helper⟧(); } )"); - auto content = sources.all_files.find("main.cpp")->second.content; - auto shard = make_shard(write_fresh(tu_index.main_file_index, content, /*with_symbols=*/true)); + auto shard = make_shard(main_blob()); auto local = hash_at(shard, point("use")); ASSERT_TRUE(local != 0); @@ -453,12 +444,14 @@ TEST_CASE(LocalSymbolNames) { // External names live in the ProjectIndex, never in the blob. auto external = [&] { - for(auto& [hash, symbol]: tu_index.symbols) { - if(symbol.name == "visible") { - return hash; - } - } - return index::SymbolHash(0); + index::SymbolHash result = 0; + tu_index.iterate_symbols( + [&](index::SymbolHash hash, const index::SymbolIdentity& symbol, llvm::StringRef) { + if(symbol.name == "visible") { + result = hash; + } + }); + return result; }(); ASSERT_TRUE(external != 0); ASSERT_FALSE(shard.find_symbol(external, name, kind)); @@ -472,7 +465,7 @@ TEST_CASE(MergedLocalNames) { int visible() { return helper(); } )"); auto content = sources.all_files.find("main.cpp")->second.content; - auto first = make_shard(write_fresh(tu_index.main_file_index, content, /*with_symbols=*/true)); + auto first = make_shard(main_blob()); auto extra = simple_rows({ {{0, 3}, 424242} diff --git a/tests/unit/index/tu_index_tests.cpp b/tests/unit/index/tu_index_tests.cpp index ec3a7de71..2b8aa9a69 100644 --- a/tests/unit/index/tu_index_tests.cpp +++ b/tests/unit/index/tu_index_tests.cpp @@ -6,10 +6,12 @@ #include "test/tester.h" #include "feature/feature.h" #include "index/serialization.h" +#include "index/shard.h" #include "index/tu_index.h" #include "semantic/selection.h" #include "llvm/Support/thread.h" +#include "llvm/Support/xxhash.h" #include "clang/Basic/Stack.h" namespace clice::testing { @@ -20,21 +22,79 @@ namespace { TEST_SUITE(tu_index, Tester) { -index::TUIndex tu_index; +/// One file's rows read back out of its envelope section through the +/// Shard reader — every assertion below therefore exercises the full +/// build → encode → read roundtrip, not builder-internal state. +struct DecodedRows { + std::vector occurrences; + llvm::DenseMap> relations{}; + + bool empty() const { + return occurrences.empty() && relations.empty(); + } +}; + +struct DecodedIndex { + index::TUIndex view; + DecodedRows main_file_index; + /// Non-main sections keyed by path id. + std::vector> file_indices; + index::SymbolTable symbols{}; +}; + +DecodedIndex tu_index; + +DecodedRows decode_rows(const index::Shard& shard) { + DecodedRows rows; + shard.for_each_occurrence([&](const index::Occurrence& occurrence) { + rows.occurrences.push_back(occurrence); + return true; + }); + shard.for_each_relation([&](index::SymbolHash hash, const index::Relation& relation) { + rows.relations[hash].push_back(relation); + return true; + }); + return rows; +} + +void decode_index(const std::string& envelope) { + tu_index = {}; + tu_index.view = index::TUIndex::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(envelope)); + auto& view = tu_index.view; + ASSERT_TRUE(view.loaded()); + + auto main_path = view.path_count() - 1; + for(std::uint32_t i = 0; i < view.section_count(); i += 1) { + auto rows = decode_rows(view.shard_of(view.section_path(i))); + if(view.section_path(i) == main_path) { + tu_index.main_file_index = std::move(rows); + } else { + tu_index.file_indices.emplace_back(view.section_path(i), std::move(rows)); + } + } + + view.iterate_symbols( + [&](index::SymbolHash hash, const index::SymbolIdentity& identity, llvm::StringRef bitmap) { + auto& symbol = tu_index.symbols[hash]; + symbol.name = identity.name.str(); + symbol.kind = identity.kind; + symbol.scope = identity.scope; + symbol.reference_files = + index::read_bitmap(bitmap.data(), bitmap.size()).value_or(Bitmap{}); + }); +} void build_index(llvm::StringRef code, std::source_location location = std::source_location::current()) { add_main("main.cpp", code); ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); + decode_index(index::build_tu_index(*unit)); } -auto select(llvm::StringRef pos, llvm::StringRef file = "") -> std::vector { - auto offset = point(pos, file); - auto fid = file.empty() ? unit->interested_file() : unit->file_id(file); - auto& index = - fid == unit->interested_file() ? tu_index.main_file_index : tu_index.file_indices[fid]; +auto select(llvm::StringRef pos) -> std::vector { + auto offset = point(pos); + auto& index = tu_index.main_file_index; auto it = std::ranges::lower_bound(index.occurrences, offset, {}, [](index::Occurrence& occurrence) { @@ -56,15 +116,11 @@ auto select(llvm::StringRef pos, llvm::StringRef file = "") -> std::vectorinterested_file() : unit->file_id(file); - auto& index = - fid == unit->interested_file() ? tu_index.main_file_index : tu_index.file_indices[fid]; + auto& index = tu_index.main_file_index; auto it = index.relations.find(occurrences.front().target); ASSERT_TRUE(it != index.relations.end()); ///<< std::format("Cannot find target: {}", occurrences.front().target); @@ -687,7 +736,7 @@ export int x = 1; module §(m)⟦§(m)foo⟧; )"); ASSERT_TRUE(compile_with_modules()); - tu_index = index::TUIndex::build(*unit); + decode_index(index::build_tu_index(*unit)); // An implementation unit's declaration is a Reference, not a Definition. auto occs = select("m"); @@ -728,7 +777,7 @@ TEST_CASE(OverrideRelation) { bool found_interface = false; bool found_implementation = false; - auto check_relations = [&](index::FileIndex& idx) { + auto check_relations = [&](DecodedRows& idx) { for(auto& [hash, rels]: idx.relations) { for(auto& r: rels) { if(r.kind == RelationKind::Interface) @@ -740,7 +789,7 @@ TEST_CASE(OverrideRelation) { }; check_relations(tu_index.main_file_index); - for(auto& [fid, idx]: tu_index.file_indices) { + for(auto& [path_id, idx]: tu_index.file_indices) { check_relations(idx); } @@ -792,7 +841,7 @@ TEST_CASE(CrossFileHeaderIndex) { } )"); ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); + decode_index(index::build_tu_index(*unit)); // The header should have its own FileIndex (separate from main). ASSERT_TRUE(tu_index.file_indices.size() >= 1U); @@ -814,9 +863,9 @@ TEST_CASE(CrossFileHeaderIndex) { auto helper_hash = it->target; ASSERT_TRUE(tu_index.symbols.contains(helper_hash)); - // The helper's declaration should be in the header FileIndex. + // The helper's declaration should be in the header's rows. bool found_in_header = false; - for(auto& [fid, file_index]: tu_index.file_indices) { + for(auto& [path_id, file_index]: tu_index.file_indices) { for(auto& [sym, rels]: file_index.relations) { if(sym == helper_hash) { found_in_header = true; @@ -861,31 +910,33 @@ TEST_CASE(LookupOccurrence) { auto& fi = tu_index.main_file_index; ASSERT_FALSE(fi.occurrences.empty()); + const index::Shard& shard = tu_index.view.shard_of(tu_index.view.path_count() - 1); + ASSERT_TRUE(shard.loaded()); auto x_range = range("x"); - const index::Occurrence* found = nullptr; - fi.lookup(point("x"), [&](const index::Occurrence& occ) { - found = &occ; + std::optional found; + shard.lookup(point("x"), [&](const index::Occurrence& occ) { + found = occ; return true; }); - ASSERT_TRUE(found); + ASSERT_TRUE(found.has_value()); EXPECT_EQ(found->range.begin, x_range.begin); EXPECT_EQ(found->range.end, x_range.end); - found = nullptr; - fi.lookup(point("ref"), [&](const index::Occurrence& occ) { - found = &occ; + found.reset(); + shard.lookup(point("ref"), [&](const index::Occurrence& occ) { + found = occ; return true; }); - ASSERT_TRUE(found); + ASSERT_TRUE(found.has_value()); EXPECT_EQ(found->target, fi.occurrences.front().target); - found = nullptr; - fi.lookup(0, [&](const index::Occurrence& occ) { - found = &occ; + found.reset(); + shard.lookup(0, [&](const index::Occurrence& occ) { + found = occ; return false; }); - EXPECT_FALSE(found); + EXPECT_FALSE(found.has_value()); } TEST_CASE(LookupRelation) { @@ -894,18 +945,19 @@ TEST_CASE(LookupRelation) { void §(def)⟦fo§(def)o⟧() {} )"); - auto& fi = tu_index.main_file_index; + const index::Shard& shard = tu_index.view.shard_of(tu_index.view.path_count() - 1); + ASSERT_TRUE(shard.loaded()); - const index::Occurrence* occ = nullptr; - fi.lookup(point("decl"), [&](const index::Occurrence& o) { - occ = &o; + std::optional occ; + shard.lookup(point("decl"), [&](const index::Occurrence& o) { + occ = o; return false; }); - ASSERT_TRUE(occ); + ASSERT_TRUE(occ.has_value()); auto def_range = range("def"); bool found_def = false; - fi.lookup(occ->target, RelationKind::Definition, [&](const index::Relation& r) { + shard.lookup(occ->target, RelationKind::Definition, [&](const index::Relation& r) { found_def = true; EXPECT_EQ(r.range.begin, def_range.begin); EXPECT_EQ(r.range.end, def_range.end); @@ -914,7 +966,7 @@ TEST_CASE(LookupRelation) { EXPECT_TRUE(found_def); bool found_any = false; - fi.lookup(occ->target, RelationKind::Caller, [&](const index::Relation&) { + shard.lookup(occ->target, RelationKind::Caller, [&](const index::Relation&) { found_any = true; return false; }); @@ -1004,14 +1056,14 @@ int Foo::§(def)⟦§(1)find⟧(int x) const { return 0; } // A full build keeps the preamble rows: this proves the row exists // (so the gate test below cannot pass vacuously) and that the loaded // fid resolves through the graph to the header's path. - tu_index = index::TUIndex::build(*unit); + decode_index(index::build_tu_index(*unit)); bool found = false; - for(auto& [fid, index]: tu_index.file_indices) { - found |= tu_index.graph.path(tu_index.graph.path_id(fid)).ends_with("foo.h"); + for(auto& [path_id, index]: tu_index.file_indices) { + found |= tu_index.view.path(path_id).ends_with("foo.h"); } ASSERT_TRUE(found); - tu_index = index::TUIndex::build(*unit, true); + decode_index(index::build_tu_index(*unit, true)); // Rows resolving into the preamble are dropped: the preamble's own // index covers them. Only the main file's rows remain. @@ -1036,14 +1088,14 @@ Derived* use(); // Full build: the base-specifier rows land in the preamble header // and its loaded fid resolves to the header's path. - tu_index = index::TUIndex::build(*unit); + decode_index(index::build_tu_index(*unit)); bool found = false; - for(auto& [fid, index]: tu_index.file_indices) { - found |= tu_index.graph.path(tu_index.graph.path_id(fid)).ends_with("bar.h"); + for(auto& [path_id, index]: tu_index.file_indices) { + found |= tu_index.view.path(path_id).ends_with("bar.h"); } ASSERT_TRUE(found); - tu_index = index::TUIndex::build(*unit, true); + decode_index(index::build_tu_index(*unit, true)); ASSERT_TRUE(tu_index.file_indices.empty()); ASSERT_FALSE(tu_index.main_file_index.occurrences.empty()); @@ -1062,7 +1114,7 @@ int x = §(1)BAZ; )"); ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit, true); + decode_index(index::build_tu_index(*unit, true)); ASSERT_TRUE(tu_index.file_indices.empty()); ASSERT_FALSE(tu_index.main_file_index.occurrences.empty()); } @@ -1139,7 +1191,7 @@ TEST_CASE(DeepExpressionChain) { llvm::thread index_thread(std::optional(clang::DesiredStackSize / 4), [&] { // Mirror the stateful worker's post-compile sequence. scan = feature::inactive_regions(*unit); - tu_index = index::TUIndex::build(*unit, true); + decode_index(index::build_tu_index(*unit, true)); // The semantic map must also serve token classification and a // selection at the giant expansion's invocation on this stack: @@ -1180,12 +1232,12 @@ TEST_CASE(SuperQualifierRef) { params.arguments.push_back(arg.c_str()); } ASSERT_TRUE(try_compile()); - tu_index = index::TUIndex::build(*unit); + decode_index(index::build_tu_index(*unit)); GO_TO_DEFINITION("use", "def"); } -TEST_CASE(SerializeRoundTrip) { +TEST_CASE(EnvelopeSections) { add_file("header.h", R"( #pragma once inline int §(hdr)helper() { return 1; } @@ -1195,92 +1247,79 @@ TEST_CASE(SerializeRoundTrip) { int main() { return §(use)helper(); } )"); ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); + decode_index(index::build_tu_index(*unit)); ASSERT_FALSE(tu_index.file_indices.empty()); - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - tu_index.serialize(os); - - auto loaded = index::TUIndex::from(buf); - ASSERT_TRUE(loaded.has_value()); - - ASSERT_EQ(loaded->built_at.count(), tu_index.built_at.count()); - ASSERT_TRUE(loaded->graph.paths == tu_index.graph.paths); - ASSERT_TRUE(loaded->graph.locations == tu_index.graph.locations); - ASSERT_TRUE(loaded->graph.path_hashes == tu_index.graph.path_hashes); - - // The persisted per-file rows travel as wire sections keyed by path id; - // recompute the expected conversion from the build-time FileID-keyed - // state. Empty rows get no section. - llvm::DenseMap> expected; - for(auto& [fid, file_index]: tu_index.file_indices) { - if(!file_index.empty()) { - expected[tu_index.graph.path_id(fid)] = {file_index.occurrences.size(), - file_index.relations.size()}; - } + auto& view = tu_index.view; + ASSERT_TRUE(view.built_at() > 0); + + // Sections ascend by path id and the interested file's rows are the + // last path id's section. + for(std::uint32_t i = 1; i < view.section_count(); i += 1) { + ASSERT_TRUE(view.section_path(i - 1) < view.section_path(i)); } - ASSERT_FALSE(expected.empty()); - // The interested file's rows travel as the last section. - ASSERT_EQ(loaded->sections.size(), expected.size() + 1); - for(auto& [path_id, counts]: expected) { - auto it = std::ranges::find(loaded->sections, path_id, &index::FileSection::path_id); - ASSERT_TRUE(it != loaded->sections.end()); - auto rows = index::TUIndex::decode_rows(*it); - ASSERT_TRUE(rows.has_value()); - ASSERT_EQ(rows->occurrences.size(), counts.first); - ASSERT_EQ(rows->relations.size(), counts.second); + auto main_section = view.section_of(view.path_count() - 1); + ASSERT_TRUE(main_section.has_value()); + + // Every section's hash is the byte identity of its blob, the blob + // loads as a self-contained single-variant shard under that identity, + // and the whole envelope passes the persisted-load gate. + ASSERT_TRUE(view.shards_verify()); + for(std::uint32_t i = 0; i < view.section_count(); i += 1) { + auto blob = view.section_blob(i); + ASSERT_EQ(view.section_hash(i), llvm::xxh3_64bits(blob)); + auto shard = index::Shard::from_bytes(blob); + ASSERT_TRUE(shard.loaded()); + ASSERT_EQ(shard.variants().size(), std::size_t(1)); + ASSERT_EQ(shard.variants().front(), view.section_hash(i)); } - auto* main_sec = loaded->main_section(); - ASSERT_TRUE(main_sec != nullptr); - auto main_rows = index::TUIndex::decode_rows(*main_sec); - ASSERT_TRUE(main_rows.has_value()); - ASSERT_TRUE(main_rows->occurrences == tu_index.main_file_index.occurrences); - ASSERT_EQ(main_rows->relations.size(), tu_index.main_file_index.relations.size()); + // The graph travels with the envelope: every path resolves and the + // consumed-content hash column covers the whole table. + for(std::uint32_t id = 0; id < view.path_count(); id += 1) { + ASSERT_FALSE(view.path(id).empty()); + ASSERT_TRUE(view.path_hash(id) != 0); + } - ASSERT_EQ(loaded->symbols.size(), tu_index.symbols.size()); + // Symbols read back identically through iteration and point lookup. + ASSERT_FALSE(tu_index.symbols.empty()); for(auto& [hash, symbol]: tu_index.symbols) { - auto it = loaded->symbols.find(hash); - ASSERT_TRUE(it != loaded->symbols.end()); - ASSERT_EQ(it->second.name, symbol.name); - ASSERT_EQ(it->second.kind.value(), symbol.kind.value()); - ASSERT_EQ(static_cast(it->second.scope), static_cast(symbol.scope)); - ASSERT_TRUE(it->second.reference_files == symbol.reference_files); + auto identity = view.find_symbol(hash); + ASSERT_TRUE(identity.has_value()); + ASSERT_EQ(identity->name, llvm::StringRef(symbol.name)); + ASSERT_EQ(identity->kind.value(), symbol.kind.value()); + ASSERT_EQ(static_cast(identity->scope), static_cast(symbol.scope)); } } TEST_CASE(FromRejectsHostileInput) { - ASSERT_FALSE(index::TUIndex::from("not a flatbuffer at all").has_value()); + ASSERT_FALSE(index::TUIndex::from_bytes("not a flatbuffer at all").loaded()); build_index(R"( int foo() { return 42; } )"); + auto bytes = tu_index.view.bytes(); - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - tu_index.serialize(os); - - // Sanity: the intact blob loads, so the rejections below are earned. - ASSERT_TRUE(index::TUIndex::from(buf).has_value()); + // Sanity: the intact envelope loads, so the rejections below are earned. + ASSERT_TRUE(index::TUIndex::from_bytes(bytes).loaded()); - ASSERT_FALSE(index::TUIndex::from(llvm::StringRef(buf.data(), buf.size() / 2)).has_value()); + ASSERT_FALSE(index::TUIndex::from_bytes(bytes.substr(0, bytes.size() / 2)).loaded()); // Bytes 4-7 carry the buffer identifier; a blob from another format // must be rejected up front. - ASSERT_TRUE(buf.size() > 8); - std::string clobbered(buf.data(), buf.size()); - for(std::size_t i = 4; i < 8; ++i) { + ASSERT_TRUE(bytes.size() > 8); + std::string clobbered(bytes.data(), bytes.size()); + for(std::size_t i = 4; i < 8; i += 1) { clobbered[i] = 'X'; } - ASSERT_FALSE(index::TUIndex::from(clobbered).has_value()); + ASSERT_FALSE(index::TUIndex::from_bytes(clobbered).loaded()); } TEST_CASE(FromRejectsStaleFormatVersion) { // Only the version slot is written: every other field reads back // absent, which is structurally valid — the verdict must hinge on the - // value. Field order MUST mirror TUIndex (tu_index.h): format_version - // is slot 0. + // value. Field order MUST mirror the envelope layout (tu_index.cpp): + // format_version is slot 0. struct VersionOnly { std::uint32_t format_version = 0; }; @@ -1291,117 +1330,83 @@ TEST_CASE(FromRejectsStaleFormatVersion) { auto stale = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version + 1}); ASSERT_TRUE(stale.has_value()); - ASSERT_FALSE(index::TUIndex::from(bytes_of(*stale)).has_value()); + ASSERT_FALSE(index::TUIndex::from_bytes(bytes_of(*stale)).loaded()); // Positive control: the same shape carrying the current version loads, // so the rejection above comes from the value, not the blob's shape. auto current = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); ASSERT_TRUE(current.has_value()); - ASSERT_TRUE(index::TUIndex::from(bytes_of(*current)).has_value()); + ASSERT_TRUE(index::TUIndex::from_bytes(bytes_of(*current)).loaded()); +} + +/// Hand-built envelopes for hostile-input tests. Field order MUST mirror +/// the envelope layout (tu_index.cpp). +struct MirrorSection { + std::uint32_t path_id = 0; + std::uint64_t hash = 0; + std::vector blob; +}; + +struct MirrorEnvelope { + std::uint32_t format_version = index::index_format_version; + std::int64_t built_at = 0; + std::vector paths; + std::vector path_hashes; + std::vector locations; + index::SymbolTable symbols{}; + std::vector sections; +}; + +std::string mirror_bytes(const MirrorEnvelope& envelope) { + auto bytes = kota::codec::fbs::to_bytes(envelope); + if(!bytes) { + return {}; + } + return std::string(bytes->begin(), bytes->end()); } TEST_CASE(FromRejectsOutOfRangePathIds) { // Structural verification does not constrain field values, and the // merge pipeline dereferences every decoded path id against the path - // table without further checks (Indexer::merge indexes paths and - // path_hashes, ProjectIndex::merge indexes file_ids_map with - // reference_files values) — a blob pointing outside its own table must - // be rejected as a whole. - auto serialized = [](index::TUIndex& index) { - std::string buf; - llvm::raw_string_ostream os(buf); - index.serialize(os); - return buf; - }; + // table without further checks — an envelope pointing outside its own + // table must be rejected as a whole. Symbol reference-file ids are + // deliberately not gated here: ProjectIndex::merge bounds them, pinned + // by project_index_tests. // Positive control first: the same shapes with in-range ids load, so // the rejections below come from the hostile values. - index::TUIndex honest; - honest.built_at = std::chrono::milliseconds(0); - honest.graph.paths = {"/proj/main.cpp"}; - honest.graph.locations.push_back({.path_id = 0, .line = 1, .include = 0}); + MirrorEnvelope honest; + honest.paths = {"/proj/main.cpp"}; + honest.locations.push_back({.path_id = 0, .line = 1, .include = 0}); honest.sections.push_back({.path_id = 0}); - honest.symbols[42].reference_files.add(0); - ASSERT_TRUE(index::TUIndex::from(serialized(honest)).has_value()); + ASSERT_TRUE(index::TUIndex::from_bytes(mirror_bytes(honest)).loaded()); { - index::TUIndex hostile; - hostile.built_at = std::chrono::milliseconds(0); - hostile.graph.paths = {"/proj/main.cpp"}; - hostile.graph.locations.push_back({.path_id = 7, .line = 1, .include = 0}); - ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); + MirrorEnvelope hostile; + hostile.paths = {"/proj/main.cpp"}; + hostile.locations.push_back({.path_id = 7, .line = 1, .include = 0}); + ASSERT_FALSE(index::TUIndex::from_bytes(mirror_bytes(hostile)).loaded()); } { - index::TUIndex hostile; - hostile.built_at = std::chrono::milliseconds(0); - hostile.graph.paths = {"/proj/main.cpp"}; + MirrorEnvelope hostile; + hostile.paths = {"/proj/main.cpp"}; hostile.sections.push_back({.path_id = 7}); // Only path id 0 exists. - ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); - } - { - index::TUIndex hostile; - hostile.built_at = std::chrono::milliseconds(0); - hostile.graph.paths = {"/proj/main.cpp"}; - hostile.symbols[42].reference_files.add(7); - ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); - } -} - -TEST_CASE(FromNormalizesPathHashes) { - build_index(R"( - int foo() { return 42; } - )"); - ASSERT_FALSE(tu_index.graph.paths.empty()); - - // A blob without path hashes (structurally valid: the field reads back - // empty) must come back resized to the path table, all "unavailable". - tu_index.graph.path_hashes.clear(); - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - tu_index.serialize(os); - - auto loaded = index::TUIndex::from(buf); - ASSERT_TRUE(loaded.has_value()); - ASSERT_EQ(loaded->graph.path_hashes.size(), loaded->graph.paths.size()); - for(auto hash: loaded->graph.path_hashes) { - ASSERT_EQ(hash, 0u); + ASSERT_FALSE(index::TUIndex::from_bytes(mirror_bytes(hostile)).loaded()); } } -TEST_CASE(ReserializeKeepsSections) { - add_file("header.h", R"( - #pragma once - inline int helper() { return 1; } - )"); - add_main("main.cpp", R"( - #include "header.h" - int main() { return helper(); } - )"); - ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); - - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - tu_index.serialize(os); - - auto loaded = index::TUIndex::from(buf); - ASSERT_TRUE(loaded.has_value()); - ASSERT_TRUE(loaded->file_indices.empty()); - ASSERT_FALSE(loaded->sections.empty()); - - // A deserialized index has no FileID-keyed state; re-serializing must - // keep the wire sections instead of wiping them from the empty maps. - llvm::SmallString<4096> again; - llvm::raw_svector_ostream os2(again); - loaded->serialize(os2); - - auto reloaded = index::TUIndex::from(again); - ASSERT_TRUE(reloaded.has_value()); - ASSERT_EQ(reloaded->sections.size(), loaded->sections.size()); - for(std::size_t i = 0; i < loaded->sections.size(); i += 1) { - ASSERT_EQ(reloaded->sections[i].path_id, loaded->sections[i].path_id); - ASSERT_EQ(reloaded->sections[i].rows_hash, loaded->sections[i].rows_hash); - } +TEST_CASE(AbsentPathHashesReadZero) { + // The hash column may be shorter than the path table on a foreign + // envelope (structurally valid: the field reads back empty); absent + // entries read as 0, "unavailable". + MirrorEnvelope envelope; + envelope.paths = {"/proj/a.h", "/proj/main.cpp"}; + envelope.path_hashes = {7}; + auto bytes = mirror_bytes(envelope); + auto view = index::TUIndex::from_bytes(bytes); + ASSERT_TRUE(view.loaded()); + ASSERT_EQ(view.path_hash(0), 7u); + ASSERT_EQ(view.path_hash(1), 0u); } }; // TEST_SUITE(tu_index) diff --git a/tests/unit/server/indexer_tests.cpp b/tests/unit/server/indexer_tests.cpp index e7e3117d2..730c3f2ba 100644 --- a/tests/unit/server/indexer_tests.cpp +++ b/tests/unit/server/indexer_tests.cpp @@ -7,6 +7,7 @@ #include "command/argument_parser.h" #include "compile/compilation.h" #include "index/manifest.h" +#include "index/serialization.h" #include "index/shard.h" #include "index/storage.h" #include "index/tu_index.h" @@ -16,6 +17,7 @@ #include "server/state/workspace.h" #include "server/worker/worker_pool.h" +#include "kota/ipc/lsp/text.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Support/xxhash.h" @@ -100,11 +102,11 @@ struct IndexerFixture { namespace { struct IndexedTU { - std::string data; ///< Serialized TUIndex, as a worker would ship it. + std::string data; ///< Envelope bytes, as a worker would ship them. std::string tu_path; ///< The TU's canonical path inside the index. }; -/// Index a real on-disk file in-process and serialize its TUIndex. +/// Index a real on-disk file in-process into its envelope bytes. IndexedTU index_file(TempDir& tmp, llvm::StringRef file, std::vector extra_args = {}) { std::string resource = std::string(resource_dir()); std::vector args = @@ -122,14 +124,93 @@ IndexedTU index_file(TempDir& tmp, llvm::StringRef file, std::vector reference_files; + }; + + struct SectionMirror { + std::uint32_t path_id = 0; + std::uint64_t hash = 0; + std::vector blob; + }; + + struct EnvelopeMirror { + std::uint32_t format_version = index::index_format_version; + std::int64_t built_at = 0; + std::vector paths; + std::vector path_hashes; + std::vector locations; + llvm::DenseMap symbols{}; + std::vector sections; + }; + + auto view = index::TUIndex::from_bytes(data); + EnvelopeMirror mirror; + mirror.built_at = view.built_at(); + for(std::uint32_t i = 0; i < view.path_count(); i += 1) { + mirror.paths.emplace_back(view.path(i)); + } + for(std::uint32_t i = 0; i < view.location_count(); i += 1) { + mirror.locations.push_back(view.location(i)); + } + view.iterate_symbols( + [&](index::SymbolHash hash, const index::SymbolIdentity& id, llvm::StringRef bitmap) { + auto& symbol = mirror.symbols[hash]; + symbol.name = std::string(id.name); + symbol.kind = id.kind.value(); + symbol.scope = static_cast(id.scope); + const auto* begin = reinterpret_cast(bitmap.data()); + symbol.reference_files.assign(begin, begin + bitmap.size()); + }); + for(std::uint32_t i = 0; i < view.section_count(); i += 1) { + auto blob = view.section_blob(i); + mirror.sections.push_back({view.section_path(i), + view.section_hash(i), + std::vector(blob.begin(), blob.end())}); + } + + auto bytes = kota::codec::fbs::to_bytes(mirror); + if(!bytes) { + return {}; + } + return std::string(bytes->begin(), bytes->end()); +} + +/// A structurally valid blob of `text`'s content generation carrying an +/// explicit variant list — the shapes load()'s healing paths probe. +std::string planted_blob(llvm::StringRef text, std::uint64_t variant) { + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content_hash = llvm::xxh3_64bits(text); + blob.content_size = static_cast(text.size()); + auto starts = kota::ipc::lsp::build_line_starts(std::string_view(text.data(), text.size())); + for(std::size_t i = 0; i < starts.size(); i += 1) { + auto next = i + 1 < starts.size() ? starts[i + 1] : blob.content_size; + blob.line_lengths.push_back(static_cast(next - starts[i])); + } + blob.variants = {variant}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; +} + void open_store(TempDir& tmp, Workspace& workspace) { auto store = CacheStore::open(tmp.path("cache"), 1); ASSERT_TRUE(store.has_value()); @@ -164,7 +245,7 @@ TEST_CASE(MergeRejectsGarbage) { ASSERT_TRUE(workspace.project_index.symbols.empty()); } -TEST_CASE(MergeSkipsMovedDisk) { +TEST_CASE(MergeIgnoresDiskDrift) { TempDir tmp; tmp.touch("main.cpp", "int value() { return 1; }\n"); auto src = tmp.path("main.cpp"); @@ -176,21 +257,23 @@ TEST_CASE(MergeSkipsMovedDisk) { auto path_id = workspace.path_pool.intern(indexed.tu_path); auto it = workspace.shards.find(path_id); ASSERT_TRUE(it != workspace.shards.end()); - ASSERT_EQ(it->second.content(), "int value() { return 1; }\n"); + ASSERT_EQ(it->second.content_hash(), llvm::xxh3_64bits("int value() { return 1; }\n")); - // The disk moved on since the rows were indexed: merging them would - // pair offsets with bytes they were not built from, so the merge must - // be skipped and the last-known snapshot kept serving. + // The disk moved on since the rows were indexed. The blob is + // self-contained — its rows pair with the generation it embeds, never + // with the disk — so the re-merge is a pure variant hit and freshness + // gating owns the drift. tmp.touch("main.cpp", "int renamed() { return 2; }\n"); indexer.merge(indexed.data.data(), indexed.data.size()); - ASSERT_EQ(it->second.content(), "int value() { return 1; }\n"); + ASSERT_EQ(it->second.content_hash(), llvm::xxh3_64bits("int value() { return 1; }\n")); + ASSERT_EQ(it->second.variants().size(), std::size_t(1)); ASSERT_TRUE(workspace.project_index.contributions.lookup(path_id).contains(path_id)); - // Once the rows describe the settled content again, the merge lands. + // Rows built from the settled content open a new generation. auto fresh = index_file(tmp, src); ASSERT_FALSE(fresh.data.empty()); indexer.merge(fresh.data.data(), fresh.data.size()); - ASSERT_EQ(it->second.content(), "int renamed() { return 2; }\n"); + ASSERT_EQ(it->second.content_hash(), llvm::xxh3_64bits("int renamed() { return 2; }\n")); } TEST_CASE(SaveCommitsDirtyShard) { @@ -221,7 +304,7 @@ TEST_CASE(SaveCommitsDirtyShard) { ASSERT_TRUE(it != workspace.shards.end()); ASSERT_EQ(indexer.pending_shard_writes(), 0u); ASSERT_EQ(indexer.last_save_shards(), 1u); - ASSERT_EQ(it->second.content(), "int flip_value() { return 1; }\n"); + ASSERT_EQ(it->second.content_hash(), llvm::xxh3_64bits("int flip_value() { return 1; }\n")); ASSERT_TRUE(workspace.project_index.contributions.lookup(path_id).contains(path_id)); } @@ -270,7 +353,7 @@ TEST_CASE(MidSaveMergeKept) { auto it = workspace.shards.find(path_id); ASSERT_TRUE(it != workspace.shards.end()); ASSERT_EQ(indexer.pending_shard_writes(), 1u); - ASSERT_EQ(it->second.content(), "int second_value() { return 2; }\n"); + ASSERT_EQ(it->second.content_hash(), llvm::xxh3_64bits("int second_value() { return 2; }\n")); auto again_body = [&]() -> kota::task<> { co_await indexer.save(); @@ -281,7 +364,7 @@ TEST_CASE(MidSaveMergeKept) { it = workspace.shards.find(path_id); ASSERT_EQ(indexer.pending_shard_writes(), 0u); - ASSERT_EQ(it->second.content(), "int second_value() { return 2; }\n"); + ASSERT_EQ(it->second.content_hash(), llvm::xxh3_64bits("int second_value() { return 2; }\n")); } TEST_CASE(MergeHitWritesNothing) { @@ -349,7 +432,7 @@ TEST_CASE(SharedHeaderVariants) { ASSERT_EQ(workspace.project_index.contributions.lookup(header_id).size(), std::size_t(3)); } -TEST_CASE(HeaderSkipCarriesContribution) { +TEST_CASE(HeaderRegenerationReplaces) { TempDir tmp; tmp.touch("dep.h", "#pragma once\ninline int dep() { return 1; }\n"); tmp.touch("main.cpp", "#include \"dep.h\"\nint use() { return dep(); }\n"); @@ -364,17 +447,23 @@ TEST_CASE(HeaderSkipCarriesContribution) { ASSERT_TRUE(old_hash != 0); // The header changes, a reindex captures it — and the header changes - // AGAIN before the result merges. The stale section must not land, but - // the previous contribution keeps serving (its rows still match the - // shard) until a follow-up pass settles. + // AGAIN before the result merges. The worker's bytes are their own + // generation: they land verbatim regardless of the disk moving on, and + // freshness gating owns the remaining drift. tmp.touch("dep.h", "#pragma once\ninline int dep() { return 2; }\n"); auto v2 = index_file(tmp, src); ASSERT_FALSE(v2.data.empty()); tmp.touch("dep.h", "#pragma once\ninline int dep() { return 3; }\n"); indexer.merge(v2.data.data(), v2.data.size()); - ASSERT_EQ(workspace.project_index.contributions.lookup(header_id).lookup(tu_id), old_hash); - ASSERT_TRUE(workspace.shards[header_id].has_variant(old_hash)); + auto new_hash = workspace.project_index.contributions.lookup(header_id).lookup(tu_id); + ASSERT_TRUE(new_hash != 0); + ASSERT_TRUE(new_hash != old_hash); + ASSERT_TRUE(workspace.shards[header_id].has_variant(new_hash)); + // A new content generation never shares row storage with the old one. + ASSERT_FALSE(workspace.shards[header_id].has_variant(old_hash)); + ASSERT_EQ(workspace.shards[header_id].content_hash(), + llvm::xxh3_64bits("#pragma once\ninline int dep() { return 2; }\n")); } TEST_CASE(SaveCompactsAndRetires) { @@ -534,25 +623,25 @@ TEST_CASE(RejectsCorruptSection) { auto indexed = index_file(tmp, tmp.path("cor_main.cpp")); ASSERT_FALSE(indexed.data.empty()); - // Corrupt the main file's nested rows section: the outer wire still - // verifies (sections are opaque bytes to it), only the nested decode - // fails. - auto tampered = index::TUIndex::from(indexed.data); - ASSERT_TRUE(tampered.has_value()); - auto main_id = static_cast(tampered->graph.paths.size() - 1); - for(auto& section: tampered->sections) { - if(section.path_id == main_id) { - section.rows = {0, 1, 2, 3}; - } + // Corrupt the main file's blob bytes in place: the outer wire still + // verifies (sections are opaque bytes to it), only the blob's byte + // identity check against the recorded section hash fails. + std::string corrupt = indexed.data; + auto tampered = index::TUIndex::from_bytes(corrupt); + ASSERT_TRUE(tampered.loaded()); + auto main_section = tampered.section_of(tampered.path_count() - 1); + ASSERT_TRUE(main_section.has_value()); + auto blob = tampered.section_blob(*main_section); + auto pos = llvm::StringRef(corrupt).find(blob); + ASSERT_TRUE(pos != llvm::StringRef::npos); + for(std::size_t i = 0; i < blob.size(); i += 1) { + corrupt[pos + i] = 'X'; } - std::string corrupt; - llvm::raw_string_ostream os(corrupt); - tampered->serialize(os); - - // The header section decodes fine and is staged before the main - // section's decode fails; the reject must discard the whole result — a - // manifest whose recorded versions all match the disk would otherwise - // be judged fresh forever with the main file's rows missing. + + // The header section verifies fine and is staged before the main + // section's identity check fails; the reject must discard the whole + // result — a manifest whose recorded versions all match the disk would + // otherwise be judged fresh forever with the main file's rows missing. indexer.merge(corrupt.data(), corrupt.size()); auto tu_id = workspace.path_pool.intern(indexed.tu_path); auto header_id = workspace.path_pool.intern(tmp.path("cor.h")); @@ -571,32 +660,6 @@ TEST_CASE(RejectsCorruptSection) { ASSERT_TRUE(workspace.shards.contains(header_id)); } -TEST_CASE(UnverifiedPairingSkipped) { - TempDir tmp; - tmp.touch("unv.cpp", "int unverified_fn() { return 123456; }\n"); - auto src = tmp.path("unv.cpp"); - auto indexed = index_file(tmp, src); - ASSERT_FALSE(indexed.data.empty()); - - // Strip the consumed-content hashes (a file behind a PCM ships none) - // and shrink the file: the content arbitration cannot see the disk - // moving on, but the rows overrun the shorter content and the fresh - // blob's own range bounds must catch it — a skip, not a blob serving - // ranges past its content. - auto tampered = index::TUIndex::from(indexed.data); - ASSERT_TRUE(tampered.has_value()); - for(auto& hash: tampered->graph.path_hashes) { - hash = 0; - } - std::string wire; - llvm::raw_string_ostream os(wire); - tampered->serialize(os); - tmp.touch("unv.cpp", "int f;\n"); - - indexer.merge(wire.data(), wire.size()); - ASSERT_FALSE(workspace.shards.contains(workspace.path_pool.intern(src))); -} - TEST_CASE(HashlessRemergeHits) { TempDir tmp; tmp.touch("pcm.cpp", "int hashless_fn() { return 7; }\n"); @@ -604,16 +667,11 @@ TEST_CASE(HashlessRemergeHits) { auto indexed = index_file(tmp, src); ASSERT_FALSE(indexed.data.empty()); - // A file behind a PCM ships no consumed-content hash, so the no-IO - // fast path cannot vouch for a stored variant; only the disk read can. - auto tampered = index::TUIndex::from(indexed.data); - ASSERT_TRUE(tampered.has_value()); - for(auto& hash: tampered->graph.path_hashes) { - hash = 0; - } - std::string wire; - llvm::raw_string_ostream os(wire); - tampered->serialize(os); + // A file behind a PCM ships no consumed-content hash; the variant + // identity is the blob's own byte hash, so membership needs no + // content vouching at all. + auto wire = strip_path_hashes(indexed.data); + ASSERT_FALSE(wire.empty()); indexer.merge(wire.data(), wire.size()); auto path_id = workspace.path_pool.intern(src); @@ -865,12 +923,8 @@ TEST_CASE(LoadHealsMissingVariant) { // Replace the header's blob with one that verifies but stores a variant // no manifest contributed — the residue of a crash or failed write that // landed the manifest without its shard. - index::FileIndex no_rows; - index::VariantInput stranger{.hash = 0x1234, .rows = &no_rows}; - std::string bytes; - llvm::raw_string_ostream os(bytes); - index::write_shard({}, {}, stranger, "", 0, os); - tmp.touch("cache/cache/v1/index/" + header_key + ".idx", bytes); + tmp.touch("cache/cache/v1/index/" + header_key + ".idx", + planted_blob("#pragma once\ninline int dep() { return 1; }\n", 0x1234)); IndexerFixture f; open_store(tmp, f.workspace); @@ -907,16 +961,11 @@ TEST_CASE(LoadHealsWrongGeneration) { } // Replace the header's blob with one from ANOTHER content generation - // that still stores the contributed variant — the residue of a failed - // shard write when an edit past every indexed row keeps the rows hash - // identical. Every recorded FileVersion matches the disk, so only the - // generation pin can tell that positions would map through stale text. - index::FileIndex no_rows; - index::VariantInput same_rows{.hash = rows_hash, .rows = &no_rows}; - std::string bytes; - llvm::raw_string_ostream os(bytes); - index::write_shard({}, {}, same_rows, "stale text", llvm::xxh3_64bits("stale text"), os); - tmp.touch("cache/cache/v1/index/" + header_key + ".idx", bytes); + // whose explicit variant list still claims the contributed identity — + // the residue of a crash between shard and manifest writes. Every + // recorded FileVersion matches the disk, so only the generation pin + // can tell that positions would map through stale text. + tmp.touch("cache/cache/v1/index/" + header_key + ".idx", planted_blob("stale text", rows_hash)); IndexerFixture f; open_store(tmp, f.workspace); diff --git a/tests/unit/server/pch_worker_tests.cpp b/tests/unit/server/pch_worker_tests.cpp index eb05d3408..831ef8959 100644 --- a/tests/unit/server/pch_worker_tests.cpp +++ b/tests/unit/server/pch_worker_tests.cpp @@ -3,8 +3,8 @@ #include #include "test/test.h" -#include "index/preamble_state.h" #include "server/protocol/worker.h" +#include "server/state/workspace.h" #include "server/worker_test_helpers.h" #include "syntax/scan.h" @@ -74,9 +74,9 @@ TEST_CASE(BuildPCHThenCompile) { // Verify the PCH file exists on disk. ASSERT_TRUE(llvm::sys::fs::exists(pch_path)); - // The worker wrote the paired PreambleState blob: it must load and + // The worker wrote the paired preamble envelope: it must load and // carry the preamble's document links (the #include of common.h). - auto state = index::PreambleState::load(tmp.path("preamble.pch.idx")); + auto state = load_pch_envelope(tmp.path("preamble.pch.idx")); ASSERT_TRUE(state != nullptr); bool has_common_link = std::ranges::any_of(state->links(), [&](auto& link) { return llvm::StringRef(link.target).ends_with("common.h"); diff --git a/tests/unit/server/query_freshness_tests.cpp b/tests/unit/server/query_freshness_tests.cpp index 449809808..54f066e22 100644 --- a/tests/unit/server/query_freshness_tests.cpp +++ b/tests/unit/server/query_freshness_tests.cpp @@ -34,44 +34,25 @@ IndexQuery agent_query{workspace, store, indexer, {.disk_only = true}}; std::uint32_t main_id = 0; std::uint32_t header_id = 0; -/// Build a TUIndex from the added sources and merge it into the workspace -/// with real contents, so shards can map their rows to positions. +/// Build an envelope from the added sources and merge it into the +/// workspace, installing each section's blob verbatim as the file's shard. void merge_into_workspace() { - auto tu_index = index::TUIndex::build(*unit); - std::string wire; - llvm::raw_string_ostream wos(wire); - tu_index.serialize(wos); - auto view = index::TUIndexView::from(wire); - ASSERT_TRUE(view.has_value()); + auto wire = index::build_tu_index(*unit); + auto view = index::TUIndex::from_bytes(wire); + ASSERT_TRUE(view.loaded()); llvm::SmallVector file_ids_map; - for(std::uint32_t i = 0; i < view->path_count(); i += 1) { - file_ids_map.push_back(workspace.path_pool.intern(view->path(i))); + for(std::uint32_t i = 0; i < view.path_count(); i += 1) { + file_ids_map.push_back(workspace.path_pool.intern(view.path(i))); } - ASSERT_TRUE(workspace.project_index.merge(*view, file_ids_map)); - main_id = file_ids_map[view->path_count() - 1]; - - auto content_of = [&](llvm::StringRef path) -> llvm::StringRef { - auto it = sources.all_files.find(llvm::sys::path::filename(path)); - return it != sources.all_files.end() ? llvm::StringRef(it->second.content) - : llvm::StringRef(); - }; - auto lookup_symbol = [&](index::SymbolHash hash) { - return view->find_symbol(hash); - }; - - for(std::uint32_t section = 0; section < view->section_count(); section += 1) { - auto local_id = view->section_path(section); - auto rows = view->decode_section_rows(section); - ASSERT_TRUE(rows.has_value()); - auto content = content_of(view->path(local_id)); - index::VariantInput fresh{view->section_rows_hash(section), &*rows, lookup_symbol}; - std::string bytes; - llvm::raw_string_ostream os(bytes); - index::write_shard(index::Shard(), {}, fresh, content, llvm::xxh3_64bits(content), os); - workspace.shards[file_ids_map[local_id]] = - index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); - if(llvm::sys::path::filename(view->path(local_id)) == "header.h") { + ASSERT_TRUE(workspace.project_index.merge(view, file_ids_map)); + main_id = file_ids_map[view.path_count() - 1]; + + for(std::uint32_t section = 0; section < view.section_count(); section += 1) { + auto local_id = view.section_path(section); + workspace.shards[file_ids_map[local_id]] = index::Shard::from_buffer( + llvm::MemoryBuffer::getMemBufferCopy(view.section_blob(section))); + if(llvm::sys::path::filename(view.path(local_id)) == "header.h") { header_id = file_ids_map[local_id]; } } diff --git a/tests/unit/server/query_overlay_tests.cpp b/tests/unit/server/query_overlay_tests.cpp index 565fdc1ff..4505d8445 100644 --- a/tests/unit/server/query_overlay_tests.cpp +++ b/tests/unit/server/query_overlay_tests.cpp @@ -4,8 +4,9 @@ #include "test/temp_dir.h" #include "test/test.h" #include "test/tester.h" -#include "index/preamble_state.h" +#include "index/serialization.h" #include "index/shard.h" +#include "index/tu_index.h" #include "server/compiler/context_resolver.h" #include "server/compiler/indexer.h" #include "server/service/query.h" @@ -37,28 +38,26 @@ index::TUIndex full_index; std::shared_ptr session; std::string main_path; -/// Compile the added sources, serialize the full TUIndex as a -/// PreambleState blob (exactly what the PCH build produces), and open a -/// session whose pch_key points at it. The session's own file index is -/// the interested-only index, mirroring the production per-edit index. +/// Compile the added sources, persist the preamble envelope (exactly what +/// the PCH build produces), and open a session whose pch_key points at +/// it. The session's own index is the interested-only envelope, mirroring +/// the production per-edit index. void open_with_overlay(std::source_location location = std::source_location::current()) { ASSERT_TRUE(compile()); - full_index = index::TUIndex::build(*unit); + full_index = index::TUIndex::from_buffer( + llvm::MemoryBuffer::getMemBufferCopy(index::build_tu_index(*unit))); + ASSERT_TRUE(full_index.loaded()); + auto blob_path = dir.path("overlay.pch.idx"); - { - std::error_code ec; - llvm::raw_fd_ostream os(blob_path, ec); - ASSERT_FALSE(bool(ec)); - index::PreambleState::serialize(*unit, full_index, {}, {}, {}, os); - } + dir.touch("overlay.pch.idx", index::build_preamble_index(*unit, {}, {}, {})); auto& st = workspace.pch_cache["key"]; st.path = "unused.pch"; st.index_path = blob_path; st.state = nullptr; - main_path = full_index.graph.paths.back(); + main_path = std::string(full_index.path(full_index.path_count() - 1)); auto path_id = workspace.path_pool.intern(main_path); session = session_store.open(path_id); @@ -67,9 +66,8 @@ void open_with_overlay(std::source_location location = std::source_location::cur session->text = it->second.content; session->line_starts = kota::ipc::lsp::build_line_starts(session->text); - auto session_index = index::TUIndex::build(*unit, true); - session->file_index = std::move(session_index.main_file_index); - session->symbols = std::move(session_index.symbols); + session->index = index::TUIndex::from_buffer( + llvm::MemoryBuffer::getMemBufferCopy(index::build_tu_index(*unit, true))); session->ast_dirty = false; session->pch_key = "key"; } @@ -78,60 +76,60 @@ index::SymbolHash hash_of(llvm::StringRef name, std::source_location location = std::source_location::current()) { index::SymbolHash hash = 0; std::uint32_t count = 0; - for(auto& [symbol_id, symbol]: full_index.symbols) { - if(symbol.name == name) { - hash = symbol_id; - count += 1; - } - } + full_index.iterate_symbols( + [&](index::SymbolHash symbol_id, const index::SymbolIdentity& symbol, llvm::StringRef) { + if(symbol.name == name) { + hash = symbol_id; + count += 1; + } + }); EXPECT_EQ(count, 1); return hash; } std::string header_path(llvm::StringRef basename) { - for(auto& path: full_index.graph.paths) { - if(llvm::sys::path::filename(path) == basename) - return path; + for(std::uint32_t i = 0; i < full_index.path_count(); i += 1) { + if(llvm::sys::path::filename(full_index.path(i)) == basename) + return std::string(full_index.path(i)); } return {}; } -/// Merge the full TUIndex into the workspace's disk index with real -/// contents, as background indexing would. +/// Merge the full envelope into the workspace's disk index, installing +/// each section's blob verbatim as background indexing would. void merge_disk_index() { - std::string wire; - llvm::raw_string_ostream wos(wire); - full_index.serialize(wos); - auto view = index::TUIndexView::from(wire); - ASSERT_TRUE(view.has_value()); - llvm::SmallVector file_ids_map; - for(std::uint32_t i = 0; i < view->path_count(); i += 1) { - file_ids_map.push_back(workspace.path_pool.intern(view->path(i))); + for(std::uint32_t i = 0; i < full_index.path_count(); i += 1) { + file_ids_map.push_back(workspace.path_pool.intern(full_index.path(i))); } - ASSERT_TRUE(workspace.project_index.merge(*view, file_ids_map)); + ASSERT_TRUE(workspace.project_index.merge(full_index, file_ids_map)); - auto content_of = [&](llvm::StringRef path) -> llvm::StringRef { - auto it = sources.all_files.find(llvm::sys::path::filename(path)); - return it != sources.all_files.end() ? llvm::StringRef(it->second.content) - : llvm::StringRef(); - }; - auto lookup_symbol = [&](index::SymbolHash hash) { - return view->find_symbol(hash); + for(std::uint32_t section = 0; section < full_index.section_count(); section += 1) { + auto local_id = full_index.section_path(section); + workspace.shards[file_ids_map[local_id]] = index::Shard::from_buffer( + llvm::MemoryBuffer::getMemBufferCopy(full_index.section_blob(section))); + } +} + +/// A settled, rows-empty per-edit index — what a session holds when every +/// row of its buffer lives behind the PCH. `loaded` is what the freshness +/// gate keys on; an unloaded index means "compile not settled". +index::TUIndex empty_session_index() { + // Field order MUST mirror the envelope layout (tu_index.cpp). + struct EnvelopeMirror { + std::uint32_t format_version = index::index_format_version; + std::int64_t built_at = 1; + std::vector paths; }; - for(std::uint32_t section = 0; section < view->section_count(); section += 1) { - auto local_id = view->section_path(section); - auto rows = view->decode_section_rows(section); - ASSERT_TRUE(rows.has_value()); - auto content = content_of(view->path(local_id)); - index::VariantInput fresh{view->section_rows_hash(section), &*rows, lookup_symbol}; - std::string bytes; - llvm::raw_string_ostream os(bytes); - index::write_shard(index::Shard(), {}, fresh, content, llvm::xxh3_64bits(content), os); - workspace.shards[file_ids_map[local_id]] = - index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); + EnvelopeMirror mirror; + mirror.paths = {main_path}; + auto bytes = kota::codec::fbs::to_bytes(mirror); + if(!bytes) { + return {}; } + return index::TUIndex::from_buffer(llvm::MemoryBuffer::getMemBufferCopy( + llvm::StringRef(reinterpret_cast(bytes->data()), bytes->size()))); } protocol::Position position_of(llvm::StringRef name) { @@ -200,8 +198,8 @@ int main() { return 0; } // Production per-edit indexes never see the preamble region (the PCH // swallows it); emulate that by emptying the session's own index so // the cursor can only resolve through the overlay's main-file entry. - session->file_index = index::FileIndex(); - session->symbols = index::SymbolTable(); + session->index = empty_session_index(); + ASSERT_TRUE(session->index.loaded()); auto uri = std::string("file://") + main_path; auto info = index_query.lookup_symbol(uri, main_path, position_of("macro"), session.get()); @@ -327,11 +325,16 @@ int main() { §(ref)⟦foo⟧(); return 0; } } TEST_CASE(MacroDefinitionText) { - add_main("main.cpp", R"(#define §(macro)⟦FOO⟧ 1 + // The π comment keeps the file non-ASCII, so its content is stored in + // the blob and the text path serves without touching the disk. + add_main("main.cpp", R"(// π +#define §(macro)⟦FOO⟧ 1 int main() { return 0; } )"); ASSERT_TRUE(compile()); - full_index = index::TUIndex::build(*unit); + full_index = index::TUIndex::from_buffer( + llvm::MemoryBuffer::getMemBufferCopy(index::build_tu_index(*unit))); + ASSERT_TRUE(full_index.loaded()); merge_disk_index(); // Macro Definition relations carry the full #define extent, so the @@ -341,6 +344,31 @@ int main() { return 0; } EXPECT_TRUE(llvm::StringRef(text->text).contains("FOO")); } +TEST_CASE(AsciiPreviewDegrades) { + // Pure-ASCII blobs re-read the disk for previews. This file only + // exists in the test VFS, so the read fails like a moved-on file: + // definition text degrades to nothing while references keep serving + // their positions, only without context lines. + add_main("main.cpp", R"(#define §(macro)⟦FOO⟧ 1 +int use = FOO; +int main() { return 0; } +)"); + ASSERT_TRUE(compile()); + full_index = index::TUIndex::from_buffer( + llvm::MemoryBuffer::getMemBufferCopy(index::build_tu_index(*unit))); + ASSERT_TRUE(full_index.loaded()); + merge_disk_index(); + + auto foo = hash_of("FOO"); + EXPECT_FALSE(agent_query.get_definition_text(foo).has_value()); + + auto references = agent_query.collect_references(foo, RelationKind::Reference); + ASSERT_FALSE(references.empty()); + for(auto& reference: references) { + EXPECT_TRUE(reference.context.empty()); + } +} + TEST_CASE(SharedPreambleScoped) { add_main("main.cpp", R"(#define §(macro)⟦§(macro)FOO⟧ 1 #if FOO @@ -348,8 +376,8 @@ TEST_CASE(SharedPreambleScoped) { int main() { return 0; } )"); open_with_overlay(); - session->file_index = index::FileIndex(); - session->symbols = index::SymbolTable(); + session->index = empty_session_index(); + ASSERT_TRUE(session->index.loaded()); // A second file with a byte-identical preamble shares the PCH (the // key excludes the source path), but the preamble entry carries @@ -375,8 +403,8 @@ TEST_CASE(DirtyPreambleServed) { int main() { return 0; } )"); open_with_overlay(); - session->file_index = index::FileIndex(); - session->symbols = index::SymbolTable(); + session->index = empty_session_index(); + ASSERT_TRUE(session->index.loaded()); // Body edits dirty the session but never move preamble rows: as long // as the buffer still starts with the blob's preamble text, the @@ -392,8 +420,8 @@ TEST_CASE(PreambleDriftSkipped) { int main() { return 0; } )"); open_with_overlay(); - session->file_index = index::FileIndex(); - session->symbols = index::SymbolTable(); + session->index = empty_session_index(); + ASSERT_TRUE(session->index.loaded()); // A deferred PCH rebuild keeps an old blob while the buffer's // preamble moved on; once the buffer no longer starts with the blob's @@ -424,10 +452,9 @@ int main() { §(ref)⟦foo⟧(); return 0; } relation.set_definition_range({0, 3}); fake.relations[foo].push_back(relation); auto header_id = workspace.path_pool.intern(header_path("foo.h")); - index::VariantInput fresh{.hash = fake.rows_hash(), .rows = &fake}; std::string bytes; llvm::raw_string_ostream os(bytes); - index::write_shard(index::Shard(), {}, fresh, "xxx\n", llvm::xxh3_64bits("xxx\n"), os); + index::write_shard(fake, {}, "xxx\n", os); workspace.shards[header_id] = index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); workspace.project_index.symbols[foo].reference_files.add(header_id); From 478b2685f44d6ce24d2f9b02bd1d8911eef30fea Mon Sep 17 00:00:00 2001 From: ykiko Date: Mon, 17 Aug 2026 04:41:37 +0800 Subject: [PATCH 07/10] fix(server): guard offset wrap, style cleanup, review test pins --- src/index/types.h | 3 +- src/server/protocol/position.h | 11 ++-- src/server/service/query.cpp | 7 +-- src/server/state/workspace.cpp | 2 +- src/server/transport/agent_client.cpp | 7 +-- src/server/worker/stateless_worker.cpp | 11 ++-- tests/unit/server/position_tests.cpp | 76 +++++++++++++++++++++++ tests/unit/server/query_overlay_tests.cpp | 52 ++++++++++++++++ 8 files changed, 149 insertions(+), 20 deletions(-) create mode 100644 tests/unit/server/position_tests.cpp diff --git a/src/index/types.h b/src/index/types.h index d094ce0fe..e550de236 100644 --- a/src/index/types.h +++ b/src/index/types.h @@ -56,10 +56,9 @@ struct Relation { }; struct Occurrence { - /// range of this occurrence. Range range; - /// + /// Hash of the symbol this occurrence names. SymbolHash target; friend bool operator==(const Occurrence&, const Occurrence&) = default; diff --git a/src/server/protocol/position.h b/src/server/protocol/position.h index 3887d4f68..c3ee701fe 100644 --- a/src/server/protocol/position.h +++ b/src/server/protocol/position.h @@ -2,6 +2,7 @@ /// Shared LSP position clamping for master-side buffer access. +#include #include #include #include @@ -61,11 +62,13 @@ class IndexedLineMap { if(position.line >= starts.size()) { return std::nullopt; } - auto offset = starts[position.line] + position.character; - if(offset > line_end(position.line)) { + // Compare against the line length, not the summed offset: the + // character is untrusted client input and the sum can wrap. + auto start = starts[position.line]; + if(position.character > line_end(position.line) - start) { return std::nullopt; } - return offset; + return start + position.character; } std::optional to_range(std::uint32_t begin, std::uint32_t end) const { @@ -89,7 +92,7 @@ class IndexedLineMap { } std::uint32_t line_of(std::uint32_t offset) const { - auto it = std::upper_bound(starts.begin(), starts.end(), offset); + auto it = std::ranges::upper_bound(starts, offset); return it == starts.begin() ? 0 : static_cast(it - starts.begin()) - 1; } diff --git a/src/server/service/query.cpp b/src/server/service/query.cpp index e6d7fd0bc..51b220e63 100644 --- a/src/server/service/query.cpp +++ b/src/server/service/query.cpp @@ -546,10 +546,9 @@ std::optional IndexQuery::find_definition_location(index::Sy if(!uri) continue; auto& merged_index = shard_it->second; - auto ls = merged_index.line_starts(); - if(ls.empty()) - continue; - IndexedLineMap map(merged_index.content(), merged_index.content_size(), ls); + IndexedLineMap map(merged_index.content(), + merged_index.content_size(), + merged_index.line_starts()); std::optional result; merged_index.lookup(hash, RelationKind::Definition, [&](const index::Relation& r) { if(auto range = map.to_range(r.range.begin, r.range.end)) { diff --git a/src/server/state/workspace.cpp b/src/server/state/workspace.cpp index 64e6aa0f0..f85442f89 100644 --- a/src/server/state/workspace.cpp +++ b/src/server/state/workspace.cpp @@ -465,7 +465,7 @@ void Workspace::enforce_loaded_budget() { i += 1; continue; } - LOG_DEBUG("Unloading PreambleIndex of {} (budget {})", loaded_state_lru[i], budget); + LOG_DEBUG("Unloading pch.idx envelope of {} (budget {})", loaded_state_lru[i], budget); it->second.state.reset(); loaded_state_lru.erase(loaded_state_lru.begin() + i); } diff --git a/src/server/transport/agent_client.cpp b/src/server/transport/agent_client.cpp index 545d790a2..497d2c28e 100644 --- a/src/server/transport/agent_client.cpp +++ b/src/server/transport/agent_client.cpp @@ -361,10 +361,9 @@ AgentClient::AgentClient(MasterServer& server, kota::ipc::JsonPeer& peer) : co_return result; auto& merged_index = shard_it->second; - auto ls = merged_index.line_starts(); - if(ls.empty()) - co_return result; - IndexedLineMap map(merged_index.content(), merged_index.content_size(), ls); + IndexedLineMap map(merged_index.content(), + merged_index.content_size(), + merged_index.line_starts()); for(auto& [hash, symbol]: srv.workspace.project_index.symbols) { if(symbol.name.empty()) diff --git a/src/server/worker/stateless_worker.cpp b/src/server/worker/stateless_worker.cpp index 4249814a3..90801ef7a 100644 --- a/src/server/worker/stateless_worker.cpp +++ b/src/server/worker/stateless_worker.cpp @@ -45,7 +45,8 @@ using RequestContext = kota::ipc::BincodePeer::RequestContext; /// is still in memory — the only moment the preamble's index is /// obtainable without deserializing the whole PCH. The file write /// happens separately, after the PCH itself is flushed. -static std::string serialize_preamble_state(CompilationUnit& unit, std::uint32_t preamble_bound) { +static std::string serialize_preamble_envelope(CompilationUnit& unit, + std::uint32_t preamble_bound) { ScopedTimer links_timer; auto links = feature::document_links(unit); auto inactive = feature::inactive_regions(unit, {}, 0, preamble_bound); @@ -63,8 +64,8 @@ static std::string serialize_preamble_state(CompilationUnit& unit, std::uint32_t /// Write the serialized blob next to the PCH. Returns an error description /// on failure so the master's anomaly carries the cause. -static std::optional write_preamble_state(llvm::StringRef blob, - llvm::StringRef output_path) { +static std::optional write_preamble_envelope(llvm::StringRef blob, + llvm::StringRef output_path) { std::error_code ec; llvm::raw_fd_ostream os(output_path, ec); if(ec) { @@ -129,7 +130,7 @@ static worker::BuildResult handle_build_pch(const worker::BuildParams& params, std::string blob; ScopedTimer index_timer; if(success) { - blob = serialize_preamble_state(unit, params.preamble_bound); + blob = serialize_preamble_envelope(unit, params.preamble_bound); } auto index_ms = index_timer.ms(); @@ -148,7 +149,7 @@ static worker::BuildResult handle_build_pch(const worker::BuildParams& params, bool internal_error = false; ScopedTimer state_write_timer; if(success) { - if(auto error = write_preamble_state(blob, params.index_output_path)) { + if(auto error = write_preamble_envelope(blob, params.index_output_path)) { success = false; internal_error = true; errors = std::move(*error); diff --git a/tests/unit/server/position_tests.cpp b/tests/unit/server/position_tests.cpp new file mode 100644 index 000000000..90d21780d --- /dev/null +++ b/tests/unit/server/position_tests.cpp @@ -0,0 +1,76 @@ +#include + +#include "test/test.h" +#include "server/protocol/position.h" + +namespace clice::testing { +namespace { + +TEST_SUITE(IndexedLineMap) { + +// Line starts of "ab\ncd\n" — two 2-byte lines plus the empty last line. +std::vector starts = {0, 3, 6}; + +TEST_CASE(AsciiArithmetic) { + // No stored content: byte offsets are UTF-16 offsets, mapping is pure + // line-table arithmetic bounded by the content size. + IndexedLineMap map("", 6, starts); + + auto pos = map.to_position(4); + ASSERT_TRUE(pos.has_value()); + EXPECT_EQ(pos->line, 1u); + EXPECT_EQ(pos->character, 1u); + + auto offset = map.to_offset(*pos); + ASSERT_TRUE(offset.has_value()); + EXPECT_EQ(*offset, 4u); + + // The newline offset is its line's end position, not the next line. + auto line_end = map.to_position(2); + ASSERT_TRUE(line_end.has_value()); + EXPECT_EQ(line_end->line, 0u); + EXPECT_EQ(line_end->character, 2u); + + // Past the content, past the line, inverted range: all refused. + EXPECT_FALSE(map.to_position(7).has_value()); + EXPECT_FALSE(map.to_offset({.line = 3, .character = 0}).has_value()); + EXPECT_FALSE(map.to_offset({.line = 0, .character = 3}).has_value()); + // A huge character must not wrap the offset back into bounds. + EXPECT_FALSE(map.to_offset({.line = 1, .character = 0xfffffffd}).has_value()); + EXPECT_FALSE(map.to_range(4, 2).has_value()); + + auto range = map.to_range(3, 5); + ASSERT_TRUE(range.has_value()); + EXPECT_EQ(range->start.line, 1u); + EXPECT_EQ(range->start.character, 0u); + EXPECT_EQ(range->end.character, 2u); +} + +TEST_CASE(EmptyLineTable) { + IndexedLineMap map("", 6, {}); + EXPECT_FALSE(map.to_position(0).has_value()); + EXPECT_FALSE(map.to_offset({.line = 0, .character = 0}).has_value()); +} + +TEST_CASE(StoredContentDelegates) { + // Stored (non-ASCII) content delegates to LineMap: the é on line 1 is + // two UTF-8 bytes but one UTF-16 unit, so the offset past it maps to + // a smaller character than its byte column. + llvm::StringRef content = "ab\né!\n"; + std::vector line_starts = {0, 3, 7}; + IndexedLineMap map(content, static_cast(content.size()), line_starts); + + auto pos = map.to_position(5); + ASSERT_TRUE(pos.has_value()); + EXPECT_EQ(pos->line, 1u); + EXPECT_EQ(pos->character, 1u); + + auto offset = map.to_offset(*pos); + ASSERT_TRUE(offset.has_value()); + EXPECT_EQ(*offset, 5u); +} + +}; // TEST_SUITE(IndexedLineMap) + +} // namespace +} // namespace clice::testing diff --git a/tests/unit/server/query_overlay_tests.cpp b/tests/unit/server/query_overlay_tests.cpp index 4505d8445..2b019ab03 100644 --- a/tests/unit/server/query_overlay_tests.cpp +++ b/tests/unit/server/query_overlay_tests.cpp @@ -369,6 +369,58 @@ int main() { return 0; } } } +TEST_CASE(AsciiPreviewFromDisk) { + // The ASCII preview happy path: the blob omits the text, the disk + // still holds the exact bytes, so definition text and context lines + // serve from the re-read; once the file moves on, the hash check + // degrades both back to positions-only. + llvm::StringRef text = "int value = 1;\nint other = value;\n"; + dir.touch("preview.cpp", text); + auto path = dir.path("preview.cpp"); + auto path_id = workspace.path_pool.intern(path); + + index::SymbolHash sym = 777; + index::FileIndex rows; + rows.occurrences.push_back({ + {4, 9}, + sym + }); + index::Relation def{ + .kind = RelationKind::Definition, + .range = {4, 9} + }; + def.set_definition_range({0, 14}); + rows.relations[sym].push_back(def); + rows.relations[sym].push_back({ + .kind = RelationKind::Reference, + .range = {27, 32}, + .target_symbol = 0 + }); + + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard(rows, {}, text, os); + workspace.shards[path_id] = + index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); + ASSERT_TRUE(workspace.shards[path_id].ascii()); + workspace.project_index.symbols[sym].name = "value"; + workspace.project_index.symbols[sym].reference_files.add(path_id); + + auto definition = agent_query.get_definition_text(sym); + ASSERT_TRUE(definition.has_value()); + EXPECT_EQ(definition->text, "int value = 1;"); + + auto references = agent_query.collect_references(sym, RelationKind::Reference); + ASSERT_FALSE(references.empty()); + EXPECT_EQ(references.front().context, "int other = value;"); + + dir.touch("preview.cpp", "int moved = 0;\n"); + EXPECT_FALSE(agent_query.get_definition_text(sym).has_value()); + references = agent_query.collect_references(sym, RelationKind::Reference); + ASSERT_FALSE(references.empty()); + EXPECT_TRUE(references.front().context.empty()); +} + TEST_CASE(SharedPreambleScoped) { add_main("main.cpp", R"(#define §(macro)⟦§(macro)FOO⟧ 1 #if FOO From 4a81365073477060bd439bb43c87b0b7e9cc636c Mon Sep 17 00:00:00 2001 From: ykiko Date: Mon, 17 Aug 2026 05:46:37 +0800 Subject: [PATCH 08/10] fix(index): content identity checks, order gates from review --- benchmarks/index_stats_benchmark.cpp | 1 + benchmarks/pipeline_benchmark.cpp | 5 +- src/index/include_graph.cpp | 21 ++++++- src/index/project_index.cpp | 56 +++++++++++++++--- src/index/project_index.h | 8 +-- src/index/shard.cpp | 54 ++++++++++++++++++ src/index/shard.h | 12 +++- src/index/tu_index.cpp | 24 ++++++-- src/index/tu_index.h | 8 ++- src/server/compiler/indexer.cpp | 27 +++++---- src/server/service/query.cpp | 11 ++-- src/server/service/query.h | 3 +- src/server/state/invalidator.cpp | 2 +- src/server/transport/master_server.cpp | 2 +- tests/unit/index/persisted_index_tests.cpp | 51 +++++++++++++++++ tests/unit/index/preamble_index_tests.cpp | 15 +++-- tests/unit/index/shard_tests.cpp | 66 +++++++++++++++++++++- tests/unit/index/tu_index_tests.cpp | 54 +++++++++++++++--- tests/unit/server/indexer_tests.cpp | 4 ++ tests/unit/server/invalidator_tests.cpp | 23 ++++++-- tests/unit/server/query_overlay_tests.cpp | 1 + 21 files changed, 386 insertions(+), 62 deletions(-) diff --git a/benchmarks/index_stats_benchmark.cpp b/benchmarks/index_stats_benchmark.cpp index 271476d0b..22e059940 100644 --- a/benchmarks/index_stats_benchmark.cpp +++ b/benchmarks/index_stats_benchmark.cpp @@ -410,6 +410,7 @@ struct Stats { scope_n[scope] += 1; } scope_map.try_emplace(hash, symbol.scope); + return true; }); for(std::uint32_t i = 0; i < tu.section_count(); i += 1) { diff --git a/benchmarks/pipeline_benchmark.cpp b/benchmarks/pipeline_benchmark.cpp index c10463761..d6afb10e8 100644 --- a/benchmarks/pipeline_benchmark.cpp +++ b/benchmarks/pipeline_benchmark.cpp @@ -220,7 +220,10 @@ FileResult profile_file(llvm::StringRef file, auto view = index::TUIndex::from_bytes(serialized); std::uint64_t symbols = 0; - view.iterate_symbols([&](auto, auto&, auto) { symbols += 1; }); + view.iterate_symbols([&](auto, auto&, auto) { + symbols += 1; + return true; + }); result.symbols = symbols; return true; }); diff --git a/src/index/include_graph.cpp b/src/index/include_graph.cpp index 3532abbf0..6cf6bdf4e 100644 --- a/src/index/include_graph.cpp +++ b/src/index/include_graph.cpp @@ -3,6 +3,8 @@ #include "compile/compilation_unit.h" #include "support/logging.h" +#include "llvm/ADT/STLExtras.h" +#include "llvm/ADT/SmallVector.h" #include "llvm/Support/xxhash.h" namespace clice::index { @@ -53,8 +55,19 @@ IncludeGraph IncludeGraph::from(CompilationUnitRef unit, llvm::StringMap path_table; IncludeGraph graph; - for(auto& [fid, directive]: unit.directives()) { - for(auto& include: directive.includes) { + // Path and location ids are assigned in first-visit order and the + // envelope's byte hash is an identity, so the visit order must be a + // pure function of the parse — sort every fid set that arrives in + // DenseMap iteration order. + auto& directives = unit.directives(); + llvm::SmallVector directive_fids; + directive_fids.reserve(directives.size()); + for(auto fid: llvm::make_first_range(directives)) { + directive_fids.push_back(fid); + } + llvm::sort(directive_fids); + for(auto fid: directive_fids) { + for(auto& include: directives.find(fid)->second.includes) { if(!include.skipped && include.fid.isValid()) { graph.file_table[include.fid] = addIncludeChain(unit, include.fid, graph, path_table); @@ -62,7 +75,9 @@ IncludeGraph IncludeGraph::from(CompilationUnitRef unit, } } - for(auto fid: indexed_fids) { + llvm::SmallVector sorted_indexed(indexed_fids.begin(), indexed_fids.end()); + llvm::sort(sorted_indexed); + for(auto fid: sorted_indexed) { graph.file_table[fid] = addIncludeChain(unit, fid, graph, path_table); } diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index d55a456db..40c4fad1d 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -14,6 +14,16 @@ namespace clice::index { namespace { +/// DenseMap reserves two sentinel key values per type, so the in-memory +/// tables can never hold them and the writer can never emit them; wire or +/// disk bytes carrying one are corrupt, and inserting one would corrupt +/// (or assert in) the very containers doing the loading. +template +bool reserved_key(T value) { + return value == llvm::DenseMapInfo::getEmptyKey() || + value == llvm::DenseMapInfo::getTombstoneKey(); +} + /// The global layer's persisted form: the FileVersion table as parallel /// columns plus the symbol table with a self-contained path table for its /// reference bitmaps. @@ -57,7 +67,7 @@ struct GlobalBlob { } // namespace bool ProjectIndex::merge(this ProjectIndex& self, - const TUIndex& view, + const TUIndex& index, llvm::ArrayRef file_ids_map) { // Decode and bound every reference bitmap before touching the table: // merged bits persist in the global blob while the result's recorded @@ -75,25 +85,30 @@ bool ProjectIndex::merge(this ProjectIndex& self, std::vector staged; bool valid = true; - view.iterate_symbols( + index.iterate_symbols( [&](SymbolHash hash, const SymbolIdentity& identity, llvm::StringRef bitmap) { - if(!valid || identity.scope != SymbolScope::External) { - return; + if(identity.scope != SymbolScope::External) { + return true; + } + if(reserved_key(hash)) { + valid = false; + return false; } Bitmap references; if(!bitmap.empty()) { auto decoded = read_bitmap(bitmap.data(), bitmap.size()); if(!decoded) { valid = false; - return; + return false; } references = std::move(*decoded); } if(!references.isEmpty() && references.maximum() >= file_ids_map.size()) { valid = false; - return; + return false; } staged.push_back({hash, identity, std::move(references)}); + return true; }); if(!valid) { return false; @@ -329,6 +344,30 @@ bool ProjectIndex::load_global(this ProjectIndex& self, } } + // Every id and hash below becomes a DenseMap or DenseSet key, first in + // the duplicate checks here and then in the tables themselves — see + // reserved_key. + for(auto id: blob.fv_ids) { + if(reserved_key(id)) { + return false; + } + } + for(auto hash: blob.sym_hashes) { + if(reserved_key(hash)) { + return false; + } + } + for(auto id: llvm::make_first_range(blob.sym_paths)) { + if(reserved_key(id)) { + return false; + } + } + for(auto fv: blob.manifest_fvs) { + if(reserved_key(fv)) { + return false; + } + } + // Ids, (path, hash) pairs and symbol hashes are all map keys in the // writer, so a repeat of any marks a corrupt blob. A repeated id in // particular would leave fv_ids interning the earlier pair to an id @@ -374,7 +413,10 @@ bool ProjectIndex::load_global(this ProjectIndex& self, return false; } for(auto id: *decoded) { - if(!covered.contains(id)) { + // Reserved keys first: probing a DenseSet FOR a sentinel value + // matches empty or tombstoned buckets, so contains() could + // spuriously accept exactly the ids the tables cannot hold. + if(reserved_key(id) || !covered.contains(id)) { return false; } } diff --git a/src/index/project_index.h b/src/index/project_index.h index 024029726..a573eb596 100644 --- a/src/index/project_index.h +++ b/src/index/project_index.h @@ -77,16 +77,16 @@ struct ProjectIndex { contributions; /// Merge a TU's external symbols straight off the wire; `file_ids_map` - /// maps the TU-local ids of `view`'s path table to pool ids. Symbol + /// maps the TU-local ids of `index`'s path table to pool ids. Symbol /// names are copied only for symbols new to the table. Returns false — /// with the table untouched — when a reference bitmap fails to decode - /// or carries an id past the path table (the bound TUIndex::from - /// enforces; the zero-copy view leaves it to this consumer): the + /// or carries an id past the path table (the bound TUIndex::from_bytes + /// enforces; the zero-copy reader leaves it to this consumer): the /// caller rejects the whole result, because merged bits persist while /// the result's recorded versions match the disk, so lost bits would /// never be rebuilt. bool merge(this ProjectIndex& self, - const TUIndex& view, + const TUIndex& index, llvm::ArrayRef file_ids_map); /// The FileVersion id for (path, content hash), interning a new record diff --git a/src/index/shard.cpp b/src/index/shard.cpp index 667543318..8e5346108 100644 --- a/src/index/shard.cpp +++ b/src/index/shard.cpp @@ -325,16 +325,26 @@ bool validate(BlobView root) { // on every query, forever — reject the blob so it is rebuilt instead. // Ends are bounded by the content size too: every decoded range is // served as a source range into the content. + // The merge also two-way merges rows under the full (begin, end, sym) + // key with equal-key rows combined at write time, so equal ranges must + // carry strictly ascending symbols — sym_hashes is strictly sorted, so + // id order stands in for hash order. std::uint32_t prev_begin = 0; std::uint32_t prev_end = 0; + std::uint32_t prev_sym = 0; for(std::uint32_t row = 0; row < occ_count; row += 1) { auto begin = occ.begin_of(row); auto end = occ.end_of(row); + auto sym = occ_sym_id(root, row); if(begin < prev_begin || end < prev_end || end < begin || end > content_size) { return false; } + if(row != 0 && begin == prev_begin && end == prev_end && sym <= prev_sym) { + return false; + } prev_begin = begin; prev_end = end; + prev_sym = sym; } // Relation ranges carry no query order to enforce, but are served as // source ranges all the same — bound them like the occurrence ends. @@ -386,6 +396,46 @@ bool validate(BlobView root) { } } + // Rows of one relation group two-way merge under the full (kind, + // begin, end, payload) key with equal-key rows combined at write time, + // so the key must ascend strictly within each group — out-of-order or + // repeated rows would mis-merge silently instead of being rejected. + // The payload mirrors decode_relation_group: a def range, a target + // symbol's hash, or 0. + { + std::size_t sym_cursor = 0; + std::size_t def_cursor = 0; + auto payload_of = [&](std::uint32_t row) -> std::uint64_t { + while(sym_cursor < rel_sym_rows.size() && rel_sym_rows[sym_cursor] < row) { + sym_cursor += 1; + } + while(def_cursor < rel_def_rows.size() && rel_def_rows[def_cursor] < row) { + def_cursor += 1; + } + if(def_cursor < rel_def_rows.size() && rel_def_rows[def_cursor] == row) { + return std::bit_cast( + LocalSourceRange{rel_def_begins[def_cursor], rel_def_ends[def_cursor]}); + } + if(sym_cursor < rel_sym_rows.size() && rel_sym_rows[sym_cursor] == row) { + auto id = !rel_sym8.empty() ? rel_sym8[sym_cursor] + : !rel_sym16.empty() ? rel_sym16[sym_cursor] + : rel_sym32[sym_cursor]; + return sym_hashes[id]; + } + return 0; + }; + for(std::size_t id = 0; id < sym_hashes.size(); id += 1) { + std::tuple prev{}; + for(auto row = offsets[id]; row < offsets[id + 1]; row += 1) { + std::tuple key{rel_kinds[row], rel.begin_of(row), rel.end_of(row), payload_of(row)}; + if(row != offsets[id] && key <= prev) { + return false; + } + prev = key; + } + } + } + auto local_syms = to_array_ref(root[&ShardBlob::local_syms]); auto local_kinds = to_array_ref(root[&ShardBlob::local_kinds]); auto local_scopes = to_array_ref(root[&ShardBlob::local_scopes]); @@ -509,6 +559,10 @@ bool Shard::ascii() const { return loaded() && content().empty(); } +bool Shard::matches_content(llvm::StringRef text) const { + return loaded() && text.size() == content_size() && llvm::xxh3_64bits(text) == content_hash(); +} + std::vector Shard::variants() const { if(!buffer) { return {}; diff --git a/src/index/shard.h b/src/index/shard.h index 24dc8f285..a37dc03da 100644 --- a/src/index/shard.h +++ b/src/index/shard.h @@ -72,6 +72,11 @@ class Shard { /// Whether the content is pure ASCII (and therefore not stored). bool ascii() const; + /// Whether `text` is the exact content the rows were built from — + /// the freshness comparison for disk state. Compares by size and + /// hash, so it works for ASCII blobs, whose text is not stored. + bool matches_content(llvm::StringRef text) const; + /// All variant identities stored in the blob, in mask-bit order. A /// worker-emitted blob holds one anonymous variant identified by its /// own byte hash. @@ -148,8 +153,11 @@ void write_shard(const FileIndex& rows, /// `fresh` as one new variant. Rows shared between variants merge into /// one row; masks are re-encoded for the surviving variant set. `old` may /// be an empty shard (all-fresh merge) and `fresh` may be empty (pure -/// compaction); every fresh shard must be a loaded single-variant blob of -/// the same content generation as the other inputs. +/// compaction), but at least one variant must survive overall — a blob +/// holds at least one variant, and a file whose last variant died is +/// retired by the caller, not compacted to nothing. Every fresh shard +/// must be a loaded single-variant blob of the same content generation as +/// the other inputs. void merge_shards(const Shard& old, llvm::ArrayRef keep, llvm::ArrayRef fresh, diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index 75d4d0f3b..6456008c5 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -35,7 +35,8 @@ struct FileSection { /// The envelope's wire layout. Only the builder below ever materializes /// it; every consumer reads the bytes through the TUIndex reader. struct EnvelopeBlob { - /// Wire schema version (index_format_version), gated by TUIndex::from. + /// Wire schema version (index_format_version), gated by + /// TUIndex::from_bytes. /// A worker respawned after the binary on disk changed can be one /// build ahead of the server, and a layout change need not be /// structurally detectable. @@ -768,8 +769,13 @@ TUIndex TUIndex::from_bytes(llvm::StringRef data) { // Structural verification does not constrain field values; every path // id the merge dereferences against the path table is bounded here so - // the accessors stay check-free. + // the accessors stay check-free. The builder ends every path table + // with the interested file, so consumers address path_count() - 1 + // unchecked — an empty table marks a corrupt envelope. auto count = root[&EnvelopeBlob::paths].size(); + if(count == 0) { + return {}; + } auto locations = root[&EnvelopeBlob::locations]; for(std::size_t i = 0; i < locations.size(); i += 1) { IncludeLocation location = locations.at(i); @@ -777,11 +783,17 @@ TUIndex TUIndex::from_bytes(llvm::StringRef data) { return {}; } } + // section_of binary-searches the section table by path id and shard_of + // trusts the result, so the ids must ascend strictly — a repeated or + // out-of-order id would attribute one file's rows to another. auto sections = root[&EnvelopeBlob::sections]; + std::uint32_t previous_path_id = 0; for(std::size_t i = 0; i < sections.size(); i += 1) { - if(sections.at(i)[&FileSection::path_id] >= count) { + auto path_id = sections.at(i)[&FileSection::path_id]; + if(path_id >= count || (i != 0 && path_id <= previous_path_id)) { return {}; } + previous_path_id = path_id; } TUIndex result; @@ -895,14 +907,16 @@ bool TUIndex::shards_verify() const { } void TUIndex::iterate_symbols( - llvm::function_ref callback) const { + llvm::function_ref callback) const { if(!loaded()) { return; } auto symbols = wire_root(data)[&EnvelopeBlob::symbols]; for(std::size_t i = 0; i < symbols.size(); i += 1) { auto entry = symbols.at(i); - callback(entry.get<0>(), identity_of(entry.get<1>()), bitmap_bytes(entry.get<1>())); + if(!callback(entry.get<0>(), identity_of(entry.get<1>()), bitmap_bytes(entry.get<1>()))) { + return; + } } } diff --git a/src/index/tu_index.h b/src/index/tu_index.h index 198231ed1..c8a290f0d 100644 --- a/src/index/tu_index.h +++ b/src/index/tu_index.h @@ -49,8 +49,9 @@ std::string build_preamble_index(CompilationUnitRef unit, /// and the blob bytes themselves are read straight off the wire — a new /// variant's bytes are sliced out and written or merged without ever /// decoding the envelope around them — and symbol names are touched only -/// when a consumer genuinely needs them. Accessors on an empty reader -/// answer empty/zero. +/// when a consumer genuinely needs them. Whole-envelope accessors on an +/// empty reader answer empty/zero; per-element accessors require a valid +/// index, and no index is valid on an empty reader (every count is 0). class TUIndex { public: TUIndex() = default; @@ -118,8 +119,9 @@ class TUIndex { /// Visit every symbol: hash, identity, and the raw serialized /// reference-files bitmap (a read_bitmap'able portable image). + /// Return false from the callback to stop. void iterate_symbols( - llvm::function_ref + llvm::function_ref callback) const; /// Look up one symbol's identity by hash. diff --git a/src/server/compiler/indexer.cpp b/src/server/compiler/indexer.cpp index 742d789f2..efd562354 100644 --- a/src/server/compiler/indexer.cpp +++ b/src/server/compiler/indexer.cpp @@ -44,10 +44,6 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { LOG_WARN("Ignoring TUIndex that failed verification"); return; } - if(view.path_count() == 0) { - LOG_WARN("Ignoring TUIndex with empty path graph"); - return; - } auto main_local_id = view.path_count() - 1; llvm::StringRef main_tu_path = view.path(main_local_id); @@ -154,10 +150,10 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { // Intern a FileVersion per file of the parse. The freshness baseline is // two-part and lives on the version, shared by every TU that consumed // it: the consumed-content hash from the compiler's own buffers, and a - // stat fast path recorded only for files that provably did not change - // since before the build started — for the rest the stat could describe - // content the rows were never built from, so they re-earn their fast - // path through a hash check instead (see file_version_stale). + // stat fast path recorded only when the disk provably still holds the + // consumed bytes — otherwise the stat could describe content the rows + // were never built from, so those files re-earn their fast path + // through a hash check instead (see file_version_stale). auto baseline_before_ns = fs::stat_baseline_before_ns(view.built_at()); llvm::SmallVector fv_of; fv_of.resize_for_overwrite(view.path_count()); @@ -176,10 +172,19 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { } auto fv = project.intern_file_version(file_ids_map[i], hash); - if(untouched) { + if(untouched && hash != 0) { + // The untouched mtime alone is no proof: a rewrite during the + // build that preserves the size and backdates the mtime + // (rsync -t) would stamp a stat describing bytes the rows were + // never built from, and file_version_stale's equality fast + // path would then judge them fresh forever. Stamp only under a + // disk-hash match; an already-stamped version earned its stamp + // the same way, so it need not re-prove it every merge. auto& record = project.file_versions.find(fv)->second; - record.size = status.getSize(); - record.mtime_ns = fs::mtime_ns(status); + if(record.mtime_ns == 0 && hash_file(path) == hash) { + record.size = status.getSize(); + record.mtime_ns = fs::mtime_ns(status); + } } fv_of[i] = fv; } diff --git a/src/server/service/query.cpp b/src/server/service/query.cpp index 51b220e63..bd8339afc 100644 --- a/src/server/service/query.cpp +++ b/src/server/service/query.cpp @@ -931,16 +931,16 @@ std::vector IndexQuery::search_symbols(llvm::String session.index.iterate_symbols( [&](index::SymbolHash hash, const index::SymbolIdentity& symbol, llvm::StringRef) { if(results.size() >= max_results) - return; + return false; if(seen.contains(hash)) - return; + return true; if(!is_indexable_kind(symbol.kind) || symbol.name.empty()) - return; + return true; if(!matches_query(symbol.name)) - return; + return true; auto def_loc = find_definition_location(hash); if(!def_loc) - return; + return true; protocol::SymbolInformation info; info.name = std::string(symbol.name); @@ -948,6 +948,7 @@ std::vector IndexQuery::search_symbols(llvm::String info.location = std::move(*def_loc); results.push_back(std::move(info)); seen.insert(hash); + return true; }); return true; }); diff --git a/src/server/service/query.h b/src/server/service/query.h index 000ad3ce1..190e364bb 100644 --- a/src/server/service/query.h +++ b/src/server/service/query.h @@ -246,7 +246,8 @@ class IndexQuery { /// Whether a session's overlay preamble entry may serve: the blob was /// built from this very file (identical preambles share one PCH, but /// macro USRs embed the source path) and the buffer still starts with - /// the blob's stored preamble text. + /// the exact preamble the blob was built from (compared by hash — the + /// text itself is not stored). bool serves_preamble(const Session& session, const index::TUIndex& state) const; /// Whether an overlay file entry may contribute results. Filters diff --git a/src/server/state/invalidator.cpp b/src/server/state/invalidator.cpp index 00ec49dfa..30f366505 100644 --- a/src/server/state/invalidator.cpp +++ b/src/server/state/invalidator.cpp @@ -194,7 +194,7 @@ DirtySet Invalidator::apply(llvm::ArrayRef events) { } auto shard_it = workspace.shards.find(event.path_id); bool shard_current = - shard_it != workspace.shards.end() && *disk == shard_it->second.content(); + shard_it != workspace.shards.end() && shard_it->second.matches_content(*disk); if(shard_current) { dirty.add_reindex_deps_only(event.path_id); } else { diff --git a/src/server/transport/master_server.cpp b/src/server/transport/master_server.cpp index aa822c340..418aa3faf 100644 --- a/src/server/transport/master_server.cpp +++ b/src/server/transport/master_server.cpp @@ -308,7 +308,7 @@ void MasterServer::on_agentic_query() { } auto shard_it = workspace.shards.find(path_id); bool shard_current = - shard_it != workspace.shards.end() && *disk == shard_it->second.content(); + shard_it != workspace.shards.end() && shard_it->second.matches_content(*disk); indexer.enqueue(path_id, shard_current ? ReindexReason::DepsOnly : ReindexReason::ContentChanged); } diff --git a/tests/unit/index/persisted_index_tests.cpp b/tests/unit/index/persisted_index_tests.cpp index d57b71b76..e97a69ad2 100644 --- a/tests/unit/index/persisted_index_tests.cpp +++ b/tests/unit/index/persisted_index_tests.cpp @@ -358,6 +358,57 @@ TEST_CASE(GlobalDuplicateSymbolRejected) { ASSERT_EQ(loaded.symbols.size(), std::size_t(2)); } +TEST_CASE(GlobalReservedKeysRejected) { + // The two DenseMap sentinel key values can never sit in the in-memory + // tables, so the writer can never emit them; a blob carrying one is + // corrupt, and inserting it would corrupt (or assert in) the loader's + // own containers. + clice::PathPool pool; + llvm::DenseMap pins; + index::ProjectIndex loaded; + + { + GlobalBlobMirror mirror; + mirror.format_version = index::index_format_version; + mirror.fv_ids = {0xffffffffu}; + mirror.fv_paths = {"/proj/a.h"}; + mirror.fv_hashes = {0x1}; + mirror.fv_sizes = {1}; + mirror.fv_mtimes = {1}; + auto bytes = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(bytes.has_value()); + ASSERT_FALSE(loaded.load_global(bytes_of(*bytes), pool, pins)); + } + { + GlobalBlobMirror mirror; + mirror.format_version = index::index_format_version; + mirror.sym_hashes = {~std::uint64_t(0)}; + mirror.sym_names = {"sym"}; + mirror.sym_kinds = {0}; + clice::Bitmap bits; + bits.add(3); + mirror.sym_bitmaps = {index::write_bitmap(bits)}; + mirror.sym_paths = { + {3, "/proj/ref.h"} + }; + auto bytes = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(bytes.has_value()); + ASSERT_FALSE(loaded.load_global(bytes_of(*bytes), pool, pins)); + } + { + GlobalBlobMirror mirror; + mirror.format_version = index::index_format_version; + mirror.sym_paths = { + {0xfffffffeu, "/proj/ref.h"} + }; + auto bytes = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(bytes.has_value()); + ASSERT_FALSE(loaded.load_global(bytes_of(*bytes), pool, pins)); + } + ASSERT_TRUE(loaded.file_versions.empty()); + ASSERT_TRUE(loaded.symbols.empty()); +} + TEST_CASE(UnknownFileVersionsDetected) { index::ProjectIndex project; auto known = project.intern_file_version(0, 0x1); diff --git a/tests/unit/index/preamble_index_tests.cpp b/tests/unit/index/preamble_index_tests.cpp index 81a59f197..97fc0693d 100644 --- a/tests/unit/index/preamble_index_tests.cpp +++ b/tests/unit/index/preamble_index_tests.cpp @@ -47,6 +47,7 @@ index::SymbolHash hash_of(llvm::StringRef name, hash = symbol_id; count += 1; } + return true; }); EXPECT_EQ(count, 1); return hash; @@ -258,14 +259,18 @@ TEST_CASE(RejectVersionMismatch) { } TEST_CASE(AcceptCurrentVersionBlob) { - // Positive control for RejectVersionMismatch: the same single-slot shape - // carrying the CURRENT version loads — slot 0 really is the version slot - // and the rejection comes from its value, not from the blob's shape. - struct VersionOnly { + // Positive control for RejectVersionMismatch: the same leading slots + // carrying the CURRENT version (plus the minimal valid path table, which + // verification demands) load — slot 0 really is the version slot and the + // rejection comes from its value, not from the blob's shape. + struct VersionAndPaths { std::uint32_t format_version = 0; + std::int64_t built_at = 0; + std::vector paths = {"/proj/main.cpp"}; }; - auto blob = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); + auto blob = + kota::codec::fbs::to_bytes(VersionAndPaths{.format_version = index::index_format_version}); ASSERT_TRUE(blob.has_value()); dir.touch("current.pch.idx", diff --git a/tests/unit/index/shard_tests.cpp b/tests/unit/index/shard_tests.cpp index 3d793044f..a6d94cf96 100644 --- a/tests/unit/index/shard_tests.cpp +++ b/tests/unit/index/shard_tests.cpp @@ -449,7 +449,9 @@ TEST_CASE(LocalSymbolNames) { [&](index::SymbolHash hash, const index::SymbolIdentity& symbol, llvm::StringRef) { if(symbol.name == "visible") { result = hash; + return false; } + return true; }); return result; }(); @@ -577,7 +579,9 @@ TEST_CASE(ContentHashMismatchRejected) { }; ASSERT_TRUE(make_shard(bytes_of()).loaded()); - blob.content = "aaåb"; + // Same byte length (ä is two UTF-8 bytes like å): only the hash + // differs, so the size check cannot be what rejects the blob. + blob.content = "aaåä"; ASSERT_FALSE(make_shard(bytes_of()).loaded()); } @@ -670,6 +674,66 @@ TEST_CASE(MisorderedRowsRejected) { ASSERT_FALSE(make_shard(bytes_of()).loaded()); } +TEST_CASE(DuplicateOccKeyRejected) { + // The merge two-way merges occurrence runs under the full (begin, end, + // sym) key and the writer combines equal keys, so a repeated or + // descending symbol under one range is non-canonical and would + // mis-merge silently instead of being rejected. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + fill_content(blob, "aaaaaaaaaaaaaaaa"); + blob.variants = {1}; + blob.sym_hashes = {111, 222}; + blob.sym_rel_offsets = {0, 0, 0}; + blob.occs.packed = {index::pack_range(0, 3), index::pack_range(0, 3)}; + blob.occ_syms8 = {0, 1}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + // Positive control: one range with ascending symbols loads. + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.occ_syms8 = {0, 0}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + blob.occ_syms8 = {1, 0}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(UnsortedRelationRowsRejected) { + // Rows of one relation group merge under the full (kind, begin, end, + // payload) key and the writer sorts and combines equal keys, so + // out-of-order or repeated rows are non-canonical and would mis-merge + // silently instead of being rejected. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + fill_content(blob, "aaaaaaaaaaaaaaaa"); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 2}; + blob.rel_kinds = {static_cast(RelationKind::Reference), + static_cast(RelationKind::Reference)}; + blob.rels.packed = {index::pack_range(0, 3), index::pack_range(4, 3)}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.rels.packed = {index::pack_range(4, 3), index::pack_range(0, 3)}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + blob.rels.packed = {index::pack_range(0, 3), index::pack_range(0, 3)}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + TEST_CASE(EscapeTableMismatchRejected) { // A sentinel length without its sparse entry decodes as begin + 255 // (end_of's fallback) and a stray entry is silently ignored: with diff --git a/tests/unit/index/tu_index_tests.cpp b/tests/unit/index/tu_index_tests.cpp index 2b8aa9a69..cb264ac7d 100644 --- a/tests/unit/index/tu_index_tests.cpp +++ b/tests/unit/index/tu_index_tests.cpp @@ -81,6 +81,7 @@ void decode_index(const std::string& envelope) { symbol.scope = identity.scope; symbol.reference_files = index::read_bitmap(bitmap.data(), bitmap.size()).value_or(Bitmap{}); + return true; }); } @@ -1316,25 +1317,29 @@ TEST_CASE(FromRejectsHostileInput) { } TEST_CASE(FromRejectsStaleFormatVersion) { - // Only the version slot is written: every other field reads back - // absent, which is structurally valid — the verdict must hinge on the - // value. Field order MUST mirror the envelope layout (tu_index.cpp): - // format_version is slot 0. - struct VersionOnly { + // Only the version slot and the path table are written: every other + // field reads back absent, which is structurally valid — the verdict + // must hinge on the version value. Field order MUST mirror the + // envelope layout (tu_index.cpp): format_version is slot 0. + struct VersionAndPaths { std::uint32_t format_version = 0; + std::int64_t built_at = 0; + std::vector paths = {"/proj/main.cpp"}; }; auto bytes_of = [](const std::vector& blob) { return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); }; - auto stale = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version + 1}); + auto stale = kota::codec::fbs::to_bytes( + VersionAndPaths{.format_version = index::index_format_version + 1}); ASSERT_TRUE(stale.has_value()); ASSERT_FALSE(index::TUIndex::from_bytes(bytes_of(*stale)).loaded()); // Positive control: the same shape carrying the current version loads, // so the rejection above comes from the value, not the blob's shape. - auto current = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); + auto current = + kota::codec::fbs::to_bytes(VersionAndPaths{.format_version = index::index_format_version}); ASSERT_TRUE(current.has_value()); ASSERT_TRUE(index::TUIndex::from_bytes(bytes_of(*current)).loaded()); } @@ -1395,6 +1400,41 @@ TEST_CASE(FromRejectsOutOfRangePathIds) { } } +TEST_CASE(FromRejectsEmptyPathTable) { + // The builder ends every path table with the interested file, and + // consumers address path_count() - 1 unchecked — an envelope with no + // paths at all is corrupt. + MirrorEnvelope hostile; + ASSERT_FALSE(index::TUIndex::from_bytes(mirror_bytes(hostile)).loaded()); +} + +TEST_CASE(FromRejectsUnsortedSections) { + // section_of binary-searches the section table by path id; a repeated + // or out-of-order id would attribute one file's rows to another. + + // Positive control: the ascending shape loads. + MirrorEnvelope honest; + honest.paths = {"/proj/a.h", "/proj/main.cpp"}; + honest.sections.push_back({.path_id = 0}); + honest.sections.push_back({.path_id = 1}); + ASSERT_TRUE(index::TUIndex::from_bytes(mirror_bytes(honest)).loaded()); + + { + MirrorEnvelope hostile; + hostile.paths = {"/proj/a.h", "/proj/main.cpp"}; + hostile.sections.push_back({.path_id = 1}); + hostile.sections.push_back({.path_id = 0}); + ASSERT_FALSE(index::TUIndex::from_bytes(mirror_bytes(hostile)).loaded()); + } + { + MirrorEnvelope hostile; + hostile.paths = {"/proj/a.h", "/proj/main.cpp"}; + hostile.sections.push_back({.path_id = 1}); + hostile.sections.push_back({.path_id = 1}); + ASSERT_FALSE(index::TUIndex::from_bytes(mirror_bytes(hostile)).loaded()); + } +} + TEST_CASE(AbsentPathHashesReadZero) { // The hash column may be shorter than the path table on a foreign // envelope (structurally valid: the field reads back empty); absent diff --git a/tests/unit/server/indexer_tests.cpp b/tests/unit/server/indexer_tests.cpp index 730c3f2ba..e7a78e2ad 100644 --- a/tests/unit/server/indexer_tests.cpp +++ b/tests/unit/server/indexer_tests.cpp @@ -127,6 +127,9 @@ IndexedTU index_file(TempDir& tmp, llvm::StringRef file, std::vector(id.scope); const auto* begin = reinterpret_cast(bitmap.data()); symbol.reference_files.assign(begin, begin + bitmap.size()); + return true; }); for(std::uint32_t i = 0; i < view.section_count(); i += 1) { auto blob = view.section_blob(i); diff --git a/tests/unit/server/invalidator_tests.cpp b/tests/unit/server/invalidator_tests.cpp index 73a017241..4459f4c10 100644 --- a/tests/unit/server/invalidator_tests.cpp +++ b/tests/unit/server/invalidator_tests.cpp @@ -8,6 +8,19 @@ namespace clice::testing { namespace { +/// A loaded shard whose rows were built from `content`, for the +/// disk-vs-shard freshness comparisons below. +index::Shard shard_of(llvm::StringRef content) { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard( + {}, + [](index::SymbolHash) -> std::optional { return std::nullopt; }, + content, + os); + return index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); +} + TEST_SUITE(Invalidator) { TEST_CASE(EmptyBatchNoEffects) { @@ -226,13 +239,13 @@ TEST_CASE(CloseCurrentShardDepsOnly) { Workspace workspace; SessionStore store; auto closed = workspace.path_pool.intern("/proj/a.cpp"); - workspace.shards[closed]; + workspace.shards[closed] = shard_of("int x;"); ContextResolver resolver(workspace); - // Disk matches the shard's stored content: a browse-and-close must not - // blank the file's rows for the reindex queue's latency. + // Disk matches the content the shard was built from: a browse-and-close + // must not blank the file's rows for the reindex queue's latency. Invalidator invalidator(workspace, store, resolver, [](llvm::StringRef) { - return std::optional{""}; + return std::optional{"int x;"}; }); auto dirty = invalidator.apply(FileEvent::buffer_closed(closed)); @@ -244,7 +257,7 @@ TEST_CASE(CloseDivergentShardContentChanged) { Workspace workspace; SessionStore store; auto closed = workspace.path_pool.intern("/proj/a.cpp"); - workspace.shards[closed]; + workspace.shards[closed] = shard_of("int x;"); ContextResolver resolver(workspace); // Disk holds edits the shard never saw (saved while open): the shard's diff --git a/tests/unit/server/query_overlay_tests.cpp b/tests/unit/server/query_overlay_tests.cpp index 2b019ab03..09c93b866 100644 --- a/tests/unit/server/query_overlay_tests.cpp +++ b/tests/unit/server/query_overlay_tests.cpp @@ -82,6 +82,7 @@ index::SymbolHash hash_of(llvm::StringRef name, hash = symbol_id; count += 1; } + return true; }); EXPECT_EQ(count, 1); return hash; From 8bee7368c1eb2bb41eaa6356b1d411aaa7e21e47 Mon Sep 17 00:00:00 2001 From: ykiko Date: Mon, 17 Aug 2026 06:38:41 +0800 Subject: [PATCH 09/10] fix(index): pch gate hashes blobs, payload partition, fv counter --- src/index/project_index.cpp | 15 ++++---- src/index/shard.cpp | 21 +++++++++++ src/index/tu_index.cpp | 7 ++++ src/index/tu_index.h | 13 ++++--- src/server/compiler/indexer.cpp | 34 +++++++++++++++--- src/server/service/query.cpp | 22 +++++++----- tests/unit/index/persisted_index_tests.cpp | 38 ++++++++++++++++++++ tests/unit/index/shard_tests.cpp | 42 ++++++++++++++++++++++ tests/unit/index/tu_index_tests.cpp | 34 ++++++++++++++++++ 9 files changed, 202 insertions(+), 24 deletions(-) diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index 40c4fad1d..14f81668b 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -346,9 +346,15 @@ bool ProjectIndex::load_global(this ProjectIndex& self, // Every id and hash below becomes a DenseMap or DenseSet key, first in // the duplicate checks here and then in the tables themselves — see - // reserved_key. + // reserved_key. The counter is one intern away from becoming a key + // itself, and the writer hands ids out from it, so ids must sit below + // it — a bound that (with the sentinels at the top of the id space) + // also keeps every id non-reserved. + if(reserved_key(blob.next_fv_id)) { + return false; + } for(auto id: blob.fv_ids) { - if(reserved_key(id)) { + if(id >= blob.next_fv_id) { return false; } } @@ -430,11 +436,6 @@ bool ProjectIndex::load_global(this ProjectIndex& self, auto id = blob.fv_ids[i]; self.file_versions[id] = {path_id, blob.fv_hashes[i], blob.fv_sizes[i], blob.fv_mtimes[i]}; self.fv_ids[{path_id, blob.fv_hashes[i]}] = id; - // Ids must stay unique forever; a blob whose counter lags its own - // table (corruption) must not hand out ids that alias stored ones. - if(id >= self.next_fv_id) { - self.next_fv_id = id + 1; - } } for(std::size_t k = 0; k < blob.manifest_fvs.size(); k += 1) { diff --git a/src/index/shard.cpp b/src/index/shard.cpp index 8e5346108..3a73b66d2 100644 --- a/src/index/shard.cpp +++ b/src/index/shard.cpp @@ -396,6 +396,27 @@ bool validate(BlobView root) { } } + // The writer splits payloads by kind — decl/def rows carry a + // definition range, every other payload names a target symbol — and + // the readers decode whichever sparse table holds the row without + // consulting its kind. A row in the wrong table would serve one + // payload's bit pattern as the other (a source range as a symbol + // hash, or vice versa), so enforce the partition, which also keeps + // the tables disjoint. + auto kind_of = [&](std::uint32_t row) { + return RelationKind(static_cast(rel_kinds[row])); + }; + for(auto row: rel_sym_rows) { + if(kind_of(row).isDeclOrDef()) { + return false; + } + } + for(auto row: rel_def_rows) { + if(!kind_of(row).isDeclOrDef()) { + return false; + } + } + // Rows of one relation group two-way merge under the full (kind, // begin, end, payload) key with equal-key rows combined at write time, // so the key must ascend strictly within each group — out-of-order or diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index 6456008c5..591c27fec 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -896,6 +896,13 @@ bool TUIndex::shards_verify() const { shards.resize(section_count()); } for(std::uint32_t i = 0; i < section_count(); i += 1) { + // Structural verification alone accepts flipped bits that still + // form a valid shard (an in-bounds range, another symbol id); + // only the byte hash catches those, so a persisted envelope must + // fail here and rebuild instead of serving corrupted rows. + if(llvm::xxh3_64bits(section_blob(i)) != section_hash(i)) { + return false; + } if(!shards[i].loaded()) { shards[i] = Shard::from_bytes(section_blob(i)); if(!shards[i].loaded()) { diff --git a/src/index/tu_index.h b/src/index/tu_index.h index c8a290f0d..4d3b9a463 100644 --- a/src/index/tu_index.h +++ b/src/index/tu_index.h @@ -61,8 +61,9 @@ class TUIndex { /// every path id the graph and sections carry; corrupt bytes load as /// an empty reader. Symbol reference-file ids are NOT validated — /// iterate_symbols hands them out raw and the consumer bounds them. - /// Section blob bytes are verified per section, by shard_of on first - /// use or by shards_verify in one pass. + /// Section blob bytes are verified per section: structurally by + /// shard_of on first use, or hash-checked and wrapped by + /// shards_verify in one pass. static TUIndex from_bytes(llvm::StringRef data); /// Adopt an owning buffer of envelope bytes (a mapped `.pch.idx`, a @@ -112,9 +113,11 @@ class TUIndex { /// section or its blob fails verification. const Shard& shard_of(std::uint32_t path_id) const; - /// Wrap and verify every section's blob in one pass — the load gate - /// for persisted envelopes, where a corrupt blob must read as "pair - /// missing" and rebuild instead of silently serving nothing. + /// Wrap every section's blob in one pass, checking its bytes against + /// the recorded section hash on top of structural verification — the + /// load gate for persisted envelopes, where a corrupt blob must read + /// as "pair missing" and rebuild instead of silently serving wrong + /// rows or nothing. bool shards_verify() const; /// Visit every symbol: hash, identity, and the raw serialized diff --git a/src/server/compiler/indexer.cpp b/src/server/compiler/indexer.cpp index efd562354..dcbda26b4 100644 --- a/src/server/compiler/indexer.cpp +++ b/src/server/compiler/indexer.cpp @@ -73,6 +73,24 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { // FileVersions these will reference are interned only at commit. llvm::SmallVector> section_contributions; llvm::SmallVector rebuilt_ids; + // TU-local path id -> content hash of the bytes each section's rows + // were built from (0 = no section). A section's shard already records + // that hash, so the FileVersion baseline below adopts it: pairing the + // rows with any other hash — the file behind a PCM whose disk moved + // on under a preserved or backdated mtime — would keep the baseline + // fresh while queries serve another generation's rows. + llvm::SmallVector consumed_hashes(view.path_count(), 0); + auto record_consumed = [&](std::uint32_t local_id, std::uint64_t content_hash) { + auto path_hash = view.path_hash(local_id); + if(path_hash != 0 && path_hash != content_hash) { + LOG_WARN("Reject merge for {}: rows for {} consumed other content than the compiler", + main_tu_path, + workspace.path_pool.resolve(file_ids_map[local_id])); + return false; + } + consumed_hashes[local_id] = content_hash; + return true; + }; for(std::uint32_t section = 0; section < view.section_count(); section += 1) { auto local_id = view.section_path(section); auto blob_hash = view.section_hash(section); @@ -85,6 +103,9 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { // hashes the blob bytes, which embed the content generation, so // one membership test is the whole check — no IO, no bytes read. if(shard && shard->loaded() && shard->has_variant(blob_hash)) { + if(!record_consumed(local_id, shard->content_hash())) { + return; + } section_contributions.emplace_back(local_id, blob_hash); hits += 1; continue; @@ -112,6 +133,9 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { workspace.path_pool.resolve(global_id)); return; } + if(!record_consumed(local_id, fresh.content_hash())) { + return; + } index::Shard replacement; if(shard && shard->loaded() && shard->content_hash() == fresh.content_hash()) { @@ -159,15 +183,17 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { fv_of.resize_for_overwrite(view.path_count()); for(std::uint32_t i = 0; i < view.path_count(); i += 1) { llvm::StringRef path = view.path(i); - auto hash = view.path_hash(i); + // The section's own record wins: for a hashless path (behind a + // PCM) it is the only hash naming the bytes the rows describe. + auto hash = consumed_hashes[i] != 0 ? consumed_hashes[i] : view.path_hash(i); fs::file_status status; bool stat_ok = !fs::status(path, status); bool untouched = stat_ok && fs::mtime_ns(status) <= baseline_before_ns; if(hash == 0 && untouched) { - // The worker had no buffer to hash (e.g. behind a PCM); the - // unchanged mtime proves the disk still holds the consumed - // bytes, so hash it here. + // The worker had no buffer to hash (e.g. behind a PCM) and no + // rows recorded one; the unchanged mtime proves the disk still + // holds the consumed bytes, so hash it here. hash = hash_file(path); } diff --git a/src/server/service/query.cpp b/src/server/service/query.cpp index bd8339afc..c70c5e0ad 100644 --- a/src/server/service/query.cpp +++ b/src/server/service/query.cpp @@ -709,13 +709,16 @@ std::optional IndexQuery::resolve_symbol(index::SymbolHash hash) { return SymbolInfo{hash, std::move(name), kind, def_loc->uri, def_loc->range}; } -/// The stored text of an indexed file, for preview slicing. ASCII blobs -/// do not store it: re-read the disk and serve it only while its hash +/// The stored text of an indexed file, for preview slicing. Non-ASCII +/// shards lend out the content they store; ASCII blobs do not store it: +/// re-read the disk into `storage` and serve it only while its hash /// still matches what the rows were built from — a moved-on file /// degrades to no preview rather than slicing mismatched text. -static std::optional indexed_text(llvm::StringRef path, const index::Shard& shard) { +static std::optional indexed_text(llvm::StringRef path, + const index::Shard& shard, + std::unique_ptr& storage) { if(!shard.ascii()) { - return shard.content().str(); + return shard.content(); } auto buffer = llvm::MemoryBuffer::getFile(path); if(!buffer) { @@ -725,7 +728,8 @@ static std::optional indexed_text(llvm::StringRef path, const index if(llvm::xxh3_64bits(text) != shard.content_hash()) { return std::nullopt; } - return text.str(); + storage = std::move(*buffer); + return text; } static std::string extract_line(llvm::StringRef content, std::uint32_t offset) { @@ -756,7 +760,8 @@ std::optional IndexQuery::get_definition_text(index: continue; auto& merged_index = shard_it->second; auto file_path = workspace.path_pool.resolve(file_id); - auto text = indexed_text(file_path, merged_index); + std::unique_ptr storage; + auto text = indexed_text(file_path, merged_index, storage); if(!text) continue; llvm::StringRef content = *text; @@ -802,8 +807,9 @@ std::vector IndexQuery::collect_references(ind auto file_path = workspace.path_pool.resolve(file_id); // A moved-on ASCII file yields no text: positions still map // through the line table, only the context line degrades. - auto text = indexed_text(file_path, merged_index); - llvm::StringRef content = text ? llvm::StringRef(*text) : llvm::StringRef(); + std::unique_ptr storage; + auto text = indexed_text(file_path, merged_index, storage); + llvm::StringRef content = text.value_or(llvm::StringRef()); IndexedLineMap map(content, merged_index.content_size(), merged_index.line_starts()); merged_index.lookup(hash, kind, [&](const index::Relation& r) { diff --git a/tests/unit/index/persisted_index_tests.cpp b/tests/unit/index/persisted_index_tests.cpp index e97a69ad2..c9cbdf182 100644 --- a/tests/unit/index/persisted_index_tests.cpp +++ b/tests/unit/index/persisted_index_tests.cpp @@ -238,6 +238,7 @@ TEST_CASE(GlobalBitmapPayloadGate) { // A malformed image after columns that decoded fine: the reject must // leave no partial state — file versions or symbols — that later // merges would build on and the next save persist. + mirror.next_fv_id = 8; mirror.fv_ids = {7}; mirror.fv_paths = {"/proj/partial.h"}; mirror.fv_hashes = {0x1}; @@ -297,6 +298,7 @@ TEST_CASE(GlobalDuplicateVersionsRejected) { // to the wrong file. GlobalBlobMirror mirror; mirror.format_version = index::index_format_version; + mirror.next_fv_id = 9; mirror.fv_ids = {7, 7}; mirror.fv_paths = {"/proj/a.h", "/proj/b.h"}; mirror.fv_hashes = {0x1, 0x2}; @@ -327,6 +329,42 @@ TEST_CASE(GlobalDuplicateVersionsRejected) { ASSERT_EQ(loaded.file_versions.size(), std::size_t(2)); } +TEST_CASE(GlobalBadCounterRejected) { + // Ids are handed out from the counter, so a legit writer's ids all sit + // below it; a lagging counter would alias stored ids on the next + // intern, and a sentinel one would insert a DenseMap reserved key. + GlobalBlobMirror mirror; + mirror.format_version = index::index_format_version; + mirror.next_fv_id = 8; + mirror.fv_ids = {7}; + mirror.fv_paths = {"/proj/a.h"}; + mirror.fv_hashes = {0x1}; + mirror.fv_sizes = {1}; + mirror.fv_mtimes = {1}; + + clice::PathPool pool; + llvm::DenseMap pins; + auto ahead = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(ahead.has_value()); + index::ProjectIndex loaded; + ASSERT_TRUE(loaded.load_global(bytes_of(*ahead), pool, pins)); + + mirror.next_fv_id = 7; + auto lagging = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(lagging.has_value()); + ASSERT_FALSE(loaded.load_global(bytes_of(*lagging), pool, pins)); + + mirror.next_fv_id = llvm::DenseMapInfo::getTombstoneKey(); + mirror.fv_ids = {}; + mirror.fv_paths = {}; + mirror.fv_hashes = {}; + mirror.fv_sizes = {}; + mirror.fv_mtimes = {}; + auto reserved = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(reserved.has_value()); + ASSERT_FALSE(loaded.load_global(bytes_of(*reserved), pool, pins)); +} + TEST_CASE(GlobalDuplicateSymbolRejected) { // Symbol hashes are map keys in the writer; a structurally valid blob // repeating one would silently replace the earlier entry's identity diff --git a/tests/unit/index/shard_tests.cpp b/tests/unit/index/shard_tests.cpp index a6d94cf96..3a3cdd309 100644 --- a/tests/unit/index/shard_tests.cpp +++ b/tests/unit/index/shard_tests.cpp @@ -950,6 +950,48 @@ TEST_CASE(StraySymbolIdRejected) { ASSERT_FALSE(make_shard(bytes_of()).loaded()); } +TEST_CASE(MismatchedPayloadTableRejected) { + // Readers decode whichever sparse table holds a row without consulting + // its kind: a decl/def row in the symbol table (or the reverse) would + // serve one payload's bit pattern as the other. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + fill_content(blob, "aaaaaaaaaaaaaaaa"); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 1}; + blob.rel_kinds = {static_cast(RelationKind::Definition)}; + blob.rels.packed = {index::pack_range(4, 3)}; + blob.rel_def_rows = {0}; + blob.rel_def_begins = {0}; + blob.rel_def_ends = {8}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.rel_def_rows = {}; + blob.rel_def_begins = {}; + blob.rel_def_ends = {}; + blob.rel_sym_rows = {0}; + blob.rel_sym8 = {0}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + blob.rel_kinds = {static_cast(RelationKind::Base)}; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.rel_sym_rows = {}; + blob.rel_sym8 = {}; + blob.rel_def_rows = {0}; + blob.rel_def_begins = {0}; + blob.rel_def_ends = {8}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + TEST_CASE(OwnerlessMaskRejected) { // A mask owning no stored variant serves its row unconditionally while // every variant is live (row_live's live.all fast path never consults diff --git a/tests/unit/index/tu_index_tests.cpp b/tests/unit/index/tu_index_tests.cpp index cb264ac7d..bec718d45 100644 --- a/tests/unit/index/tu_index_tests.cpp +++ b/tests/unit/index/tu_index_tests.cpp @@ -1293,6 +1293,40 @@ TEST_CASE(EnvelopeSections) { } } +TEST_CASE(CorruptSectionBytesRejected) { + // The persisted-load gate must reject any flipped section byte, in + // particular flips that still form a structurally valid shard (an + // opaque hash field, say) — only the byte hash catches those, and + // serving them would navigate by corrupted rows across restarts. + add_main("main.cpp", R"( + int value = 1; + int main() { return value; } + )"); + ASSERT_TRUE(compile()); + decode_index(index::build_tu_index(*unit)); + + auto& view = tu_index.view; + ASSERT_TRUE(view.section_count() > 0); + auto blob = view.section_blob(0); + auto offset = static_cast(blob.data() - view.bytes().data()); + + bool structurally_valid_flip = false; + std::string bytes = view.bytes().str(); + for(std::size_t i = 0; i < blob.size(); i += 1) { + std::string mutated = bytes; + mutated[offset + i] ^= 0x01; + auto reloaded = index::TUIndex::from_bytes(mutated); + if(!reloaded.loaded()) { + continue; + } + if(index::Shard::from_bytes(reloaded.section_blob(0)).loaded()) { + structurally_valid_flip = true; + } + ASSERT_FALSE(reloaded.shards_verify()); + } + ASSERT_TRUE(structurally_valid_flip); +} + TEST_CASE(FromRejectsHostileInput) { ASSERT_FALSE(index::TUIndex::from_bytes("not a flatbuffer at all").loaded()); From 5e9c5eeb2d3127b63e36d9d4b343abc1aca1f45b Mon Sep 17 00:00:00 2001 From: ykiko Date: Mon, 17 Aug 2026 07:20:50 +0800 Subject: [PATCH 10/10] fix(index): line table prefix sums, zero-copy variant membership --- src/index/shard.cpp | 27 ++++++++++++++++++++++++--- tests/unit/index/shard_tests.cpp | 8 ++++++++ 2 files changed, 32 insertions(+), 3 deletions(-) diff --git a/src/index/shard.cpp b/src/index/shard.cpp index 3a73b66d2..5a39975a1 100644 --- a/src/index/shard.cpp +++ b/src/index/shard.cpp @@ -218,7 +218,9 @@ bool validate(BlobView root) { // The line table reconstructs every line start by prefix sum, so it // must both pair with its escape table and add up to exactly the // content size — a drifted sum would shift every position mapping - // below the corruption. + // below the corruption. When the content is stored, the sum check is + // not enough: a table redistributing bytes between lines keeps the + // sum intact, so every start must match the one the content derives. auto line_lengths = to_array_ref(root[&ShardBlob::line_lengths]); auto long_line_rows = to_array_ref(root[&ShardBlob::long_line_rows]); auto long_line_lengths = to_array_ref(root[&ShardBlob::long_line_lengths]); @@ -227,9 +229,21 @@ bool validate(BlobView root) { !std::ranges::is_sorted(long_line_rows, std::less_equal{})) { return false; } + std::vector starts; + if(!content.empty()) { + starts = + kota::ipc::lsp::build_line_starts(std::string_view(content.data(), content.size())); + if(starts.size() != line_lengths.size()) { + return false; + } + } std::uint64_t line_sum = 0; std::size_t line_escape_cursor = 0; - for(auto length: line_lengths) { + for(std::size_t row = 0; row < line_lengths.size(); row += 1) { + if(!starts.empty() && starts[row] != line_sum) { + return false; + } + auto length = line_lengths[row]; if(length == length_escape) { auto value = long_line_lengths[line_escape_cursor]; line_escape_cursor += 1; @@ -596,7 +610,14 @@ std::vector Shard::variants() const { } bool Shard::has_variant(RowsHash hash) const { - return loaded() && llvm::is_contained(variants(), hash); + if(!buffer) { + return false; + } + auto stored = to_array_ref(root_of(*buffer)[&ShardBlob::variants]); + if(stored.empty()) { + return hash == blob_hash; + } + return llvm::is_contained(stored, hash); } void Shard::set_live(llvm::ArrayRef live_hashes) { diff --git a/tests/unit/index/shard_tests.cpp b/tests/unit/index/shard_tests.cpp index 3a3cdd309..a9e57e302 100644 --- a/tests/unit/index/shard_tests.cpp +++ b/tests/unit/index/shard_tests.cpp @@ -631,6 +631,14 @@ TEST_CASE(LineTableMismatchRejected) { blob.line_lengths = {}; ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + // Redistributing bytes between lines preserves the sum; with stored + // content every line start must match the one the content derives. + fill_content(blob, "aå\nbb\n"); + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.line_lengths = {3, 4}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); } TEST_CASE(MisorderedRowsRejected) {