diff --git a/CMakeLists.txt b/CMakeLists.txt index def7cf71a..63f462f61 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -234,7 +234,7 @@ if(CLICE_ENABLE_TEST) endif() if(CLICE_ENABLE_BENCHMARK) - foreach(benchmark scan_benchmark pipeline_benchmark pch_chain_benchmark) + foreach(benchmark scan_benchmark pipeline_benchmark pch_chain_benchmark index_stats_benchmark) add_executable(${benchmark} "${PROJECT_SOURCE_DIR}/benchmarks/${benchmark}.cpp" ) diff --git a/benchmarks/index_stats_benchmark.cpp b/benchmarks/index_stats_benchmark.cpp new file mode 100644 index 000000000..49e53b768 --- /dev/null +++ b/benchmarks/index_stats_benchmark.cpp @@ -0,0 +1,1235 @@ +/// In-process measurement probe for the on-disk index redesign. Compiles +/// every TU in a compilation database exactly like the background-index +/// worker does (full parse without PCH + TUIndex::build), serializes the +/// index the way production consumes it, then walks the resulting structures +/// and accumulates the distributions the new "merged blob" format needs to +/// pick its column tiers. +/// +/// Compiles run on a few worker threads, each accumulating into its own +/// Stats; the only shared state is a per-(path, variant-hash) registry +/// deciding which thread walks a distinct variant's rows — touched once per +/// FileIndex, never per row — so "per distinct variant" populations are not +/// double-counted across threads. Accumulators merge after join. +/// +/// This is measurement scratch code: it favours being obvious over being +/// fast or general. +/// +/// Usage: +/// index_stats_benchmark [OPTIONS] +/// +/// Example: +/// ./build/RelWithDebInfo/bin/index_stats_benchmark \ +/// build/RelWithDebInfo/compile_commands.json + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "command/command.h" +#include "command/toolchain.h" +#include "compile/compilation.h" +#include "index/tu_index.h" +#include "support/filesystem.h" +#include "support/format.h" +#include "support/logging.h" + +#include "kota/deco/deco.h" +#include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/DenseSet.h" +#include "llvm/ADT/SmallString.h" +#include "llvm/ADT/StringMap.h" +#include "llvm/ADT/StringSet.h" +#include "llvm/Support/FileSystem.h" +#include "llvm/Support/raw_ostream.h" +#include "llvm/Support/xxhash.h" + +using namespace clice; + +namespace { + +struct BenchmarkOptions { + DecoKV(names = {"--log-level"}; help = "Log level: trace, debug, info, warn, error, off"; + required = false;) + log_level = "off"; + + DecoKV(names = {"--filter"}; help = "Only process files whose path contains this substring"; + required = false;) + filter; + + DecoKV(names = {"--limit"}; help = "Process only the N largest files by source size (0 = all)"; + required = false;) + limit = 0; + + DecoKV(names = {"--threads"}; help = "Worker threads (1-4)"; required = false;) + threads = 4; + + DecoKV(names = {"--out-dir"}; help = "Directory for stats.json and REPORT.md"; + required = false;) + out_dir = "temp/index-stats"; + + DecoFlag(names = {"-h", "--help"}; help = "Show help message"; required = false;) + help; + + DecoInput(meta_var = "CDB"; help = "Path to compile_commands.json"; required = false;) + cdb_path; +}; + +constexpr std::array relation_names = {"Invalid", + "Declaration", + "Definition", + "Reference", + "WeakReference", + "Read", + "Write", + "Interface", + "Implementation", + "TypeDef", + "Base", + "Derived", + "Constructor", + "Destructor", + "Caller", + "Callee"}; + +/// Percentiles over a raw sample, nearest-rank (matches bench::percentile and +/// the TS harness). Finalize before querying. +struct Dist { + std::vector samples; + + void add(std::uint64_t v) { + samples.push_back(v); + } + + void merge(const Dist& other) { + samples.insert(samples.end(), other.samples.begin(), other.samples.end()); + } + + void finalize() { + std::ranges::sort(samples); + } + + std::uint64_t pct(double p) const { + if(samples.empty()) { + return 0; + } + auto index = static_cast(p * static_cast(samples.size() - 1)); + return samples[index]; + } + + std::uint64_t max() const { + return samples.empty() ? 0 : samples.back(); + } +}; + +/// A bucketed distribution for high-volume per-row samples (token lengths, +/// begin deltas), where holding one entry per row would be wasteful. Values +/// at or above `cap` fold into the last bucket but are counted in `over_cap` +/// and the exact `max_value` is kept. +template +struct Hist { + std::array buckets{}; + std::uint64_t total = 0; + std::uint64_t over_cap = 0; + std::uint64_t max_value = 0; + + void add(std::uint64_t v) { + buckets[std::min(v, Cap - 1)] += 1; + total += 1; + if(v >= Cap) { + over_cap += 1; + } + if(v > max_value) { + max_value = v; + } + } + + void merge(const Hist& other) { + for(std::size_t i = 0; i < Cap; i += 1) { + buckets[i] += other.buckets[i]; + } + total += other.total; + over_cap += other.over_cap; + max_value = std::max(max_value, other.max_value); + } + + std::uint64_t pct(double p) const { + if(total == 0) { + return 0; + } + auto rank = static_cast(p * static_cast(total - 1)); + std::uint64_t cum = 0; + for(std::size_t i = 0; i < Cap; i += 1) { + cum += buckets[i]; + if(cum > rank) { + return i; + } + } + return Cap - 1; + } + + std::uint64_t count_ge(std::uint64_t threshold) const { + std::uint64_t count = 0; + for(std::size_t i = threshold; i < Cap; i += 1) { + count += buckets[i]; + } + return count; + } +}; + +/// Per source-file aggregates, keyed by path string and folded across every +/// TU that touched the file. +struct PathAgg { + /// Distinct FileIndex rows hashes this thread claimed for this path; + /// claims are globally unique, so the merged union is the file's M. + std::set variants; + /// Number of TU contributions (one FileIndex per TU per path) — N. + std::uint32_t contributions = 0; + + bool size_probed = false; + bool size_known = false; + std::uint64_t file_size = 0; + + /// Distinct symbols (occurrence targets, relation keys and symbol + /// payloads, mirroring the shard writer's referenced set) over all + /// variants — S, the blob-local symbol id space. + llvm::DenseSet symbols; + + std::uint64_t max_range_end = 0; + + /// Row counts summed over distinct variants (what the merged blob stores). + std::uint64_t occ_m = 0, rel_m = 0, decldef_m = 0; + /// Row counts summed over all contributions (no dedup). + std::uint64_t occ_n = 0, rel_n = 0, decldef_n = 0; +}; + +/// Distinct outgoing include-list shapes seen for one (parent path, content +/// hash) key, plus how many TUs entered it. +struct DirAgg { + llvm::DenseSet shapes; + std::uint32_t contributions = 0; +}; + +/// Arbitrates which thread accumulates a distinct (path, variant) exactly +/// once. Touched once per FileIndex per TU — a coarse coordination point, +/// not the per-row hot path. +struct VariantRegistry { + std::mutex mutex; + std::set> seen; + + bool try_claim(llvm::StringRef path, std::uint64_t hash) { + std::scoped_lock lock(mutex); + return seen.insert({path.str(), hash}).second; + } +}; + +struct Stats { + VariantRegistry* registry = nullptr; + + std::unordered_map paths; + + /// Globally deduped symbol scope, first-seen wins. + llvm::DenseMap scope_map; + /// Scope counts summed over all TU symbol tables (per-TU population). + std::array scope_n{}; + + /// Relation kind counts over distinct variants. + std::array kind_hist{}; + + Dist occ_counts, rel_counts, s_variant, wire_sizes; + Hist<4096> length_hist; + Hist<8192> delta_hist; + + llvm::StringMap directive; + + std::uint64_t skipped_missing = 0; + std::uint64_t skipped_compile = 0; + std::uint64_t indexed = 0; + std::uint64_t had_diagnostics = 0; + + void add_file_index(index::FileIndex& fi, llvm::StringRef path) { + if(fi.occurrences.empty() && fi.relations.empty()) { + return; + } + + auto& agg = paths[path.str()]; + if(!agg.size_probed) { + agg.size_probed = true; + std::uint64_t size = 0; + if(!llvm::sys::fs::file_size(path, size)) { + agg.file_size = size; + agg.size_known = true; + } + } + agg.contributions += 1; + + auto hash = fi.rows_hash(); + bool inserted = registry->try_claim(path, hash); + if(inserted) { + agg.variants.insert(hash); + } + + std::uint64_t occ = fi.occurrences.size(); + std::uint64_t rel = 0, decldef = 0; + for(auto& [symbol, relations]: fi.relations) { + rel += relations.size(); + for(auto& r: relations) { + auto k = static_cast(r.kind); + if(r.kind == RelationKind::Declaration || r.kind == RelationKind::Definition) { + decldef += 1; + } + if(inserted && k < kind_hist.size()) { + kind_hist[k] += 1; + } + } + } + + agg.occ_n += occ; + agg.rel_n += rel; + agg.decldef_n += decldef; + + if(!inserted) { + return; + } + + agg.occ_m += occ; + agg.rel_m += rel; + agg.decldef_m += decldef; + occ_counts.add(occ); + rel_counts.add(rel); + + llvm::DenseSet variant_symbols; + for(auto& o: fi.occurrences) { + variant_symbols.insert(o.target); + } + for(auto& [symbol, relations]: fi.relations) { + variant_symbols.insert(symbol); + // Non-decl/def payloads are symbol hashes the shard's table + // must also cover (decl/def payloads are definition ranges). + for(auto& r: relations) { + if(r.target_symbol != 0 && !RelationKind(r.kind).isDeclOrDef()) { + variant_symbols.insert(r.target_symbol); + } + } + } + s_variant.add(variant_symbols.size()); + for(auto s: variant_symbols) { + agg.symbols.insert(s); + } + + bool first = true; + std::uint32_t prev_begin = 0; + for(auto& o: fi.occurrences) { + length_hist.add(o.range.length()); + if(o.range.end > agg.max_range_end) { + agg.max_range_end = o.range.end; + } + if(!first) { + delta_hist.add(o.range.begin >= prev_begin ? o.range.begin - prev_begin : 0); + } + prev_begin = o.range.begin; + first = false; + } + } + + void add_directives(index::TUIndex& tu) { + auto& graph = tu.graph; + if(graph.paths.empty()) { + return; + } + auto root_path_id = static_cast(graph.paths.size() - 1); + + // Outgoing edges keyed by parent location index (-1 = TU root). + llvm::DenseMap>> outgoing; + for(auto& loc: graph.locations) { + std::int64_t parent = loc.include == static_cast(-1) + ? -1 + : static_cast(loc.include); + outgoing[parent].emplace_back(loc.line, loc.path_id); + } + + llvm::StringMap> local; + for(auto& [parent, list]: outgoing) { + std::uint32_t parent_path_id = + parent < 0 ? root_path_id : graph.locations[parent].path_id; + if(parent_path_id >= graph.paths.size()) { + continue; + } + std::uint64_t parent_hash = + parent_path_id < graph.path_hashes.size() ? graph.path_hashes[parent_path_id] : 0; + + std::ranges::sort(list, [&](auto& a, auto& b) { + if(a.first != b.first) { + return a.first < b.first; + } + return graph.paths[a.second] < graph.paths[b.second]; + }); + + std::string shape; + for(auto& [line, child_path_id]: list) { + shape += std::format("{},{}\n", + line, + child_path_id < graph.paths.size() + ? llvm::StringRef(graph.paths[child_path_id]) + : llvm::StringRef()); + } + + auto key = std::format("{}#{:016x}", graph.paths[parent_path_id], parent_hash); + local[key].insert(llvm::xxh3_64bits(shape)); + } + + for(auto& entry: local) { + auto& agg = directive[entry.getKey()]; + for(auto shape: entry.getValue()) { + agg.shapes.insert(shape); + } + agg.contributions += 1; + } + } + + void add_tu(index::TUIndex& tu) { + indexed += 1; + + for(auto& [hash, symbol]: tu.symbols) { + auto scope = static_cast(symbol.scope); + if(scope < scope_n.size()) { + scope_n[scope] += 1; + } + scope_map.try_emplace(hash, symbol.scope); + } + + if(!tu.graph.paths.empty()) { + add_file_index(tu.main_file_index, tu.graph.paths.back()); + } + // Multiple FileIDs can share a path id (repeated header contexts); + // last-wins, like the wire sections. + llvm::DenseMap by_path; + for(auto& [fid, fi]: tu.file_indices) { + by_path[tu.graph.path_id(fid)] = &fi; + } + for(auto& [path_id, fi]: by_path) { + add_file_index(*fi, tu.graph.paths[path_id]); + } + + add_directives(tu); + } + + void merge(Stats& other) { + for(auto& [path, o]: other.paths) { + auto& a = paths[path]; + a.variants.merge(o.variants); + a.contributions += o.contributions; + if(!a.size_probed && o.size_probed) { + a.size_probed = true; + a.size_known = o.size_known; + a.file_size = o.file_size; + } + for(auto s: o.symbols) { + a.symbols.insert(s); + } + a.max_range_end = std::max(a.max_range_end, o.max_range_end); + a.occ_m += o.occ_m; + a.rel_m += o.rel_m; + a.decldef_m += o.decldef_m; + a.occ_n += o.occ_n; + a.rel_n += o.rel_n; + a.decldef_n += o.decldef_n; + } + + for(auto& [hash, scope]: other.scope_map) { + scope_map.try_emplace(hash, scope); + } + for(std::size_t i = 0; i < scope_n.size(); i += 1) { + scope_n[i] += other.scope_n[i]; + } + for(std::size_t i = 0; i < kind_hist.size(); i += 1) { + kind_hist[i] += other.kind_hist[i]; + } + + occ_counts.merge(other.occ_counts); + rel_counts.merge(other.rel_counts); + s_variant.merge(other.s_variant); + wire_sizes.merge(other.wire_sizes); + length_hist.merge(other.length_hist); + delta_hist.merge(other.delta_hist); + + for(auto& entry: other.directive) { + auto& agg = directive[entry.getKey()]; + for(auto shape: entry.getValue().shapes) { + agg.shapes.insert(shape); + } + agg.contributions += entry.getValue().contributions; + } + + skipped_missing += other.skipped_missing; + skipped_compile += other.skipped_compile; + indexed += other.indexed; + had_diagnostics += other.had_diagnostics; + } +}; + +/// Per-row mask cost of the shard writer's tier ladder (tier_of in +/// src/index/shard.cpp): none, u32, u64, then roaring — a 4-byte offset +/// column entry plus the serialized bitmap, whose floor is 18 bytes (the +/// 8-byte portable header plus one single-value array container). +std::uint32_t mask_bytes(std::uint32_t m) { + if(m <= 1) { + return 0; + } + if(m <= 32) { + return 4; + } + if(m <= 64) { + return 8; + } + return 4 + 18; +} + +CompilationParams make_params(const std::vector& arguments, + llvm::StringRef file, + llvm::StringRef content) { + CompilationParams params; + params.kind = CompilationKind::Indexing; + params.arguments = arguments; + params.add_remapped_file(file, content); + return params; +} + +/// Everything derived from a finalized Stats, computed once and shared by the +/// stdout summary, REPORT.md and stats.json. +struct Report { + // M / N + std::array m_buckets{}; // 1 / 2-8 / 9-16 / 17-32 / 33-64 / >64 + Dist m_dist, n_dist; + std::uint64_t sum_m = 0, sum_n = 0; + double variant_hit_rate = 0; + + // S + Dist s_dist; + std::uint64_t s_over_65535 = 0; + std::uint64_t path_count = 0; + + // occurrences / relations per variant + Dist occ_dist, rel_dist, s_variant_dist; + + // token length + std::uint64_t length_p50 = 0, length_p99 = 0, length_max = 0, length_total = 0; + std::uint64_t length_over_255 = 0, length_over_cap = 0; + + // begin deltas + std::uint64_t delta_p50 = 0, delta_p90 = 0, delta_over_cap = 0, delta_total = 0; + + // max range end + Dist max_end_dist; + std::uint64_t max_end_under_64k = 0; + + // u16 offset coverage + std::uint64_t files_under_64k = 0, files_sized = 0; + std::uint64_t occ_total = 0, occ_under_64k = 0; + + // relations + std::array kind_hist{}; + std::uint64_t rel_total = 0, range_payload = 0; + + // scope + std::array scope_dedup{}; + std::array scope_n{}; + + // directive sharing + std::uint64_t dir_keys = 0, dir_identical = 0, dir_shapes_sum = 0; + std::uint64_t dir_multi_keys = 0, dir_multi_identical = 0, dir_multi_shapes_sum = 0; + + // size accounting + std::uint64_t today_m = 0, new_m = 0, today_n = 0, new_n = 0; + std::uint64_t wire_total = 0; + Dist wire_dist; +}; + +Report build_report(Stats& stats) { + Report r; + r.path_count = stats.paths.size(); + + for(auto& [path, agg]: stats.paths) { + auto m = static_cast(agg.variants.size()); + r.m_dist.add(m); + r.n_dist.add(agg.contributions); + r.sum_m += m; + r.sum_n += agg.contributions; + + std::size_t bucket = m <= 1 ? 0 : m <= 8 ? 1 : m <= 16 ? 2 : m <= 32 ? 3 : m <= 64 ? 4 : 5; + r.m_buckets[bucket] += 1; + + auto s = static_cast(agg.symbols.size()); + r.s_dist.add(s); + if(s > 65535) { + r.s_over_65535 += 1; + } + + r.max_end_dist.add(agg.max_range_end); + if(agg.max_range_end < 65536) { + r.max_end_under_64k += 1; + } + + bool small = agg.size_known && agg.file_size < 65536; + if(agg.size_known) { + r.files_sized += 1; + if(small) { + r.files_under_64k += 1; + } + } + r.occ_total += agg.occ_m; + if(small) { + r.occ_under_64k += agg.occ_m; + } + + // The shard writer emits fixed u32 begin columns at every file size; + // the u16 coverage stats above measure what a size-tiered offset + // column would win, not what the implemented format spends. + std::uint32_t off = 4; + std::uint32_t sid = s <= 65535 ? 2 : 4; + std::uint32_t mask = mask_bytes(m); + + r.today_m += agg.occ_m * 16 + agg.rel_m * 24; + r.new_m += agg.occ_m * (off + 1 + sid + mask); + r.new_m += agg.rel_m * (1 + off + 1 + mask) + agg.decldef_m * 8 + + (agg.rel_m - agg.decldef_m) * sid; + + r.today_n += agg.occ_n * 16 + agg.rel_n * 24; + r.new_n += agg.occ_n * (off + 1 + sid + mask); + r.new_n += agg.rel_n * (1 + off + 1 + mask) + agg.decldef_n * 8 + + (agg.rel_n - agg.decldef_n) * sid; + } + r.variant_hit_rate = r.sum_n == 0 ? 0 : 1.0 - static_cast(r.sum_m) / r.sum_n; + + r.m_dist.finalize(); + r.n_dist.finalize(); + r.s_dist.finalize(); + r.max_end_dist.finalize(); + + stats.occ_counts.finalize(); + stats.rel_counts.finalize(); + stats.s_variant.finalize(); + stats.wire_sizes.finalize(); + r.occ_dist = stats.occ_counts; + r.rel_dist = stats.rel_counts; + r.s_variant_dist = stats.s_variant; + r.wire_dist = stats.wire_sizes; + + r.length_total = stats.length_hist.total; + r.length_p50 = stats.length_hist.pct(0.5); + r.length_p99 = stats.length_hist.pct(0.99); + r.length_max = stats.length_hist.max_value; + r.length_over_255 = stats.length_hist.count_ge(256); + r.length_over_cap = stats.length_hist.over_cap; + + r.delta_total = stats.delta_hist.total; + r.delta_p50 = stats.delta_hist.pct(0.5); + r.delta_p90 = stats.delta_hist.pct(0.9); + r.delta_over_cap = stats.delta_hist.over_cap; + + r.kind_hist = stats.kind_hist; + for(auto c: stats.kind_hist) { + r.rel_total += c; + } + r.range_payload = + stats.kind_hist[RelationKind::Declaration] + stats.kind_hist[RelationKind::Definition]; + + for(auto& [hash, scope]: stats.scope_map) { + auto s = static_cast(scope); + if(s < r.scope_dedup.size()) { + r.scope_dedup[s] += 1; + } + } + r.scope_n = stats.scope_n; + + for(auto& entry: stats.directive) { + auto shapes = entry.getValue().shapes.size(); + r.dir_keys += 1; + r.dir_shapes_sum += shapes; + if(shapes == 1) { + r.dir_identical += 1; + } + if(entry.getValue().contributions >= 2) { + r.dir_multi_keys += 1; + r.dir_multi_shapes_sum += shapes; + if(shapes == 1) { + r.dir_multi_identical += 1; + } + } + } + + for(auto w: stats.wire_sizes.samples) { + r.wire_total += w; + } + + return r; +} + +std::string format_report_md(const Stats& stats, const Report& r, llvm::StringRef cdb) { + auto frac = [](std::uint64_t num, std::uint64_t den) { + return den == 0 ? 0.0 : 100.0 * static_cast(num) / static_cast(den); + }; + std::string o; + auto line = [&](std::string s) { + o += s; + o += '\n'; + }; + + line("# Index format distribution report"); + line(""); + line(std::format("- CDB: `{}`", cdb)); + line( + std::format("- TUs indexed: {} (skipped: {} missing file, {} did not compile; {} indexed " + "with diagnostics)", + stats.indexed, + stats.skipped_missing, + stats.skipped_compile, + stats.had_diagnostics)); + line(std::format("- Distinct source files (paths): {}", r.path_count)); + line(""); + line( + "> Caveat: this CDB is single-config, so **M is a lower bound** — a " + "multi-config project spreads over more preprocessing variants per file."); + line(""); + + line("## 1. Variants per file (M) and TU count (N)"); + line(""); + line("| M bucket | files |"); + line("|---|---|"); + constexpr static std::array labels = + {"1", "2-8", "9-16", "17-32", "33-64", ">64"}; + for(std::size_t i = 0; i < 6; i += 1) { + line(std::format("| {} | {} |", labels[i], r.m_buckets[i])); + } + line(""); + line("| metric | p50 | p90 | p99 | max |"); + line("|---|---|---|---|---|"); + line(std::format("| M | {} | {} | {} | {} |", + r.m_dist.pct(0.5), + r.m_dist.pct(0.9), + r.m_dist.pct(0.99), + r.m_dist.max())); + line(std::format("| N | {} | {} | {} | {} |", + r.n_dist.pct(0.5), + r.n_dist.pct(0.9), + r.n_dist.pct(0.99), + r.n_dist.max())); + line(""); + line(std::format("- sum(M) = {}, sum(N) = {}", r.sum_m, r.sum_n)); + line(std::format("- variant hit rate (1 - sum(M)/sum(N)) = {:.2f}%", + 100.0 * r.variant_hit_rate)); + line(""); + + line("## 2. Distinct symbols per file (S)"); + line(""); + line("| population | p50 | p99 | max |"); + line("|---|---|---|---|"); + line(std::format("| per path (all variants) | {} | {} | {} |", + r.s_dist.pct(0.5), + r.s_dist.pct(0.99), + r.s_dist.max())); + line(std::format("| per single variant | {} | {} | {} |", + r.s_variant_dist.pct(0.5), + r.s_variant_dist.pct(0.99), + r.s_variant_dist.max())); + line(""); + line(std::format("- files with S > 65535: {} / {} ({:.3f}%) → **u16 symbol id {}**", + r.s_over_65535, + r.path_count, + frac(r.s_over_65535, r.path_count), + r.s_over_65535 == 0 ? "safe" : "NOT safe")); + line(""); + + line("## 3. Occurrences"); + line(""); + line("| metric | p50 | p90 | p99 | max |"); + line("|---|---|---|---|---|"); + line(std::format("| occurrences / variant | {} | {} | {} | {} |", + r.occ_dist.pct(0.5), + r.occ_dist.pct(0.9), + r.occ_dist.pct(0.99), + r.occ_dist.max())); + line(std::format("| token length | {} | - | {} | {} |", + r.length_p50, + r.length_p99, + r.length_max)); + line(std::format("| begin delta | {} | {} | - | - |", r.delta_p50, r.delta_p90)); + line(std::format("| max range.end / file | {} | {} | {} | {} |", + r.max_end_dist.pct(0.5), + r.max_end_dist.pct(0.9), + r.max_end_dist.pct(0.99), + r.max_end_dist.max())); + line(""); + line(std::format("- token length > 255: {} / {} ({:.3f}%) → **u8 length {}**", + r.length_over_255, + r.length_total, + frac(r.length_over_255, r.length_total), + r.length_over_255 == 0 ? "safe" : "NOT safe (some tokens exceed 255)")); + line(std::format("- token length >= 4096 (histogram overflow): {}", r.length_over_cap)); + line(std::format("- begin delta >= 8192 (histogram overflow): {}", r.delta_over_cap)); + line(std::format("- files with max range.end < 64KB: {} / {} ({:.2f}%)", + r.max_end_under_64k, + r.path_count, + frac(r.max_end_under_64k, r.path_count))); + line(""); + + line("## 4. u16 offset coverage"); + line(""); + line(std::format("- files < 64KB on disk: {} / {} sized ({:.2f}%)", + r.files_under_64k, + r.files_sized, + frac(r.files_under_64k, r.files_sized))); + line(std::format("- occurrences residing in files < 64KB: {} / {} ({:.2f}%)", + r.occ_under_64k, + r.occ_total, + frac(r.occ_under_64k, r.occ_total))); + line(""); + + line("## 5. Relations"); + line(""); + line("| kind | count |"); + line("|---|---|"); + for(std::size_t i = 0; i < 16; i += 1) { + if(r.kind_hist[i] > 0) { + line(std::format("| {} | {} |", relation_names[i], r.kind_hist[i])); + } + } + line(""); + line("| metric | p50 | p90 | p99 | max |"); + line("|---|---|---|---|---|"); + line(std::format("| relations / variant | {} | {} | {} | {} |", + r.rel_dist.pct(0.5), + r.rel_dist.pct(0.9), + r.rel_dist.pct(0.99), + r.rel_dist.max())); + line(""); + line( + "- range-payload kinds (carry a definition range in the target slot " + "instead of a symbol hash): **Declaration, Definition**"); + line(std::format("- range-payload relations: {} / {} ({:.2f}%)", + r.range_payload, + r.rel_total, + frac(r.range_payload, r.rel_total))); + line(""); + + line("## 6. Symbol scope split"); + line(""); + line("| scope | deduped by hash | per-TU sum |"); + line("|---|---|---|"); + constexpr static std::array scope_labels = {"External", + "TULocal", + "FileLocal"}; + for(std::size_t i = 0; i < 3; i += 1) { + line(std::format("| {} | {} | {} |", scope_labels[i], r.scope_dedup[i], r.scope_n[i])); + } + std::uint64_t scope_total = r.scope_dedup[0] + r.scope_dedup[1] + r.scope_dedup[2]; + line(""); + line(std::format("- deduped total {} → External {:.1f}%, TULocal {:.1f}%, FileLocal {:.1f}%", + scope_total, + frac(r.scope_dedup[0], scope_total), + frac(r.scope_dedup[1], scope_total), + frac(r.scope_dedup[2], scope_total))); + line(""); + + line("## 7. Include directive sharing"); + line(""); + line(std::format("- (parent path, content hash) keys: {}", r.dir_keys)); + line(std::format("- keys with one shape across all TUs: {} / {} ({:.2f}%)", + r.dir_identical, + r.dir_keys, + frac(r.dir_identical, r.dir_keys))); + line(std::format("- avg distinct shapes per key: {:.4f}", + r.dir_keys == 0 ? 0.0 : static_cast(r.dir_shapes_sum) / r.dir_keys)); + line(std::format("- keys entered by >= 2 TUs: {}", r.dir_multi_keys)); + line(std::format("- of those, identical in every TU: {} / {} ({:.2f}%); avg shapes {:.4f}", + r.dir_multi_identical, + r.dir_multi_keys, + frac(r.dir_multi_identical, r.dir_multi_keys), + r.dir_multi_keys == 0 + ? 0.0 + : static_cast(r.dir_multi_shapes_sum) / r.dir_multi_keys)); + line(""); + + line("## 8. Size accounting"); + line(""); + line("| population | today (occ*16 + rel*24) | new-format estimate | ratio |"); + line("|---|---|---|---|"); + line(std::format("| distinct variants (M, merged blob) | {} | {} | {:.3f} |", + r.today_m, + r.new_m, + r.today_m == 0 ? 0.0 : static_cast(r.new_m) / r.today_m)); + line(std::format("| all contributions (N) | {} | {} | {:.3f} |", + r.today_n, + r.new_n, + r.today_n == 0 ? 0.0 : static_cast(r.new_n) / r.today_n)); + line(""); + line(std::format("- total serialized TUIndex wire size (all TUs): {}", r.wire_total)); + line(std::format("- per-TU wire size: p50 {}, p90 {}, max {}", + r.wire_dist.pct(0.5), + r.wire_dist.pct(0.9), + r.wire_dist.max())); + line(""); + line( + "> New-format estimate assigns no cross-variant row-dedup credit, so " + "for M>1 files it is an upper bound."); + line(""); + + line("## Answers"); + line(""); + line(std::format("- (a) mask tiers given observed M (max {}): {}", + r.m_dist.max(), + r.m_dist.max() <= 1 + ? "every file is single-variant here → 0-bit masks suffice; the " + "u32/u64/roaring tier ladder is exercised only by multi-config projects" + : "M spreads beyond 1 → tiered masks pay off")); + line(std::format("- (b) u16 symbol id: {} ({} files exceed 65535 symbols)", + r.s_over_65535 == 0 ? "SAFE" : "UNSAFE", + r.s_over_65535)); + line(std::format("- (c) u8 token length: {} ({:.3f}% of occurrences exceed 255)", + r.length_over_255 == 0 ? "SAFE" : "needs escape/overflow path", + frac(r.length_over_255, r.length_total))); + line( + std::format("- (d) u16 offset coverage: {:.2f}% of sized files < 64KB, " + "{:.2f}% of occurrences in files < 64KB", + frac(r.files_under_64k, r.files_sized), + frac(r.occ_under_64k, r.occ_total))); + line(std::format("- (e) range-payload relation fraction: {:.2f}%", + frac(r.range_payload, r.rel_total))); + line( + std::format("- (f) directive sharing: {:.2f}% of all keys single-shape, " + "{:.2f}% of multi-TU keys single-shape", + frac(r.dir_identical, r.dir_keys), + frac(r.dir_multi_identical, r.dir_multi_keys))); + line( + std::format("- (g) size totals (merged M population): today {} bytes vs new-format " + "estimate {} bytes ({:.1f}% of today)", + r.today_m, + r.new_m, + r.today_m == 0 ? 0.0 : 100.0 * static_cast(r.new_m) / r.today_m)); + line(""); + line( + "Optional merged-blob replay (exact today shard sizes): **skipped** — it " + "needs the workspace path pool and per-shard disk content plumbing, out of " + "scope for this scratch probe."); + + return o; +} + +std::string format_stats_json(const Stats& stats, const Report& r, llvm::StringRef cdb) { + std::string o = "{\n"; + auto kv = [&](llvm::StringRef k, auto v, bool comma = true) { + o += std::format(" \"{}\": {}{}\n", k, v, comma ? "," : ""); + }; + + // The only string value in the output; a Windows path would otherwise + // break the JSON. + std::string escaped_cdb; + for(char c: cdb) { + if(c == '"' || c == '\\') { + escaped_cdb += '\\'; + } + escaped_cdb += c; + } + o += std::format(" \"cdb\": \"{}\",\n", escaped_cdb); + kv("tus_indexed", stats.indexed); + kv("skipped_missing", stats.skipped_missing); + kv("skipped_compile", stats.skipped_compile); + kv("indexed_with_diagnostics", stats.had_diagnostics); + kv("paths", r.path_count); + + o += " \"m_buckets\": {"; + constexpr static std::array labels = + {"1", "2_8", "9_16", "17_32", "33_64", "gt64"}; + for(std::size_t i = 0; i < 6; i += 1) { + o += std::format("\"{}\": {}{}", labels[i], r.m_buckets[i], i + 1 < 6 ? ", " : ""); + } + o += "},\n"; + + kv("m_p50", r.m_dist.pct(0.5)); + kv("m_p90", r.m_dist.pct(0.9)); + kv("m_p99", r.m_dist.pct(0.99)); + kv("m_max", r.m_dist.max()); + kv("n_p50", r.n_dist.pct(0.5)); + kv("n_p90", r.n_dist.pct(0.9)); + kv("n_p99", r.n_dist.pct(0.99)); + kv("n_max", r.n_dist.max()); + kv("sum_m", r.sum_m); + kv("sum_n", r.sum_n); + kv("variant_hit_rate", std::format("{:.6f}", r.variant_hit_rate)); + + kv("s_p50", r.s_dist.pct(0.5)); + kv("s_p99", r.s_dist.pct(0.99)); + kv("s_max", r.s_dist.max()); + kv("s_variant_p50", r.s_variant_dist.pct(0.5)); + kv("s_variant_p99", r.s_variant_dist.pct(0.99)); + kv("s_variant_max", r.s_variant_dist.max()); + kv("s_over_65535", r.s_over_65535); + + kv("occ_per_variant_p50", r.occ_dist.pct(0.5)); + kv("occ_per_variant_p99", r.occ_dist.pct(0.99)); + kv("occ_per_variant_max", r.occ_dist.max()); + kv("length_p50", r.length_p50); + kv("length_p99", r.length_p99); + kv("length_max", r.length_max); + kv("length_total", r.length_total); + kv("length_over_255", r.length_over_255); + kv("delta_p50", r.delta_p50); + kv("delta_p90", r.delta_p90); + kv("max_range_end_p99", r.max_end_dist.pct(0.99)); + kv("max_range_end_max", r.max_end_dist.max()); + kv("max_range_end_under_64k", r.max_end_under_64k); + + kv("files_under_64k", r.files_under_64k); + kv("files_sized", r.files_sized); + kv("occ_total", r.occ_total); + kv("occ_under_64k", r.occ_under_64k); + + o += " \"relation_kinds\": {"; + bool first = true; + for(std::size_t i = 0; i < 16; i += 1) { + if(r.kind_hist[i] > 0) { + o += std::format("{}\"{}\": {}", first ? "" : ", ", relation_names[i], r.kind_hist[i]); + first = false; + } + } + o += "},\n"; + kv("rel_total", r.rel_total); + kv("range_payload_relations", r.range_payload); + kv("rel_per_variant_p50", r.rel_dist.pct(0.5)); + kv("rel_per_variant_p99", r.rel_dist.pct(0.99)); + kv("rel_per_variant_max", r.rel_dist.max()); + + o += std::format( + " \"scope_dedup\": {{\"External\": {}, \"TULocal\": {}, \"FileLocal\": {}}},\n", + r.scope_dedup[0], + r.scope_dedup[1], + r.scope_dedup[2]); + o += std::format( + " \"scope_per_tu\": {{\"External\": {}, \"TULocal\": {}, \"FileLocal\": {}}},\n", + r.scope_n[0], + r.scope_n[1], + r.scope_n[2]); + + kv("dir_keys", r.dir_keys); + kv("dir_identical", r.dir_identical); + kv("dir_shapes_sum", r.dir_shapes_sum); + kv("dir_multi_keys", r.dir_multi_keys); + kv("dir_multi_identical", r.dir_multi_identical); + kv("dir_multi_shapes_sum", r.dir_multi_shapes_sum); + + kv("size_today_m", r.today_m); + kv("size_new_m", r.new_m); + kv("size_today_n", r.today_n); + kv("size_new_n", r.new_n); + kv("wire_total", r.wire_total); + kv("wire_p50", r.wire_dist.pct(0.5)); + kv("wire_p90", r.wire_dist.pct(0.9)); + kv("wire_max", r.wire_dist.max(), false); + + o += "}\n"; + return o; +} + +} // namespace + +int main(int argc, const char** argv) { + auto args = kota::deco::util::argvify(argc, argv); + auto result = kota::deco::cli::parse(args); + if(!result.has_value()) { + std::println(stderr, "Error: {}", result.error().message); + return 1; + } + auto& opts = result->options; + + if(opts.help.value_or(false) || !opts.cdb_path.has_value()) { + std::ostringstream oss; + kota::deco::cli::write_usage_for(oss, + "index_stats_benchmark [OPTIONS] "); + std::print("{}", oss.str()); + return opts.help.value_or(false) ? 0 : 1; + } + + clice::logging::options.level = spdlog::level::from_str(*opts.log_level); + clice::logging::stderr_logger("index_stats_benchmark", clice::logging::options); + + CompilationDatabase cdb; + Toolchain toolchain; + auto count = cdb.load(*opts.cdb_path); + if(!count) { + std::println(stderr, "Error: failed to load {}", *opts.cdb_path); + return 1; + } + std::println("CDB loaded: {} entries", *count); + + // Distinct source paths: lookup() below returns every command of a + // file at once, and the --limit selection sizes files, not entries. + std::vector files; + llvm::StringSet<> seen_files; + for(auto& entry: cdb.get_entries()) { + auto path = cdb.resolve_path(entry.file); + if(opts.filter.has_value() && !path.contains(*opts.filter)) { + continue; + } + if(!seen_files.insert(path).second) { + continue; + } + files.push_back(path); + } + + if(*opts.limit > 0 && files.size() > static_cast(*opts.limit)) { + std::vector> sized; + sized.reserve(files.size()); + for(auto path: files) { + std::uint64_t size = 0; + if(llvm::sys::fs::file_size(path, size)) { + size = 0; + } + sized.emplace_back(size, path); + } + std::ranges::stable_sort(sized, std::greater{}, [](auto& pair) { return pair.first; }); + files.clear(); + for(auto& [size, path]: sized | std::views::take(*opts.limit)) { + files.push_back(path); + } + } + + /// Command lookup and toolchain resolution stay on the main thread; + /// workers receive ready-to-use argv (interned, pointer-stable). + struct Job { + llvm::StringRef file; + std::vector argv; + }; + + // One job per distinct command line: literal CDB duplicates would + // recompile the same variant and inflate the N-side distributions, + // while distinct commands for one file (per-config -D/-I/language + // spreads) are exactly the population that creates preprocessing + // variants and must all be compiled. + std::vector jobs; + jobs.reserve(files.size()); + llvm::DenseSet seen_commands; + for(auto file: files) { + for(auto& command: cdb.lookup(file)) { + toolchain.resolve_or_warn(command); + auto argv = command.to_argv(); + std::string joined; + for(auto* arg: argv) { + joined += arg; + joined += '\0'; + } + if(seen_commands.insert(llvm::xxh3_64bits(joined)).second) { + jobs.push_back({file, std::move(argv)}); + } + } + } + + auto thread_count = std::clamp(*opts.threads, 1, 4); + std::println("Processing {} file(s) on {} thread(s)\n", jobs.size(), thread_count); + + VariantRegistry registry; + std::vector stats_by_thread(thread_count); + for(auto& stats: stats_by_thread) { + stats.registry = ®istry; + } + + std::atomic next{0}; + std::atomic done{0}; + auto worker = [&](Stats& stats) { + while(true) { + auto i = next.fetch_add(1); + if(i >= jobs.size()) { + break; + } + auto& job = jobs[i]; + + auto finish = [&](llvm::StringRef verdict) { + auto d = done.fetch_add(1) + 1; + if(!verdict.empty()) { + std::println(stderr, " [{}/{}] {} {}", d, jobs.size(), verdict, job.file); + } else if(d % 10 == 0) { + std::println(stderr, " [{}/{}] last {}", d, jobs.size(), job.file); + } + }; + + auto content = fs::read(job.file); + if(!content) { + stats.skipped_missing += 1; + finish("skip (unreadable)"); + continue; + } + + auto params = make_params(job.argv, job.file, *content); + auto unit = compile(params); + if(!unit.completed()) { + stats.skipped_compile += 1; + finish("skip (did not compile)"); + continue; + } + if(auto errors = collect_errors(unit); !errors.empty()) { + stats.had_diagnostics += 1; + } + + auto tu_index = index::TUIndex::build(unit); + + llvm::SmallString<0> buffer; + llvm::raw_svector_ostream os(buffer); + tu_index.serialize(os); + stats.wire_sizes.add(buffer.size()); + + stats.add_tu(tu_index); + finish(""); + } + }; + + std::vector threads; + for(auto& stats: stats_by_thread) { + threads.emplace_back(worker, std::ref(stats)); + } + for(auto& thread: threads) { + thread.join(); + } + + Stats stats = std::move(stats_by_thread[0]); + for(std::size_t i = 1; i < stats_by_thread.size(); i += 1) { + stats.merge(stats_by_thread[i]); + } + + std::println(stderr, ""); + std::println("Indexed {} TU(s); skipped {} missing, {} non-compiling; {} had diagnostics", + stats.indexed, + stats.skipped_missing, + stats.skipped_compile, + stats.had_diagnostics); + + auto report = build_report(stats); + auto md = format_report_md(stats, report, *opts.cdb_path); + auto json = format_stats_json(stats, report, *opts.cdb_path); + + if(auto error = llvm::sys::fs::create_directories(*opts.out_dir)) { + std::println(stderr, "Error: cannot create {}: {}", *opts.out_dir, error.message()); + return 1; + } + auto md_path = std::format("{}/REPORT.md", *opts.out_dir); + auto json_path = std::format("{}/stats.json", *opts.out_dir); + if(auto w = fs::write(md_path, md); !w) { + std::println(stderr, "Error: cannot write {}: {}", md_path, w.error().message()); + return 1; + } + if(auto w = fs::write(json_path, json); !w) { + std::println(stderr, "Error: cannot write {}: {}", json_path, w.error().message()); + return 1; + } + + std::print("\n{}", md); + std::println("Wrote {} and {}", md_path, json_path); + return 0; +} diff --git a/src/index/manifest.cpp b/src/index/manifest.cpp new file mode 100644 index 000000000..641b9800f --- /dev/null +++ b/src/index/manifest.cpp @@ -0,0 +1,169 @@ +#include "index/manifest.h" + +#include +#include +#include + +#include "index/serialization.h" + +namespace clice::index { + +namespace { + +/// The manifest's persisted form. Node and contribution payloads are +/// varint-packed by hand: a big TU enters thousands of files, and LEB128 +/// on the small ids and lines roughly halves the blob against fixed-width +/// columns. +struct ManifestBlob { + std::uint32_t format_version = 0; + + std::uint64_t global_gen = 0; + + std::uint64_t built_at = 0; + + std::uint32_t tu_fv = 0; + + std::uint32_t node_count = 0; + + std::uint32_t contribution_count = 0; + + /// node_count × (varint fv, varint parent + 1 with 0 = root, varint line) + std::vector nodes; + + /// contribution_count × (varint fv, 8-byte little-endian rows hash — + /// hashes are random bits, varint would only inflate them) + std::vector contributions; +}; + +void write_varint(std::vector& out, std::uint64_t value) { + while(value >= 0x80) { + out.push_back(static_cast(value) | 0x80); + value >>= 7; + } + out.push_back(static_cast(value)); +} + +bool read_varint(std::span data, std::size_t& pos, std::uint64_t& value) { + value = 0; + for(std::uint32_t shift = 0; shift < 64; shift += 7) { + if(pos >= data.size()) { + return false; + } + auto byte = data[pos]; + pos += 1; + // The tenth byte holds only value bit 63; greater payloads would + // shift out silently and decode to an unrelated small value. + if(shift == 63 && byte > 1) { + return false; + } + value |= static_cast(byte & 0x7f) << shift; + if((byte & 0x80) == 0) { + return true; + } + } + return false; +} + +} // namespace + +void serialize_manifest(const TUManifest& manifest, llvm::raw_ostream& os) { + ManifestBlob blob; + blob.format_version = index_format_version; + blob.global_gen = manifest.global_gen; + blob.built_at = manifest.built_at; + blob.tu_fv = manifest.tu_fv; + + blob.node_count = static_cast(manifest.nodes.size()); + blob.nodes.reserve(manifest.nodes.size() * 6); + for(auto& node: manifest.nodes) { + write_varint(blob.nodes, node.fv); + write_varint(blob.nodes, node.parent + 1); + write_varint(blob.nodes, node.line); + } + + blob.contribution_count = static_cast(manifest.contributions.size()); + blob.contributions.reserve(manifest.contributions.size() * 10); + for(auto& [fv, hash]: manifest.contributions) { + write_varint(blob.contributions, fv); + std::uint8_t bytes[8]; + std::memcpy(bytes, &hash, sizeof(hash)); + blob.contributions.insert(blob.contributions.end(), bytes, bytes + sizeof(bytes)); + } + + serialize_blob(blob, os); +} + +std::optional deserialize_manifest(llvm::StringRef data) { + ManifestBlob blob; + if(!deserialize_blob(data, blob) || blob.format_version != index_format_version) { + return std::nullopt; + } + + // The counts size reserves below and are untrusted; a node occupies at + // least 3 payload bytes (three varints) and a contribution at least 9 + // (varint + 8-byte hash), so a count beyond these bounds cannot be + // honest and must not reach an allocator. + if(blob.node_count > blob.nodes.size() / 3 || + blob.contribution_count > blob.contributions.size() / 9) { + return std::nullopt; + } + + TUManifest manifest; + manifest.global_gen = blob.global_gen; + manifest.built_at = blob.built_at; + manifest.tu_fv = blob.tu_fv; + + constexpr std::uint64_t id_max = std::numeric_limits::max(); + std::span nodes(blob.nodes); + std::size_t pos = 0; + manifest.nodes.reserve(blob.node_count); + for(std::uint32_t i = 0; i < blob.node_count; i += 1) { + std::uint64_t fv = 0; + std::uint64_t parent = 0; + std::uint64_t line = 0; + if(!read_varint(nodes, pos, fv) || !read_varint(nodes, pos, parent) || + !read_varint(nodes, pos, line)) { + return std::nullopt; + } + if(fv > id_max || line > id_max) { + return std::nullopt; + } + // Parents may follow their children (the include graph resolves + // parent chains after appending the child), so only bounds are + // checked; consumers walking parents must carry their own visited + // set. + if(parent > blob.node_count) { + return std::nullopt; + } + manifest.nodes.push_back({static_cast(fv), + static_cast(parent) - 1, + static_cast(line)}); + } + if(pos != nodes.size()) { + return std::nullopt; + } + + std::span contributions(blob.contributions); + pos = 0; + manifest.contributions.reserve(blob.contribution_count); + for(std::uint32_t i = 0; i < blob.contribution_count; i += 1) { + std::uint64_t fv = 0; + if(!read_varint(contributions, pos, fv) || fv > id_max) { + return std::nullopt; + } + if(pos + 8 > contributions.size()) { + return std::nullopt; + } + std::uint64_t hash = 0; + std::memcpy(&hash, contributions.data() + pos, sizeof(hash)); + pos += 8; + manifest.contributions.emplace_back(static_cast(fv), hash); + } + if(pos != contributions.size()) { + return std::nullopt; + } + + return manifest; +} + +} // namespace clice::index diff --git a/src/index/manifest.h b/src/index/manifest.h new file mode 100644 index 000000000..d84802734 --- /dev/null +++ b/src/index/manifest.h @@ -0,0 +1,71 @@ +#pragma once + +#include +#include +#include +#include + +#include "llvm/ADT/StringRef.h" +#include "llvm/Support/raw_ostream.h" + +namespace clice::index { + +/// One entered file of a TU's include tree: which file version was entered, +/// through which include directive (the parent node's file at `line`), and +/// where. Multiple entries of one file (headers without guards) are +/// distinct nodes. +struct ManifestNode { + /// FileVersion id (ProjectIndex::file_versions). + std::uint32_t fv = 0; + + /// Index of the including node, ~0 when the directive sits in the TU's + /// own file (the TU root is not itself a node). + std::uint32_t parent = ~0u; + + /// 1-based line of the include directive in the parent. + std::uint32_t line = 0; + + friend bool operator==(const ManifestNode&, const ManifestNode&) = default; +}; + +/// What one TU's indexing produced, replaced wholesale by its next reindex: +/// the include tree over file versions (which doubles as the TU's +/// dependency set for staleness) and the rows each file received, keyed by +/// content-identity so a re-merge can tell "already stored" from "new +/// variant" without touching any shard. +struct TUManifest { + /// ProjectIndex::global_generation stamped by the save that persisted + /// this manifest. The global blob pins the stamp it expects per TU and + /// the loader adopts a manifest only on an exact match: a manifest + /// that outran a lost global write would otherwise serve against a + /// symbol table that never learned its symbols, and a manifest whose + /// own write failed would pass off the previous reindex's dependency + /// set and rows as current. + std::uint64_t global_gen = 0; + + /// Milliseconds since epoch, sampled before the indexed build started. + std::uint64_t built_at = 0; + + /// The TU's own file version. + std::uint32_t tu_fv = 0; + + std::vector nodes; + + /// FileVersion -> rows hash for every file this TU contributed rows to, + /// the TU's own file included. Deduplicated: one entry per file version + /// even when the file was entered several times. + std::vector> contributions; + + friend bool operator==(const TUManifest&, const TUManifest&) = default; +}; + +/// Serialize a manifest (varint-packed nodes inside a small reflected +/// wrapper). +void serialize_manifest(const TUManifest& manifest, llvm::raw_ostream& os); + +/// Verify and decode a manifest blob; nullopt for corrupt, truncated or +/// old-format data. FileVersion ids are not resolved here — the loader +/// drops manifests referencing ids the global table does not know. +std::optional deserialize_manifest(llvm::StringRef data); + +} // namespace clice::index diff --git a/src/index/merged_index.cpp b/src/index/merged_index.cpp deleted file mode 100644 index 6fc7eddef..000000000 --- a/src/index/merged_index.cpp +++ /dev/null @@ -1,903 +0,0 @@ -#include "index/merged_index.h" - -#include -#include -#include -#include - -#include "compile/dep_file.h" -#include "index/path_pool.h" -#include "index/serialization.h" -#include "support/filesystem.h" - -#include "kota/ipc/lsp/position.h" -#include "llvm/ADT/DenseSet.h" -#include "llvm/ADT/STLExtras.h" -#include "llvm/Support/raw_os_ostream.h" -#include "llvm/Support/xxhash.h" - -namespace llvm { - -template -unsigned dense_hash(const Ts&... ts) { - return llvm::DenseMapInfo>::getHashValue(std::tuple{ts...}); -} - -template <> -struct DenseMapInfo { - using R = clice::LocalSourceRange; - using V = clice::index::Occurrence; - - inline static V getEmptyKey() { - return V(R(-1, 0), 0); - } - - inline static V getTombstoneKey() { - return V(R(-2, 0), 0); - } - - static auto getHashValue(const V& v) { - return dense_hash(v.range.begin, v.range.end, v.target); - } - - static bool isEqual(const V& lhs, const V& rhs) { - return lhs.range == rhs.range && lhs.target == rhs.target; - } -}; - -template <> -struct DenseMapInfo { - using R = clice::index::Relation; - - inline static R getEmptyKey() { - return R{ - .kind = clice::RelationKind::Invalid, - .range = clice::LocalSourceRange(-1, 0), - .target_symbol = 0, - }; - } - - inline static R getTombstoneKey() { - return R{ - .kind = clice::RelationKind::Invalid, - .range = clice::LocalSourceRange(-2, 0), - .target_symbol = 0, - }; - } - - /// Contextual doesn’t take part in hashing and equality. - static auto getHashValue(const R& relation) { - return dense_hash(static_cast(relation.kind), - relation.range.begin, - relation.range.end, - relation.target_symbol); - } - - static bool isEqual(const R& lhs, const R& rhs) { - return lhs.kind == rhs.kind && lhs.range == rhs.range && - lhs.target_symbol == rhs.target_symbol; - } -}; - -} // namespace llvm - -namespace clice::index { - -/// (path_id, content_hash) captured for one dependency at index-build time. -struct DepHash { - std::uint32_t path_id; - std::uint64_t content_hash; - - friend bool operator==(const DepHash&, const DepHash&) = default; -}; - -/// Stat fast path for one dependency, recorded at merge time only when the -/// file provably did not change since before the indexed build started. -/// The server's mutable, in-place-repairing form of the same fast path is -/// `DepState` (server/state/workspace.h). -struct DepStamp { - std::uint32_t path_id; - std::uint64_t size; - std::int64_t mtime_ns; - - friend bool operator==(const DepStamp&, const DepStamp&) = default; -}; - -namespace { - -/// Hash a file's content with the same scheme the server layer uses for its -/// dependency snapshots (`workspace::hash_file`). Returns 0 on read failure. -std::uint64_t hash_file(llvm::StringRef path) { - auto buffer = llvm::MemoryBuffer::getFile(path); - if(!buffer) { - return 0; - } - return llvm::xxh3_64bits((*buffer)->getBuffer()); -} - -/// Two-layer staleness test for a single dependency, mirroring the server's -/// `deps_changed`: Layer 1 trusts a stat EQUAL to the recorded stamp (no -/// file read) — equality, not a watermark, so backdated or preserved mtimes -/// cannot masquerade as fresh; Layer 2 re-hashes the disk against the -/// consumed-content hash and treats a match as a mere touch, not an edit. -bool dep_stale(llvm::StringRef path, - const std::optional& stamp, - std::optional stored_hash) { - fs::file_status status; - if(auto err = fs::status(path, status)) { - return true; - } - - if(stamp && stamp->size == status.getSize() && stamp->mtime_ns == fs::mtime_ns(status)) { - return false; - } - - // The stat moved (or no stamp was recorded): without a baseline hash we - // cannot prove the content unchanged, so fall back to the conservative - // rebuild. - if(!stored_hash) { - return true; - } - // A matching hash means the file was only touched, not edited. We do NOT - // refresh the stored stamp on a match: the baseline lives inside the - // immutable serialized shard, and updating it would mean re-serializing - // the whole shard just to skip a rebuild. So a touched-but-unchanged file - // is re-hashed on every check until a real edit forces a genuine reindex - // — a cheap single read, far cheaper than a needless full reindex. - return hash_file(path) != *stored_hash; -} - -} // namespace - -struct IncludeContext { - std::uint32_t include_id; - - std::uint32_t canonical_id; - - friend bool operator==(const IncludeContext&, const IncludeContext&) = default; -}; - -struct HeaderContext { - std::uint32_t version = 0; - - llvm::SmallVector includes; - - friend bool operator==(const HeaderContext&, const HeaderContext&) = default; -}; - -struct CompilationContext { - std::uint32_t version = 0; - - std::uint32_t canonical_id = 0; - - std::uint64_t build_at; - - std::vector include_locations; - - /// Consumed-content hash of each distinct dependency (first-seen order), - /// used by the Layer 2 staleness check to distinguish an edit from a - /// touch. Reported by the indexing worker from the compiler's buffers. - llvm::SmallVector dep_hashes; - - /// Stat fast paths for the dependencies that provably did not change - /// since before the indexed build started (sparse, like dep_hashes). - llvm::SmallVector dep_stamps; - - friend bool operator==(const CompilationContext&, const CompilationContext&) = default; -}; - -struct MergedIndex::Impl { - /// On-disk shard schema version (index_format_version), stamped by - /// serialize() and gated by load(); version-less blobs from older builds - /// read back as 0. - std::uint32_t format_version = 0; - - /// Shard-local path table: every path id stored in this shard indexes - /// into it, so shards are self-contained across sessions (runtime pool - /// ids never persist). - PathPool paths; - - /// The content of corresponding source file. - std::string content; - - /// Line start offsets for position mapping. - std::vector line_starts; - - /// If this file is included by other source file, then it has header contexts. - /// The key represents the source file id, value represents the context in the - /// source file. - llvm::SmallDenseMap header_contexts; - - /// If this file is compiled as source file, then it has compilation contexts. - /// The key represents the compilation command id. File with compilation content - /// could provide header contexts for other files. - llvm::SmallDenseMap compilation_contexts; - - /// We use the value of SHA256 to judge whether two indices are same. - /// The same indices will be given same canonical id. - llvm::StringMap canonical_cache; - - /// The max canonical id we have allocated. - std::uint32_t max_canonical_id = 0; - - /// The reference count of each canonical id. Derived state: rebuilt from - /// the context tables when a blob loads in memory. - KOTATSU_ANNOTATE(skip = true) - > canonical_ref_counts; - - /// The canonical id set of removed index. Never persisted: compact() - /// erases the masked rows for real before a shard reaches disk. - KOTATSU_ANNOTATE(skip = true) - removed; - - /// All merged symbol occurrences. - llvm::DenseMap occurrences; - - /// All merged symbol relations. - llvm::DenseMap> relations; - - /// Symbols local to this file (FileLocal) or TU (TULocal). - SymbolTable symbols; - - /// Sorted occurrences cache for fast lookup. - KOTATSU_ANNOTATE(skip = true) - > occurrences_cache; - - /// Drop one reference to a canonical index; the last reference masks its - /// occurrences and relations via the removed bitmap. A later re-merge of - /// identical content resurrects the id instead of re-adding rows. - void release_canonical(this Impl& self, std::uint32_t canonical_id) { - auto& ref_count = self.canonical_ref_counts[canonical_id]; - ref_count -= 1; - if(ref_count == 0) { - self.removed.add(canonical_id); - } - } - - void merge(this Impl& self, std::uint32_t path_id, FileIndex& index, auto&& add_context) { - auto hash = index.hash(); - auto hash_key = llvm::StringRef(reinterpret_cast(hash.data()), hash.size()); - auto [it, success] = self.canonical_cache.try_emplace(hash_key, self.max_canonical_id); - - auto canonical_id = it->second; - add_context(self, canonical_id); - - if(!success) { - self.canonical_ref_counts[canonical_id] += 1; - self.removed.remove(canonical_id); - return; - } - - for(auto& occurrence: index.occurrences) { - self.occurrences[occurrence].add(canonical_id); - } - - for(auto& [symbol_id, relations]: index.relations) { - auto& target = self.relations[symbol_id]; - for(auto& relation: relations) { - target[relation].add(canonical_id); - } - } - - self.canonical_ref_counts.emplace_back(1); - self.max_canonical_id += 1; - } - - /// Erase the rows masked by the removed bitmap for real. Queries are - /// unaffected (masked rows were already invisible), but a later re-merge - /// of identical content mints a fresh canonical instead of resurrecting - /// the id — the same behavior a save/load cycle produces. - void compact(this Impl& self) { - if(self.removed.isEmpty()) { - return; - } - - llvm::SmallVector dead_hashes; - for(const auto& entry: self.canonical_cache) { - if(self.removed.contains(entry.getValue())) { - dead_hashes.push_back(entry.getKey()); - } - } - for(auto hash: dead_hashes) { - self.canonical_cache.erase(hash); - } - - llvm::SmallVector dead_occurrences; - for(auto& [occurrence, bitmap]: self.occurrences) { - bitmap -= self.removed; - if(bitmap.isEmpty()) { - dead_occurrences.push_back(occurrence); - } - } - for(auto& occurrence: dead_occurrences) { - self.occurrences.erase(occurrence); - } - - llvm::SmallVector dead_symbols; - for(auto& [symbol, entries]: self.relations) { - llvm::SmallVector dead_relations; - for(auto& [relation, bitmap]: entries) { - bitmap -= self.removed; - if(bitmap.isEmpty()) { - dead_relations.push_back(relation); - } - } - for(auto& relation: dead_relations) { - entries.erase(relation); - } - if(entries.empty()) { - dead_symbols.push_back(symbol); - } - } - for(auto symbol: dead_symbols) { - self.relations.erase(symbol); - } - - self.removed = roaring::Roaring(); - self.occurrences_cache.clear(); - } - - friend bool operator==(const Impl&, const Impl&) = default; -}; - -namespace { - -using ShardView = kota::codec::fbs::table_view; - -/// The blob was fully verified at load(); per-query views skip that cost. -ShardView root_of(const llvm::MemoryBuffer& buffer) { - return ShardView::from_verified_bytes(blob_bytes(buffer.getBuffer())); -} - -} // namespace - -MergedIndex::MergedIndex(std::unique_ptr buffer, std::unique_ptr impl) : - buffer(std::move(buffer)), impl(std::move(impl)) {} - -/// Revision values come from one process-wide monotonic source, so a value -/// never repeats across shard objects: an entry erased and re-created at -/// the same key while a save's commit is in flight cannot alias the -/// revision that save snapshotted (a per-object counter restarting at the -/// same small numbers could). -static std::uint64_t next_revision() { - static std::atomic counter{0}; - return counter.fetch_add(1, std::memory_order_relaxed) + 1; -} - -MergedIndex::MergedIndex() = default; - -MergedIndex::MergedIndex(llvm::StringRef data) : - MergedIndex(llvm::MemoryBuffer::getMemBuffer(data, "", false), nullptr) {} - -MergedIndex::MergedIndex(MergedIndex&& other) = default; - -MergedIndex& MergedIndex::operator=(MergedIndex&& other) = default; - -MergedIndex::~MergedIndex() = default; - -void MergedIndex::load_in_memory(this Self& self) { - if(self.impl) { - return; - } - - self.impl = std::make_unique(); - if(!self.buffer) { - return; - } - - auto& index = *self.impl; - // The buffer's structure was verified at load(), but structural - // verification does not constrain field values: a canonical id at or - // past max_canonical_id would index canonical_ref_counts out of bounds - // below (and in every later release_canonical). A blob carrying one — - // like a blob that fails to decode outright — is dropped, so the shard - // reads as empty and the background indexer rebuilds it. - auto usable = [&] { - if(!deserialize_blob(self.buffer->getBuffer(), index)) { - return false; - } - auto in_range = [&](std::uint32_t canonical_id) { - return canonical_id < index.max_canonical_id; - }; - for(const auto& entry: index.canonical_cache) { - if(!in_range(entry.getValue())) { - return false; - } - } - for(auto& context: llvm::make_second_range(index.header_contexts)) { - for(auto& include: context.includes) { - if(!in_range(include.canonical_id)) { - return false; - } - } - } - for(auto& context: llvm::make_second_range(index.compilation_contexts)) { - if(!in_range(context.canonical_id)) { - return false; - } - } - return true; - }; - if(!usable()) { - self.impl = std::make_unique(); - self.buffer.reset(); - return; - } - - index.canonical_ref_counts.resize(index.max_canonical_id, 0); - - for(auto& context: llvm::make_second_range(index.header_contexts)) { - for(auto& include: context.includes) { - index.canonical_ref_counts[include.canonical_id] += 1; - } - } - for(auto& context: llvm::make_second_range(index.compilation_contexts)) { - index.canonical_ref_counts[context.canonical_id] += 1; - } - - self.buffer.reset(); -} - -MergedIndex MergedIndex::load(llvm::StringRef path) { - auto buffer = llvm::MemoryBuffer::getFile(path); - if(!buffer) { - return MergedIndex(); - } - - // A stale cache directory from an older build must never crash the - // server or be misread. from_bytes deep-verifies every offset, string, - // vector and table the views can reach; shards whose format version - // differs are discarded (version-less shards read back 0). A discarded - // shard is treated as "not on disk" and the background indexer rebuilds - // it. - auto root = ShardView::from_bytes(blob_bytes((*buffer)->getBuffer())); - if(!root.valid() || root[&Impl::format_version] != index_format_version) { - return MergedIndex(); - } - - return MergedIndex(std::move(*buffer), nullptr); -} - -void MergedIndex::serialize(this Self& self, llvm::raw_ostream& out) { - if(self.buffer) { - out.write(self.buffer->getBufferStart(), self.buffer->getBufferSize()); - return; - } - - if(!self.impl) { - return; - } - - // The serialized shard is served through buffer-only lookups that never - // consult the removed bitmap, so masked state must not reach disk at - // all: compact first, then reflect the impl directly onto the wire. - self.impl->compact(); - self.impl->format_version = index_format_version; - serialize_blob(*self.impl, out); -} - -void MergedIndex::lookup(this const Self& self, - std::uint32_t offset, - llvm::function_ref callback) { - if(self.impl) { - auto& index = *self.impl; - auto& occurrences = index.occurrences_cache; - if(occurrences.empty()) { - for(auto& [o, _]: index.occurrences) { - occurrences.emplace_back(o); - } - std::ranges::sort(occurrences, [](const Occurrence& lhs, const Occurrence& rhs) { - return std::tuple(lhs.range.begin, lhs.range.end, lhs.target) < - std::tuple(rhs.range.begin, rhs.range.end, rhs.target); - }); - } - - auto it = std::ranges::lower_bound(occurrences, offset, {}, [](index::Occurrence& o) { - return o.range.end; - }); - - while(it != occurrences.end()) { - if(it->range.contains(offset)) { - // Skip occurrences whose canonical_ids are all removed. - if(!index.removed.isEmpty()) { - auto bitmap_it = index.occurrences.find(*it); - if(bitmap_it != index.occurrences.end()) { - auto remaining = bitmap_it->second - index.removed; - if(remaining.isEmpty()) { - it++; - continue; - } - } - } - - if(!callback(*it)) { - break; - } - - it++; - continue; - } - - break; - } - } else if(self.buffer) { - auto occurrences = root_of(*self.buffer)[&Impl::occurrences]; - scan_occurrences_at( - occurrences.size(), - offset, - [&](std::size_t i) { return occurrences.at(i).get<0>(); }, - callback); - } -} - -void MergedIndex::lookup(this const Self& self, - SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback) { - if(self.impl) { - auto it = self.impl->relations.find(symbol); - if(it == self.impl->relations.end()) [[unlikely]] { - return; - } - - auto& relations = it->second; - for(auto& [relation, bitmap]: relations) { - if(RelationKind(relation.kind) & kind) { - // Skip relations whose canonical_ids are all removed. - if(!self.impl->removed.isEmpty()) { - auto remaining = bitmap - self.impl->removed; - if(remaining.isEmpty()) { - continue; - } - } - - if(!callback(relation)) { - break; - } - } - } - } else if(self.buffer) { - auto found = root_of(*self.buffer)[&Impl::relations].find(symbol); - if(!found) [[unlikely]] { - return; - } - - auto entries = found->get<1>(); - for(std::size_t i = 0; i < entries.size(); ++i) { - Relation relation = entries.at(i).get<0>(); - if(RelationKind(relation.kind) & kind) { - if(!callback(relation)) { - break; - } - } - } - } -} - -bool MergedIndex::need_update(this const Self& self) { - if(self.impl) { - if(self.impl->compilation_contexts.empty()) { - return true; - } - - auto& paths = self.impl->paths.paths; - - // Every context must be validated: shards normally hold one, but - // whichever a partial iteration skipped would keep serving stale - // rows behind a fresh verdict. - for(auto& entry: self.impl->compilation_contexts) { - auto& context = entry.getSecond(); - - llvm::DenseMap hashes; - for(auto& dep: context.dep_hashes) { - hashes.try_emplace(dep.path_id, dep.content_hash); - } - llvm::DenseMap stamps; - for(auto& stamp: context.dep_stamps) { - stamps.try_emplace(stamp.path_id, stamp); - } - - llvm::DenseSet deps; - for(auto& location: context.include_locations) { - if(!deps.insert(location.path_id).second) { - continue; - } - // A dep the table does not cover cannot be validated: rebuild. - if(location.path_id >= paths.size()) { - return true; - } - auto stamp_it = stamps.find(location.path_id); - auto it = hashes.find(location.path_id); - if(dep_stale(paths[location.path_id], - stamp_it != stamps.end() ? std::optional(stamp_it->second) - : std::nullopt, - it != hashes.end() ? std::optional(it->second) : std::nullopt)) { - return true; - } - } - } - - return false; - } else if(self.buffer) { - auto root = root_of(*self.buffer); - auto contexts = root[&Impl::compilation_contexts]; - if(contexts.empty()) { - return true; - } - - auto paths = root[&Impl::paths]; - - for(std::size_t c = 0; c < contexts.size(); ++c) { - auto context = contexts.at(c).get<1>(); - - llvm::DenseMap hashes; - auto dep_hashes = context[&CompilationContext::dep_hashes]; - for(std::size_t i = 0; i < dep_hashes.size(); ++i) { - DepHash dep = dep_hashes[i]; - hashes.try_emplace(dep.path_id, dep.content_hash); - } - llvm::DenseMap stamps; - auto dep_stamps = context[&CompilationContext::dep_stamps]; - for(std::size_t i = 0; i < dep_stamps.size(); ++i) { - DepStamp stamp = dep_stamps[i]; - stamps.try_emplace(stamp.path_id, stamp); - } - - llvm::DenseSet deps; - auto locations = context[&CompilationContext::include_locations]; - for(std::size_t i = 0; i < locations.size(); ++i) { - IncludeLocation location = locations[i]; - if(!deps.insert(location.path_id).second) { - continue; - } - // A dep the table does not cover cannot be validated: rebuild. - if(location.path_id >= paths.size()) { - return true; - } - auto stamp_it = stamps.find(location.path_id); - auto it = hashes.find(location.path_id); - if(dep_stale(to_ref(paths[location.path_id]), - stamp_it != stamps.end() ? std::optional(stamp_it->second) - : std::nullopt, - it != hashes.end() ? std::optional(it->second) : std::nullopt)) { - return true; - } - } - } - - return false; - } - - return true; -} - -bool MergedIndex::has_contribution(this const Self& self, llvm::StringRef context_path) { - // Match the path table's normalization so Windows separators compare. - llvm::SmallString<256> normalized; - if(context_path.contains('\\')) { - normalized = context_path; - std::replace(normalized.begin(), normalized.end(), '\\', '/'); - context_path = normalized; - } - - if(self.impl) { - auto it = self.impl->paths.find(context_path); - if(it == self.impl->paths.cache.end()) { - return false; - } - return self.impl->header_contexts.contains(it->second) || - self.impl->compilation_contexts.contains(it->second); - } - - if(self.buffer) { - auto root = root_of(*self.buffer); - auto paths = root[&Impl::paths]; - std::optional local; - for(std::uint32_t i = 0; i < paths.size(); ++i) { - if(to_ref(paths[i]) == context_path) { - local = i; - break; - } - } - if(!local) { - return false; - } - return root[&Impl::header_contexts].contains(*local) || - root[&Impl::compilation_contexts].contains(*local); - } - - return false; -} - -void MergedIndex::remove(this Self& self, llvm::StringRef context_path) { - self.rev = next_revision(); - self.load_in_memory(); - auto& index = *self.impl; - - auto path_it = index.paths.find(context_path); - if(path_it == index.paths.cache.end()) { - return; - } - auto path_id = path_it->second; - - // Handle header context removal. - auto hc_it = index.header_contexts.find(path_id); - if(hc_it != index.header_contexts.end()) { - for(auto& [_, canonical_id]: hc_it->second.includes) { - index.release_canonical(canonical_id); - } - index.header_contexts.erase(hc_it); - } - - // Handle compilation context removal. - auto cc_it = index.compilation_contexts.find(path_id); - if(cc_it != index.compilation_contexts.end()) { - index.release_canonical(cc_it->second.canonical_id); - index.compilation_contexts.erase(cc_it); - } - - // Invalidate cached occurrences. - index.occurrences_cache.clear(); -} - -bool MergedIndex::find_symbol(this const Self& self, - SymbolHash hash, - std::string& name, - SymbolKind& kind) { - if(self.impl) { - auto it = self.impl->symbols.find(hash); - if(it != self.impl->symbols.end()) { - name = it->second.name; - kind = it->second.kind; - return true; - } - } else if(self.buffer) { - auto found = root_of(*self.buffer)[&Impl::symbols].find(hash); - if(found) { - auto symbol = found->get<1>(); - name = std::string(symbol[&Symbol::name]); - kind = SymbolKind(symbol[&Symbol::kind]); - return true; - } - } - return false; -} - -void MergedIndex::merge_symbols(this Self& self, const SymbolTable& symbols) { - self.rev = next_revision(); - self.load_in_memory(); - for(auto& [hash, symbol]: symbols) { - auto [it, inserted] = self.impl->symbols.try_emplace(hash); - if(inserted) { - it->second.name = symbol.name; - it->second.kind = symbol.kind; - it->second.scope = symbol.scope; - } - } -} - -void MergedIndex::merge(this Self& self, - llvm::StringRef tu_path, - std::chrono::milliseconds build_at, - llvm::ArrayRef deps, - FileIndex& index, - llvm::StringRef content) { - self.rev = next_revision(); - self.load_in_memory(); - self.impl->content = content.str(); - self.impl->line_starts = kota::ipc::lsp::build_line_starts(self.impl->content); - - // Intern the dependencies into the shard's own path table. The staleness - // baseline per distinct dep is two-part: the consumed-content hash the - // worker computed from the compiler's own buffers (describing exactly - // what the rows were built from), and a stat fast path recorded only for - // files that provably did not change since before the build started — - // for the rest the stat could describe content the rows were never built - // from, so they re-earn their fast path through a hash check instead. - std::vector include_locations; - llvm::SmallVector dep_hashes; - llvm::SmallVector dep_stamps; - llvm::DenseSet seen; - include_locations.reserve(deps.size()); - auto baseline_before_ns = fs::stat_baseline_before_ns(build_at.count()); - for(auto& dep: deps) { - auto local_id = self.impl->paths.path_id(dep.path); - include_locations.push_back({local_id, dep.line, dep.include_id}); - if(!seen.insert(local_id).second) { - continue; - } - - fs::file_status status; - bool stat_ok = !fs::status(dep.path, status); - bool untouched = stat_ok && fs::mtime_ns(status) <= baseline_before_ns; - - auto hash = dep.content_hash; - if(hash == 0 && untouched) { - // The worker had no buffer to hash (e.g. behind a PCH); the - // unchanged mtime proves the disk still holds the consumed - // bytes, so hash it here. - hash = hash_file(dep.path); - } - if(hash != 0) { - dep_hashes.emplace_back(local_id, hash); - } - if(untouched) { - dep_stamps.push_back({local_id, status.getSize(), fs::mtime_ns(status)}); - } - } - - auto path_id = self.impl->paths.path_id(tu_path); - self.impl->merge(path_id, index, [&](Impl& self, std::uint32_t canonical_id) { - // A reindex of the same TU replaces its previous contribution: - // without the release, the old canonical's occurrences and relations - // stay live and queries serve pre-edit state alongside the new one. - auto [it, inserted] = self.compilation_contexts.try_emplace(path_id); - if(!inserted) { - self.release_canonical(it->second.canonical_id); - } - auto& context = it->second; - context.canonical_id = canonical_id; - context.build_at = build_at.count(); - context.include_locations = std::move(include_locations); - context.dep_hashes = std::move(dep_hashes); - context.dep_stamps = std::move(dep_stamps); - }); - self.impl->occurrences_cache.clear(); -} - -void MergedIndex::merge(this Self& self, - llvm::StringRef tu_path, - std::uint32_t include_id, - FileIndex& index, - llvm::StringRef content) { - self.rev = next_revision(); - self.load_in_memory(); - auto path_id = self.impl->paths.path_id(tu_path); - // The stored content is the position-mapping truth for this file; a - // reindex after an edit must refresh it, not just fill it once. - if(!content.empty() && self.impl->content != content) { - self.impl->content = content.str(); - self.impl->line_starts = kota::ipc::lsp::build_line_starts(self.impl->content); - } - self.impl->merge(path_id, index, [&](Impl& self, std::uint32_t canonical_id) { - // Keyed by the including TU: a reindex of that TU replaces its - // previous contribution to this file wholesale, while contributions - // from other TUs stay untouched. - auto [it, inserted] = self.header_contexts.try_emplace(path_id); - if(!inserted) { - for(auto& [_, old_canonical]: it->second.includes) { - self.release_canonical(old_canonical); - } - it->second.includes.clear(); - } - it->second.includes.emplace_back(include_id, canonical_id); - }); - self.impl->occurrences_cache.clear(); -} - -llvm::StringRef MergedIndex::content(this const Self& self) { - if(self.impl) { - return self.impl->content; - } else if(self.buffer) { - return to_ref(root_of(*self.buffer)[&Impl::content]); - } - return {}; -} - -std::span MergedIndex::line_starts(this const Self& self) { - if(self.impl) { - return self.impl->line_starts; - } else if(self.buffer) { - auto starts = to_array_ref(root_of(*self.buffer)[&Impl::line_starts]); - return {starts.data(), starts.size()}; - } - return {}; -} - -bool operator==(MergedIndex& lhs, MergedIndex& rhs) { - lhs.load_in_memory(); - rhs.load_in_memory(); - return *lhs.impl == *rhs.impl; -} - -} // namespace clice::index diff --git a/src/index/merged_index.h b/src/index/merged_index.h deleted file mode 100644 index 69d09d00f..000000000 --- a/src/index/merged_index.h +++ /dev/null @@ -1,155 +0,0 @@ -#pragma once - -#include -#include -#include -#include -#include - -#include "index/tu_index.h" - -#include "llvm/Support/Allocator.h" -#include "llvm/Support/MemoryBuffer.h" - -namespace clice::index { - -/// A dependency of a compilation context: where it was included and the -/// file's path, interned into the shard's own path table on merge, plus the -/// hash of the bytes the indexing compile consumed for it (0 = unavailable; -/// the staleness baseline then stays conservative for this file). -struct DepLocation { - llvm::StringRef path; - std::uint32_t line = 0; - std::uint32_t include_id = 0; - std::uint64_t content_hash = 0; -}; - -class MergedIndex { -public: - /// The in-memory shard state, defined in merged_index.cpp. Its reflected - /// layout doubles as the persisted shard schema: serialization reflects - /// an Impl directly and the buffer-backed query paths read the blob - /// through a zero-copy view of the same layout. - struct Impl; - -private: - using Self = MergedIndex; - - MergedIndex(std::unique_ptr buffer, std::unique_ptr impl); - - void load_in_memory(this Self& self); - -public: - MergedIndex(); - - MergedIndex(llvm::StringRef data); - - MergedIndex(const MergedIndex&) = delete; - - MergedIndex(MergedIndex&& other); - - MergedIndex& operator=(const MergedIndex&) = delete; - - MergedIndex& operator=(MergedIndex&& other); - - ~MergedIndex(); - - /// Load merged index from disk - static MergedIndex load(llvm::StringRef path); - - /// Serialize it to binary format. Compacts rows masked by removals in - /// place first, so the serialized blob is a direct reflection of Impl. - void serialize(this Self& self, llvm::raw_ostream& out); - - /// Lookup the occurrence in corresponding offset. - void lookup(this const Self& self, - std::uint32_t offset, - llvm::function_ref callback); - - /// Lookup the relations of given symbol. - void lookup(this const Self& self, - SymbolHash symbol, - RelationKind kind, - llvm::function_ref callback); - - /// Whether this index needs rebuilding. Dependency paths come from the - /// shard's own path table; shards are fully self-contained. - bool need_update(this const Self& self); - - bool need_rewrite() { - return impl != nullptr; - } - - /// Mutation stamp, reassigned by every merge/remove from a process-wide - /// monotonic source (values never repeat across objects, so an erase + - /// re-create at the same key cannot alias an older snapshot). 0 = not - /// mutated since construction. save() snapshots it before serializing - /// and flips the shard back to its committed blob only if no mutation - /// landed across the commit await — otherwise the flip would silently - /// drop the newer contribution. - std::uint64_t revision() const { - return rev; - } - - /// Whether this index holds any data (a rejected or missing blob loads - /// as an empty index). - bool loaded() const { - return buffer != nullptr || impl != nullptr; - } - - /// Remove the contribution keyed by `context_path` (a TU for header - /// shards, the file itself for compilation shards). - void remove(this Self& self, llvm::StringRef context_path); - - /// Whether this shard holds a contribution keyed by `context_path`. - /// Cheap on serialized shards: scans the small context tables without - /// deserializing the shard. - bool has_contribution(this const Self& self, llvm::StringRef context_path); - - /// Get the stored source content for position mapping. - llvm::StringRef content(this const Self& self); - - /// Get line starts for position mapping. - std::span line_starts(this const Self& self); - - /// Look up a symbol in this shard's local symbol table. - bool find_symbol(this const Self& self, SymbolHash hash, std::string& name, SymbolKind& kind); - - /// Add symbols to this shard's local symbol table (idempotent by hash). - void merge_symbols(this Self& self, const SymbolTable& symbols); - - /// Merge the index with given compilation context, keyed by `tu_path`. - /// Dependency paths are interned into the shard's own table and a content - /// hash is captured per distinct dependency for the staleness check. - void merge(this Self& self, - llvm::StringRef tu_path, - std::chrono::milliseconds build_at, - llvm::ArrayRef deps, - FileIndex& index, - llvm::StringRef content); - - /// Merge the index with given header context. @param tu_path is the - /// including TU: a later merge with the same TU replaces that TU's - /// previous contribution, other TUs' contributions are untouched. - void merge(this Self& self, - llvm::StringRef tu_path, - std::uint32_t include_id, - FileIndex& index, - llvm::StringRef content); - - friend bool operator==(MergedIndex& lhs, MergedIndex& rhs); - -private: - /// The binary serialization data of index. If you load merged index - /// from disk, we use directly access the data without deserialization - /// unless you want to modify it. - std::unique_ptr buffer; - - /// The in memory data of the index. - std::unique_ptr impl; - - /// See revision(). - std::uint64_t rev = 0; -}; - -} // namespace clice::index diff --git a/src/index/path_pool.h b/src/index/path_pool.h deleted file mode 100644 index d639a3157..000000000 --- a/src/index/path_pool.h +++ /dev/null @@ -1,83 +0,0 @@ -#pragma once - -#include -#include -#include -#include - -#include "llvm/ADT/DenseMap.h" -#include "llvm/ADT/SmallString.h" -#include "llvm/ADT/StringRef.h" -#include "llvm/Support/Allocator.h" - -namespace clice::index { - -/// Intern pool mapping paths to dense local ids. Used for the self-contained -/// path tables of serialized index artifacts: every persisted path id is an -/// index into its artifact's own table, never a runtime pool id (those are -/// per-session). -/// -/// FIXME: like the runtime pool (support/path_pool.h), only -/// separators are normalized — case-variant spellings of one file intern -/// (and persist) as distinct paths. -struct PathPool { - llvm::BumpPtrAllocator allocator; - - std::vector paths; - - llvm::DenseMap cache; - - llvm::StringRef save(llvm::StringRef s) { - auto data = allocator.Allocate(s.size() + 1); - std::ranges::copy(s, data); - data[s.size()] = '\0'; - return llvm::StringRef(data, s.size()); - } - - auto path_id(llvm::StringRef path) { - assert(!path.empty()); - - // Normalize backslashes to forward slashes so that paths from different - // sources (URI decoding, CDB, clang FileManager) compare equal on - // Windows where native separators are backslashes. - llvm::SmallString<256> normalized; - if(path.contains('\\')) { - normalized = path; - std::replace(normalized.begin(), normalized.end(), '\\', '/'); - path = normalized; - } - - auto [it, success] = cache.try_emplace(path, paths.size()); - if(!success) { - return it->second; - } - - auto& [k, v] = *it; - k = save(path); - paths.emplace_back(k); - return it->second; - } - - llvm::StringRef path(std::uint32_t id) const { - return paths[id]; - } - - /// Look up a path in the cache, normalizing backslashes first. - /// Returns cache.end() if the path is not interned. - auto find(llvm::StringRef path) const { - llvm::SmallString<256> normalized; - if(path.contains('\\')) { - normalized = path; - std::replace(normalized.begin(), normalized.end(), '\\', '/'); - path = normalized; - } - return cache.find(path); - } - - /// Tables are equal when they intern the same paths in the same order. - friend bool operator==(const PathPool& lhs, const PathPool& rhs) { - return lhs.paths == rhs.paths; - } -}; - -} // namespace clice::index diff --git a/src/index/preamble_state.h b/src/index/preamble_state.h index d33f7a2fa..c04ce72a2 100644 --- a/src/index/preamble_state.h +++ b/src/index/preamble_state.h @@ -28,7 +28,7 @@ constexpr inline std::uint32_t preamble_format_version = 5; /// queries run directly on the serialized data, nothing is deserialized up /// front. It carries the preamble's full symbol index — every header the /// PCH covers plus the main file's preamble region — with per-file content -/// and line starts for position mapping (mirroring MergedIndex shards), +/// and line starts for position mapping (mirroring the disk shards), /// and the PCH-derived feature state that is spliced into main-file /// results: document links, inactive regions and the open conditional /// stack at the preamble bound. diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index 567f93e47..45cd010d3 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -1,104 +1,420 @@ #include "index/project_index.h" +#include +#include +#include +#include + #include "index/serialization.h" -#include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/DenseSet.h" #include "llvm/ADT/STLExtras.h" namespace clice::index { -llvm::SmallVector ProjectIndex::merge(this ProjectIndex& self, - TUIndex& index, - clice::PathPool& pool) { - auto& paths = index.graph.paths; - llvm::SmallVector file_ids_map; - file_ids_map.resize_for_overwrite(paths.size()); +namespace { + +/// The global layer's persisted form: the FileVersion table as parallel +/// columns plus the symbol table with a self-contained path table for its +/// reference bitmaps. +struct GlobalBlob { + std::uint32_t format_version = 0; + + /// See ProjectIndex::global_generation. + std::uint64_t generation = 0; + + std::uint32_t next_fv_id = 0; + + std::vector fv_ids; + std::vector fv_paths; + std::vector fv_hashes; + std::vector fv_sizes; + std::vector fv_mtimes; + + /// The external symbol table as parallel columns. Reference bitmaps + /// travel as raw portable images rather than through the Bitmap repr: + /// the repr has no failure channel and would normalize a malformed + /// image to empty — silently losing the symbol's reference files with + /// nothing ever rebuilding them — while raw images let load_global + /// reject the blob so everything is reindexed. + std::vector sym_hashes; + std::vector sym_names; + std::vector sym_kinds; + std::vector> sym_bitmaps; + + /// tu_fv -> generation stamp of every manifest current at this save. + /// The loader adopts a manifest blob only on an exact stamp match, so + /// a manifest whose write failed while the global landed (or one that + /// outran a lost global write) reads as lost and its TU reindexes, + /// instead of an older on-disk manifest serving as current. + std::vector manifest_fvs; + std::vector manifest_gens; + + /// Pool id -> path for every id the symbol bitmaps reference. + std::vector> sym_paths; +}; + +} // namespace + +bool ProjectIndex::merge(this ProjectIndex& self, + const TUIndexView& view, + llvm::ArrayRef file_ids_map) { + // Decode and bound every reference bitmap before touching the table: + // merged bits persist in the global blob while the result's recorded + // versions all match the disk, so a malformed image normalized to + // empty — or a silently dropped out-of-range id, whose relations would + // sit in a shard the symbol's fan-out never visits — would lose + // reference files with nothing ever rebuilding them. Either rejects + // the whole result instead — and the reject must leave no partial + // names or bits behind, hence the staging. + struct StagedSymbol { + SymbolHash hash; + SymbolIdentity identity; + Bitmap references; + }; + + std::vector staged; + bool valid = true; + view.iterate_symbols( + [&](SymbolHash hash, const SymbolIdentity& identity, llvm::StringRef bitmap) { + if(!valid || identity.scope != SymbolScope::External) { + return; + } + Bitmap references; + if(!bitmap.empty()) { + auto decoded = read_bitmap(bitmap.data(), bitmap.size()); + if(!decoded) { + valid = false; + return; + } + references = std::move(*decoded); + } + if(!references.isEmpty() && references.maximum() >= file_ids_map.size()) { + valid = false; + return; + } + staged.push_back({hash, identity, std::move(references)}); + }); + if(!valid) { + return false; + } + + for(auto& [hash, identity, references]: staged) { + auto& target = self.symbols[hash]; + if(target.name.empty()) { + target.name = std::string(identity.name); + target.kind = identity.kind; + } + for(auto ref: references) { + target.reference_files.add(file_ids_map[ref]); + } + } + + return true; +} + +std::uint32_t ProjectIndex::intern_file_version(this ProjectIndex& self, + std::uint32_t path_id, + std::uint64_t content_hash) { + auto [it, inserted] = self.fv_ids.try_emplace({path_id, content_hash}, self.next_fv_id); + if(inserted) { + self.file_versions.try_emplace( + self.next_fv_id, + FileVersionRecord{.path_id = path_id, .content_hash = content_hash}); + self.next_fv_id += 1; + } + return it->second; +} + +bool ProjectIndex::knows_file_versions(this const ProjectIndex& self, const TUManifest& manifest) { + if(!self.file_versions.contains(manifest.tu_fv)) { + return false; + } + for(auto& node: manifest.nodes) { + if(!self.file_versions.contains(node.fv)) { + return false; + } + } + for(auto& [fv, hash]: manifest.contributions) { + if(!self.file_versions.contains(fv)) { + return false; + } + } + return true; +} + +llvm::SmallVector ProjectIndex::apply_manifest(this ProjectIndex& self, + std::uint32_t tu_path_id, + TUManifest manifest) { + llvm::SmallVector affected = self.remove_manifest(tu_path_id); + + for(auto& [fv, hash]: manifest.contributions) { + auto path_id = self.file_versions.at(fv).path_id; + self.contributions[path_id][tu_path_id] = hash; + affected.push_back(path_id); + } + self.manifests[tu_path_id] = std::move(manifest); + + llvm::sort(affected); + affected.erase(llvm::unique(affected), affected.end()); + return affected; +} - for(std::uint32_t i = 0; i < paths.size(); i++) { - file_ids_map[i] = pool.intern(paths[i]); +llvm::SmallVector ProjectIndex::remove_manifest(this ProjectIndex& self, + std::uint32_t tu_path_id) { + llvm::SmallVector affected; + auto it = self.manifests.find(tu_path_id); + if(it == self.manifests.end()) { + return affected; } - for(auto& [symbol_id, symbol]: index.symbols) { - if(symbol.scope != SymbolScope::External) + for(auto& [fv, hash]: it->second.contributions) { + auto path_id = self.file_versions.at(fv).path_id; + auto contribution_it = self.contributions.find(path_id); + if(contribution_it == self.contributions.end()) { continue; - auto& target_symbol = self.symbols[symbol_id]; - if(target_symbol.name.empty()) { - target_symbol.name = symbol.name; - target_symbol.kind = symbol.kind; } - for(auto ref: symbol.reference_files) { - target_symbol.reference_files.add(file_ids_map[ref]); + contribution_it->second.erase(tu_path_id); + if(contribution_it->second.empty()) { + self.contributions.erase(contribution_it); } + affected.push_back(path_id); } + self.manifests.erase(it); - return file_ids_map; + llvm::sort(affected); + affected.erase(llvm::unique(affected), affected.end()); + return affected; } -void ProjectIndex::serialize(this ProjectIndex& self, - llvm::raw_ostream& os, - const clice::PathPool& pool, - llvm::ArrayRef shards) { - self.format_version = index_format_version; - self.shards.assign(shards.begin(), shards.end()); +llvm::SmallVector ProjectIndex::live_variants(this const ProjectIndex& self, + std::uint32_t path_id) { + llvm::SmallVector variants; + auto it = self.contributions.find(path_id); + if(it == self.contributions.end()) { + return variants; + } + for(auto hash: llvm::make_second_range(it->second)) { + variants.push_back(hash); + } + llvm::sort(variants); + variants.erase(llvm::unique(variants), variants.end()); + return variants; +} - Bitmap referenced(shards.size(), shards.data()); - for(auto& symbol: llvm::make_second_range(self.symbols)) { - referenced |= symbol.reference_files; +void ProjectIndex::serialize_global(this ProjectIndex& self, + llvm::raw_ostream& os, + const clice::PathPool& pool) { + // Garbage-collect the FileVersion table down to what some manifest + // still references — in memory too, so ids of dead versions stop + // accumulating across the session (they are never reused either way). + llvm::DenseSet referenced; + for(auto& manifest: llvm::make_second_range(self.manifests)) { + referenced.insert(manifest.tu_fv); + for(auto& node: manifest.nodes) { + referenced.insert(node.fv); + } + for(auto& [fv, hash]: manifest.contributions) { + referenced.insert(fv); + } + } + + llvm::SmallVector dead; + for(auto id: llvm::make_first_range(self.file_versions)) { + if(!referenced.contains(id)) { + dead.push_back(id); + } + } + for(auto id: dead) { + auto& record = self.file_versions.at(id); + self.fv_ids.erase({record.path_id, record.content_hash}); + self.file_versions.erase(id); + } + + GlobalBlob blob; + blob.format_version = index_format_version; + blob.generation = self.global_generation; + blob.next_fv_id = self.next_fv_id; + + llvm::SmallVector ids; + ids.reserve(self.file_versions.size()); + for(auto id: llvm::make_first_range(self.file_versions)) { + ids.push_back(id); + } + llvm::sort(ids); + for(auto id: ids) { + auto& record = self.file_versions.at(id); + blob.fv_ids.push_back(id); + blob.fv_paths.emplace_back(pool.resolve(record.path_id)); + blob.fv_hashes.push_back(record.content_hash); + blob.fv_sizes.push_back(record.size); + blob.fv_mtimes.push_back(record.mtime_ns); } - self.paths.clear(); - self.paths.reserve(referenced.cardinality()); - for(auto id: referenced) { - self.paths.emplace_back(id, pool.resolve(id).str()); + blob.manifest_fvs.reserve(self.manifests.size()); + blob.manifest_gens.reserve(self.manifests.size()); + for(auto& manifest: llvm::make_second_range(self.manifests)) { + blob.manifest_fvs.push_back(manifest.tu_fv); + blob.manifest_gens.push_back(manifest.global_gen); } - serialize_blob(self, os); + Bitmap bitmap_referenced; + blob.sym_hashes.reserve(self.symbols.size()); + blob.sym_names.reserve(self.symbols.size()); + blob.sym_kinds.reserve(self.symbols.size()); + blob.sym_bitmaps.reserve(self.symbols.size()); + for(auto& [hash, symbol]: self.symbols) { + blob.sym_hashes.push_back(hash); + // Moved out for the write and moved back below; the table itself + // stays untouched in between so the two iterations pair up. + blob.sym_names.push_back(std::move(symbol.name)); + blob.sym_kinds.push_back(symbol.kind.value()); + blob.sym_bitmaps.push_back(write_bitmap(symbol.reference_files)); + bitmap_referenced |= symbol.reference_files; + } + blob.sym_paths.reserve(bitmap_referenced.cardinality()); + for(auto id: bitmap_referenced) { + blob.sym_paths.emplace_back(id, pool.resolve(id).str()); + } + + serialize_blob(blob, os); + + std::size_t i = 0; + for(auto& symbol: llvm::make_second_range(self.symbols)) { + symbol.name = std::move(blob.sym_names[i]); + i += 1; + } } -std::optional ProjectIndex::from(llvm::StringRef data, - clice::PathPool& pool, - llvm::SmallVectorImpl& shards) { - std::optional index{std::in_place}; - if(!deserialize_blob(data, *index) || index->format_version != index_format_version) { - return std::nullopt; +bool ProjectIndex::load_global(this ProjectIndex& self, + llvm::StringRef data, + clice::PathPool& pool, + llvm::DenseMap& manifest_pins) { + GlobalBlob blob; + if(!deserialize_blob(data, blob) || blob.format_version != index_format_version) { + return false; } - // The blob's ids are the writing session's pool ids: intern its path - // table and remap every decoded id into this session's pool. Ids the - // table does not cover are dropped, not misresolved. - llvm::DenseMap remap; - remap.reserve(index->paths.size()); - for(auto& [id, path]: index->paths) { - // The writer only emits interned paths, which are never empty; an - // empty entry marks a corrupt blob and must not become a real pool - // entry. + auto count = blob.fv_ids.size(); + if(blob.fv_paths.size() != count || blob.fv_hashes.size() != count || + blob.fv_sizes.size() != count || blob.fv_mtimes.size() != count) { + return false; + } + auto sym_count = blob.sym_hashes.size(); + if(blob.sym_names.size() != sym_count || blob.sym_kinds.size() != sym_count || + blob.sym_bitmaps.size() != sym_count) { + return false; + } + if(blob.manifest_gens.size() != blob.manifest_fvs.size()) { + return false; + } + + // Every value check runs before the first mutation: a blob rejected + // halfway through would otherwise leave partial state behind — file + // versions whose corrupt stat stamps feed the freshness fast path, and + // symbols the next global save would persist — while the caller treats + // the failed load as "no index on disk". + + // The writer only emits interned paths, which are never empty; an + // empty entry marks a corrupt blob and must not become a real pool + // entry. + for(auto& path: blob.fv_paths) { if(path.empty()) { - return std::nullopt; + return false; + } + } + for(auto& path: llvm::make_second_range(blob.sym_paths)) { + if(path.empty()) { + return false; } - remap.try_emplace(id, pool.intern(path)); } - for(auto& symbol: llvm::make_second_range(index->symbols)) { - Bitmap remapped; - for(auto id: symbol.reference_files) { - if(auto it = remap.find(id); it != remap.end()) { - remapped.add(it->second); + // Ids and (path, hash) pairs are both map keys in the writer, so a + // repeat of either marks a corrupt blob. A repeated id in particular + // would leave fv_ids interning the earlier pair to an id whose record + // names the later path, attributing contributions to the wrong file. + llvm::DenseSet blob_fvs(blob.fv_ids.begin(), blob.fv_ids.end()); + if(blob_fvs.size() != count) { + return false; + } + llvm::DenseSet> blob_versions; + for(std::size_t i = 0; i < count; i += 1) { + if(!blob_versions.insert({llvm::StringRef(blob.fv_paths[i]), blob.fv_hashes[i]}).second) { + return false; + } + } + + // The writer only pins manifests whose tu_fv survived the same save's + // garbage collection, so an unresolvable pin marks a corrupt blob. + for(auto fv: blob.manifest_fvs) { + if(!blob_fvs.contains(fv)) { + return false; + } + } + + // The writer emits a path-table entry for every id its bitmaps + // reference; an uncovered id dropped here would silently lose the + // symbol's reference files with every manifest still fresh — reject + // the blob so everything is reindexed instead. + llvm::DenseSet covered; + for(auto id: llvm::make_first_range(blob.sym_paths)) { + covered.insert(id); + } + std::vector bitmaps; + bitmaps.reserve(sym_count); + for(auto& image: blob.sym_bitmaps) { + auto decoded = read_bitmap(image.data(), image.size()); + if(!decoded) { + return false; + } + for(auto id: *decoded) { + if(!covered.contains(id)) { + return false; } } - symbol.reference_files = std::move(remapped); + bitmaps.push_back(std::move(*decoded)); } - for(auto id: index->shards) { - if(auto it = remap.find(id); it != remap.end()) { - shards.push_back(it->second); + self.global_generation = blob.generation; + self.next_fv_id = blob.next_fv_id; + for(std::size_t i = 0; i < count; i += 1) { + auto path_id = pool.intern(blob.fv_paths[i]); + auto id = blob.fv_ids[i]; + self.file_versions[id] = {path_id, blob.fv_hashes[i], blob.fv_sizes[i], blob.fv_mtimes[i]}; + self.fv_ids[{path_id, blob.fv_hashes[i]}] = id; + // Ids must stay unique forever; a blob whose counter lags its own + // table (corruption) must not hand out ids that alias stored ones. + if(id >= self.next_fv_id) { + self.next_fv_id = id + 1; } } - // The table and manifest were only the wire form; the runtime state is - // the pool and the caller's shard list. - index->paths.clear(); - index->shards.clear(); - return index; + for(std::size_t k = 0; k < blob.manifest_fvs.size(); k += 1) { + manifest_pins[blob.manifest_fvs[k]] = blob.manifest_gens[k]; + } + + // The blob's bitmap ids are the writing session's pool ids: intern its + // path table and remap every decoded id into this session's pool. Every + // id was proven covered by the table above. + llvm::DenseMap remap; + remap.reserve(blob.sym_paths.size()); + for(auto& [id, path]: blob.sym_paths) { + remap.try_emplace(id, pool.intern(path)); + } + + self.symbols.reserve(sym_count); + for(std::size_t k = 0; k < sym_count; k += 1) { + Bitmap remapped; + for(auto id: bitmaps[k]) { + remapped.add(remap.find(id)->second); + } + auto& symbol = self.symbols[blob.sym_hashes[k]]; + symbol.name = std::move(blob.sym_names[k]); + symbol.kind = SymbolKind(blob.sym_kinds[k]); + symbol.reference_files = std::move(remapped); + } + + return true; } } // namespace clice::index diff --git a/src/index/project_index.h b/src/index/project_index.h index 7c0b29a0e..b74c5bec8 100644 --- a/src/index/project_index.h +++ b/src/index/project_index.h @@ -1,69 +1,140 @@ #pragma once #include -#include -#include #include -#include +#include "index/manifest.h" #include "index/tu_index.h" #include "support/path_pool.h" +#include "llvm/ADT/ArrayRef.h" +#include "llvm/ADT/DenseMap.h" #include "llvm/ADT/SmallVector.h" #include "llvm/Support/raw_ostream.h" namespace clice::index { -/// Project-wide symbol table accumulated from background indexing. +/// One observed version of a file on disk: the identity every manifest +/// dependency points at, and the single place its freshness baseline +/// lives — the consumed-content hash plus a stat fast path, shared by +/// every TU that consumed this version instead of being copied per TU. /// -/// There is a single path-id space at runtime: the server-wide -/// clice::PathPool. Symbol reference bitmaps carry those ids directly, so -/// queries never translate between pools, and they persist as-is — the blob -/// stays self-contained through a path table mapping every referenced id to -/// its path (which is also the garbage collection: paths no longer -/// referenced by any symbol or shard are simply not written). Loading -/// interns the table into the running pool and remaps every id. +/// `size`/`mtime_ns` are recorded only when the file provably did not +/// change since before the indexed build started; mtime_ns == 0 means "no +/// fast path" and the check falls through to the hash comparison, which +/// repairs the fast path in place on a match — once, for all consumers. +struct FileVersionRecord { + /// Runtime path pool id; persisted as the path string. + std::uint32_t path_id = 0; + + /// xxh3 of the bytes the indexing compile consumed (0 = the worker had + /// no buffer to hash; freshness stays conservative for this version). + std::uint64_t content_hash = 0; + + std::uint64_t size = 0; + std::int64_t mtime_ns = 0; +}; + +/// The index's global layer: everything mutable, everything shared across +/// files. +/// +/// - the project-wide external symbol table with per-symbol reference-file +/// bitmaps (the cross-file query fan-out), +/// - the FileVersion table anchoring freshness and content identity, +/// - one manifest per indexed TU (replaced wholesale by its reindex), +/// - `contributions`, derived from the manifests at load: per file, which +/// TU contributed which rows variant. Its distinct hashes per file are +/// the file's live variants — the mask Shard queries filter by — and its +/// emptiness is what retires a shard blob. /// -/// Serialization reflects this object directly; `format_version`, `paths` -/// and `shards` are serialize-time state populated by serialize() and -/// consumed by from(). +/// There is a single path-id space at runtime (clice::PathPool); persisted +/// blobs are self-contained through path tables and remap on load. struct ProjectIndex { - /// Persisted-blob schema version (index_format_version), stamped by - /// serialize() and gated by from(). - std::uint32_t format_version = 0; + SymbolTable symbols; - /// The blob's self-contained path table: pool id → path for every id - /// the symbol bitmaps and the shard manifest reference. - std::vector> paths; + llvm::DenseMap file_versions; - SymbolTable symbols; + /// (path_id, content_hash) -> FileVersion id. + llvm::DenseMap, std::uint32_t> fv_ids; + + /// Ids are monotonic and never reused, so a manifest on disk stays + /// resolvable against any later global blob (or is detected as stale). + std::uint32_t next_fv_id = 0; + + /// Generation of the persisted global blob, bumped once per save that + /// writes it. Manifests are stamped with the generation they were + /// saved under (TUManifest::global_gen), and the global blob pins the + /// stamp expected of every TU's manifest; the loader adopts a manifest + /// only on an exact match, so a lost or failed manifest write cannot + /// leave an older on-disk manifest serving as current. + std::uint64_t global_generation = 0; + + /// TU path_id -> its manifest. + llvm::DenseMap manifests; + + /// Derived from `manifests`: file path_id -> (TU path_id -> rows hash). + llvm::DenseMap> + contributions; + + /// Merge a TU's external symbols straight off the wire; `file_ids_map` + /// maps the TU-local ids of `view`'s path table to pool ids. Symbol + /// names are copied only for symbols new to the table. Returns false — + /// with the table untouched — when a reference bitmap fails to decode + /// or carries an id past the path table (the bound TUIndex::from + /// enforces; the zero-copy view leaves it to this consumer): the + /// caller rejects the whole result, because merged bits persist while + /// the result's recorded versions match the disk, so lost bits would + /// never be rebuilt. + bool merge(this ProjectIndex& self, + const TUIndexView& view, + llvm::ArrayRef file_ids_map); + + /// The FileVersion id for (path, content hash), interning a new record + /// on first sight. + std::uint32_t intern_file_version(this ProjectIndex& self, + std::uint32_t path_id, + std::uint64_t content_hash); + + /// Whether every FileVersion id the manifest references is known — + /// the loader's staleness gate for manifests read from disk. + bool knows_file_versions(this const ProjectIndex& self, const TUManifest& manifest); + + /// Install (or replace) a TU's manifest and rederive the affected + /// contribution entries. Returns the file path_ids whose contribution + /// set changed — the caller refreshes those shards' live-variant masks. + llvm::SmallVector apply_manifest(this ProjectIndex& self, + std::uint32_t tu_path_id, + TUManifest manifest); + + /// Drop a TU's manifest and its contribution entries. Returns the + /// affected file path_ids, like apply_manifest. + llvm::SmallVector remove_manifest(this ProjectIndex& self, + std::uint32_t tu_path_id); + + /// The distinct rows hashes contributed to `path_id` — the file's live + /// variant set. + llvm::SmallVector live_variants(this const ProjectIndex& self, + std::uint32_t path_id); + + /// Serialize the global blob: the FileVersion table (garbage-collected + /// down to the versions some manifest still references, in memory too), + /// the symbol table with a self-contained path table, and a per-TU pin + /// of every manifest's generation stamp. + void serialize_global(this ProjectIndex& self, + llvm::raw_ostream& os, + const clice::PathPool& pool); - /// Pool ids of the files owning a MergedIndex shard blob, persisted so - /// the loader knows which blobs to fetch. - std::vector shards; - - /// Merge a TU's external symbols, interning the TU's paths into `pool`. - /// Returns the TU-local id → pool id mapping for the TU's path graph. - llvm::SmallVector merge(this ProjectIndex& self, - TUIndex& index, - clice::PathPool& pool); - - /// Serialize with a path table covering exactly the ids used by the - /// symbol bitmaps plus `shards`, the pool ids of the files owning a - /// MergedIndex shard blob. - void serialize(this ProjectIndex& self, - llvm::raw_ostream& os, - const clice::PathPool& pool, - llvm::ArrayRef shards); - - /// Restore from a serialized blob, interning its path table into `pool` - /// and filling `shards` with the pool ids of the files whose shard blobs - /// the loader should fetch. Returns nullopt for an unreadable or - /// old-format blob — the caller treats that as "no index on disk" and - /// rebuilds in the background. - static std::optional from(llvm::StringRef data, - clice::PathPool& pool, - llvm::SmallVectorImpl& shards); + /// Restore the global blob, interning its paths into `pool`. Returns + /// false for an unreadable or old-format blob, leaving the index (and + /// `pool`) untouched — the caller treats that as "no index on disk" + /// and rebuilds in the background. Manifests are loaded separately + /// (apply_manifest per blob); `manifest_pins` maps each pinned TU's + /// tu_fv to the generation stamp its manifest must carry to be + /// adopted. + bool load_global(this ProjectIndex& self, + llvm::StringRef data, + clice::PathPool& pool, + llvm::DenseMap& manifest_pins); }; } // namespace clice::index diff --git a/src/index/serialization.h b/src/index/serialization.h index b8b6dfb77..c890bae60 100644 --- a/src/index/serialization.h +++ b/src/index/serialization.h @@ -4,40 +4,76 @@ #include #include #include +#include #include #include #include #include #include -#include "index/path_pool.h" #include "index/tu_index.h" #include "semantic/symbol.h" #include "support/bitmap.h" #include "kota/codec/fbs/fbs.h" #include "llvm/ADT/ArrayRef.h" -#include "llvm/ADT/STLExtras.h" -#include "llvm/ADT/StringMap.h" #include "llvm/ADT/StringRef.h" #include "llvm/Support/raw_ostream.h" +namespace clice::index { + +/// Decode a serialized bitmap without trusting its bytes: bounded by the +/// buffer, nullopt on a failed parse — croaring's C++ read wrappers abort +/// on one, and blob bitmaps are untrusted disk/wire input. The caller +/// chooses what a failure means: anything feeding persisted state (the +/// project merge, blob loaders) rejects the whole input — normalized to +/// empty, reference bits would read as fresh and stay lost forever — +/// while the session-scoped full decode degrades to an empty bitmap, +/// rebuilt by the next parse. +inline std::optional read_bitmap(const void* data, std::size_t size) { + auto* decoded = + roaring::api::roaring_bitmap_portable_deserialize_safe(static_cast(data), + size); + if(!decoded) { + return std::nullopt; + } + // deserialize_safe only bounds the reads; the bitmap it hands back can + // still violate internal invariants (unsorted containers), on which + // croaring's operations are undefined. + if(!roaring::api::roaring_bitmap_internal_validate(decoded, nullptr)) { + roaring::api::roaring_bitmap_free(decoded); + return std::nullopt; + } + return Bitmap(decoded); +} + +/// Encode a bitmap as its portable image — the only format with a bounded +/// deserializer. +inline std::vector write_bitmap(const Bitmap& bitmap) { + std::vector buffer(bitmap.getSizeInBytes(true)); + bitmap.write(reinterpret_cast(buffer.data()), true); + return buffer; +} + +} // namespace clice::index + namespace kota::meta { -/// Roaring bitmaps travel as their own serialized image (the non-portable -/// format, matching every in-process reader). +/// Roaring bitmaps travel in the portable format — the only one with a +/// bounded deserializer. Only the session-scoped full decode goes through +/// this repr (it has no failure channel, so a malformed image degrades to +/// empty); the merge path and persisted blobs read raw images and reject +/// unparseable ones. template <> struct repr { using type = std::vector; static type to(const clice::Bitmap& bitmap) { - type buffer(bitmap.getSizeInBytes(false)); - bitmap.write(reinterpret_cast(buffer.data()), false); - return buffer; + return clice::index::write_bitmap(bitmap); } static clice::Bitmap from(const type& buffer) { - return clice::Bitmap::read(reinterpret_cast(buffer.data()), false); + return clice::index::read_bitmap(buffer.data(), buffer.size()).value_or(clice::Bitmap{}); } }; @@ -56,80 +92,6 @@ struct repr { } }; -/// A PathPool persists as its path table; ids are the dense indices, so -/// interning the table back in order reproduces them. Both directions drive -/// the visitor: encoding writes the interned StringRefs straight to the -/// wire, decoding interns one path at a time. -template <> -struct repr { - using type = std::vector; - - template - static bool serialize(auto& vis, const clice::index::PathPool& pool) { - return codec::encode_value(vis, pool.paths); - } - - template - static bool deserialize(auto& vis, clice::index::PathPool& pool) { - type shape; - return vis.visit_seq(shape, [&](auto& sv) -> bool { - while(sv.has_element()) { - std::string path; - if(!sv.visit_element( - [&](auto& ev) -> bool { return codec::decode_value(ev, path); })) { - return false; - } - // The pool never interns an empty path, so a blob carrying - // one is corrupt; reject it instead of tripping the intern - // precondition. - if(path.empty()) { - return false; - } - pool.path_id(path); - } - return true; - }); - } -}; - -/// A StringMap iterates as StringMapEntry, which no codec understands; -/// persist the entries as key/value pairs, sorted by value for -/// deterministic blobs (values are unique canonical ids). Format-agnostic, -/// unlike the reprs above: the pair-list form is not fbs-specific, and the -/// schema layer classifies fields without a format tag — a format-scoped -/// repr would leave it staring at StringMapEntry, which it rejects. -template <> -struct repr> { - using type = std::vector>; - - template - static bool serialize(auto& vis, const llvm::StringMap& map) { - llvm::SmallVector> entries; - entries.reserve(map.size()); - for(const auto& entry: map) { - entries.emplace_back(entry.getKey(), entry.getValue()); - } - llvm::sort(entries, llvm::less_second{}); - return codec::encode_value(vis, entries); - } - - template - static bool deserialize(auto& vis, llvm::StringMap& map) { - type shape; - return vis.visit_seq(shape, [&](auto& sv) -> bool { - while(sv.has_element()) { - std::pair entry; - if(!sv.visit_element( - [&](auto& ev) -> bool { return codec::decode_value(ev, entry); })) { - return false; - } - map.try_emplace(entry.first, entry.second); - } - return true; - }); - } -}; - template <> struct repr { using type = std::int64_t; @@ -151,7 +113,7 @@ namespace clice::index { /// regular field and every loader discards blobs with a different value — /// including version-less blobs from older builds, which read back as 0. /// Bump it whenever a persisted type's reflected layout changes. -constexpr inline std::uint32_t index_format_version = 3; +constexpr inline std::uint32_t index_format_version = 5; /// Serialize a reflected index blob to `os` as a verified-readable /// flatbuffer. Encoding only fails on structural impossibilities (e.g. more diff --git a/src/index/shard.cpp b/src/index/shard.cpp new file mode 100644 index 000000000..ee4cbc06b --- /dev/null +++ b/src/index/shard.cpp @@ -0,0 +1,1188 @@ +#include "index/shard.h" + +#include +#include +#include +#include +#include + +#include "index/serialization.h" + +#include "kota/ipc/lsp/position.h" +#include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/DenseSet.h" +#include "llvm/ADT/STLExtras.h" +#include "llvm/Support/xxhash.h" + +namespace clice::index { + +namespace { + +using ShardView = kota::codec::fbs::table_view; + +/// Sentinel in the length column: the real end lives in the sparse +/// (row, end) escape table. +constexpr std::uint8_t length_escape = 0xff; + +enum class MaskTier : std::uint8_t { + /// One variant: no mask columns at all. + Single, + U32, + U64, + Roaring, +}; + +MaskTier tier_of(std::size_t variant_count) { + if(variant_count <= 1) { + return MaskTier::Single; + } + if(variant_count <= 32) { + return MaskTier::U32; + } + if(variant_count <= 64) { + return MaskTier::U64; + } + return MaskTier::Roaring; +} + +/// The blob was fully verified at load; per-query views skip that cost. +ShardView root_of(const llvm::MemoryBuffer& buffer) { + return ShardView::from_verified_bytes(blob_bytes(buffer.getBuffer())); +} + +/// One side of the blob's row storage (occurrences or relations) as +/// contiguous column refs. +struct RowColumns { + llvm::ArrayRef begins; + llvm::ArrayRef lengths; + llvm::ArrayRef long_rows; + llvm::ArrayRef long_ends; + llvm::ArrayRef masks32; + llvm::ArrayRef masks64; + llvm::ArrayRef roaring_offsets; + llvm::ArrayRef roaring; + + std::uint32_t end_of(std::uint32_t row) const { + auto length = lengths[row]; + if(length == length_escape) { + // validate() proves every sentinel owns exactly one escape + // entry, so the search always lands. + auto it = std::ranges::lower_bound(long_rows, row); + return long_ends[it - long_rows.begin()]; + } + return begins[row] + length; + } +}; + +RowColumns occ_columns(ShardView root) { + return { + to_array_ref(root[&ShardBlob::occ_begins]), + to_array_ref(root[&ShardBlob::occ_lengths]), + to_array_ref(root[&ShardBlob::occ_long_rows]), + to_array_ref(root[&ShardBlob::occ_long_ends]), + to_array_ref(root[&ShardBlob::occ_masks32]), + to_array_ref(root[&ShardBlob::occ_masks64]), + to_array_ref(root[&ShardBlob::occ_roaring_offsets]), + to_array_ref(root[&ShardBlob::occ_roaring]), + }; +} + +RowColumns rel_columns(ShardView root) { + return { + to_array_ref(root[&ShardBlob::rel_begins]), + to_array_ref(root[&ShardBlob::rel_lengths]), + to_array_ref(root[&ShardBlob::rel_long_rows]), + to_array_ref(root[&ShardBlob::rel_long_ends]), + to_array_ref(root[&ShardBlob::rel_masks32]), + to_array_ref(root[&ShardBlob::rel_masks64]), + to_array_ref(root[&ShardBlob::rel_roaring_offsets]), + to_array_ref(root[&ShardBlob::rel_roaring]), + }; +} + +/// The slice bounds were validated monotonic and in-bounds at load, and +/// every slice proven to decode, so this cannot fail. +Bitmap read_row_bitmap(const RowColumns& columns, std::uint32_t row) { + auto begin = columns.roaring_offsets[row]; + return *read_bitmap(columns.roaring.data() + begin, columns.roaring_offsets[row + 1] - begin); +} + +/// Structural verification does not constrain field values; everything the +/// readers dereference through raw column pointers or binary-search must be +/// proven in-bounds and in order here, once, so queries stay check-free. +bool validate(ShardView root) { + auto variants = to_array_ref(root[&ShardBlob::variants]); + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + auto offsets = to_array_ref(root[&ShardBlob::sym_rel_offsets]); + + if(variants.empty()) { + return false; + } + // A variant is identified by its rows hash everywhere (set_live, + // write_shard's keep filter), so hashes must be unique: rows owned only + // by a duplicated entry would serve and survive compaction with no + // contribution owning them. + llvm::SmallVector sorted_variants(variants.begin(), variants.end()); + std::ranges::sort(sorted_variants); + if(std::ranges::adjacent_find(sorted_variants) != sorted_variants.end()) { + return false; + } + + // Every freshness decision compares the advertised content hash + // (manifest FileVersions, the merge's generation checks), so content + // bytes corrupted under an intact structure would keep loading as fresh + // while position mapping reads text the rows were not built from. + if(llvm::xxh3_64bits(to_ref(root[&ShardBlob::content])) != root[&ShardBlob::content_hash]) { + return false; + } + + auto occ = occ_columns(root); + auto rel = rel_columns(root); + auto occ_count = occ.begins.size(); + auto rel_count = rel.begins.size(); + + if(offsets.size() != sym_hashes.size() + 1) { + return false; + } + if(!std::ranges::is_sorted(offsets) || offsets.back() != rel_count) { + return false; + } + // Strictly: symbol lookups lower-bound the hash column and read only + // the first match's slices, so a duplicated hash would strand the later + // id's relations and local name unreachably. + if(!std::ranges::is_sorted(sym_hashes, std::less_equal{})) { + return false; + } + + auto sym_ids_ok = [&](auto ids) { + return llvm::all_of(ids, [&](std::uint32_t id) { return id < sym_hashes.size(); }); + }; + auto occ_syms16 = to_array_ref(root[&ShardBlob::occ_syms16]); + auto occ_syms32 = to_array_ref(root[&ShardBlob::occ_syms32]); + if(occ.lengths.size() != occ_count) { + return false; + } + if(occ_syms16.size() + occ_syms32.size() != occ_count || + (!occ_syms16.empty() && !occ_syms32.empty())) { + return false; + } + if(!sym_ids_ok(occ_syms16) || !sym_ids_ok(occ_syms32)) { + return false; + } + + auto rel_kinds = to_array_ref(root[&ShardBlob::rel_kinds]); + if(rel_kinds.size() != rel_count || rel.lengths.size() != rel_count) { + return false; + } + + auto sparse_ok = [](llvm::ArrayRef rows, std::size_t values, std::size_t count) { + return rows.size() == values && std::ranges::is_sorted(rows, std::less_equal{}) && + (rows.empty() || rows.back() < count); + }; + // The escape table must pair one-to-one, in row order, with the + // sentinel lengths: end_of trusts the pairing, and a sentinel missing + // its entry (or a stray entry masking one elsewhere) can pass every + // range bound below while serving a wrong end forever. + auto escapes_ok = [](llvm::ArrayRef lengths, + llvm::ArrayRef long_rows, + llvm::ArrayRef long_ends) { + if(long_ends.size() != long_rows.size()) { + return false; + } + std::size_t cursor = 0; + for(std::uint32_t row = 0; row < lengths.size(); row += 1) { + if(lengths[row] != length_escape) { + continue; + } + if(cursor == long_rows.size() || long_rows[cursor] != row) { + return false; + } + cursor += 1; + } + return cursor == long_rows.size(); + }; + if(!escapes_ok(occ.lengths, occ.long_rows, occ.long_ends)) { + return false; + } + // lookup(offset) binary-searches the decoded end column and stops its + // containment walk on begin order; rows out of either order (a corrupt + // escaped end included) would silently miss or misresolve occurrences + // on every query, forever — reject the blob so it is rebuilt instead. + // Ends are bounded by the stored content too: every decoded range is + // served as a source range into it. + auto content_size = root[&ShardBlob::content].size(); + std::uint32_t prev_begin = 0; + std::uint32_t prev_end = 0; + for(std::uint32_t row = 0; row < occ_count; row += 1) { + auto begin = occ.begins[row]; + auto end = occ.end_of(row); + if(begin < prev_begin || end < prev_end || end < begin || end > content_size) { + return false; + } + prev_begin = begin; + prev_end = end; + } + if(!escapes_ok(rel.lengths, rel.long_rows, rel.long_ends)) { + return false; + } + // Relation ranges carry no query order to enforce, but are served as + // source ranges all the same — bound them like the occurrence ends. + // The exception is the default LocalSourceRange, the writer's sentinel + // for pair relations, which carry no range of their own; every other + // kind is written with a real range, so a sentinel there is corruption + // that would serve an invalid source range forever. + for(std::uint32_t row = 0; row < rel_count; row += 1) { + auto begin = rel.begins[row]; + auto end = rel.end_of(row); + if((LocalSourceRange{begin, end}) == LocalSourceRange{}) { + if(!RelationKind(static_cast(rel_kinds[row])).isBetweenSymbol()) { + return false; + } + continue; + } + if(end < begin || end > content_size) { + return false; + } + } + + auto rel_sym_rows = to_array_ref(root[&ShardBlob::rel_sym_rows]); + auto rel_sym16 = to_array_ref(root[&ShardBlob::rel_sym16]); + auto rel_sym32 = to_array_ref(root[&ShardBlob::rel_sym32]); + if(!sparse_ok(rel_sym_rows, rel_sym16.size() + rel_sym32.size(), rel_count) || + (!rel_sym16.empty() && !rel_sym32.empty())) { + return false; + } + if(!sym_ids_ok(rel_sym16) || !sym_ids_ok(rel_sym32)) { + return false; + } + auto rel_def_rows = to_array_ref(root[&ShardBlob::rel_def_rows]); + auto rel_def_begins = to_array_ref(root[&ShardBlob::rel_def_begins]); + auto rel_def_ends = to_array_ref(root[&ShardBlob::rel_def_ends]); + if(!sparse_ok(rel_def_rows, rel_def_begins.size(), rel_count) || + rel_def_ends.size() != rel_def_begins.size()) { + return false; + } + for(std::uint32_t k = 0; k < rel_def_begins.size(); k += 1) { + if(rel_def_ends[k] < rel_def_begins[k] || rel_def_ends[k] > content_size) { + return false; + } + } + + auto local_syms = to_array_ref(root[&ShardBlob::local_syms]); + auto local_kinds = to_array_ref(root[&ShardBlob::local_kinds]); + auto local_scopes = to_array_ref(root[&ShardBlob::local_scopes]); + if(!sparse_ok(local_syms, local_kinds.size(), sym_hashes.size()) || + local_scopes.size() != local_kinds.size() || + root[&ShardBlob::local_names].size() != local_kinds.size()) { + return false; + } + + // Beyond the per-tier column shape, every mask must own at least one + // stored variant and no bits past the variant table: an ownerless row + // serves unconditionally while every stored variant is live (row_live's + // live.all fast path skips the mask), vanishes once any variant dies, + // and the next compaction erases it for real — every manifest still + // fresh throughout. + auto masks_ok = [&](const RowColumns& columns, std::size_t count) { + switch(tier_of(variants.size())) { + case MaskTier::Single: { + return columns.masks32.empty() && columns.masks64.empty() && + columns.roaring_offsets.empty() && columns.roaring.empty(); + } + case MaskTier::U32: { + if(columns.masks32.size() != count || !columns.masks64.empty() || + !columns.roaring_offsets.empty()) { + return false; + } + auto stray = variants.size() < 32 ? ~std::uint32_t(0) << variants.size() : 0; + return llvm::all_of(columns.masks32, [&](std::uint32_t mask) { + return mask != 0 && (mask & stray) == 0; + }); + } + case MaskTier::U64: { + if(columns.masks64.size() != count || !columns.masks32.empty() || + !columns.roaring_offsets.empty()) { + return false; + } + auto stray = variants.size() < 64 ? ~std::uint64_t(0) << variants.size() : 0; + return llvm::all_of(columns.masks64, [&](std::uint64_t mask) { + return mask != 0 && (mask & stray) == 0; + }); + } + case MaskTier::Roaring: { + if(!columns.masks32.empty() || !columns.masks64.empty() || + columns.roaring_offsets.size() != count + 1 || + !std::ranges::is_sorted(columns.roaring_offsets) || + columns.roaring_offsets.back() != columns.roaring.size() || + (count != 0 && columns.roaring_offsets.front() != 0)) { + return false; + } + // A slice failing decode would read as an empty mask — the + // ownerless-row corruption above in another coat. Prove + // each slice decodes once here so queries stay check-free + // and the blob rebuilds instead. + for(std::uint32_t row = 0; row < count; row += 1) { + auto begin = columns.roaring_offsets[row]; + auto mask = read_bitmap(columns.roaring.data() + begin, + columns.roaring_offsets[row + 1] - begin); + if(!mask || mask->isEmpty() || mask->maximum() >= variants.size()) { + return false; + } + } + return true; + } + } + std::unreachable(); + }; + return masks_ok(occ, occ_count) && masks_ok(rel, rel_count); +} + +std::uint32_t occ_sym_id(ShardView root, std::uint32_t row) { + auto syms16 = to_array_ref(root[&ShardBlob::occ_syms16]); + if(!syms16.empty()) { + return syms16[row]; + } + return to_array_ref(root[&ShardBlob::occ_syms32])[row]; +} + +} // namespace + +Shard::Shard(std::unique_ptr buffer) : buffer(std::move(buffer)) {} + +Shard Shard::from_bytes(llvm::StringRef data) { + return from_buffer(llvm::MemoryBuffer::getMemBuffer(data, "", false)); +} + +Shard Shard::from_buffer(std::unique_ptr buffer) { + if(!buffer) { + return {}; + } + + // Stale or corrupt bytes (an older build's cache directory) must never + // crash the server or be misread: deep structural verification first, + // then the format-version gate, then the cross-field size checks the + // raw column readers rely on. Anything failing loads as "not on disk" + // and the background indexer rebuilds it. + auto root = ShardView::from_bytes(blob_bytes(buffer->getBuffer())); + if(!root.valid() || root[&ShardBlob::format_version] != index_format_version || + !validate(root)) { + return {}; + } + return Shard(std::move(buffer)); +} + +std::uint64_t Shard::content_hash() const { + if(!buffer) { + return 0; + } + return root_of(*buffer)[&ShardBlob::content_hash]; +} + +std::vector Shard::variants() const { + if(!buffer) { + return {}; + } + auto stored = to_array_ref(root_of(*buffer)[&ShardBlob::variants]); + return {stored.begin(), stored.end()}; +} + +bool Shard::has_variant(RowsHash hash) const { + if(!buffer) { + return false; + } + return llvm::is_contained(to_array_ref(root_of(*buffer)[&ShardBlob::variants]), hash); +} + +void Shard::set_live(llvm::ArrayRef live_hashes) { + live = {}; + if(!buffer) { + return; + } + + auto stored = to_array_ref(root_of(*buffer)[&ShardBlob::variants]); + std::size_t matched = 0; + Live next; + for(std::uint32_t id = 0; id < stored.size(); id += 1) { + if(!llvm::is_contained(live_hashes, stored[id])) { + continue; + } + matched += 1; + if(id < 64) { + next.bits |= std::uint64_t(1) << id; + } + next.big.add(id); + } + next.all = matched == stored.size(); + live = std::move(next); +} + +bool Shard::has_dead_variants() const { + return loaded() && !live.all; +} + +bool Shard::row_live(bool occurrence, std::uint32_t row) const { + if(live.all) { + return true; + } + auto root = root_of(*buffer); + auto columns = occurrence ? occ_columns(root) : rel_columns(root); + switch(tier_of(to_array_ref(root[&ShardBlob::variants]).size())) { + case MaskTier::Single: { + return (live.bits & 1) != 0; + } + case MaskTier::U32: { + return (columns.masks32[row] & static_cast(live.bits)) != 0; + } + case MaskTier::U64: { + return (columns.masks64[row] & live.bits) != 0; + } + case MaskTier::Roaring: { + return read_row_bitmap(columns, row).intersect(live.big); + } + } + std::unreachable(); +} + +void Shard::lookup(std::uint32_t offset, + llvm::function_ref callback) const { + if(!buffer) { + return; + } + auto root = root_of(*buffer); + auto columns = occ_columns(root); + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + + // Binary search the first row whose end reaches the offset, then walk + // while rows contain it. Occurrence ranges are name-token spans, + // pairwise disjoint or identical, so under (begin, end) order the end + // column is monotonic too. + std::size_t lo = 0; + std::size_t hi = columns.begins.size(); + while(lo < hi) { + auto mid = lo + (hi - lo) / 2; + if(columns.end_of(mid) < offset) { + lo = mid + 1; + } else { + hi = mid; + } + } + + for(; lo < columns.begins.size(); lo += 1) { + auto row = static_cast(lo); + LocalSourceRange range{columns.begins[row], columns.end_of(row)}; + if(!range.contains(offset)) { + break; + } + if(!row_live(true, row)) { + continue; + } + Occurrence result{range, sym_hashes[occ_sym_id(root, row)]}; + if(!callback(result)) { + break; + } + } +} + +void Shard::lookup(SymbolHash symbol, + RelationKind kind, + llvm::function_ref callback) const { + if(!buffer) { + return; + } + auto root = root_of(*buffer); + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + auto it = std::ranges::lower_bound(sym_hashes, symbol); + if(it == sym_hashes.end() || *it != symbol) [[unlikely]] { + return; + } + auto id = static_cast(it - sym_hashes.begin()); + + auto offsets = to_array_ref(root[&ShardBlob::sym_rel_offsets]); + auto begin_row = offsets[id]; + auto end_row = offsets[id + 1]; + + auto columns = rel_columns(root); + auto kinds = to_array_ref(root[&ShardBlob::rel_kinds]); + auto sym_rows = to_array_ref(root[&ShardBlob::rel_sym_rows]); + auto sym16 = to_array_ref(root[&ShardBlob::rel_sym16]); + auto sym32 = to_array_ref(root[&ShardBlob::rel_sym32]); + auto def_rows = to_array_ref(root[&ShardBlob::rel_def_rows]); + auto def_begins = to_array_ref(root[&ShardBlob::rel_def_begins]); + auto def_ends = to_array_ref(root[&ShardBlob::rel_def_ends]); + + auto sym_cursor = std::ranges::lower_bound(sym_rows, begin_row) - sym_rows.begin(); + auto def_cursor = std::ranges::lower_bound(def_rows, begin_row) - def_rows.begin(); + + for(auto row = begin_row; row < end_row; row += 1) { + while(sym_cursor < static_cast(sym_rows.size()) && + sym_rows[sym_cursor] < row) { + sym_cursor += 1; + } + while(def_cursor < static_cast(def_rows.size()) && + def_rows[def_cursor] < row) { + def_cursor += 1; + } + + auto row_kind = static_cast(kinds[row]); + if(!(RelationKind(row_kind) & kind)) { + continue; + } + if(!row_live(false, row)) { + continue; + } + + Relation relation{ + .kind = row_kind, + .range = {columns.begins[row], columns.end_of(row)}, + .target_symbol = 0, + }; + if(def_cursor < static_cast(def_rows.size()) && + def_rows[def_cursor] == row) { + relation.set_definition_range({def_begins[def_cursor], def_ends[def_cursor]}); + } else if(sym_cursor < static_cast(sym_rows.size()) && + sym_rows[sym_cursor] == row) { + auto payload = sym16.empty() ? sym32[sym_cursor] : sym16[sym_cursor]; + relation.target_symbol = sym_hashes[payload]; + } + + if(!callback(relation)) { + break; + } + } +} + +bool Shard::find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const { + if(!buffer) { + return false; + } + auto root = root_of(*buffer); + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + auto it = std::ranges::lower_bound(sym_hashes, hash); + if(it == sym_hashes.end() || *it != hash) { + return false; + } + auto id = static_cast(it - sym_hashes.begin()); + + auto local_syms = to_array_ref(root[&ShardBlob::local_syms]); + auto local_it = std::ranges::lower_bound(local_syms, id); + if(local_it == local_syms.end() || *local_it != id) { + return false; + } + auto local = local_it - local_syms.begin(); + name = std::string(root[&ShardBlob::local_names].at(local)); + kind = SymbolKind(to_array_ref(root[&ShardBlob::local_kinds])[local]); + return true; +} + +llvm::StringRef Shard::content() const { + if(!buffer) { + return {}; + } + return to_ref(root_of(*buffer)[&ShardBlob::content]); +} + +std::span Shard::line_starts() const { + if(!buffer) { + return {}; + } + if(line_starts_cache.empty()) { + line_starts_cache = kota::ipc::lsp::build_line_starts(content()); + } + return line_starts_cache; +} + +namespace { + +/// Working row forms during a write; masks stay wide, the tier is chosen +/// at emit time from the final variant count. +template +struct OccRow { + std::uint32_t begin; + std::uint32_t end; + std::uint64_t sym; + MaskT mask; +}; + +template +struct RelRow { + std::uint8_t kind; + std::uint32_t begin; + std::uint32_t end; + /// Raw Relation::target_symbol bits; whether they mean a definition + /// range, a symbol hash or nothing is decided by `kind` (mirroring the + /// in-memory encoding). + std::uint64_t payload; + MaskT mask; +}; + +template +MaskT single_bit(std::uint32_t id) { + if constexpr(std::same_as) { + return std::uint64_t(1) << id; + } else { + MaskT mask; + mask.add(id); + return mask; + } +} + +template +bool mask_empty(const MaskT& mask) { + if constexpr(std::same_as) { + return mask == 0; + } else { + return mask.isEmpty(); + } +} + +template +void mask_or(MaskT& into, const MaskT& from) { + into |= from; +} + +/// The old blob's mask of one row, remapped through old-id -> new-id (a +/// dropped variant's bit vanishes; an all-dropped row reads as empty and +/// is skipped by the caller). +template +MaskT remap_mask(ShardView root, + const RowColumns& columns, + std::uint32_t row, + llvm::ArrayRef id_map) { + MaskT result{}; + auto apply = [&](std::uint32_t old_id) { + if(id_map[old_id] >= 0) { + mask_or(result, single_bit(static_cast(id_map[old_id]))); + } + }; + switch(tier_of(to_array_ref(root[&ShardBlob::variants]).size())) { + case MaskTier::Single: { + apply(0); + break; + } + case MaskTier::U32: { + auto bits = columns.masks32[row]; + while(bits != 0) { + auto id = static_cast(std::countr_zero(bits)); + apply(id); + bits &= bits - 1; + } + break; + } + case MaskTier::U64: { + auto bits = columns.masks64[row]; + while(bits != 0) { + auto id = static_cast(std::countr_zero(bits)); + apply(id); + bits &= bits - 1; + } + break; + } + case MaskTier::Roaring: { + for(auto id: read_row_bitmap(columns, row)) { + apply(id); + } + break; + } + } + return result; +} + +/// Everything write_shard accumulates before choosing column tiers. +template +struct MergedRows { + std::vector> occurrences; + /// Groups sorted by symbol hash; rows sorted by (kind, begin, end, + /// payload) within each. + std::vector>>> relations; +}; + +/// Two-way merge of runs sorted under `key`; rows with equal keys are one +/// row and OR their masks — the cross-variant dedup. +template +void merge_sorted(std::vector old_rows, + std::vector fresh_rows, + Key key, + std::vector& out) { + out.reserve(old_rows.size() + fresh_rows.size()); + auto lhs = old_rows.begin(); + auto rhs = fresh_rows.begin(); + while(lhs != old_rows.end() || rhs != fresh_rows.end()) { + if(rhs == fresh_rows.end() || (lhs != old_rows.end() && key(*lhs) < key(*rhs))) { + out.push_back(std::move(*lhs)); + lhs += 1; + } else if(lhs == old_rows.end() || key(*rhs) < key(*lhs)) { + out.push_back(std::move(*rhs)); + rhs += 1; + } else { + mask_or(lhs->mask, rhs->mask); + out.push_back(std::move(*lhs)); + lhs += 1; + rhs += 1; + } + } +} + +template +void merge_occurrences(ShardView old_root, + llvm::ArrayRef id_map, + const VariantInput& fresh, + std::int64_t fresh_id, + std::vector>& out) { + std::vector> old_rows; + if(old_root.valid()) { + auto columns = occ_columns(old_root); + auto sym_hashes = to_array_ref(old_root[&ShardBlob::sym_hashes]); + old_rows.reserve(columns.begins.size()); + for(std::uint32_t row = 0; row < columns.begins.size(); row += 1) { + auto mask = remap_mask(old_root, columns, row, id_map); + if(mask_empty(mask)) { + continue; + } + old_rows.push_back({columns.begins[row], + columns.end_of(row), + sym_hashes[occ_sym_id(old_root, row)], + std::move(mask)}); + } + } + + std::vector> fresh_rows; + if(fresh.rows) { + fresh_rows.reserve(fresh.rows->occurrences.size()); + auto bit = single_bit(static_cast(fresh_id)); + for(auto& occurrence: fresh.rows->occurrences) { + fresh_rows.push_back( + {occurrence.range.begin, occurrence.range.end, occurrence.target, bit}); + } + std::ranges::sort(fresh_rows, [](const auto& lhs, const auto& rhs) { + return std::tuple(lhs.begin, lhs.end, lhs.sym) < + std::tuple(rhs.begin, rhs.end, rhs.sym); + }); + } + + merge_sorted( + std::move(old_rows), + std::move(fresh_rows), + [](const auto& row) { return std::tuple(row.begin, row.end, row.sym); }, + out); +} + +template +std::vector> decode_relation_group(ShardView root, + const RowColumns& columns, + std::uint32_t begin_row, + std::uint32_t end_row, + llvm::ArrayRef id_map) { + auto sym_hashes = to_array_ref(root[&ShardBlob::sym_hashes]); + auto kinds = to_array_ref(root[&ShardBlob::rel_kinds]); + auto sym_rows = to_array_ref(root[&ShardBlob::rel_sym_rows]); + auto sym16 = to_array_ref(root[&ShardBlob::rel_sym16]); + auto sym32 = to_array_ref(root[&ShardBlob::rel_sym32]); + auto def_rows = to_array_ref(root[&ShardBlob::rel_def_rows]); + auto def_begins = to_array_ref(root[&ShardBlob::rel_def_begins]); + auto def_ends = to_array_ref(root[&ShardBlob::rel_def_ends]); + + auto sym_cursor = std::ranges::lower_bound(sym_rows, begin_row) - sym_rows.begin(); + auto def_cursor = std::ranges::lower_bound(def_rows, begin_row) - def_rows.begin(); + + std::vector> rows; + rows.reserve(end_row - begin_row); + for(auto row = begin_row; row < end_row; row += 1) { + while(sym_cursor < static_cast(sym_rows.size()) && + sym_rows[sym_cursor] < row) { + sym_cursor += 1; + } + while(def_cursor < static_cast(def_rows.size()) && + def_rows[def_cursor] < row) { + def_cursor += 1; + } + + auto mask = remap_mask(root, columns, row, id_map); + if(mask_empty(mask)) { + continue; + } + + std::uint64_t payload = 0; + if(def_cursor < static_cast(def_rows.size()) && + def_rows[def_cursor] == row) { + payload = std::bit_cast( + LocalSourceRange{def_begins[def_cursor], def_ends[def_cursor]}); + } else if(sym_cursor < static_cast(sym_rows.size()) && + sym_rows[sym_cursor] == row) { + auto id = sym16.empty() ? sym32[sym_cursor] : sym16[sym_cursor]; + payload = sym_hashes[id]; + } + + rows.push_back( + {kinds[row], columns.begins[row], columns.end_of(row), payload, std::move(mask)}); + } + return rows; +} + +template +void merge_relation_rows(std::vector> old_rows, + std::vector> fresh_rows, + std::vector>& out) { + merge_sorted( + std::move(old_rows), + std::move(fresh_rows), + [](const auto& row) { return std::tuple(row.kind, row.begin, row.end, row.payload); }, + out); +} + +template +void merge_relations(ShardView old_root, + llvm::ArrayRef id_map, + const VariantInput& fresh, + std::int64_t fresh_id, + std::vector>>>& out) { + // Fresh groups, sorted by symbol hash; rows in a group follow the + // builder's canonical (kind, begin, end, payload) order already, but a + // hand-built FileIndex (tests) may not — sort defensively, it is cheap + // relative to the merge. + std::vector>>> fresh_groups; + if(fresh.rows) { + auto bit = single_bit(static_cast(fresh_id)); + fresh_groups.reserve(fresh.rows->relations.size()); + for(auto& [hash, relations]: fresh.rows->relations) { + std::vector> rows; + rows.reserve(relations.size()); + for(auto& relation: relations) { + rows.push_back({static_cast(relation.kind), + relation.range.begin, + relation.range.end, + relation.target_symbol, + bit}); + } + std::ranges::sort(rows, [](const auto& lhs, const auto& rhs) { + return std::tuple(lhs.kind, lhs.begin, lhs.end, lhs.payload) < + std::tuple(rhs.kind, rhs.begin, rhs.end, rhs.payload); + }); + fresh_groups.emplace_back(hash, std::move(rows)); + } + std::ranges::sort(fresh_groups, {}, [](const auto& group) { return group.first; }); + } + + struct OldGroup { + std::uint64_t hash; + std::uint32_t begin_row; + std::uint32_t end_row; + }; + + std::vector old_groups; + RowColumns old_columns; + if(old_root.valid()) { + old_columns = rel_columns(old_root); + auto sym_hashes = to_array_ref(old_root[&ShardBlob::sym_hashes]); + auto offsets = to_array_ref(old_root[&ShardBlob::sym_rel_offsets]); + for(std::uint32_t id = 0; id < sym_hashes.size(); id += 1) { + if(offsets[id] != offsets[id + 1]) { + old_groups.push_back({sym_hashes[id], offsets[id], offsets[id + 1]}); + } + } + } + + auto lhs = old_groups.begin(); + auto rhs = fresh_groups.begin(); + while(lhs != old_groups.end() || rhs != fresh_groups.end()) { + if(rhs == fresh_groups.end() || (lhs != old_groups.end() && lhs->hash < rhs->first)) { + auto rows = decode_relation_group(old_root, + old_columns, + lhs->begin_row, + lhs->end_row, + id_map); + if(!rows.empty()) { + out.emplace_back(lhs->hash, std::move(rows)); + } + lhs += 1; + } else if(lhs == old_groups.end() || rhs->first < lhs->hash) { + out.emplace_back(rhs->first, std::move(rhs->second)); + rhs += 1; + } else { + auto old_rows = decode_relation_group(old_root, + old_columns, + lhs->begin_row, + lhs->end_row, + id_map); + std::vector> merged; + merge_relation_rows(std::move(old_rows), std::move(rhs->second), merged); + if(!merged.empty()) { + out.emplace_back(lhs->hash, std::move(merged)); + } + lhs += 1; + rhs += 1; + } + } +} + +template +void emit_mask(ShardBlob& blob, bool occurrence, MaskTier tier, const MaskT& mask) { + auto& masks32 = occurrence ? blob.occ_masks32 : blob.rel_masks32; + auto& masks64 = occurrence ? blob.occ_masks64 : blob.rel_masks64; + auto& roaring_offsets = occurrence ? blob.occ_roaring_offsets : blob.rel_roaring_offsets; + auto& roaring = occurrence ? blob.occ_roaring : blob.rel_roaring; + + switch(tier) { + case MaskTier::Single: { + break; + } + case MaskTier::U32: { + if constexpr(std::same_as) { + masks32.push_back(static_cast(mask)); + } + break; + } + case MaskTier::U64: { + if constexpr(std::same_as) { + masks64.push_back(mask); + } + break; + } + case MaskTier::Roaring: { + if constexpr(std::same_as) { + auto size = mask.getSizeInBytes(true); + auto offset = roaring.size(); + roaring.resize(offset + size); + mask.write(reinterpret_cast(roaring.data() + offset), true); + roaring_offsets.push_back(static_cast(offset)); + } + break; + } + } +} + +void emit_range(std::vector& lengths, + std::vector& long_rows, + std::vector& long_ends, + std::uint32_t row, + std::uint32_t begin, + std::uint32_t end) { + auto length = end - begin; + if(length >= length_escape) { + lengths.push_back(length_escape); + long_rows.push_back(row); + long_ends.push_back(end); + } else { + lengths.push_back(static_cast(length)); + } +} + +template +void write_shard_impl(ShardView old_root, + llvm::ArrayRef id_map, + const VariantInput& fresh, + std::int64_t fresh_id, + std::vector variants, + llvm::StringRef content, + std::uint64_t content_hash, + llvm::raw_ostream& os) { + MergedRows merged; + merge_occurrences(old_root, id_map, fresh, fresh_id, merged.occurrences); + merge_relations(old_root, id_map, fresh, fresh_id, merged.relations); + + // The symbol table covers exactly what the merged rows reference: + // occurrence targets, relation group keys, and symbol payloads. + llvm::DenseSet referenced; + for(auto& row: merged.occurrences) { + referenced.insert(row.sym); + } + for(auto& [hash, rows]: merged.relations) { + referenced.insert(hash); + for(auto& row: rows) { + if(row.payload != 0 && + !RelationKind(static_cast(row.kind)).isDeclOrDef()) { + referenced.insert(row.payload); + } + } + } + + ShardBlob blob; + blob.format_version = index_format_version; + blob.content_hash = content_hash; + blob.content = content.str(); + blob.variants = std::move(variants); + + blob.sym_hashes.assign(referenced.begin(), referenced.end()); + std::ranges::sort(blob.sym_hashes); + auto sym_id = [&](std::uint64_t hash) { + return static_cast(std::ranges::lower_bound(blob.sym_hashes, hash) - + blob.sym_hashes.begin()); + }; + + // Local symbol names: survivors from the old blob, plus the fresh + // variant's non-External symbols its rows referenced. External names + // live in the ProjectIndex and are never stored here. + struct LocalInfo { + std::string name; + std::uint8_t kind; + std::uint8_t scope; + }; + + llvm::DenseMap locals; + if(old_root.valid()) { + auto old_sym_hashes = to_array_ref(old_root[&ShardBlob::sym_hashes]); + auto old_local_syms = to_array_ref(old_root[&ShardBlob::local_syms]); + auto old_kinds = to_array_ref(old_root[&ShardBlob::local_kinds]); + auto old_scopes = to_array_ref(old_root[&ShardBlob::local_scopes]); + auto old_names = old_root[&ShardBlob::local_names]; + for(std::uint32_t k = 0; k < old_local_syms.size(); k += 1) { + auto hash = old_sym_hashes[old_local_syms[k]]; + if(referenced.contains(hash)) { + locals.try_emplace( + hash, + LocalInfo{std::string(old_names.at(k)), old_kinds[k], old_scopes[k]}); + } + } + } + if(fresh.symbols) { + for(auto hash: referenced) { + auto found = fresh.symbols(hash); + if(!found || found->scope == SymbolScope::External) { + continue; + } + locals.try_emplace(hash, + LocalInfo{std::string(found->name), + found->kind.value(), + static_cast(found->scope)}); + } + } + + llvm::SmallVector> sorted_locals; + sorted_locals.reserve(locals.size()); + for(auto& [hash, info]: locals) { + sorted_locals.emplace_back(sym_id(hash), &info); + } + std::ranges::sort(sorted_locals, {}, [](const auto& entry) { return entry.first; }); + for(auto& [id, info]: sorted_locals) { + blob.local_syms.push_back(id); + blob.local_names.push_back(info->name); + blob.local_kinds.push_back(info->kind); + blob.local_scopes.push_back(info->scope); + } + + auto tier = tier_of(blob.variants.size()); + bool wide_syms = blob.sym_hashes.size() > 0xffff; + + for(std::uint32_t row = 0; row < merged.occurrences.size(); row += 1) { + auto& occurrence = merged.occurrences[row]; + blob.occ_begins.push_back(occurrence.begin); + emit_range(blob.occ_lengths, + blob.occ_long_rows, + blob.occ_long_ends, + row, + occurrence.begin, + occurrence.end); + auto id = sym_id(occurrence.sym); + if(wide_syms) { + blob.occ_syms32.push_back(id); + } else { + blob.occ_syms16.push_back(static_cast(id)); + } + emit_mask(blob, true, tier, occurrence.mask); + } + if(tier == MaskTier::Roaring) { + blob.occ_roaring_offsets.push_back(static_cast(blob.occ_roaring.size())); + } + + // Relation groups follow symbol-table order; a symbol with occurrences + // only gets an empty slice. + blob.sym_rel_offsets.reserve(blob.sym_hashes.size() + 1); + auto group = merged.relations.begin(); + std::uint32_t rel_row = 0; + for(auto hash: blob.sym_hashes) { + blob.sym_rel_offsets.push_back(rel_row); + if(group == merged.relations.end() || group->first != hash) { + continue; + } + for(auto& row: group->second) { + blob.rel_kinds.push_back(row.kind); + blob.rel_begins.push_back(row.begin); + emit_range(blob.rel_lengths, + blob.rel_long_rows, + blob.rel_long_ends, + rel_row, + row.begin, + row.end); + if(row.payload != 0) { + if(RelationKind(static_cast(row.kind)).isDeclOrDef()) { + auto range = std::bit_cast(row.payload); + blob.rel_def_rows.push_back(rel_row); + blob.rel_def_begins.push_back(range.begin); + blob.rel_def_ends.push_back(range.end); + } else { + auto id = sym_id(row.payload); + blob.rel_sym_rows.push_back(rel_row); + if(wide_syms) { + blob.rel_sym32.push_back(id); + } else { + blob.rel_sym16.push_back(static_cast(id)); + } + } + } + emit_mask(blob, false, tier, row.mask); + rel_row += 1; + } + group += 1; + } + blob.sym_rel_offsets.push_back(rel_row); + if(tier == MaskTier::Roaring) { + blob.rel_roaring_offsets.push_back(static_cast(blob.rel_roaring.size())); + } + + serialize_blob(blob, os); +} + +} // namespace + +void write_shard(const Shard& old, + llvm::ArrayRef keep, + const VariantInput& fresh, + llvm::StringRef content, + std::uint64_t content_hash, + llvm::raw_ostream& os) { + ShardView old_root; + std::vector old_variants = old.variants(); + + // old-id -> new-id; -1 drops the variant. + llvm::SmallVector id_map(old_variants.size(), -1); + std::vector variants; + for(std::uint32_t id = 0; id < old_variants.size(); id += 1) { + if(llvm::is_contained(keep, old_variants[id])) { + id_map[id] = static_cast(variants.size()); + variants.push_back(old_variants[id]); + } + } + if(!variants.empty()) { + old_root = root_of(*old.buffer); + } + + std::int64_t fresh_id = -1; + if(fresh.rows) { + assert(!llvm::is_contained(variants, fresh.hash) && + "a variant already stored must not be re-appended"); + fresh_id = static_cast(variants.size()); + variants.push_back(fresh.hash); + } + assert(!variants.empty() && "a shard blob holds at least one variant"); + + if(variants.size() <= 64) { + write_shard_impl(old_root, + id_map, + fresh, + fresh_id, + std::move(variants), + content, + content_hash, + os); + } else { + write_shard_impl(old_root, + id_map, + fresh, + fresh_id, + std::move(variants), + content, + content_hash, + os); + } +} + +} // namespace clice::index diff --git a/src/index/shard.h b/src/index/shard.h new file mode 100644 index 000000000..e151c3392 --- /dev/null +++ b/src/index/shard.h @@ -0,0 +1,215 @@ +#pragma once + +#include +#include +#include +#include +#include +#include + +#include "index/tu_index.h" +#include "support/bitmap.h" + +#include "llvm/ADT/ArrayRef.h" +#include "llvm/Support/MemoryBuffer.h" +#include "llvm/Support/raw_ostream.h" + +namespace clice::index { + +/// Identity of one preprocessing variant of a file: xxh3 over the file's +/// rows in canonical order (see FileIndex::rows_hash). Two compilation +/// contexts whose preprocessing of the file agrees produce the same value +/// and share one stored variant. +using RowsHash = std::uint64_t; + +/// A file's persisted index rows: every variant produced from one content +/// generation, merged so that a row shared by several variants is stored +/// once. The blob knows nothing about which TU contributed which variant — +/// that mapping is global state (ProjectIndex) — so re-indexing a TU whose +/// rows are unchanged never touches the blob. +/// +/// Columnar layout, chosen at serialize time from known bounds and +/// self-described by which columns are populated: +/// - symbol ids: u16 when the table has at most 65535 entries, else u32 +/// - row ranges: begin u32 + length u8, lengths >= 255 escape to a +/// sparse (row, end) table +/// - variant masks: absent when there is one variant, u32 up to 32, +/// u64 up to 64, concatenated roaring bitmaps beyond +/// +/// Relations are grouped by symbol: `sym_rel_offsets` slices the relation +/// columns per symbol-table entry, so a per-symbol lookup is a binary +/// search plus a slice walk. A relation's payload (`Relation::target_symbol`) +/// is sparse by class: a definition range for decl/def rows that recorded +/// one, a symbol reference for symbol-pair rows, nothing otherwise. +struct ShardBlob { + /// Persisted-blob schema version (index_format_version), stamped by the + /// writer and gated by Shard::from_bytes. + std::uint32_t format_version = 0; + + /// xxh3 of `content`: the content generation these rows were built + /// from. A variant produced from different bytes of the file starts a + /// new blob instead of merging in — offsets from two generations must + /// never share row storage. + std::uint64_t content_hash = 0; + + /// The file's text, for position mapping. Line starts are derived at + /// load time, never persisted. + std::string content; + + /// Variant id (mask bit position) -> rows hash. + std::vector variants; + + /// Referenced symbols, sorted by hash; the index into this table is the + /// symbol id the row columns use. + std::vector sym_hashes; + + /// Relation-column slice per symbol: entry i's relations occupy rows + /// [sym_rel_offsets[i], sym_rel_offsets[i + 1]). Size is table size + 1. + std::vector sym_rel_offsets; + + /// Symbols local to this file (FileLocal) or its TU (TULocal), whose + /// names live nowhere else: sparse over the symbol table, ascending. + std::vector local_syms; + std::vector local_names; + std::vector local_kinds; + std::vector local_scopes; + + /// Occurrences sorted by (begin, end, symbol hash). + std::vector occ_begins; + std::vector occ_lengths; + std::vector occ_long_rows; + std::vector occ_long_ends; + std::vector occ_syms16; + std::vector occ_syms32; + std::vector occ_masks32; + std::vector occ_masks64; + std::vector occ_roaring_offsets; + std::vector occ_roaring; + + /// Relations in symbol-table order, sorted by (kind, begin, end) within + /// each group. + std::vector rel_kinds; + std::vector rel_begins; + std::vector rel_lengths; + std::vector rel_long_rows; + std::vector rel_long_ends; + std::vector rel_sym_rows; + std::vector rel_sym16; + std::vector rel_sym32; + std::vector rel_def_rows; + std::vector rel_def_begins; + std::vector rel_def_ends; + std::vector rel_masks32; + std::vector rel_masks64; + std::vector rel_roaring_offsets; + std::vector rel_roaring; +}; + +/// A variant to write into a shard blob: the rows plus a resolver for the +/// symbols they reference. Only non-External entries land in the blob's +/// local-name table (External names live in the ProjectIndex); returned +/// name refs must stay valid for the duration of the write call. +struct VariantInput { + RowsHash hash = 0; + const FileIndex* rows = nullptr; + llvm::function_ref(SymbolHash)> symbols; +}; + +/// Zero-copy reader over a shard blob, plus the live-variant mask the +/// indexer maintains: a variant whose last contributing TU was removed or +/// replaced stops serving immediately, and its rows are erased for real by +/// the next write_shard covering the blob. +class Shard { +public: + Shard() = default; + + /// Wrap verified blob bytes without owning them (the caller keeps the + /// bytes alive). Returns an empty shard when verification fails or the + /// format version differs. + static Shard from_bytes(llvm::StringRef data); + + /// Adopt an owning buffer of blob bytes (storage reads, freshly + /// written blobs). A null buffer, corrupt bytes or a different format + /// version load as an empty shard; the caller treats that as "not on + /// disk". + static Shard from_buffer(std::unique_ptr buffer); + + /// Whether this shard holds a blob. + bool loaded() const { + return buffer != nullptr; + } + + /// The serialized blob bytes backing this shard (what save persists). + llvm::StringRef bytes() const { + return buffer ? buffer->getBuffer() : llvm::StringRef(); + } + + std::uint64_t content_hash() const; + + /// All variants stored in the blob, in variant-id order. + std::vector variants() const; + + bool has_variant(RowsHash hash) const; + + /// Restrict queries to the given variants (the file's live + /// contributions). Hashes the blob does not store are ignored. + void set_live(llvm::ArrayRef live); + + /// Whether any stored variant was masked out by set_live — the signal + /// that the next serialization of this file should compact. + bool has_dead_variants() const; + + void lookup(std::uint32_t offset, llvm::function_ref callback) const; + + void lookup(SymbolHash symbol, + RelationKind kind, + llvm::function_ref callback) const; + + /// Look up a local symbol's name and kind. + bool find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const; + + llvm::StringRef content() const; + + /// Line start offsets for position mapping, derived from the content on + /// first use. + std::span line_starts() const; + +private: + explicit Shard(std::unique_ptr buffer); + + friend void write_shard(const Shard& old, + llvm::ArrayRef keep, + const VariantInput& fresh, + llvm::StringRef content, + std::uint64_t content_hash, + llvm::raw_ostream& os); + + struct Live { + /// Fast path: every stored variant is live, no per-row filtering. + bool all = true; + std::uint64_t bits = 0; + Bitmap big; + }; + + bool row_live(bool occurrence, std::uint32_t row) const; + + std::unique_ptr buffer; + Live live; + /// Lazily derived from the blob's content; content is immutable for the + /// shard's lifetime, so the cache never invalidates. + mutable std::vector line_starts_cache; +}; + +/// Write a shard blob for one content generation: the variants of `old` +/// whose hash is in `keep` (in stored order), then `fresh` if its rows are +/// non-null. Rows shared between variants merge; masks are re-encoded for +/// the surviving variant set. `old` may be an empty shard (fresh build) and +/// `fresh.rows` may be null (pure compaction). +void write_shard(const Shard& old, + llvm::ArrayRef keep, + const VariantInput& fresh, + llvm::StringRef content, + std::uint64_t content_hash, + llvm::raw_ostream& os); + +} // namespace clice::index diff --git a/src/index/shared.h b/src/index/shared.h deleted file mode 100644 index a943671b9..000000000 --- a/src/index/shared.h +++ /dev/null @@ -1,17 +0,0 @@ -#pragma once - -#include "llvm/ADT/DenseMap.h" -#include "clang/Basic/SourceLocation.h" - -namespace clice { - -class CompilationUnitRef; - -} - -namespace clice::index { - -template -using Shared = llvm::DenseMap; - -} diff --git a/src/index/storage.cpp b/src/index/storage.cpp new file mode 100644 index 000000000..fd52293c4 --- /dev/null +++ b/src/index/storage.cpp @@ -0,0 +1,105 @@ +#include "index/storage.h" + +#include "support/cache_store.h" +#include "support/logging.h" + +#include "llvm/Support/raw_ostream.h" + +namespace clice::index { + +namespace { + +llvm::StringRef namespace_of(IndexBlobKind kind) { + switch(kind) { + case IndexBlobKind::Shard: return "index"; + case IndexBlobKind::Manifest: return "index-manifest"; + case IndexBlobKind::Global: return "index-global"; + } + std::unreachable(); +} + +class FsIndexStorage final : public IndexStorage { +public: + explicit FsIndexStorage(CacheStore& store) : store(store) { + for(auto kind: {IndexBlobKind::Shard, IndexBlobKind::Manifest, IndexBlobKind::Global}) { + store.register_namespace({ + .name = std::string(namespace_of(kind)), + .extension = ".idx", + .policy = CachePolicy::Persistent, + }); + } + } + + std::unique_ptr read(IndexBlobKind kind, llvm::StringRef key) override { + auto path = store.lookup(namespace_of(kind), key); + if(!path) { + return nullptr; + } + auto buffer = llvm::MemoryBuffer::getFile(*path); + if(!buffer) { + return nullptr; + } + return std::move(*buffer); + } + + bool contains(IndexBlobKind kind, llvm::StringRef key) override { + return store.lookup(namespace_of(kind), key).has_value(); + } + + llvm::SmallVector write(llvm::ArrayRef batch) override { + llvm::SmallVector failed; + for(std::size_t i = 0; i < batch.size(); i += 1) { + auto& blob = batch[i]; + auto ns = namespace_of(blob.kind); + auto pending = store.begin_store(ns, blob.key); + std::error_code ec; + llvm::raw_fd_ostream os(pending.tmp_path, ec); + if(ec) { + LOG_WARN("Failed to write index blob {}/{}: {}", ns, blob.key, ec.message()); + failed.push_back(i); + continue; + } + os.write(blob.bytes.data(), blob.bytes.size()); + os.close(); + // A truncated blob (disk full) must never be committed: the + // namespaces are Persistent, so it would be served forever. + if(os.has_error()) { + LOG_WARN("Failed to write index blob {}/{}: {}", + ns, + blob.key, + os.error().message()); + os.clear_error(); + failed.push_back(i); + continue; + } + if(auto committed = store.commit(std::move(pending)); !committed) { + LOG_WARN("Failed to commit index blob {}/{}: {}", + ns, + blob.key, + committed.error().message()); + failed.push_back(i); + continue; + } + } + return failed; + } + + void remove(IndexBlobKind kind, llvm::StringRef key) override { + store.invalidate(namespace_of(kind), key); + } + + void for_each_key(IndexBlobKind kind, llvm::function_ref fn) override { + store.for_each_key(namespace_of(kind), fn); + } + +private: + CacheStore& store; +}; + +} // namespace + +std::unique_ptr make_fs_index_storage(CacheStore& store) { + return std::make_unique(store); +} + +} // namespace clice::index diff --git a/src/index/storage.h b/src/index/storage.h new file mode 100644 index 000000000..d4286c0f4 --- /dev/null +++ b/src/index/storage.h @@ -0,0 +1,72 @@ +#pragma once + +#include +#include +#include +#include + +#include "llvm/ADT/ArrayRef.h" +#include "llvm/ADT/STLFunctionalExtras.h" +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringRef.h" +#include "llvm/Support/MemoryBuffer.h" + +namespace clice { + +class CacheStore; + +} + +namespace clice::index { + +/// The three blob families the index persists. +enum class IndexBlobKind : std::uint8_t { + /// Per-file row blobs (ShardBlob), keyed by a path hash. + Shard, + /// Per-TU manifests, keyed by a path hash. + Manifest, + /// The single global blob (FileVersion table + symbols), key "global". + Global, +}; + +/// Storage backend for index blobs. The filesystem implementation below is +/// the default; a database-backed one plugs in behind the same interface. +/// +/// All methods are thread-safe. `write` does the heavy IO (fsync) and +/// belongs off the event loop; reads are cheap. +class IndexStorage { +public: + virtual ~IndexStorage() = default; + + /// The blob's bytes (memory-mapped where the backend allows), or + /// nullptr when missing or unreadable. + virtual std::unique_ptr read(IndexBlobKind kind, llvm::StringRef key) = 0; + + /// Whether a blob exists under the key, even when unreadable — how the + /// loader tells a missing global blob (sweep everything) from a + /// transient read failure (touch nothing). + virtual bool contains(IndexBlobKind kind, llvm::StringRef key) = 0; + + struct Blob { + IndexBlobKind kind; + std::string key; + std::string bytes; + }; + + /// Persist a batch in order. Atomicity is per blob, not per batch — a + /// crash can land a prefix; every load path treats any partially + /// written combination as stale data to rebuild. An entry that fails + /// to persist is logged and skipped, and its batch index returned so + /// the caller can re-dirty it for a later save. + virtual llvm::SmallVector write(llvm::ArrayRef batch) = 0; + + virtual void remove(IndexBlobKind kind, llvm::StringRef key) = 0; + + virtual void for_each_key(IndexBlobKind kind, llvm::function_ref fn) = 0; +}; + +/// Filesystem implementation over the cache store; registers the index +/// namespaces on construction. +std::unique_ptr make_fs_index_storage(CacheStore& store); + +} // namespace clice::index diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index ad31a9aa6..387d3647e 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -12,7 +12,6 @@ #include "support/logging.h" #include "support/timer.h" -#include "llvm/Support/SHA256.h" #include "llvm/Support/xxhash.h" #include "clang/AST/DeclCXX.h" @@ -610,33 +609,42 @@ void FileIndex::lookup(SymbolHash symbol, } } -std::array FileIndex::hash() { - llvm::SHA256 hasher; - - using u8 = std::uint8_t; - - if(!occurrences.empty()) { - static_assert(sizeof(Occurrence) == sizeof(Range) + sizeof(SymbolHash)); - static_assert(sizeof(Occurrence) % 8 == 0); - auto data = reinterpret_cast(occurrences.data()); - auto size = occurrences.size() * sizeof(Occurrence); - hasher.update(llvm::ArrayRef(data, size)); +std::uint64_t FileIndex::rows_hash() const { + static_assert(sizeof(Occurrence) == sizeof(Range) + sizeof(SymbolHash)); + static_assert(sizeof(Relation) == + sizeof(RelationKind) + 4 + sizeof(Range) + sizeof(SymbolHash)); + + // One flat buffer in a deterministic order: the sorted occurrences, + // then each relation group in ascending symbol order (DenseMap + // iteration order must never leak into the hash). + std::vector buffer; + std::size_t size = occurrences.size() * sizeof(Occurrence); + for(auto& [symbol, group]: relations) { + size += sizeof(symbol) + group.size() * sizeof(Relation); } + buffer.reserve(size); - for(auto& [symbol_id, relations]: relations) { - hasher.update(std::bit_cast>(symbol_id)); - static_assert(sizeof(Relation) == - sizeof(RelationKind) + 4 + sizeof(Range) + sizeof(SymbolHash)); - static_assert(sizeof(Relation) % 8 == 0); + auto append = [&](const void* data, std::size_t bytes) { + auto* raw = static_cast(data); + buffer.insert(buffer.end(), raw, raw + bytes); + }; - if(!relations.empty()) { - auto data = reinterpret_cast(relations.data()); - auto size = relations.size() * sizeof(Relation); - hasher.update(llvm::ArrayRef(data, size)); - } + append(occurrences.data(), occurrences.size() * sizeof(Occurrence)); + + llvm::SmallVector keys; + keys.reserve(relations.size()); + for(auto symbol: llvm::make_first_range(relations)) { + keys.push_back(symbol); + } + llvm::sort(keys); + for(auto symbol: keys) { + append(&symbol, sizeof(symbol)); + auto& group = relations.find(symbol)->second; + append(group.data(), group.size() * sizeof(Relation)); } - return hasher.final(); + return llvm::xxh3_64bits( + llvm::StringRef(reinterpret_cast(buffer.data()), buffer.size())); } TUIndex TUIndex::build(CompilationUnitRef unit, bool interested_only) { @@ -652,22 +660,46 @@ TUIndex TUIndex::build(CompilationUnitRef unit, bool interested_only) { void TUIndex::serialize(llvm::raw_ostream& os) { format_version = index_format_version; - /// Convert the FileID-keyed working state into the persisted - /// path_id-keyed form; multiple FileIDs can share a path id (repeated - /// header contexts), last-wins. A deserialized index has no FileID-keyed - /// state at all — its path-keyed rows already are the persisted form, so - /// re-serializing must not wipe them. + /// Convert the FileID-keyed working state into wire sections; multiple + /// FileIDs can share a path id (repeated header contexts), last-wins. + /// A deserialized index has no FileID-keyed state at all — its sections + /// already are the persisted form, so re-serializing must not wipe them. ScopedTimer copy_timer; - if(!file_indices.empty()) { - path_file_indices.clear(); + if(!file_indices.empty() || !main_file_index.empty()) { + sections.clear(); + llvm::DenseMap positions; + auto add = [&](std::uint32_t path_id, const FileIndex& index) { + if(index.empty()) { + return; + } + auto encoded = kota::codec::fbs::to_bytes(index); + assert(encoded.has_value()); + FileSection section{path_id, index.rows_hash(), std::move(*encoded)}; + auto [it, inserted] = positions.try_emplace(path_id, sections.size()); + if(inserted) { + sections.push_back(std::move(section)); + } else { + sections[it->second] = std::move(section); + } + }; for(auto& [fid, file_index]: file_indices) { - path_file_indices[graph.path_id(fid)] = file_index; + add(graph.path_id(fid), file_index); } + // size() - 1 would wrap on an empty path table and name the main + // file with an id every reader rejects, losing the rows silently. + assert(!graph.paths.empty() && "rows cannot exist without a path table naming their file"); + add(static_cast(graph.paths.size() - 1), main_file_index); } auto copy_ms = copy_timer.ms_f(); + // The interested file's rows travel only as their section; the + // reflected field is written empty and restored after the pack. + auto main_rows = std::move(main_file_index); + main_file_index = FileIndex(); + ScopedTimer pack_timer; serialize_blob(*this, os); + main_file_index = std::move(main_rows); LOG_PERF("index_detail", "op=serialize copy_ms={:.2f} pack_ms={:.2f}", copy_ms, @@ -686,9 +718,9 @@ std::optional TUIndex::from(llvm::StringRef data) { // Nor does it constrain field values, and every decoded path id is // dereferenced against the path table without further checks — graph - // locations and per-file rows in Indexer::merge, reference_files through - // ProjectIndex::merge's file_ids_map. A blob carrying an out-of-range - // one is rejected as a whole. + // locations and wire sections in Indexer::merge, reference_files + // through ProjectIndex::merge's file_ids_map. A blob carrying an + // out-of-range one is rejected as a whole. auto in_range = [count = index->graph.paths.size()](std::uint32_t path_id) { return path_id < count; }; @@ -697,8 +729,8 @@ std::optional TUIndex::from(llvm::StringRef data) { return std::nullopt; } } - for(auto& [path_id, _]: index->path_file_indices) { - if(!in_range(path_id)) { + for(auto& section: index->sections) { + if(!in_range(section.path_id)) { return std::nullopt; } } @@ -710,4 +742,167 @@ std::optional TUIndex::from(llvm::StringRef data) { return index; } +const FileSection* TUIndex::main_section() const { + if(graph.paths.empty()) { + return nullptr; + } + auto main_id = static_cast(graph.paths.size() - 1); + // The interested file's section is appended last by serialize(). + for(auto& section: std::ranges::reverse_view(sections)) { + if(section.path_id == main_id) { + return §ion; + } + } + return nullptr; +} + +std::optional TUIndex::decode_rows(const FileSection& section) { + std::optional rows{std::in_place}; + auto data = + llvm::StringRef(reinterpret_cast(section.rows.data()), section.rows.size()); + if(!deserialize_blob(data, *rows)) { + return std::nullopt; + } + return rows; +} + +namespace { + +using WireView = kota::codec::fbs::table_view; + +/// The buffer was fully verified at TUIndexView::from; per-accessor views +/// skip that cost. +WireView wire_root(llvm::StringRef data) { + return WireView::from_verified_bytes(blob_bytes(data)); +} + +SymbolIdentity identity_of(kota::codec::fbs::table_view symbol) { + return {to_ref(symbol[&Symbol::name]), + SymbolKind(symbol[&Symbol::kind]), + symbol[&Symbol::scope]}; +} + +/// The symbol's serialized reference bitmap (the Bitmap repr's byte image) +/// as a StringRef borrowing the wire. +llvm::StringRef bitmap_bytes(kota::codec::fbs::table_view symbol) { + const auto* raw = symbol[&Symbol::reference_files].raw(); + if(!raw) { + return {}; + } + return llvm::StringRef(reinterpret_cast(raw->data()), raw->size()); +} + +} // namespace + +std::optional TUIndexView::from(llvm::StringRef data) { + auto root = WireView::from_bytes(blob_bytes(data)); + if(!root.valid() || root[&TUIndex::format_version] != index_format_version) { + return std::nullopt; + } + + // Structural verification does not constrain field values; every path + // id the merge dereferences against the path table is bounded here so + // the accessors stay check-free. + auto graph = root[&TUIndex::graph]; + auto count = graph[&IncludeGraph::paths].size(); + auto locations = graph[&IncludeGraph::locations]; + for(std::size_t i = 0; i < locations.size(); i += 1) { + IncludeLocation location = locations.at(i); + if(location.path_id >= count) { + return std::nullopt; + } + } + auto sections = root[&TUIndex::sections]; + for(std::size_t i = 0; i < sections.size(); i += 1) { + if(sections.at(i)[&FileSection::path_id] >= count) { + return std::nullopt; + } + } + return TUIndexView(data); +} + +std::int64_t TUIndexView::built_at() const { + return wire_root(data)[&TUIndex::built_at]; +} + +std::uint32_t TUIndexView::path_count() const { + return static_cast( + wire_root(data)[&TUIndex::graph][&IncludeGraph::paths].size()); +} + +llvm::StringRef TUIndexView::path(std::uint32_t id) const { + return to_ref(wire_root(data)[&TUIndex::graph][&IncludeGraph::paths].at(id)); +} + +std::uint64_t TUIndexView::path_hash(std::uint32_t id) const { + // The hash column may be shorter than the path table on a foreign + // blob; an absent hash reads as 0, "unavailable" — the same + // normalization TUIndex::from applies. + auto hashes = wire_root(data)[&TUIndex::graph][&IncludeGraph::path_hashes]; + return id < hashes.size() ? hashes.at(id) : 0; +} + +std::uint32_t TUIndexView::location_count() const { + return static_cast( + wire_root(data)[&TUIndex::graph][&IncludeGraph::locations].size()); +} + +IncludeLocation TUIndexView::location(std::uint32_t i) const { + return wire_root(data)[&TUIndex::graph][&IncludeGraph::locations].at(i); +} + +std::uint32_t TUIndexView::section_count() const { + return static_cast(wire_root(data)[&TUIndex::sections].size()); +} + +std::uint32_t TUIndexView::section_path(std::uint32_t i) const { + return wire_root(data)[&TUIndex::sections].at(i)[&FileSection::path_id]; +} + +std::uint64_t TUIndexView::section_rows_hash(std::uint32_t i) const { + return wire_root(data)[&TUIndex::sections].at(i)[&FileSection::rows_hash]; +} + +std::optional TUIndexView::decode_section_rows(std::uint32_t i) const { + auto rows = to_array_ref(wire_root(data)[&TUIndex::sections].at(i)[&FileSection::rows]); + std::optional decoded{std::in_place}; + auto bytes = llvm::StringRef(reinterpret_cast(rows.data()), rows.size()); + if(!deserialize_blob(bytes, *decoded)) { + return std::nullopt; + } + return decoded; +} + +std::optional TUIndexView::main_section_index() const { + auto count = path_count(); + if(count == 0) { + return std::nullopt; + } + auto main_id = count - 1; + // The interested file's section is appended last by serialize(). + for(auto i = section_count(); i > 0; i -= 1) { + if(section_path(i - 1) == main_id) { + return i - 1; + } + } + return std::nullopt; +} + +void TUIndexView::iterate_symbols( + llvm::function_ref callback) const { + auto symbols = wire_root(data)[&TUIndex::symbols]; + for(std::size_t i = 0; i < symbols.size(); i += 1) { + auto entry = symbols.at(i); + callback(entry.get<0>(), identity_of(entry.get<1>()), bitmap_bytes(entry.get<1>())); + } +} + +std::optional TUIndexView::find_symbol(SymbolHash hash) const { + auto found = wire_root(data)[&TUIndex::symbols].find(hash); + if(!found) { + return std::nullopt; + } + return identity_of(found->get<1>()); +} + } // namespace clice::index diff --git a/src/index/tu_index.h b/src/index/tu_index.h index c727d54c5..7a397b14e 100644 --- a/src/index/tu_index.h +++ b/src/index/tu_index.h @@ -1,6 +1,5 @@ #pragma once -#include #include #include #include @@ -29,10 +28,10 @@ enum class SymbolScope : std::uint8_t { External = 0, /// Can be referenced across files within one TU but not across TUs /// (internal linkage: static, anonymous namespace). Stored in the main - /// file's MergedIndex shard. + /// file's Shard blob. TULocal = 1, /// Cannot be referenced from any other file (local variables, parameters, - /// labels). Stored in the defining file's MergedIndex shard. + /// labels). Stored in the defining file's Shard blob. FileLocal = 2, }; @@ -82,7 +81,16 @@ struct FileIndex { RelationKind kind, llvm::function_ref callback) const; - std::array hash(); + bool empty() const { + return occurrences.empty() && relations.empty(); + } + + /// Content identity of the rows: xxh3 over the occurrences and the + /// relation groups in ascending symbol order. Requires the canonical + /// row order build() establishes (sorted, deduplicated); two files + /// preprocessed identically hash equal, and that equality is what + /// deduplicates variants across compilation contexts. + std::uint64_t rows_hash() const; }; struct Symbol { @@ -100,6 +108,19 @@ struct Symbol { using SymbolTable = llvm::DenseMap; +/// One file's rows on the wire: the hash first, so the master can skip the +/// nested decode for rows it already stores, and the rows themselves as a +/// self-contained nested blob decoded per miss. +struct FileSection { + std::uint32_t path_id = 0; + + /// FileIndex::rows_hash of the nested rows. + std::uint64_t rows_hash = 0; + + /// Nested fbs FileIndex blob (TUIndex::decode_rows). + std::vector rows; +}; + struct TUIndex { /// Persisted-blob schema version (index_format_version), stamped by /// serialize() and gated by from(). These blobs never touch disk — they @@ -122,12 +143,17 @@ struct TUIndex { KOTATSU_ANNOTATE(skip = true) > file_indices; - /// File indices keyed by path_id: populated from file_indices by - /// serialize(), and directly by from() for deserialized data. - llvm::DenseMap path_file_indices; - + /// The interested file's rows, used in memory (sessions, preamble + /// state). serialize() moves it into its wire section for the duration + /// of the write, so the reflected field always travels empty. FileIndex main_file_index; + /// The wire form of the per-file rows: populated by serialize() from + /// file_indices and main_file_index (the interested file's section is + /// last), kept raw by from(). Files whose rows are empty get no + /// section — no rows means no contribution. + std::vector sections; + /// Build the index for `unit`. With interested_only, only rows in /// the interested file are kept. Note that a full build over a unit /// compiled with a preamble PCH is not a production combination @@ -136,14 +162,90 @@ struct TUIndex { /// under the main path id. static TUIndex build(CompilationUnitRef unit, bool interested_only = false); - /// Serialization reflects this object directly (path_file_indices is - /// populated from file_indices first — hence non-const). + /// Serialization reflects this object directly (sections are populated + /// from the row state first — hence non-const). void serialize(llvm::raw_ostream& os); /// Verify and deserialize a buffer; nullopt when structural /// verification fails, the format version differs, or a decoded path id - /// falls outside the blob's own path table. + /// falls outside the blob's own path table. Section rows stay raw — + /// decode them per file with decode_rows. static std::optional from(llvm::StringRef data); + + /// The interested file's wire section, or nullptr when its rows were + /// empty. + const FileSection* main_section() const; + + /// Verify and decode one section's rows; nullopt for a corrupt nested + /// blob. + static std::optional decode_rows(const FileSection& section); +}; + +/// A symbol's identity as a merge consumer needs it; the name borrows the +/// wire buffer. +struct SymbolIdentity { + llvm::StringRef name; + SymbolKind kind; + SymbolScope scope; +}; + +/// Zero-copy reader over a serialized TUIndex, for the master's merge path: +/// the graph and the per-file rows hashes are read straight off the wire, +/// section rows are decoded only for actual misses, and symbol names are +/// touched only when a symbol is genuinely new to the global table. The +/// view borrows the wire bytes; keep them alive while using it. +/// +/// TUIndex::from stays the full-decode entry for consumers that need the +/// whole object (sessions, tests). +class TUIndexView { +public: + /// Verify the buffer, gate the format version, and bound every path id + /// the graph and sections carry. Symbol reference-file ids are NOT + /// validated here — iterate_symbols hands them out raw and the consumer + /// bounds them (decoding every bitmap twice just to validate would + /// defeat the view). + static std::optional from(llvm::StringRef data); + + std::int64_t built_at() const; + + std::uint32_t path_count() const; + + llvm::StringRef path(std::uint32_t id) const; + + std::uint64_t path_hash(std::uint32_t id) const; + + std::uint32_t location_count() const; + + IncludeLocation location(std::uint32_t i) const; + + std::uint32_t section_count() const; + + std::uint32_t section_path(std::uint32_t i) const; + + std::uint64_t section_rows_hash(std::uint32_t i) const; + + /// Verify and decode one section's rows straight from the wire buffer. + std::optional decode_section_rows(std::uint32_t i) const; + + /// The section index of the interested file (path_count() - 1), or + /// nullopt when its rows were empty. + std::optional main_section_index() const; + + /// Visit every symbol: hash, identity, and the raw serialized + /// reference-files bitmap (a read_bitmap'able portable image). + void iterate_symbols( + llvm::function_ref + callback) const; + + /// Look up one symbol's identity by hash. + std::optional find_symbol(SymbolHash hash) const; + +private: + explicit TUIndexView(llvm::StringRef data) : data(data) {} + + /// The verified wire bytes; accessors rebuild the (pointer-sized) fbs + /// view from them on demand. + llvm::StringRef data; }; } // namespace clice::index diff --git a/src/server/compiler/compiler.cpp b/src/server/compiler/compiler.cpp index 8380d6a29..07a04f785 100644 --- a/src/server/compiler/compiler.cpp +++ b/src/server/compiler/compiler.cpp @@ -1193,7 +1193,12 @@ kota::task<> Compiler::run_compile(std::shared_ptr session) { ? std::nullopt : index::TUIndex::from(result.value().tu_index_data); if(tu_index) { - session->file_index = std::move(tu_index->main_file_index); + // The interested file's rows travel as a wire section; a file + // with no rows at all has no section and gets an empty index. + auto* section = tu_index->main_section(); + auto rows = section ? index::TUIndex::decode_rows(*section) + : std::optional{std::in_place}; + session->file_index = rows ? std::move(*rows) : index::FileIndex(); session->symbols = std::move(tu_index->symbols); } else { // The AST and the file index settle together — that pairing is diff --git a/src/server/compiler/indexer.cpp b/src/server/compiler/indexer.cpp index d9642b3dd..7fb9cf4f6 100644 --- a/src/server/compiler/indexer.cpp +++ b/src/server/compiler/indexer.cpp @@ -7,6 +7,9 @@ #include #include +#include "index/manifest.h" +#include "index/shard.h" +#include "index/storage.h" #include "index/tu_index.h" #include "server/compiler/context_resolver.h" #include "server/protocol/worker.h" @@ -17,48 +20,53 @@ #include "support/timer.h" #include "llvm/ADT/DenseSet.h" +#include "llvm/ADT/STLExtras.h" #include "llvm/ADT/StringSet.h" -#include "llvm/Support/FileSystem.h" #include "llvm/Support/MemoryBuffer.h" -#include "llvm/Support/Path.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Support/xxhash.h" namespace clice { +/// Stable blob key for a file's shard or a TU's manifest: runtime pool ids +/// are per-session, so blobs are named by a hash of the path instead. +static std::string blob_key(llvm::StringRef path) { + return std::format("{:016x}", llvm::xxh3_64bits(path)); +} + void Indexer::merge(const void* tu_index_data, std::size_t size) { + // Zero-copy consumption: the wire stays serialized; only miss sections + // and genuinely new symbol names are ever materialized. auto loaded = - index::TUIndex::from(llvm::StringRef(static_cast(tu_index_data), size)); + index::TUIndexView::from(llvm::StringRef(static_cast(tu_index_data), size)); if(!loaded) { LOG_WARN("Ignoring TUIndex that failed verification"); return; } - auto& tu_index = *loaded; - if(tu_index.graph.paths.empty()) { + auto& view = *loaded; + if(view.path_count() == 0) { LOG_WARN("Ignoring TUIndex with empty path graph"); return; } - auto main_tu_path_id = static_cast(tu_index.graph.paths.size() - 1); - llvm::StringRef main_tu_path = tu_index.graph.paths[main_tu_path_id]; + auto main_local_id = view.path_count() - 1; + llvm::StringRef main_tu_path = view.path(main_local_id); // Shards pair the worker's rows with content read from disk here; if the // disk moved on since the worker read it, the rows' offsets describe // bytes that no longer exist and merging would misplace every position // until the next reindex. The worker's consumed-content hash arbitrates. // A missing hash (0) proceeds as before. - auto content_matches = [&](std::uint32_t tu_path_id, llvm::StringRef disk_content) { - auto consumed = tu_index.graph.path_hashes[tu_path_id]; + auto content_matches = [&](std::uint32_t local_id, llvm::StringRef disk_content) { + auto consumed = view.path_hash(local_id); return consumed == 0 || llvm::xxh3_64bits(disk_content) == consumed; }; - // The main file's verdict gates the WHOLE result: its per-header shards - // and the trailing sweep all describe this one compile, so applying any - // of it against a moved-on main file would mix two generations (and the - // sweep would strip contributions the stale result merely no longer - // mentions). Skipping everything keeps the last-known state consistent; - // the changed file fails the next staleness check (or is already - // pending), and a follow-up pass redoes the merge against settled - // content. + // The main file's verdict gates the WHOLE result: every section and the + // manifest describe this one compile, so applying any of it against a + // moved-on main file would mix two generations. Skipping everything + // keeps the last-known state consistent; the changed file fails the + // next staleness check (or is already pending), and a follow-up pass + // redoes the merge against settled content. auto main_buf = llvm::MemoryBuffer::getFile(main_tu_path); if(!main_buf) { LOG_WARN("Skip merge for {}: cannot read content: {}", @@ -66,150 +74,272 @@ void Indexer::merge(const void* tu_index_data, std::size_t size) { main_buf.getError().message()); return; } - if(!content_matches(main_tu_path_id, (*main_buf)->getBuffer())) { + if(!content_matches(main_local_id, (*main_buf)->getBuffer())) { LOG_INFO("Skip merge for {}: disk moved on since it was indexed", main_tu_path); return; } - auto file_ids_map = workspace.project_index.merge(tu_index, workspace.path_pool); + // Interning paths only names them — pool ids left behind by a rejected + // result are inert. Everything that is index STATE (symbols, + // FileVersions, the manifest, shards) commits only below the section + // loop, once every part of the result validated. + auto& project = workspace.project_index; + llvm::SmallVector file_ids_map; + file_ids_map.resize_for_overwrite(view.path_count()); + for(std::uint32_t i = 0; i < view.path_count(); i += 1) { + file_ids_map[i] = workspace.path_pool.intern(view.path(i)); + } + auto tu_path_id = file_ids_map[main_local_id]; + + index::TUManifest manifest; + manifest.built_at = static_cast(view.built_at()); - // Collect non-External symbols referenced in a FileIndex. Each file's - // MergedIndex shard stores exactly the local symbols its occurrences - // reference, so lookup is a single shard check — no scanning. - auto collect_local_symbols = [&](const index::FileIndex& file_idx) { - index::SymbolTable result; - for(auto& occ: file_idx.occurrences) { - auto it = tu_index.symbols.find(occ.target); - if(it != tu_index.symbols.end() && it->second.scope != index::SymbolScope::External) { - result.try_emplace(occ.target, it->second); + // The previous contribution entry for a file this pass must skip (disk + // moved on under a section): its rows stay consistent with the shard + // they live in, so it keeps serving; a later pass redoes the file. + auto carry_old_contribution = [&](std::uint32_t global_id) { + auto manifest_it = project.manifests.find(tu_path_id); + if(manifest_it == project.manifests.end()) { + return; + } + for(auto& [fv, hash]: manifest_it->second.contributions) { + if(project.file_versions.find(fv)->second.path_id == global_id) { + manifest.contributions.emplace_back(fv, hash); + return; } } - return result; }; - // Only shards that actually receive this TU's new contribution count as - // touched; a file still in the include graph but with an empty FileIndex - // must be swept below like a dropped one. - llvm::DenseSet touched; - - auto merge_file_index = [&](std::uint32_t tu_path_id, index::FileIndex& file_idx) { - auto global_path_id = file_ids_map[tu_path_id]; - auto& shard = workspace.merged_indices[global_path_id]; - - if(tu_path_id == main_tu_path_id) { - auto& path_hashes = tu_index.graph.path_hashes; - llvm::SmallVector deps; - deps.reserve(tu_index.graph.locations.size() + 1); - // The TU's own content is a dependency of its shard too: without - // it, a closed TU edited on disk with unchanged includes would - // never look stale. - deps.push_back({main_tu_path, 0, 0, path_hashes[main_tu_path_id]}); - for(auto& loc: tu_index.graph.locations) { - deps.push_back({tu_index.graph.paths[loc.path_id], - loc.line, - loc.include, - path_hashes[loc.path_id]}); - } - // Read and verified against the consumed hash before anything - // was merged: an unreadable or moved-on main file skips this - // whole result up front. - shard.merge(main_tu_path, tu_index.built_at, deps, file_idx, (*main_buf)->getBuffer()); - shard.merge_symbols(collect_local_symbols(file_idx)); - touched.insert(global_path_id); + std::size_t hits = 0; + std::size_t appended = 0; + std::size_t rebuilt = 0; + auto lookup_symbol = [&](index::SymbolHash hash) { + return view.find_symbol(hash); + }; + // Staged, not committed: a section that fails to decode rejects the + // whole result mid-loop, and shards installed before that point would + // leave the surviving manifest referencing variants the new blobs no + // longer store. + llvm::SmallVector> replacements; + // (TU-local path id, rows hash) per serving section; the FileVersions + // these will reference are interned only at commit. + llvm::SmallVector> section_contributions; + for(std::uint32_t section = 0; section < view.section_count(); section += 1) { + auto local_id = view.section_path(section); + auto rows_hash = view.section_rows_hash(section); + auto global_id = file_ids_map[local_id]; + auto consumed = view.path_hash(local_id); + bool is_main = local_id == main_local_id; + + auto shard_it = workspace.shards.find(global_id); + auto* shard = shard_it != workspace.shards.end() ? &shard_it->second : nullptr; + + // Fast path: this content generation's blob already stores these + // rows — recording the contribution is the only work, no IO at all. + if(consumed != 0 && shard && shard->loaded() && shard->content_hash() == consumed && + shard->has_variant(rows_hash)) { + section_contributions.emplace_back(local_id, rows_hash); + hits += 1; + continue; + } + + // Read and arbitrate the content the blob will pair the rows with. + std::string header_content; + llvm::StringRef content; + if(is_main) { + content = (*main_buf)->getBuffer(); } else { - std::optional include_id; - for(std::uint32_t i = 0; i < tu_index.graph.locations.size(); ++i) { - if(tu_index.graph.locations[i].path_id == tu_path_id) { - include_id = i; - break; - } - } - if(!include_id) { - LOG_WARN("Skip merge for path {}: include location not found", global_path_id); - return; - } - auto header_path = workspace.path_pool.resolve(global_path_id); - llvm::StringRef header_content; - std::string header_content_storage; - auto header_buf = llvm::MemoryBuffer::getFile(header_path); - if(header_buf) { - header_content_storage = (*header_buf)->getBuffer().str(); - header_content = header_content_storage; + auto path = workspace.path_pool.resolve(global_id); + auto buf = llvm::MemoryBuffer::getFile(path); + if(buf) { + header_content = (*buf)->getBuffer().str(); + content = header_content; } - // Unconditional, unlike the read above: an unreadable or + // Unconditional, unlike the main-file read: an unreadable or // truncated-to-empty header must not slip past the arbitration - // and pair the rows with content they were not built from. - if(!content_matches(tu_path_id, header_content)) { - LOG_INFO("Skip merge for {}: disk moved on since it was indexed", header_path); - touched.insert(global_path_id); - return; + // and pair the rows with content they were not built from. A + // failed read is checked on its own — with no worker hash the + // arbitration would otherwise wave the empty content through. + if(!buf || !content_matches(local_id, content)) { + LOG_INFO("Skip merge for {}: disk moved on since it was indexed", path); + carry_old_contribution(global_id); + continue; } - // Keyed by the including TU so a reindex of that TU replaces its - // prior contribution to this header's shard. - shard.merge(main_tu_path, *include_id, file_idx, header_content); - shard.merge_symbols(collect_local_symbols(file_idx)); - touched.insert(global_path_id); } - }; + auto generation = consumed != 0 ? consumed : llvm::xxh3_64bits(content); + + // The same hit as the fast path above, verifiable for a hash-less + // section only now that the disk read pinned the generation: + // re-appending an already-stored variant would duplicate it in the + // blob's variant table (write_shard asserts against exactly that). + if(consumed == 0 && shard && shard->loaded() && shard->content_hash() == generation && + shard->has_variant(rows_hash)) { + section_contributions.emplace_back(local_id, rows_hash); + hits += 1; + continue; + } - for(auto& [tu_path_id, file_idx]: tu_index.path_file_indices) { - merge_file_index(tu_path_id, file_idx); + // The recomputed hash guards the variant identity alongside the + // structural decode: rows installed under a hash they do not + // reproduce would satisfy every later hit-path check for that hash + // while the shard stores different rows. + auto rows = view.decode_section_rows(section); + if(!rows || rows->rows_hash() != rows_hash) { + // Not a carry-and-skip like the moved-on case above: that one + // self-corrects because the skipped file's recorded version is + // stale by hash. A decode failure leaves every recorded version + // matching the disk, so an installed manifest would be judged + // fresh forever with this file's rows missing or stale — even + // across restarts. Reject the whole result like the main-file + // gate does; nothing is committed yet. + LOG_WARN("Reject merge for {}: rows section for {} failed verification", + main_tu_path, + workspace.path_pool.resolve(global_id)); + return; + } + + index::VariantInput fresh{rows_hash, &*rows, lookup_symbol}; + std::string bytes; + llvm::raw_string_ostream os(bytes); + if(shard && shard->loaded() && shard->content_hash() == generation) { + // Same generation, new variant: merge it in, keeping every + // stored variant — dead ones stay masked until the next save + // compacts them. + write_shard(*shard, shard->variants(), fresh, content, generation, os); + appended += 1; + } else { + // New content generation (or no blob at all): rows from other + // generations must never share offset storage with these, so + // the blob starts over. Stale contributions from other TUs + // simply stop matching any stored variant. + write_shard(index::Shard(), {}, fresh, content, generation, os); + rebuilt += 1; + } + + auto replacement = index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); + if(!replacement.loaded()) { + if(consumed == 0) { + // Unverified pairing: the arbitration above cannot see the + // disk shrinking under rows built from longer content, and + // the blob's own range bounds catch it here instead. A + // carry self-corrects — the recorded version is stale by + // hash. + LOG_INFO("Skip merge for {}: disk moved on since it was indexed", + workspace.path_pool.resolve(global_id)); + carry_old_contribution(global_id); + continue; + } + // Hash-verified content always fits rows built from it: this + // blob failed on the rows themselves (inverted or out-of-range + // spans that kept wire structure and hash). Carrying would + // install a manifest whose versions all match the disk, pinning + // the stale rows as fresh forever — reject like the decode + // failure above. + LOG_WARN("Reject merge for {}: rows for {} do not form a valid shard", + main_tu_path, + workspace.path_pool.resolve(global_id)); + return; + } + replacements.emplace_back(global_id, std::move(replacement)); + section_contributions.emplace_back(local_id, rows_hash); } - merge_file_index(main_tu_path_id, tu_index.main_file_index); - // A file dropped from this TU (a removed transitive include, or one - // whose contribution became empty) is no longer merged above, but its - // shard may still hold this TU's previous contribution — sweep it, or - // references under the dropped include keep being served as if the edge - // still existed. - for(auto& [path_id, shard]: workspace.merged_indices) { - if(touched.contains(path_id)) { - continue; + // The last gate and the first commit. A malformed reference bitmap (or + // an out-of-range reference id) rejects the whole result for the same + // reason a rows section that fails decode does above: everything the + // merge would install reads as fresh forever, with the lost bits never + // rebuilt. + if(!project.merge(view, file_ids_map)) { + LOG_WARN("Reject merge for {}: symbol reference bitmap failed verification", main_tu_path); + return; + } + + // Intern a FileVersion per file of the parse. The freshness baseline is + // two-part and lives on the version, shared by every TU that consumed + // it: the consumed-content hash from the compiler's own buffers, and a + // stat fast path recorded only for files that provably did not change + // since before the build started — for the rest the stat could describe + // content the rows were never built from, so they re-earn their fast + // path through a hash check instead (see file_version_stale). + auto baseline_before_ns = fs::stat_baseline_before_ns(view.built_at()); + llvm::SmallVector fv_of; + fv_of.resize_for_overwrite(view.path_count()); + for(std::uint32_t i = 0; i < view.path_count(); i += 1) { + llvm::StringRef path = view.path(i); + auto hash = view.path_hash(i); + + fs::file_status status; + bool stat_ok = !fs::status(path, status); + bool untouched = stat_ok && fs::mtime_ns(status) <= baseline_before_ns; + if(hash == 0 && untouched) { + // The worker had no buffer to hash (e.g. behind a PCM); the + // unchanged mtime proves the disk still holds the consumed + // bytes, so hash it here. + hash = hash_file(path); } - if(shard.has_contribution(main_tu_path)) { - shard.remove(main_tu_path); + + auto fv = project.intern_file_version(file_ids_map[i], hash); + if(untouched) { + auto& record = project.file_versions.find(fv)->second; + record.size = status.getSize(); + record.mtime_ns = fs::mtime_ns(status); } + fv_of[i] = fv; } - auto external_count = std::ranges::count_if(tu_index.symbols, [](auto& kv) { - return kv.second.scope == index::SymbolScope::External; - }); - LOG_INFO("Merged TUIndex: {} paths, {} symbols ({} external), {} merged_shards", - tu_index.graph.paths.size(), - tu_index.symbols.size(), - external_count, - workspace.merged_indices.size()); -} + manifest.tu_fv = fv_of[main_local_id]; + manifest.nodes.reserve(view.location_count()); + for(std::uint32_t i = 0; i < view.location_count(); i += 1) { + auto location = view.location(i); + manifest.nodes.push_back({fv_of[location.path_id], location.include, location.line}); + } + for(auto [local_id, rows_hash]: section_contributions) { + manifest.contributions.emplace_back(fv_of[local_id], rows_hash); + } -/// Stable blob key for a file's shard: runtime pool ids are per-session, -/// so blobs are named by a hash of the path instead. -static std::string shard_key(llvm::StringRef path) { - return std::format("{:016x}", llvm::xxh3_64bits(path)); + for(auto& [global_id, replacement]: replacements) { + workspace.shards[global_id] = std::move(replacement); + dirty_shards.insert(global_id); + } + + // Replace this TU's manifest wholesale: files it no longer touches lose + // their contribution here, which is also what retires their variants — + // no sweep over other shards is needed. + auto affected = project.apply_manifest(tu_path_id, std::move(manifest)); + for(auto path_id: affected) { + auto it = workspace.shards.find(path_id); + if(it == workspace.shards.end()) { + continue; + } + it->second.set_live(project.live_variants(path_id)); + } + dirty_manifests.insert(tu_path_id); + global_dirty = true; + + LOG_INFO( + "Merged TUIndex: {} paths, {} sections ({} hits, {} appended, {} rebuilt), " + "{} merged_shards", + view.path_count(), + view.section_count(), + hits, + appended, + rebuilt, + workspace.shards.size()); } -/// Begin a two-phase store write and serialize the blob to its tmp path. -/// Returns the entry to commit, or nullopt if serialization failed. -static std::optional - serialize_blob(CacheStore& store, - llvm::StringRef key, - llvm::function_ref serialize) { - auto pending = store.begin_store("index", key); - std::error_code ec; - llvm::raw_fd_ostream os(pending.tmp_path, ec); - if(ec) { - LOG_WARN("Failed to write index blob {}: {}", key, ec.message()); - return std::nullopt; - } - serialize(os); - os.flush(); - // A truncated blob (disk full) must never be committed: the index - // namespace is Persistent, so it would be served forever. - if(os.has_error()) { - LOG_WARN("Failed to write index blob {}: {}", key, os.error().message()); - os.clear_error(); - return std::nullopt; - } - return pending; +void Indexer::drop_index(std::uint32_t tu_path_id) { + auto& project = workspace.project_index; + if(!project.manifests.contains(tu_path_id)) { + return; + } + for(auto path_id: project.remove_manifest(tu_path_id)) { + auto it = workspace.shards.find(path_id); + if(it != workspace.shards.end()) { + it->second.set_live(project.live_variants(path_id)); + } + } + dirty_manifests.insert(tu_path_id); + global_dirty = true; } kota::task<> Indexer::save() { @@ -217,204 +347,416 @@ kota::task<> Indexer::save() { // nothing, and the gauge must not keep exposing the previous round's // count as current. saved_shards = 0; - if(!workspace.store) + if(!workspace.index_storage) co_return; - auto& store = *workspace.store; + auto& storage = *workspace.index_storage; + auto& project = workspace.project_index; ScopedTimer timer; - // Phase 1, synchronous: serialize the ProjectIndex and every dirty - // shard to tmp files. No suspension point in between, so the batch is - // a consistent snapshot even if a merge runs before the commits below - // are done. Shards are only published together with the ProjectIndex - // they were built against: pairing new shards with an old project blob - // (or vice versa) would serve a mixed snapshot after restart. - // The project blob doubles as the shard manifest: it lists every file - // owning a shard so the loader knows exactly which blobs to fetch (and - // sweeps the rest). + // Compact shards whose variant set shrank: queries already mask the + // dead rows, this erases them for real before the blob reaches disk. + // A file with no live variant left in its blob is retired entirely + // below: either no contribution remains, or the remaining ones pin + // hashes a newer content generation replaced (their TUs have not + // reindexed yet) — compacting to those pins would write a blob with no + // variants at all, while retiring serves the same nothing the mask + // already does. The pinning TUs are re-enqueued: no in-process event + // would rebuild their rows otherwise (a reverted file even reads fresh + // by hash), only a restart reaching load()'s re-enqueue. + llvm::SmallVector retired; + for(auto& [path_id, shard]: workspace.shards) { + auto live = project.live_variants(path_id); + if(llvm::none_of(live, [&](std::uint64_t hash) { return shard.has_variant(hash); })) { + retired.push_back(path_id); + continue; + } + if(!shard.has_dead_variants()) { + continue; + } + std::string bytes; + llvm::raw_string_ostream os(bytes); + write_shard(shard, live, {}, shard.content(), shard.content_hash(), os); + auto replacement = index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); + assert(replacement.loaded() && "a freshly written shard blob must verify"); + shard = std::move(replacement); + dirty_shards.insert(path_id); + } + for(auto path_id: retired) { + workspace.shards.erase(path_id); + dirty_shards.erase(path_id); + auto it = project.contributions.find(path_id); + if(it == project.contributions.end()) { + continue; + } + for(auto tu: llvm::make_first_range(it->second)) { + enqueue(tu, ReindexReason::ContentChanged); + } + } + + // Snapshot the dirty state on the loop: everything below serializes + // from copies, so merges landing across the write await simply re-dirty + // for the next save. The id vectors parallel the batch so a failed + // entry can be re-dirtied by its batch index. + std::vector batch; llvm::SmallVector shard_ids; - shard_ids.reserve(workspace.merged_indices.size()); - for(auto& [path_id, shard]: workspace.merged_indices) { + llvm::SmallVector manifest_ids; + llvm::SmallVector> removals; + for(auto path_id: retired) { + removals.push_back( + {index::IndexBlobKind::Shard, blob_key(workspace.path_pool.resolve(path_id))}); + } + for(auto path_id: dirty_shards) { + auto it = workspace.shards.find(path_id); + assert(it != workspace.shards.end() && "dirty shards stay resident until retirement"); + batch.push_back({index::IndexBlobKind::Shard, + blob_key(workspace.path_pool.resolve(path_id)), + it->second.bytes().str()}); shard_ids.push_back(path_id); } - - auto project_pending = serialize_blob(store, "project", [&](llvm::raw_ostream& os) { - workspace.project_index.serialize(os, workspace.path_pool, shard_ids); - }); - if(!project_pending) { - LOG_WARN("Skipping index save: ProjectIndex serialization failed"); - co_return; + auto shard_count = batch.size(); + + // One generation per persisted global blob, stamped into every + // manifest of the batch and pinned per TU inside the global blob + // (serialize_global): load() adopts a manifest only at its pinned + // stamp. Without the pin an ordering check alone cannot tell a + // deliberately unchanged older manifest from one whose update failed + // while the global landed — both FileVersion sets can stay fully + // resolvable (a reindex that changed rows or the include tree only). + if(global_dirty) { + project.global_generation += 1; + } + for(auto tu_path_id: dirty_manifests) { + auto it = project.manifests.find(tu_path_id); + auto key = blob_key(workspace.path_pool.resolve(tu_path_id)); + // Dirty with no in-memory manifest means dropped (drop_index): the + // persisted blob must go too, or a restart resurrects the TU's + // rows as fresh. + if(it == project.manifests.end()) { + removals.push_back({index::IndexBlobKind::Manifest, std::move(key)}); + continue; + } + it->second.global_gen = project.global_generation; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_manifest(it->second, os); + batch.push_back({index::IndexBlobKind::Manifest, std::move(key), std::move(bytes)}); + manifest_ids.push_back(tu_path_id); + } + auto manifest_count = batch.size() - shard_count; + + // The global blob goes last: a crash mid-batch then strands only + // manifests, which load() drops by their generation stamp — the + // reverse order would strand a global claiming symbols in files whose + // rows never landed. + if(global_dirty) { + std::string bytes; + llvm::raw_string_ostream os(bytes); + project.serialize_global(os, workspace.path_pool); + batch.push_back({index::IndexBlobKind::Global, "global", std::move(bytes)}); } - LOG_INFO("Saved ProjectIndex ({} symbols)", workspace.project_index.symbols.size()); - - // Each write remembers which shard it snapshotted and at what - // revision, so the post-commit flip below can prove the shard is - // byte-identical to the blob it just published. - struct ShardWrite { - CacheStore::PendingEntry pending; - std::uint32_t path_id; - std::uint64_t revision; - }; - // A commit's result pairs the published path with the blob reopened in - // the same job: the mmap plus the flatbuffer verification walk the - // whole file, and neither belongs on the event loop. - struct CommitOutcome { - std::expected path; - index::MergedIndex reloaded; - }; + dirty_shards.clear(); + dirty_manifests.clear(); + global_dirty = false; - llvm::SmallVector shards; - std::size_t total = workspace.merged_indices.size(); - for(auto& [path_id, shard]: workspace.merged_indices) { - if(!shard.need_rewrite()) - continue; - if(auto pending = serialize_blob(store, - shard_key(workspace.path_pool.resolve(path_id)), - [&](llvm::raw_ostream& os) { shard.serialize(os); })) { - shards.push_back({std::move(*pending), path_id, shard.revision()}); - } - } - LOG_INFO("Serialized {} MergedIndex shards (of {} total)", shards.size(), total); - - // Phase 2: commit each blob (fsync + atomic rename) on the kota thread - // pool, keeping the heavy IO off the event loop. The project blob goes - // first; if it cannot be published, drop the shards of this snapshot. - auto committed = - co_await kota::queue([&] { return store.commit(std::move(*project_pending)); }); - if(!committed.has_value() || !committed.value().has_value()) { - // The shard entries clean their own tmp blobs up on destruction. - LOG_WARN("Failed to commit ProjectIndex blob, dropping {} shard blobs", shards.size()); + if(batch.empty() && removals.empty()) { co_return; } - // FIXME: shard commits are strictly sequential (one co_await per shard). - // For large projects this adds ~N×2ms of round-trip overhead. Consider - // batching commits or dispatching them in parallel on the thread pool. - for(auto& write: shards) { - auto key = write.pending.key; - - auto outcome = co_await kota::queue([&]() -> CommitOutcome { - CommitOutcome out{store.commit(std::move(write.pending)), {}}; - if(out.path) { - out.reloaded = index::MergedIndex::load(*out.path); - } - return out; - }); - if(!outcome.has_value() || !outcome.value().path.has_value()) { - LOG_WARN("Failed to commit index blob {}", key); - continue; - } - saved_shards += 1; - - // Flip the shard back to its committed, buffer-backed blob: the - // heap Impl a merge materialized would otherwise live until - // shutdown (F21), and need_rewrite() would keep claiming the - // shard dirty forever, turning every later save into a full - // rewrite. Only a shard untouched across the commit await may - // flip — a merge that landed meanwhile made the blob stale, and - // the shard simply stays dirty for the next save. An unreadable - // reload keeps the live Impl for the same reason. - auto it = workspace.merged_indices.find(write.path_id); - if(it == workspace.merged_indices.end() || it->second.revision() != write.revision) { - continue; + // The dirty set was snapshot-cleared above so merges landing across + // the write await re-dirty for the next save; the in-flight count + // keeps pending_shard_writes() truthful meanwhile — a stats reader + // polling for "shard writes settled" must not observe zero while the + // commit is still running (and saved_shards still holds its reset). + saving_shards = shard_count; + llvm::SmallVector failed; + co_await kota::queue([&] { + failed = storage.write(batch); + for(auto& [kind, key]: removals) { + storage.remove(kind, key); } - if(!outcome.value().reloaded.loaded()) { - LOG_WARN("Committed index blob {} did not read back; keeping the in-memory shard", key); - continue; + }); + // An entry the storage failed to commit is re-dirtied so a later save + // retries it; discarded, the cache would trail the in-memory index + // until an unrelated merge happens to dirty the same entry or a + // restart rebuilds it. (Failed removals need no retry: load() drops + // stale manifests by their generation pin and sweeps orphan shards.) + std::size_t failed_shards = 0; + for(auto i: failed) { + if(i < shard_count) { + failed_shards += 1; + dirty_shards.insert(shard_ids[i]); + } else if(i - shard_count < manifest_count) { + dirty_manifests.insert(manifest_ids[i - shard_count]); + } else { + global_dirty = true; } - it->second = std::move(outcome.value().reloaded); } + saved_shards = shard_count - failed_shards; + saving_shards = 0; LOG_PERF("index", - "phase=save shards={} total={} elapsed_ms={}", - shards.size(), - total, + "phase=save shards={} manifests={} total={} elapsed_ms={}", + shard_count, + manifest_count, + workspace.shards.size(), timer.ms()); } void Indexer::load() { - if(!workspace.store) + if(!workspace.index_storage) return; + auto& storage = *workspace.index_storage; + auto& project = workspace.project_index; ScopedTimer timer; - bool has_project = false; - llvm::StringSet<> expected_keys; - auto project_path = workspace.store->lookup("index", "project"); - if(project_path) { - auto buf = llvm::MemoryBuffer::getFile(*project_path); - if(!buf) { - // Transient read failure — don't load shards (useless without - // the project index), but don't destroy them either. - LOG_WARN("Failed to read ProjectIndex blob: {}", buf.getError().message()); + auto sweep_all = [&] { + for(auto kind: {index::IndexBlobKind::Shard, index::IndexBlobKind::Manifest}) { + llvm::SmallVector keys; + storage.for_each_key(kind, [&](llvm::StringRef key) { keys.push_back(key.str()); }); + for(auto& key: keys) { + storage.remove(kind, key); + } + } + }; + + auto global = storage.read(index::IndexBlobKind::Global, "global"); + if(!global) { + // A global blob that exists but failed to open is a transient IO + // error, not absence: sweeping would destroy an intact index and + // force a full rebuild. Run this session memory-only instead — + // saving a fresh lineage over blobs whose anchor was never read + // could alias their fv ids and generation stamps — and leave + // everything for a healthier restart to load. + if(storage.contains(index::IndexBlobKind::Global, "global")) { + LOG_WARN("Index global blob unreadable; disabling index persistence this session"); + workspace.index_storage.reset(); return; } - // An unreadable or old-format blob loads as "no index on disk": - // everything is swept and rebuilt once in the background. - llvm::SmallVector manifest; - auto loaded = index::ProjectIndex::from((*buf)->getBuffer(), workspace.path_pool, manifest); - if(loaded) { - workspace.project_index = std::move(*loaded); - has_project = true; - LOG_INFO("Loaded ProjectIndex: {} symbols", workspace.project_index.symbols.size()); - - // The manifest names every shard blob; fetch exactly those. A - // blob the loader rejects (corruption, old format) counts as - // missing: enqueue the file so the background round rebuilds it - // — for headers no CDB entry would ever re-enqueue it otherwise. - for(auto path_id: manifest) { - auto key = shard_key(workspace.path_pool.resolve(path_id)); - auto shard_path = workspace.store->lookup("index", key); - auto shard = - shard_path ? index::MergedIndex::load(*shard_path) : index::MergedIndex(); - if(shard.loaded()) { - workspace.merged_indices[path_id] = std::move(shard); - expected_keys.insert(key); - } else { - // No shard survives, so there is nothing stale to keep - // serving; ContentChanged states the truth ("the index - // does not describe this file") without effect. - LOG_INFO("Discarding unreadable shard for {}", - workspace.path_pool.resolve(path_id)); - enqueue(path_id, ReindexReason::ContentChanged); + // No global table means no resolvable manifests: everything else + // is unreachable data, swept so it cannot survive as orphans. + sweep_all(); + return; + } + llvm::DenseMap manifest_pins; + if(!project.load_global(global->getBuffer(), workspace.path_pool, manifest_pins)) { + LOG_INFO("Discarding old-format index global blob"); + sweep_all(); + storage.remove(index::IndexBlobKind::Global, "global"); + return; + } + + // Adopt exactly the manifests the global blob pins, at exactly the + // pinned generation stamp and with every FileVersion resolvable. The + // rest are stale residue — a crash between batch phases, a failed + // write under a landed global, a dropped TU whose removal was lost — + // and are swept, with their TUs re-enqueued where recoverable. + llvm::DenseSet adopted_pins; + llvm::SmallVector dead_manifests; + storage.for_each_key(index::IndexBlobKind::Manifest, [&](llvm::StringRef key) { + auto blob = storage.read(index::IndexBlobKind::Manifest, key); + auto manifest = blob ? index::deserialize_manifest(blob->getBuffer()) : std::nullopt; + auto pin = manifest ? manifest_pins.find(manifest->tu_fv) : manifest_pins.end(); + if(!manifest || pin == manifest_pins.end() || pin->second != manifest->global_gen || + !project.knows_file_versions(*manifest)) { + dead_manifests.push_back(key.str()); + // The manifest raced a crash ahead of the global blob (its own + // pin never landed). When the TU's version is still resolvable, + // re-enqueue it: the CDB sweep never covers standalone-indexed + // headers. + if(manifest) { + auto fv = project.file_versions.find(manifest->tu_fv); + if(fv != project.file_versions.end()) { + enqueue(fv->second.path_id, ReindexReason::ContentChanged); } } - } else { - LOG_INFO("Discarding old-format ProjectIndex blob"); + return; + } + adopted_pins.insert(manifest->tu_fv); + auto tu_path_id = project.file_versions.find(manifest->tu_fv)->second.path_id; + project.apply_manifest(tu_path_id, std::move(*manifest)); + }); + for(auto& key: dead_manifests) { + storage.remove(index::IndexBlobKind::Manifest, key); + } + // A pinned TU without an adopted manifest lost it to a failed write + // that the landed global outran, or to a lost removal; re-enqueue it — + // its rows are unservable until a reindex. Pinned fvs always resolve: + // load_global rejects a blob whose pins its own table cannot cover. + for(auto fv: llvm::make_first_range(manifest_pins)) { + if(!adopted_pins.contains(fv)) { + enqueue(project.file_versions.find(fv)->second.path_id, ReindexReason::ContentChanged); + } + } + + // Every contributing FileVersion pins the content generation its rows + // were built from; the shard must store that same generation. The + // variant check below cannot catch a stale shard alone when an edit + // past every indexed row left the rows hash identical: all recorded + // versions would match the disk while positions map through the old + // text, forever. A version with no consumed-content hash (0) pins + // nothing — it is permanently stale and reindexes its TU anyway. + llvm::DenseMap> generations; + for(auto& manifest: llvm::make_second_range(project.manifests)) { + for(auto fv: llvm::make_first_range(manifest.contributions)) { + auto& record = project.file_versions.find(fv)->second; + if(record.content_hash == 0) { + continue; + } + auto& pinned = generations[record.path_id]; + if(!llvm::is_contained(pinned, record.content_hash)) { + pinned.push_back(record.content_hash); + } } } - // Sweep every blob the manifest does not name (or all of them when the - // project index itself is gone or outdated — Persistent namespace - // cleanup is the caller's mark-and-sweep). + // Fetch exactly the shard blobs the contributions expect. A blob that + // is missing or fails verification leaves its contributing TUs' rows + // unservable, so those manifests are dropped and the TUs reindex (for + // headers no CDB entry would ever re-enqueue them otherwise). + llvm::StringSet<> expected_keys; + llvm::SmallVector unservable; + for(auto& [path_id, entry]: project.contributions) { + auto key = blob_key(workspace.path_pool.resolve(path_id)); + auto shard = index::Shard::from_buffer(storage.read(index::IndexBlobKind::Shard, key)); + // A blob can verify yet miss a contributed variant, or carry + // another content generation than the contributions pin (crash or + // failed write left a manifest newer than its shard); set_live + // would drop missing rows silently and stale content misplaces + // every position, so both are as unservable as an unreadable blob. + auto generation_ok = [&] { + auto it = generations.find(path_id); + return it == generations.end() || llvm::all_of(it->second, [&](std::uint64_t hash) { + return hash == shard.content_hash(); + }); + }; + bool servable = shard.loaded() && generation_ok() && + llvm::all_of(llvm::make_second_range(entry), + [&](std::uint64_t hash) { return shard.has_variant(hash); }); + if(!servable) { + LOG_INFO("Discarding unservable shard for {}", workspace.path_pool.resolve(path_id)); + unservable.push_back(path_id); + continue; + } + expected_keys.insert(key); + shard.set_live(project.live_variants(path_id)); + workspace.shards[path_id] = std::move(shard); + } + llvm::SmallVector mask_refresh; + for(auto path_id: unservable) { + auto contribution_it = project.contributions.find(path_id); + if(contribution_it == project.contributions.end()) { + continue; + } + llvm::SmallVector owners; + for(auto tu: llvm::make_first_range(contribution_it->second)) { + owners.push_back(tu); + } + for(auto tu: owners) { + // The removal retires the TU's contributions to EVERY file it + // touched, not just the unservable one; the affected set feeds + // the mask refresh below, like the merge path's. + auto affected = project.remove_manifest(tu); + mask_refresh.append(affected.begin(), affected.end()); + storage.remove(index::IndexBlobKind::Manifest, + blob_key(workspace.path_pool.resolve(tu))); + enqueue(tu, ReindexReason::ContentChanged); + } + } + for(auto path_id: mask_refresh) { + auto it = workspace.shards.find(path_id); + if(it != workspace.shards.end()) { + it->second.set_live(project.live_variants(path_id)); + } + } + + // Sweep shard blobs nothing references any more. llvm::SmallVector orphans; - workspace.store->for_each_key("index", [&](llvm::StringRef key) { - if(key == "project" && has_project) - return; + storage.for_each_key(index::IndexBlobKind::Shard, [&](llvm::StringRef key) { if(!expected_keys.contains(key)) { orphans.push_back(key.str()); } }); - for(auto& key: orphans) { - workspace.store->invalidate("index", key); + storage.remove(index::IndexBlobKind::Shard, key); } - if(!workspace.merged_indices.empty()) { - LOG_INFO("Loaded {} MergedIndex shards", workspace.merged_indices.size()); + if(!workspace.shards.empty()) { + LOG_INFO("Loaded {} index shards, {} manifests, {} symbols", + workspace.shards.size(), + project.manifests.size(), + project.symbols.size()); } LOG_PERF("startup", - "phase=index_load symbols={} shards={} elapsed_ms={}", - workspace.project_index.symbols.size(), - workspace.merged_indices.size(), + "phase=index_load symbols={} shards={} manifests={} elapsed_ms={}", + project.symbols.size(), + workspace.shards.size(), + project.manifests.size(), timer.ms()); } +bool Indexer::file_version_stale(std::uint32_t fv_id) { + auto [cached, inserted] = fv_verdicts.try_emplace(fv_id, true); + if(!inserted) { + return cached->second; + } + + auto record_it = workspace.project_index.file_versions.find(fv_id); + if(record_it == workspace.project_index.file_versions.end()) { + return true; + } + auto& record = record_it->second; + auto path = workspace.path_pool.resolve(record.path_id); + + // Two-layer test on the shared FileVersion: Layer 1 trusts a stat EQUAL + // to the recorded stamp (no file read) — equality, not a watermark, so + // backdated or preserved mtimes cannot masquerade as fresh; Layer 2 + // re-hashes the disk against the consumed-content hash and treats a + // match as a mere touch, repairing the stamp in place — once, for every + // TU that consumes this version. + auto stale = [&] { + fs::file_status status; + if(auto err = fs::status(path, status)) { + return true; + } + if(record.mtime_ns != 0 && record.size == status.getSize() && + record.mtime_ns == fs::mtime_ns(status)) { + return false; + } + if(record.content_hash == 0) { + return true; + } + if(hash_file(path) != record.content_hash) { + return true; + } + record.size = status.getSize(); + record.mtime_ns = fs::mtime_ns(status); + global_dirty = true; + return false; + }(); + fv_verdicts[fv_id] = stale; + return stale; +} + bool Indexer::need_update(llvm::StringRef file_path) { - auto merged_it = workspace.merged_indices.find(workspace.path_pool.intern(file_path)); - if(merged_it == workspace.merged_indices.end()) + auto& project = workspace.project_index; + auto manifest_it = project.manifests.find(workspace.path_pool.intern(file_path)); + if(manifest_it == project.manifests.end()) return true; - return merged_it->second.need_update(); + // Every referenced version must be validated: whichever a partial + // iteration skipped would keep serving stale rows behind a fresh + // verdict. + auto& manifest = manifest_it->second; + if(file_version_stale(manifest.tu_fv)) { + return true; + } + for(auto& node: manifest.nodes) { + if(file_version_stale(node.fv)) { + return true; + } + } + return false; } void Indexer::enqueue(std::uint32_t server_path_id, ReindexReason reason) { @@ -716,6 +1058,10 @@ kota::task<> Indexer::run_background_indexing() { LOG_DEBUG("Background indexing: starting, {} files queued", index_queue.size() - index_queue_pos); + // FileVersion verdicts hold for one round: the disk can change under a + // running round, but staleness is re-judged per round anyway. + fv_verdicts.clear(); + std::stable_partition( index_queue.begin() + index_queue_pos, index_queue.end(), @@ -778,8 +1124,6 @@ kota::task<> Indexer::run_background_indexing() { progress_data.dispatched = dispatched; on_progress_changed.emit(); - indexing_active = false; - // Safe point to compact: no dispatch loop holds an index into the queue. // Files enqueued while we awaited the workers keep the queue alive for // the next scheduled round. @@ -798,6 +1142,11 @@ kota::task<> Indexer::run_background_indexing() { timer.ms()); co_await save(); + // The round owns the "active" gate through its save: releasing it + // before the write await would let a next round's save overlap this + // one's in-flight batch, racing same-key blob writes on the pool. + indexing_active = false; + // Files enqueued while the round was joining its workers saw their // schedule() no-op against indexing_active; without this kick they // would wait for the next external event — and a content-changed diff --git a/src/server/compiler/indexer.h b/src/server/compiler/indexer.h index 74142bbd6..2f9e549c7 100644 --- a/src/server/compiler/indexer.h +++ b/src/server/compiler/indexer.h @@ -46,12 +46,13 @@ enum class ReindexReason : std::uint8_t { /// /// Indexer owns the indexing queue and drives disk files through /// the stateless workers, merging each TUIndex result into Workspace's -/// ProjectIndex and MergedIndex shards. It holds no index data of its own. +/// ProjectIndex (manifests, FileVersions, symbols) and Shard blobs. It +/// holds no index data of its own beyond the dirty bookkeeping. /// /// Responsibilities: /// - Background indexing scheduling (enqueue → idle timer → worker dispatch) -/// - Merging TUIndex results into Workspace's ProjectIndex -/// - Persisting and restoring the index shards +/// - Merging TUIndex results into Workspace's index state +/// - Persisting and restoring the index blobs /// /// NOT responsible for: /// - Index queries — handled by IndexQuery @@ -135,21 +136,38 @@ class Indexer { /// Schedule background indexing (respects idle timeout and dedup). void schedule(); - /// Merge a TUIndex result into Workspace's ProjectIndex and MergedIndex shards. + /// Merge a TUIndex result: intern FileVersions, replace the TU's + /// manifest, and write row blobs only for variants no shard stores yet + /// — a re-index whose rows are unchanged records its contributions and + /// touches nothing else. void merge(const void* tu_index_data, std::size_t size); - /// Save Workspace's ProjectIndex and MergedIndex shards to the cache - /// store ("index" namespace, Persistent policy). Serialization runs - /// on the event loop; each blob's commit (fsync + rename) is offloaded - /// to the kota thread pool. + /// Drop a TU's index wholesale: manifest and contributions now (the + /// affected shards' live masks follow), persisted blobs at the next + /// save. For invalidation content-based freshness cannot see — a + /// compile-command change — where a surviving manifest would keep + /// judging the old-command rows fresh, in this session and after a + /// restart. + void drop_index(std::uint32_t tu_path_id); + + /// Persist the dirty state (rewritten shards, replaced manifests, the + /// global blob) through the index storage. Serialization runs on the + /// event loop from copies; the write batch is offloaded to the kota + /// thread pool. Shards whose variant set shrank are compacted first. kota::task<> save(); - /// Load Workspace's ProjectIndex and MergedIndex shards from the cache - /// store, sweeping orphaned shard blobs. + /// Load the global blob, adopt every resolvable manifest, fetch the + /// shard blobs the contributions expect, and sweep the rest. void load(); - /// Check whether a file needs re-indexing (stale or missing shard). - bool need_update(llvm::StringRef file_path); + /// Shard blobs whose write has not durably completed: dirty since the + /// last save plus the batch a running save is committing. The gauge + /// reaches zero only once every shard write settled — never in the + /// window where save() has snapshot-cleared the dirty set but its + /// commit (and the last_save_shards update) is still in flight. + std::size_t pending_shard_writes() const { + return dirty_shards.size() + saving_shards; + } /// Cancel background indexing and wait for all tasks to settle. kota::task<> stop(); @@ -169,10 +187,9 @@ class Indexer { return index_queue.size(); } - /// How many shard blobs the last save() durably committed. With the - /// post-commit flip-back this is the true dirty set — a steady-state - /// save commits 0 — so the stats endpoint can pin full-rewrite - /// regressions. + /// How many shard blobs the last save() durably committed. A + /// steady-state save commits 0 — only variant-set changes rewrite a + /// blob — so the stats endpoint can pin full-rewrite regressions. std::size_t last_save_shards() const { return saved_shards; } @@ -287,11 +304,35 @@ class Indexer { friend struct testing::IndexerFixture; + /// Blobs mutated since the last save, plus whether the global blob + /// (symbols, FileVersion table) changed. + llvm::DenseSet dirty_shards; + llvm::DenseSet dirty_manifests; + bool global_dirty = false; + + /// Per-round FileVersion staleness verdicts: many TUs share the same + /// versions, and one stat (or repair) per version per round is enough. + /// Cleared when a round starts. + llvm::DenseMap fv_verdicts; + + /// Two-layer staleness test on a FileVersion, cached per round; a hash + /// match after a stat mismatch repairs the version's stat fast path in + /// place for every consumer. + bool file_version_stale(std::uint32_t fv_id); + + /// Check whether a file needs re-indexing: no manifest, or a stale + /// FileVersion among its dependencies. Valid only within one round: + /// the verdicts above are cleared when a round starts, never here. + bool need_update(llvm::StringRef file_path); + llvm::DenseMap reindex_reasons; std::uint64_t reindex_ticket = 0; bool indexing_active = false; bool indexing_scheduled = false; std::size_t saved_shards = 0; + /// Shards in the batch a running save() is committing (see + /// pending_shard_writes). + std::size_t saving_shards = 0; std::shared_ptr index_idle_timer; /// Pause/resume: when paused, new index tasks wait on this event. diff --git a/src/server/protocol/extension.h b/src/server/protocol/extension.h index cc01eb381..2b4937fde 100644 --- a/src/server/protocol/extension.h +++ b/src/server/protocol/extension.h @@ -124,9 +124,9 @@ struct StatsResult { std::uint32_t pch_loaded_states = 0; std::uint64_t pch_state_bytes = 0; - /// Index shards holding a heap Impl (need_rewrite() true), and their - /// stored-content bytes. Zero after a settled save: committed shards - /// flip back to their buffer-backed blobs. + /// Shard blobs awaiting persistence (the indexer's dirty set — zero + /// after a settled save), and the total mapped bytes of every loaded + /// shard blob. std::uint32_t index_inmemory_shards = 0; std::uint64_t index_shard_content_bytes = 0; diff --git a/src/server/service/query.cpp b/src/server/service/query.cpp index 6ae6f516a..9c0c55410 100644 --- a/src/server/service/query.cpp +++ b/src/server/service/query.cpp @@ -188,10 +188,10 @@ bool IndexQuery::find_symbol_info(index::SymbolHash hash, if(found) return true; - // Check per-file MergedIndex shards (TU-local + file-local symbols). + // Check per-file Shard blobs (TU-local + file-local symbols). // Each shard stores exactly the local symbols its occurrences reference, // so the symbol will be in the shard that produced the occurrence. - for(auto& [path_id, shard]: workspace.merged_indices) { + for(auto& [path_id, shard]: workspace.shards) { if(shard.find_symbol(hash, name, kind)) return true; } @@ -245,7 +245,7 @@ IndexQuery::CursorHit IndexQuery::resolve_cursor(llvm::StringRef path, return hit; } - // Fallback to MergedIndex. Position -> offset uses the session text when + // Fallback to the disk shard. Position -> offset uses the session text when // one exists (open but not yet compiled); for closed files the shard's // own stored content provides the mapping. auto path_id = workspace.path_pool.find(path); @@ -255,8 +255,8 @@ IndexQuery::CursorHit IndexQuery::resolve_cursor(llvm::StringRef path, // resolved against them would name the wrong symbol. if(skip_stale_contribution(*path_id)) return {}; - auto shard_it = workspace.merged_indices.find(*path_id); - if(shard_it == workspace.merged_indices.end()) + auto shard_it = workspace.shards.find(*path_id); + if(shard_it == workspace.shards.end()) return {}; auto& merged_index = shard_it->second; @@ -302,8 +302,8 @@ std::vector IndexQuery::query_relations(llvm::StringRef path for(auto file_id: sym_it->second.reference_files) { if(skip_shard(file_id)) continue; - auto shard_it = workspace.merged_indices.find(file_id); - if(shard_it == workspace.merged_indices.end()) + auto shard_it = workspace.shards.find(file_id); + if(shard_it == workspace.shards.end()) continue; auto uri = lsp::URI::from_file_path(workspace.path_pool.resolve(file_id)); if(!uri) @@ -484,8 +484,8 @@ std::optional IndexQuery::find_definition_location(index::Sy for(auto file_id: sym_it->second.reference_files) { if(skip_shard(file_id)) continue; - auto shard_it = workspace.merged_indices.find(file_id); - if(shard_it == workspace.merged_indices.end()) + auto shard_it = workspace.shards.find(file_id); + if(shard_it == workspace.shards.end()) continue; auto uri = lsp::URI::from_file_path(workspace.path_pool.resolve(file_id)); if(!uri) @@ -538,8 +538,8 @@ void IndexQuery::collect_grouped_relations( for(auto file_id: sym_it->second.reference_files) { if(skip_shard(file_id)) continue; - auto shard_it = workspace.merged_indices.find(file_id); - if(shard_it == workspace.merged_indices.end()) + auto shard_it = workspace.shards.find(file_id); + if(shard_it == workspace.shards.end()) continue; auto& merged_index = shard_it->second; auto ls = merged_index.line_starts(); @@ -607,8 +607,8 @@ void IndexQuery::collect_unique_targets(index::SymbolHash hash, for(auto file_id: sym_it->second.reference_files) { if(skip_shard(file_id)) continue; - auto shard_it = workspace.merged_indices.find(file_id); - if(shard_it == workspace.merged_indices.end()) + auto shard_it = workspace.shards.find(file_id); + if(shard_it == workspace.shards.end()) continue; shard_it->second.lookup(hash, kind, [&](const index::Relation& r) { if(seen.insert(r.target_symbol).second) { @@ -687,8 +687,8 @@ std::optional IndexQuery::get_definition_text(index: for(auto file_id: sym_it->second.reference_files) { if(skip_shard(file_id)) continue; - auto shard_it = workspace.merged_indices.find(file_id); - if(shard_it == workspace.merged_indices.end()) + auto shard_it = workspace.shards.find(file_id); + if(shard_it == workspace.shards.end()) continue; auto& merged_index = shard_it->second; auto ls = merged_index.line_starts(); @@ -730,8 +730,8 @@ std::vector IndexQuery::collect_references(ind for(auto file_id: sym_it->second.reference_files) { if(skip_shard(file_id)) continue; - auto shard_it = workspace.merged_indices.find(file_id); - if(shard_it == workspace.merged_indices.end()) + auto shard_it = workspace.shards.find(file_id); + if(shard_it == workspace.shards.end()) continue; auto& merged_index = shard_it->second; auto ls = merged_index.line_starts(); @@ -975,8 +975,8 @@ std::vector IndexQuery::locate_symbols(const agentic::ReadSymbol if(skip_stale_contribution(*path_id)) return {}; - auto shard_it = workspace.merged_indices.find(*path_id); - if(shard_it == workspace.merged_indices.end()) + auto shard_it = workspace.shards.find(*path_id); + if(shard_it == workspace.shards.end()) return {}; auto& merged_index = shard_it->second; diff --git a/src/server/service/query.h b/src/server/service/query.h index 73dabdbf3..24e6cb800 100644 --- a/src/server/service/query.h +++ b/src/server/service/query.h @@ -47,7 +47,7 @@ struct ResolvedSymbol { /// Read-only index query layer. /// /// IndexQuery holds no index data of its own. All persistent data lives in -/// Workspace (disk-derived ProjectIndex + MergedIndex shards) and per-file +/// Workspace (disk-derived ProjectIndex + Shard blobs) and per-file /// data lives in Session (file index from unsaved buffers). /// /// Responsibilities: @@ -113,7 +113,7 @@ class IndexQuery { workspace(workspace), sessions(sessions), indexer(indexer), options(options) {} /// Query relations (Definition, Reference, etc.) for a symbol at cursor. - /// @param session Active Session for this file, or nullptr to use MergedIndex only. + /// @param session Active Session for this file, or nullptr to use the disk shards only. std::vector query_relations(llvm::StringRef path, const protocol::Position& position, RelationKind kind, @@ -217,7 +217,7 @@ class IndexQuery { }; /// Resolve the symbol at (position), checking Session's file_index first - /// then falling back to Workspace's MergedIndex. + /// then falling back to Workspace's disk shards. CursorHit resolve_cursor(llvm::StringRef path, const protocol::Position& position, Session* session); diff --git a/src/server/state/invalidator.cpp b/src/server/state/invalidator.cpp index 1ed3cf918..00ec49dfa 100644 --- a/src/server/state/invalidator.cpp +++ b/src/server/state/invalidator.cpp @@ -192,9 +192,9 @@ DirtySet Invalidator::apply(llvm::ArrayRef events) { dirty.add_clear_reindex(event.path_id); break; } - auto shard_it = workspace.merged_indices.find(event.path_id); - bool shard_current = shard_it != workspace.merged_indices.end() && - *disk == shard_it->second.content(); + auto shard_it = workspace.shards.find(event.path_id); + bool shard_current = + shard_it != workspace.shards.end() && *disk == shard_it->second.content(); if(shard_current) { dirty.add_reindex_deps_only(event.path_id); } else { @@ -308,31 +308,22 @@ DirtySet Invalidator::apply(llvm::ArrayRef events) { // see, whether it appeared, changed or vanished. PCH/PCM // keys embed the canonical flags, so pull-side caches miss // naturally. - auto invalidate_entry = [&](std::uint32_t path_id, bool keep_shard) { + auto invalidate_entry = [&](std::uint32_t path_id, bool keep_index) { if(store.find(path_id)) { // The next compile re-resolves the command (added: // first real entry replaces the guessed one; - // changed: new flags; removed: fall back). The - // shard was indexed under the old command either - // way — same eviction as the closed branch. + // changed: new flags; removed: fall back). dirty.mark_ast_dirty.push_back(path_id); - if(!keep_shard) { - workspace.merged_indices.erase(path_id); - dirty.add_reindex_content_changed(path_id); - } - } else if(!keep_shard) { - // The shard was indexed under the old command, and + } + if(!keep_index) { + // The index was built under the old command, and // the indexer's freshness gate validates content - // only: evict the shard so the queued reindex is - // not filtered out as fresh. ContentChanged: a new + // only: drop the TU's index so the queued reindex + // is not filtered out as fresh — in this session + // or after a restart. ContentChanged: a new // command can rewrite the rows (macros, includes) - // as thoroughly as an edit — and the shard is gone - // anyway. - // TODO: a background index task already in flight - // can merge its old-command result back after this - // eviction; closing that window needs an index - // generation guard in the indexer. - workspace.merged_indices.erase(path_id); + // as thoroughly as an edit. + dirty.drop_index.push_back(path_id); dirty.add_reindex_content_changed(path_id); } @@ -353,28 +344,31 @@ DirtySet Invalidator::apply(llvm::ArrayRef events) { continue; } dirty.drop_context.push_back(header_id); + // A standalone-indexed header borrowed the changed + // command too; its manifest is as stale as the + // host's (no-op for headers indexed only via TUs). + dirty.drop_index.push_back(header_id); if(store.find(header_id)) { dirty.mark_ast_dirty.push_back(header_id); } else { - workspace.merged_indices.erase(header_id); dirty.add_reindex_content_changed(header_id); } } }; for(auto path_id: delta.added) { - invalidate_entry(path_id, /*keep_shard=*/false); + invalidate_entry(path_id, /*keep_index=*/false); } for(auto path_id: delta.changed) { - invalidate_entry(path_id, /*keep_shard=*/false); + invalidate_entry(path_id, /*keep_index=*/false); } for(auto path_id: delta.removed) { - // A removed entry keeps its shard: the last-known + // A removed entry keeps its index: the last-known // content still serves navigation, same conservative // semantics as DiskRemoved. The graph rebuild above // already dropped the file's source role, and the // orphan recheck cleans choices through it. - invalidate_entry(path_id, /*keep_shard=*/true); + invalidate_entry(path_id, /*keep_index=*/true); } // The first CDB of the session may have introduced C++20 @@ -423,6 +417,7 @@ DirtySet Invalidator::apply(llvm::ArrayRef events) { dedup(dirty.force_revalidate); dedup(dirty.reindex_content_changed); dedup(dirty.reindex_deps_only); + dedup(dirty.drop_index); dedup(dirty.drop_context); return dirty; } diff --git a/src/server/state/invalidator.h b/src/server/state/invalidator.h index 34326edfb..391bd2d2e 100644 --- a/src/server/state/invalidator.h +++ b/src/server/state/invalidator.h @@ -173,6 +173,12 @@ struct DirtySet { void add_clear_reindex(std::uint32_t path_id) { erase_id(reindex_content_changed, path_id); erase_id(reindex_deps_only, path_id); + // A removal retains the last-known index, so it also cancels an + // earlier entry-change drop — a surviving drop would mask the shard + // and let the next save retire it. The reverse order needs no + // handling: every drop emission is paired with a reindex adder, + // which already un-clears. + erase_id(drop_index, path_id); if(llvm::find(clear_reindex, path_id) == clear_reindex.end()) { clear_reindex.push_back(path_id); } @@ -184,6 +190,14 @@ struct DirtySet { } public: + /// TUs whose compile command changed: their index describes a compile + /// that no longer exists, and content-based freshness cannot see that. + /// The indexer drops the manifest, contributions and persisted blobs — + /// a surviving manifest would judge the queued reindex fresh and keep + /// the old-command rows serving, in this session and after a restart. + /// Follows the later-event rule above: a later removal's clear cancels + /// the drop, since the deleted file's last-known index keeps serving. + llvm::SmallVector drop_index; /// Headers whose resolved context borrows a compile command that no /// longer exists in that form (the host's CDB entry changed): drop the /// context so the next use re-resolves. Content validation cannot see @@ -206,8 +220,8 @@ struct DirtySet { return mark_ast_dirty.empty() && mark_lost.empty() && reset_trial.empty() && reset_header_mode.empty() && force_revalidate.empty() && reindex_content_changed.empty() && reindex_deps_only.empty() && - clear_reindex.empty() && drop_context.empty() && !recheck_contexts && !save_cache && - !reschedule_indexing && !ensure_compile_graph; + clear_reindex.empty() && drop_index.empty() && drop_context.empty() && + !recheck_contexts && !save_cache && !reschedule_indexing && !ensure_compile_graph; } }; diff --git a/src/server/state/workspace.h b/src/server/state/workspace.h index 035a32a89..e9ee46cd0 100644 --- a/src/server/state/workspace.h +++ b/src/server/state/workspace.h @@ -12,9 +12,10 @@ #include "command/command.h" #include "command/toolchain.h" #include "compile/dep_file.h" -#include "index/merged_index.h" #include "index/preamble_state.h" #include "index/project_index.h" +#include "index/shard.h" +#include "index/storage.h" #include "semantic/symbol.h" #include "server/compiler/compile_graph.h" #include "server/state/config.h" @@ -35,7 +36,7 @@ class ContextResolver; /// On-disk cache layout version (CacheStore root `cache/v{N}`). /// Bump to discard all cached artifacts after incompatible format changes. -constexpr inline std::uint32_t cache_format_version = 5; +constexpr inline std::uint32_t cache_format_version = 6; /// Sentinel for "no path": path pool ids start at 0, so 0 is a real file. constexpr inline std::uint32_t no_path_id = ~0u; @@ -48,8 +49,8 @@ constexpr inline std::uint32_t no_path_id = ~0u; /// since before the build started, so matching them proves the disk still /// holds the consumed content. mtime_ns == 0 means "no fast path" — the /// check falls through to the hash comparison and, on a match, repairs the -/// fast path in place. The index shard's immutable, non-repairing form of -/// the same fast path is `DepStamp` (merged_index.cpp). +/// fast path in place. The index's form of the same fast path is +/// `index::FileVersionRecord`, shared by every TU consuming the version. struct DepState { std::uint32_t path_id = no_path_id; std::uint64_t size = 0; @@ -188,7 +189,7 @@ struct PCMState { /// Workspace is the single source of truth for: /// - dependency relationships (include graph, module DAG) /// - compilation artifacts shared across files (PCH/PCM caches) -/// - symbol index (ProjectIndex + per-file MergedIndex shards) +/// - symbol index (ProjectIndex + per-file Shard blobs) /// - compilation database and configuration /// /// Workspace is NEVER modified by unsaved buffer content. The only mutation @@ -255,13 +256,19 @@ struct Workspace { /// Maps to the .pcm file on disk used as -fmodule-file argument. llvm::DenseMap pcm_paths; - /// Global symbol table across all indexed translation units. + /// The index's global layer: symbols, FileVersions, per-TU manifests + /// and the derived contribution map. index::ProjectIndex project_index; - /// Per-file index shards from background indexing, keyed by project-level - /// path_id. Contains symbol occurrences, relations, and stored content - /// for position mapping. - llvm::DenseMap merged_indices; + /// Per-file row blobs from background indexing, keyed by project-level + /// path_id: symbol occurrences, relations and stored content for + /// position mapping, served zero-copy. + llvm::DenseMap shards; + + /// Index blob persistence, opened together with the cache store. + /// Declared after `store`: the filesystem backend borrows it, so it + /// must be destroyed first. + std::unique_ptr index_storage; /// Monotonic generation of context-affecting workspace state (include /// graph, CDB, disk contents). Bumped on didSave; clice/queryContext diff --git a/src/server/transport/agent_client.cpp b/src/server/transport/agent_client.cpp index 2d1d24dfb..43c204813 100644 --- a/src/server/transport/agent_client.cpp +++ b/src/server/transport/agent_client.cpp @@ -118,7 +118,7 @@ AgentClient::AgentClient(MasterServer& server, kota::ipc::JsonPeer& peer) : } if(filter == "all" || filter == "header") { - for(auto& [path_id, shard]: ws.merged_indices) { + for(auto& [path_id, shard]: ws.shards) { if(seen.contains(path_id)) continue; auto path_str = ws.path_pool.resolve(path_id); @@ -355,8 +355,8 @@ AgentClient::AgentClient(MasterServer& server, kota::ipc::JsonPeer& peer) : if(srv.agent_query.skip_shard(*path_id)) co_return result; - auto shard_it = srv.workspace.merged_indices.find(*path_id); - if(shard_it == srv.workspace.merged_indices.end()) + auto shard_it = srv.workspace.shards.find(*path_id); + if(shard_it == srv.workspace.shards.end()) co_return result; auto& merged_index = shard_it->second; diff --git a/src/server/transport/lsp_client.cpp b/src/server/transport/lsp_client.cpp index 5831f1c41..e0358b7e5 100644 --- a/src/server/transport/lsp_client.cpp +++ b/src/server/transport/lsp_client.cpp @@ -645,11 +645,10 @@ void LSPClient::register_extensions() { } stats.pch_cache_entries = static_cast(srv.workspace.pch_cache.size()); - for(auto& [path_id, shard]: srv.workspace.merged_indices) { - if(shard.need_rewrite()) { - stats.index_inmemory_shards += 1; - stats.index_shard_content_bytes += shard.content().size(); - } + stats.index_inmemory_shards = + static_cast(srv.indexer.pending_shard_writes()); + for(auto& [path_id, shard]: srv.workspace.shards) { + stats.index_shard_content_bytes += shard.bytes().size(); } stats.last_save_shards = static_cast(srv.indexer.last_save_shards()); diff --git a/src/server/transport/master_server.cpp b/src/server/transport/master_server.cpp index 1bec108f5..aa822c340 100644 --- a/src/server/transport/master_server.cpp +++ b/src/server/transport/master_server.cpp @@ -306,9 +306,9 @@ void MasterServer::on_agentic_query() { if(!disk) { continue; } - auto shard_it = workspace.merged_indices.find(path_id); + auto shard_it = workspace.shards.find(path_id); bool shard_current = - shard_it != workspace.merged_indices.end() && *disk == shard_it->second.content(); + shard_it != workspace.shards.end() && *disk == shard_it->second.content(); indexer.enqueue(path_id, shard_current ? ReindexReason::DepsOnly : ReindexReason::ContentChanged); } @@ -367,6 +367,10 @@ void MasterServer::dispatch(llvm::ArrayRef events) { contexts.drop_header_context(path_id); } + for(auto path_id: dirty.drop_index) { + indexer.drop_index(path_id); + } + for(auto path_id: dirty.reindex_content_changed) { indexer.enqueue(path_id, ReindexReason::ContentChanged); } @@ -484,11 +488,11 @@ void MasterServer::open_cache_store() { .max_bytes = 8 * GiB}); store->register_namespace( {.name = "pcm", .extension = ".pcm", .policy = CachePolicy::LRU, .max_bytes = 8 * GiB}); - store->register_namespace( - {.name = "index", .extension = ".idx", .policy = CachePolicy::Persistent}); store->register_namespace( {.name = "header_context", .extension = ".h", .policy = CachePolicy::Scratch}); workspace.store.emplace(std::move(*store)); + // Registers the index namespaces itself. + workspace.index_storage = index::make_fs_index_storage(*workspace.store); LOG_INFO("Cache store: {}", workspace.store->base_dir()); workspace.load_cache(contexts); diff --git a/tests/integration/features/index_staleness.test.ts b/tests/integration/features/index_staleness.test.ts index 43615034d..23e15aaa7 100644 --- a/tests/integration/features/index_staleness.test.ts +++ b/tests/integration/features/index_staleness.test.ts @@ -10,7 +10,8 @@ import { expect, test } from "../fixtures.ts"; const HEADER = "#pragma once\ninline int alpha() { return 1; }\n"; const CLOSED_TU = '#include "header.h"\nint use() { return alpha(); }\n'; -/// mtimes of the per-TU shards (numeric names; excludes the project blob). +/// mtimes of the per-file shard blobs (the "index" namespace holds nothing +/// else). function shardMtimes(workspace: Workspace): Map { const dir = path.join(workspace.cacheRoot(), "index"); const shards = new Map(); @@ -18,15 +19,15 @@ function shardMtimes(workspace: Workspace): Map { return shards; } for (const name of fs.readdirSync(dir)) { - if (name.endsWith(".idx") && name.slice(0, -".idx".length) !== "project") { + if (name.endsWith(".idx")) { shards.set(name, fs.statSync(path.join(dir, name), { bigint: true }).mtimeNs); } } return shards; } -function projectMtime(workspace: Workspace): bigint { - const p = path.join(workspace.cacheRoot(), "index", "project.idx"); +function globalMtime(workspace: Workspace): bigint { + const p = path.join(workspace.cacheRoot(), "index-global", "global.idx"); return fs.existsSync(p) ? fs.statSync(p, { bigint: true }).mtimeNs : 0n; } @@ -63,13 +64,15 @@ test("touch header no reindex", async ({ session }) => { // completes (save() rewrites the project blob), then re-snapshot: the // storm filter must skip the closed TU, leaving its shard untouched. const before = shardMtimes(workspace); - const projectBefore = projectMtime(workspace); + const globalBefore = globalMtime(workspace); const c2 = session.spawn(workspace); await c2.initialize(workspace); - // save() rewrites the project blob unconditionally each round (see - // Indexer::save), so its mtime moving proves the round ran. + // The touch makes the header's stat mismatch its FileVersion stamp; the + // staleness check re-hashes, proves a mere touch, and repairs the stamp + // — which dirties the global blob, so its mtime moving proves both that + // the round ran and that the repair persisted. expect( - await poll(() => projectMtime(workspace) !== projectBefore), + await poll(() => globalMtime(workspace) !== globalBefore), "indexing round never ran in session 2", ).toBe(true); const after = shardMtimes(workspace); diff --git a/tests/unit/index/index_query_tests.cpp b/tests/unit/index/index_query_tests.cpp index bf5a80dfd..0025d67f7 100644 --- a/tests/unit/index/index_query_tests.cpp +++ b/tests/unit/index/index_query_tests.cpp @@ -1,449 +1,205 @@ +#include +#include + #include "test/test.h" #include "test/tester.h" -#include "index/merged_index.h" -#include "index/project_index.h" +#include "index/shard.h" #include "index/tu_index.h" +#include "server/compiler/context_resolver.h" +#include "server/compiler/indexer.h" +#include "server/service/query.h" +#include "server/state/session_store.h" +#include "server/worker/worker_pool.h" + +#include "llvm/ADT/SmallVector.h" +#include "llvm/Support/MemoryBuffer.h" +#include "llvm/Support/Path.h" +#include "llvm/Support/xxhash.h" namespace clice::testing { namespace { TEST_SUITE(IndexQuery, Tester) { -index::ProjectIndex project_index; -clice::PathPool pool; -llvm::DenseMap merged_indices; - -/// Build TUIndex from code and merge into ProjectIndex + MergedIndex shards. -void build_and_merge(llvm::StringRef code, - std::source_location location = std::source_location::current()) { - add_main("main.cpp", code); - ASSERT_TRUE(compile()); - +kota::event_loop loop; +Workspace workspace; +SessionStore store; +WorkerPool pool{loop}; +ContextResolver resolver{workspace}; +Indexer indexer{loop, workspace, pool, resolver, store}; +clice::IndexQuery query{workspace, store, indexer}; + +std::uint32_t main_id = 0; +std::uint32_t header_id = 0; + +/// Mirror of the indexer's merge over in-memory sources: project symbols, +/// per-section shard blobs, and the TU manifest with its contributions — +/// so live-variant masks and staleness gates behave as in production. +void merge_into_workspace() { auto tu_index = index::TUIndex::build(*unit); - auto file_ids_map = project_index.merge(tu_index, pool); - - // Merge main file index as compilation context. - auto main_tu_path_id = static_cast(tu_index.graph.paths.size() - 1); - auto main_global_id = file_ids_map[main_tu_path_id]; - llvm::StringRef main_tu_path = tu_index.graph.paths[main_tu_path_id]; - - llvm::SmallVector deps; - for(auto& loc: tu_index.graph.locations) { - deps.push_back({tu_index.graph.paths[loc.path_id], loc.line, loc.include}); - } - - merged_indices[main_global_id].merge(main_tu_path, - tu_index.built_at, - deps, - tu_index.main_file_index, - {}); - - // Merge header file indices. - for(auto& [fid, file_idx]: tu_index.file_indices) { - auto tu_pid = tu_index.graph.path_id(fid); - auto global_pid = file_ids_map[tu_pid]; - auto include_id = tu_index.graph.include_location_id(fid); - merged_indices[global_pid].merge(main_tu_path, include_id, file_idx, {}); - } -} - -/// Reset index state between test cases. -void reset() { - project_index = index::ProjectIndex(); - pool = clice::PathPool(); - merged_indices.clear(); - clear(); -} - -/// Lookup the symbol hash at a given annotation offset in any merged index. -index::SymbolHash lookup_symbol(llvm::StringRef pos) { - auto offset = point(pos); - index::SymbolHash result = 0; - for(auto& [path_id, merged]: merged_indices) { - merged.lookup(offset, [&](const index::Occurrence& o) { - if(o.range.contains(offset)) { - result = o.target; - return false; - } - return true; - }); - if(result != 0) - break; + std::string wire; + llvm::raw_string_ostream wos(wire); + tu_index.serialize(wos); + auto view = index::TUIndexView::from(wire); + ASSERT_TRUE(view.has_value()); + + auto& project = workspace.project_index; + llvm::SmallVector file_ids_map; + for(std::uint32_t i = 0; i < view->path_count(); i += 1) { + file_ids_map.push_back(workspace.path_pool.intern(view->path(i))); } - return result; -} - -/// Find all relations of a given kind for a symbol across all merged indices. -std::vector find_relations(index::SymbolHash symbol, RelationKind kind) { - std::vector results; - - auto sym_it = project_index.symbols.find(symbol); - if(sym_it == project_index.symbols.end()) - return results; + ASSERT_TRUE(project.merge(*view, file_ids_map)); + main_id = file_ids_map[view->path_count() - 1]; - // Search every shard that references this symbol. - for(auto file_id: sym_it->second.reference_files) { - auto it = merged_indices.find(file_id); - if(it == merged_indices.end()) - continue; - - it->second.lookup(symbol, kind, [&](const index::Relation& r) { - results.push_back(r); - return true; - }); - } + auto content_of = [&](llvm::StringRef path) -> llvm::StringRef { + auto it = sources.all_files.find(llvm::sys::path::filename(path)); + return it != sources.all_files.end() ? llvm::StringRef(it->second.content) + : llvm::StringRef(); + }; + auto lookup_symbol = [&](index::SymbolHash hash) { + return view->find_symbol(hash); + }; - // Also search all shards (symbol may appear in files not tracked by reference_files). - if(results.empty()) { - for(auto& [pid, merged]: merged_indices) { - merged.lookup(symbol, kind, [&](const index::Relation& r) { - results.push_back(r); - return true; - }); + index::TUManifest manifest; + manifest.tu_fv = project.intern_file_version(main_id, view->path_hash(view->path_count() - 1)); + + for(std::uint32_t section = 0; section < view->section_count(); section += 1) { + auto local_id = view->section_path(section); + auto global_id = file_ids_map[local_id]; + auto rows = view->decode_section_rows(section); + ASSERT_TRUE(rows.has_value()); + auto content = content_of(view->path(local_id)); + index::VariantInput fresh{view->section_rows_hash(section), &*rows, lookup_symbol}; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard(index::Shard(), {}, fresh, content, llvm::xxh3_64bits(content), os); + workspace.shards[global_id] = + index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); + + auto fv = project.intern_file_version(global_id, view->path_hash(local_id)); + manifest.contributions.emplace_back(fv, view->section_rows_hash(section)); + if(llvm::sys::path::filename(view->path(local_id)) == "header.h") { + header_id = global_id; } } - return results; -} - -// ============================================================ -// Test cases -// ============================================================ - -TEST_CASE(GoToDefinition) { - reset(); - build_and_merge(R"( - int §(decl)foo(); - - int §(def)⟦§(def)foo⟧() { return 42; } - - int main() { - return §(use)foo(); - } - )"); - - auto hash = lookup_symbol("use"); - ASSERT_NE(hash, 0UL); - - auto defs = find_relations(hash, RelationKind::Definition); - ASSERT_FALSE(defs.empty()); - - auto expected = range("def"); - ASSERT_EQ(dump(defs.front().range), dump(expected)); -} - -TEST_CASE(FindReferences) { - reset(); - build_and_merge(R"( - int §(decl)foo(); - - int §(def)foo() { return 42; } - - int bar() { - return §(ref1)foo() + §(ref2)foo(); - } - )"); - - auto hash = lookup_symbol("decl"); - ASSERT_NE(hash, 0UL); - - auto refs = find_relations(hash, RelationKind::Reference); - ASSERT_GE(refs.size(), 2U); -} - -TEST_CASE(DeclAndDef) { - reset(); - build_and_merge(R"( - int §(decl)foo(); - int §(def)⟦§(def)foo⟧() { return 42; } - )"); - - auto hash = lookup_symbol("decl"); - ASSERT_NE(hash, 0UL); - - auto decls = find_relations(hash, RelationKind::Declaration); - ASSERT_FALSE(decls.empty()); - - auto defs = find_relations(hash, RelationKind::Definition); - ASSERT_FALSE(defs.empty()); - - auto expected_def = range("def"); - ASSERT_EQ(dump(defs.front().range), dump(expected_def)); -} - -TEST_CASE(CallerCallee) { - reset(); - build_and_merge(R"( - void §(callee_def)callee() {} - - void §(caller_def)caller() { - §(call_site)callee(); + for(auto path_id: project.apply_manifest(main_id, std::move(manifest))) { + auto it = workspace.shards.find(path_id); + if(it != workspace.shards.end()) { + it->second.set_live(project.live_variants(path_id)); } - )"); - - auto caller_hash = lookup_symbol("caller_def"); - ASSERT_NE(caller_hash, 0UL); - - auto callees = find_relations(caller_hash, RelationKind::Callee); - ASSERT_FALSE(callees.empty()); - - auto callee_hash = lookup_symbol("callee_def"); - ASSERT_NE(callee_hash, 0UL); - - auto callers = find_relations(callee_hash, RelationKind::Caller); - ASSERT_FALSE(callers.empty()); -} - -TEST_CASE(OverrideRelation) { - reset(); - build_and_merge(R"( - struct Base { - virtual void §(base_method)method() {} - }; - - struct Derived : Base { - void §(derived_method)method() override {} - }; - )"); - - // Derived::method should have Interface relation to Base::method. - auto derived_hash = lookup_symbol("derived_method"); - ASSERT_NE(derived_hash, 0UL); - - auto interfaces = find_relations(derived_hash, RelationKind::Interface); - ASSERT_FALSE(interfaces.empty()); - - // Base::method should have Implementation relation. - auto base_hash = lookup_symbol("base_method"); - ASSERT_NE(base_hash, 0UL); - - auto impls = find_relations(base_hash, RelationKind::Implementation); - ASSERT_FALSE(impls.empty()); -} - -TEST_CASE(BaseAndDerived) { - reset(); - build_and_merge(R"( - struct §(base_cls)Animal { - virtual void speak() {} - }; - - struct §(derived_cls)Dog : §(base_ref)Animal { - void speak() override {} - }; - )"); - - auto derived_hash = lookup_symbol("derived_cls"); - ASSERT_NE(derived_hash, 0UL); - - // Look for any Base relation in any shard. - bool found_base = false; - for(auto& [pid, merged]: merged_indices) { - merged.lookup(derived_hash, RelationKind::Base, [&](const index::Relation& r) { - found_base = true; - return false; - }); } - ASSERT_TRUE(found_base); } -TEST_CASE(ClassTemplate) { - reset(); - build_and_merge(R"( - template - struct §(primary)⟦§(primary)foo⟧ {}; - - §(use)foo x; - )"); - - auto hash = lookup_symbol("use"); - ASSERT_NE(hash, 0UL); - - auto defs = find_relations(hash, RelationKind::Definition); - ASSERT_FALSE(defs.empty()); +std::string main_path() { + return std::string(workspace.path_pool.resolve(main_id)); } -TEST_CASE(SymbolKinds) { - reset(); - build_and_merge(R"( - struct §(cls)MyClass {}; - void §(func)myFunc() {} - int §(var)myVar = 0; +TEST_CASE(DefinitionAcrossFiles) { + add_file("header.h", R"( + struct §(def)⟦§(def)Widget⟧ { int value; }; )"); - - auto cls_hash = lookup_symbol("cls"); - ASSERT_NE(cls_hash, 0UL); - ASSERT_TRUE(project_index.symbols.contains(cls_hash)); - ASSERT_EQ(project_index.symbols[cls_hash].kind.value(), SymbolKind(SymbolKind::Struct).value()); - - auto func_hash = lookup_symbol("func"); - ASSERT_NE(func_hash, 0UL); - ASSERT_TRUE(project_index.symbols.contains(func_hash)); - ASSERT_EQ(project_index.symbols[func_hash].kind.value(), - SymbolKind(SymbolKind::Function).value()); - - auto var_hash = lookup_symbol("var"); - ASSERT_NE(var_hash, 0UL); - ASSERT_TRUE(project_index.symbols.contains(var_hash)); - ASSERT_EQ(project_index.symbols[var_hash].kind.value(), - SymbolKind(SymbolKind::Variable).value()); -} - -TEST_CASE(ReferenceFiles) { - reset(); - build_and_merge(R"( - int §(target)target = 42; - int a = §(ref)target + 1; + add_main("main.cpp", R"( + #include "header.h" + §(use)⟦§(use)Widget⟧ instance; )"); + ASSERT_TRUE(compile()); + merge_into_workspace(); - auto hash = lookup_symbol("target"); - ASSERT_NE(hash, 0UL); - - auto sym_it = project_index.symbols.find(hash); - ASSERT_TRUE(sym_it != project_index.symbols.end()); + auto hit_offset = point("use"); + index::SymbolHash symbol = 0; + workspace.shards[main_id].lookup(hit_offset, [&](const index::Occurrence& o) { + symbol = o.target; + return false; + }); + ASSERT_TRUE(symbol != 0); - // reference_files should contain at least the main file. - ASSERT_FALSE(sym_it->second.reference_files.isEmpty()); + auto location = query.find_definition_location(symbol); + ASSERT_TRUE(location.has_value()); + ASSERT_TRUE(llvm::StringRef(location->uri).ends_with("header.h")); } -TEST_CASE(CrossFileQuery) { - reset(); - +TEST_CASE(ReferencesAcrossFiles) { add_file("header.h", R"( - #pragma once - int §(hdr_decl)helper(); + int shared_fn(); )"); add_main("main.cpp", R"( #include "header.h" - - int main() { - return §(use_helper)helper(); - } + int call() { return §(use)⟦§(use)shared_fn⟧(); } )"); ASSERT_TRUE(compile()); + merge_into_workspace(); - auto tu_index = index::TUIndex::build(*unit); - auto file_ids_map = project_index.merge(tu_index, pool); - - // Merge main file. - auto main_tu_path_id = static_cast(tu_index.graph.paths.size() - 1); - auto main_global_id = file_ids_map[main_tu_path_id]; - - llvm::StringRef main_tu_path = tu_index.graph.paths[main_tu_path_id]; - llvm::SmallVector deps; - for(auto& loc: tu_index.graph.locations) { - deps.push_back({tu_index.graph.paths[loc.path_id], loc.line, loc.include}); - } - merged_indices[main_global_id].merge(main_tu_path, - tu_index.built_at, - deps, - tu_index.main_file_index, - {}); - - // Merge header file indices. - for(auto& [fid, file_idx]: tu_index.file_indices) { - auto tu_pid = tu_index.graph.path_id(fid); - auto global_pid = file_ids_map[tu_pid]; - auto include_id = tu_index.graph.include_location_id(fid); - merged_indices[global_pid].merge(main_tu_path, include_id, file_idx, {}); - } - - // Query: from usage in main.cpp, find the symbol via merged index. - auto use_offset = point("use_helper"); - index::SymbolHash helper_hash = 0; - merged_indices[main_global_id].lookup(use_offset, [&](const index::Occurrence& o) { - if(o.range.contains(use_offset)) { - helper_hash = o.target; - return false; - } - return true; + index::SymbolHash symbol = 0; + workspace.shards[main_id].lookup(point("use"), [&](const index::Occurrence& o) { + symbol = o.target; + return false; }); - ASSERT_NE(helper_hash, 0UL); + ASSERT_TRUE(symbol != 0); - // Find declaration across all shards -- should find it in header shard. - auto decls = find_relations(helper_hash, RelationKind::Declaration); - ASSERT_FALSE(decls.empty()); + auto references = query.collect_references(symbol, RelationKind::Reference); + ASSERT_FALSE(references.empty()); } -TEST_CASE(ImplementationDirection) { - /// Locks the relation direction used by go-to-implementation: at a base - /// virtual method, Implementation relations point to the overrides; at - /// an override, Interface relations point back to the overridden method. - reset(); - build_and_merge(R"( - struct Base { - virtual void §(base)draw(); - }; - struct Circle : Base { - void §(circle)draw() override; - }; - struct Square : Circle { - void §(square)draw() override; - }; +TEST_CASE(SearchSymbols) { + add_main("main.cpp", R"( + struct Searchable { int field; }; + Searchable instance; )"); + ASSERT_TRUE(compile()); + merge_into_workspace(); - auto base = lookup_symbol("base"); - auto circle = lookup_symbol("circle"); - auto square = lookup_symbol("square"); - ASSERT_NE(base, 0UL); - ASSERT_NE(circle, 0UL); - ASSERT_NE(square, 0UL); - - auto targets_of = [&](index::SymbolHash sym, RelationKind kind) { - std::vector targets; - for(auto& r: find_relations(sym, kind)) - targets.push_back(r.target_symbol); - return targets; - }; + auto results = query.search_symbols("Searchable", 10); + ASSERT_FALSE(results.empty()); + ASSERT_EQ(results.front().name, "Searchable"); +} - auto impls = targets_of(base, RelationKind::Implementation); - EXPECT_TRUE(std::ranges::contains(impls, circle)); +TEST_CASE(LocalSymbolName) { + add_main("main.cpp", R"( + static int §(local)⟦§(local)hidden⟧() { return 1; } + int use() { return hidden(); } + )"); + ASSERT_TRUE(compile()); + merge_into_workspace(); - auto interfaces = targets_of(circle, RelationKind::Interface); - EXPECT_TRUE(std::ranges::contains(interfaces, base)); + index::SymbolHash symbol = 0; + workspace.shards[main_id].lookup(point("local"), [&](const index::Occurrence& o) { + symbol = o.target; + return false; + }); + ASSERT_TRUE(symbol != 0); - // The chain is direct-base only: Square::draw implements Circle::draw. - auto circle_impls = targets_of(circle, RelationKind::Implementation); - EXPECT_TRUE(std::ranges::contains(circle_impls, square)); - EXPECT_FALSE(std::ranges::contains(impls, square)); + // TU-local names are not in the project table; the query falls back to + // the shard's own local-name table. + std::string name; + SymbolKind kind; + ASSERT_TRUE(query.find_symbol_info(symbol, name, kind)); + ASSERT_EQ(name, "hidden"); } -TEST_CASE(TypeDefinitionTargets) { - /// Locks the data go-to-type-definition relies on: TypeDefinition - /// relations at variable declarations carry the type's symbol hash. - reset(); - build_and_merge(R"( - struct §(widget)Widget {}; - using §(alias)Alias = Widget; - - Widget §(plain)w; - Alias §(aliased)a; - auto §(deduced)b = Widget{}; +TEST_CASE(StaleContributionSuppressed) { + add_main("main.cpp", R"( + int stale_fn() { return 1; } + int use() { return §(use)⟦§(use)stale_fn⟧(); } )"); + ASSERT_TRUE(compile()); + merge_into_workspace(); - auto widget = lookup_symbol("widget"); - auto alias = lookup_symbol("alias"); - ASSERT_NE(widget, 0UL); - ASSERT_NE(alias, 0UL); - - auto type_targets = [&](llvm::StringRef pos) { - auto sym = lookup_symbol(pos); - EXPECT_NE(sym, 0UL); - std::vector targets; - for(auto& r: find_relations(sym, RelationKind::TypeDefinition)) - targets.push_back(r.target_symbol); - return targets; - }; + index::SymbolHash symbol = 0; + workspace.shards[main_id].lookup(point("use"), [&](const index::Occurrence& o) { + symbol = o.target; + return false; + }); + ASSERT_FALSE(query.collect_references(symbol, RelationKind::Reference).empty()); - EXPECT_TRUE(std::ranges::contains(type_targets("plain"), widget)); - // Known index gaps (recorded, to be fixed in the indexer separately): - // auto-deduced and alias-typed variables do not resolve to the - // underlying record yet. Lock the current behavior so a future fix - // shows up as an intentional test update. - EXPECT_TRUE(type_targets("deduced").empty()); - EXPECT_TRUE(std::ranges::contains(type_targets("aliased"), alias)); + // A content-changed pending file's rows describe text that no longer + // exists: its contribution disappears from cross-file results until + // the reindex lands. + indexer.enqueue(main_id, ReindexReason::ContentChanged); + ASSERT_TRUE(query.collect_references(symbol, RelationKind::Reference).empty()); } }; // TEST_SUITE(IndexQuery) + } // namespace } // namespace clice::testing diff --git a/tests/unit/index/merged_index_tests.cpp b/tests/unit/index/merged_index_tests.cpp deleted file mode 100644 index 1052c33f9..000000000 --- a/tests/unit/index/merged_index_tests.cpp +++ /dev/null @@ -1,1132 +0,0 @@ -#include -#include -#include -#include - -#include "test/temp_dir.h" -#include "test/test.h" -#include "test/tester.h" -#include "index/merged_index.h" -#include "index/serialization.h" - -#include "llvm/ADT/DenseMap.h" -#include "llvm/ADT/SmallVector.h" -#include "llvm/Support/raw_ostream.h" -#include "llvm/Support/xxhash.h" - -namespace clice::testing { - -namespace { - -TEST_SUITE(MergedIndex, Tester) { - -index::TUIndex tu_index; - -void build_index(llvm::StringRef code, - std::source_location location = std::source_location::current()) { - add_main("main.cpp", code); - ASSERT_TRUE(compile()); - - tu_index = index::TUIndex::build(*unit); -}; - -void EXPECT_SELECT(llvm::StringRef pos, - llvm::StringRef expect_range, - llvm::StringRef file = "", - std::source_location location = std::source_location::current()) { - auto offset = point(pos, file); - auto expected = range(expect_range, file); - - auto fid = file.empty() ? unit->interested_file() : unit->file_id(file); - auto& index = tu_index.file_indices[fid]; - - auto it = - std::ranges::lower_bound(index.occurrences, offset, {}, [](index::Occurrence& occurrence) { - return occurrence.range.end; - }); - - auto err = std::format("Fail to find symbol for offset: {}, expected range: {}", - offset, - dump(expected)); - - ASSERT_TRUE(it != index.occurrences.end()); - - /// FIXME: Make eq pretty print reflectable struct. - ASSERT_EQ(dump(it->range), dump(expected)); -} - -TEST_CASE(Serialization) { - build_index(R"( - struct Foo { int x; int y; }; - Foo make_foo() { return Foo{1, 2}; } - int use_foo() { return make_foo().x; } - )"); - - llvm::StringMap merged_indices; - auto& graph = tu_index.graph; - for(auto& [fid, index]: tu_index.file_indices) { - llvm::StringRef path = graph.paths[graph.path_id(fid)]; - merged_indices[path].merge("tu0", graph.include_location_id(fid), index, {}); - } - - for(auto& [path, merged]: merged_indices) { - llvm::SmallString<1024> s; - llvm::raw_svector_ostream os(s); - - merged.serialize(os); - - auto view = index::MergedIndex(s); - ASSERT_TRUE(merged == view); - } -} - -TEST_CASE(RevisionAndFlipBack) { - build_index(R"( - int flip_func() { return 1; } - )"); - - index::MergedIndex merged; - ASSERT_EQ(merged.revision(), 0u); - - auto fid = unit->interested_file(); - merged.merge("tu0", tu_index.graph.include_location_id(fid), tu_index.main_file_index, {}); - auto merged_rev = merged.revision(); - ASSERT_TRUE(merged_rev != 0u); - ASSERT_TRUE(merged.need_rewrite()); - - // The flip save() performs after a commit: the serialized twin is - // buffer-backed (no heap Impl, not dirty) and answers identically. - llvm::SmallString<1024> s; - llvm::raw_svector_ostream os(s); - merged.serialize(os); - auto reloaded = index::MergedIndex(s); - ASSERT_FALSE(reloaded.need_rewrite()); - ASSERT_EQ(reloaded.revision(), 0u); - - // Every mutation bumps the revision, so a save can prove no merge - // landed across its commit await. (Ordering is load-bearing: operator== - // materializes both sides' Impl, and serialize() compacts removed rows - // and caches — the comparison is only valid before remove()/lookup() - // touch either side.) - ASSERT_TRUE(merged == reloaded); - merged.remove("tu0"); - ASSERT_TRUE(merged.revision() != merged_rev && merged.revision() != 0u); -} - -TEST_CASE(LookupByOffset) { - build_index(R"( - int §(func)⟦§(func)foo⟧() { return 42; } - int bar() { return §(ref)⟦§(ref)foo⟧(); } - )"); - - // Merge the main file index into a MergedIndex. - index::MergedIndex merged; - auto fid = unit->interested_file(); - merged.merge("tu0", tu_index.graph.include_location_id(fid), tu_index.main_file_index, {}); - - // Lookup at the reference offset should find an occurrence. - auto ref_offset = point("ref"); - bool found = false; - merged.lookup(ref_offset, [&](const index::Occurrence& occ) { - if(occ.range.contains(ref_offset)) { - found = true; - } - return true; - }); - ASSERT_TRUE(found); -} - -TEST_CASE(LookupBySymbolAndKind) { - build_index(R"( - void §(target)target_func() {} - void caller() { §(call)target_func(); } - )"); - - index::MergedIndex merged; - auto fid = unit->interested_file(); - merged.merge("tu0", tu_index.graph.include_location_id(fid), tu_index.main_file_index, {}); - - // Find the target_func symbol hash via occurrence lookup. - auto target_offset = point("target"); - index::SymbolHash target_hash = 0; - merged.lookup(target_offset, [&](const index::Occurrence& occ) { - if(occ.range.contains(target_offset)) { - target_hash = occ.target; - return false; - } - return true; - }); - ASSERT_TRUE(target_hash != 0); - - // Lookup Definition relation for the symbol. - bool found_def = false; - merged.lookup(target_hash, RelationKind::Definition, [&](const index::Relation& rel) { - found_def = true; - return true; - }); - ASSERT_TRUE(found_def); -} - -TEST_CASE(MultipleMergesDedup) { - add_file("header.h", R"( - #pragma once - inline int shared() { return 1; } - )"); - add_main("a.cpp", R"( - #include "header.h" - int use_a() { return shared(); } - )"); - ASSERT_TRUE(compile()); - auto tu_a = index::TUIndex::build(*unit); - - add_file("header.h", R"( - #pragma once - inline int shared() { return 1; } - )"); - add_main("b.cpp", R"( - #include "header.h" - int use_b() { return shared(); } - )"); - ASSERT_TRUE(compile()); - auto tu_b = index::TUIndex::build(*unit); - - // Merge header indices from both TUs into same MergedIndex. - index::MergedIndex merged_header; - for(auto& [fid, file_index]: tu_a.file_indices) { - merged_header.merge("tu0", tu_a.graph.include_location_id(fid), file_index, {}); - } - for(auto& [fid, file_index]: tu_b.file_indices) { - merged_header.merge("tu1", tu_b.graph.include_location_id(fid), file_index, {}); - } - - // Serialize and deserialize to verify dedup survives round-trip. - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - merged_header.serialize(os); - - auto restored = index::MergedIndex(buf); - ASSERT_TRUE(merged_header == restored); -} - -TEST_CASE(SerializationRoundTripInMemory) { - build_index(R"( - struct Foo { int x; }; - Foo make() { return Foo{42}; } - )"); - - // Merge using the include_id overload (same as existing Serialization test). - index::MergedIndex merged; - auto fid = unit->interested_file(); - auto include_id = tu_index.graph.include_location_id(fid); - merged.merge("tu0", include_id, tu_index.main_file_index, {}); - - // Serialize. - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - merged.serialize(os); - - // Deserialize and compare. - auto restored = index::MergedIndex(buf); - ASSERT_TRUE(merged == restored); - - // Lookup should work on the deserialized version too. - bool found = false; - for(auto& occ: tu_index.main_file_index.occurrences) { - restored.lookup(occ.range.begin, [&](const index::Occurrence& o) { - if(o.range.begin == occ.range.begin) { - found = true; - } - return true; - }); - if(found) - break; - } - ASSERT_TRUE(found); -} - -TEST_CASE(RemoveCompilationContext) { - build_index(R"( - int foo() { return 42; } - int bar() { return foo(); } - )"); - - // Merge as a compilation context (using the build_at overload). - index::MergedIndex merged; - auto fid = unit->interested_file(); - merged.merge("tu0", tu_index.built_at, {}, tu_index.main_file_index, {}); - - // Verify occurrence lookup works before remove. - bool found_before = false; - for(auto& occ: tu_index.main_file_index.occurrences) { - merged.lookup(occ.range.begin, [&](const index::Occurrence& o) { - found_before = true; - return false; - }); - if(found_before) - break; - } - ASSERT_TRUE(found_before); - - // Remove the compilation context. - merged.remove("tu0"); - - // Serialize and verify the removed data round-trips. - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - merged.serialize(os); - // Should not crash. - auto restored = index::MergedIndex(buf); -} - -TEST_CASE(RemoveHeaderContext) { - add_file("header.h", R"( - #pragma once - inline int shared() { return 1; } - )"); - add_main("main.cpp", R"( - #include "header.h" - int use() { return shared(); } - )"); - ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); - - // Merge header index as header context. - index::MergedIndex merged_header; - for(auto& [fid, file_index]: tu_index.file_indices) { - merged_header.merge("tu0", tu_index.graph.include_location_id(fid), file_index, {}); - } - - // Remove should not crash. - merged_header.remove("tu0"); - - // Serialize after remove should work. - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - merged_header.serialize(os); -} - -TEST_CASE(RemergeReplacesContribution) { - add_file("header.h", R"( - #pragma once - inline int shared() { return 1; } - )"); - add_main("main.cpp", R"( - #include "header.h" - int use() { return shared(); } - )"); - ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); - - auto header_fid = unit->file_id("header.h"); - auto& header_idx = tu_index.file_indices[header_fid]; - auto include_id = tu_index.graph.include_location_id(header_fid); - - // The symbol defined in the header: its Definition relation exists only - // in the header's file index, not in main's (which only references it). - index::SymbolHash defined{}; - for(auto& [symbol, relations]: header_idx.relations) { - for(auto& relation: relations) { - if(RelationKind(relation.kind) & RelationKind(RelationKind::Definition)) { - defined = symbol; - } - } - } - - auto has_definition = [&](index::MergedIndex& merged) { - bool found = false; - merged.lookup(defined, RelationKind::Definition, [&](const index::Relation&) { - found = true; - return false; - }); - return found; - }; - - index::MergedIndex merged; - merged.merge("tu0", include_id, header_idx, {}); - ASSERT_TRUE(has_definition(merged)); - - // Identical re-merge (a touch): the contribution is resurrected, not lost. - merged.merge("tu0", include_id, header_idx, {}); - ASSERT_TRUE(has_definition(merged)); - - // Re-merge of the same TU with different content: the old contribution - // is masked instead of being served alongside the new one. - merged.merge("tu0", include_id, tu_index.main_file_index, {}); - ASSERT_FALSE(has_definition(merged)); -} - -TEST_CASE(RemergePreservesOtherTus) { - add_file("header.h", R"( - #pragma once - inline int shared() { return 1; } - )"); - add_main("main.cpp", R"( - #include "header.h" - int use() { return shared(); } - )"); - ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); - - auto header_fid = unit->file_id("header.h"); - auto& header_idx = tu_index.file_indices[header_fid]; - auto include_id = tu_index.graph.include_location_id(header_fid); - - index::SymbolHash defined{}; - for(auto& [symbol, relations]: header_idx.relations) { - for(auto& relation: relations) { - if(RelationKind(relation.kind) & RelationKind(RelationKind::Definition)) { - defined = symbol; - } - } - } - - index::MergedIndex merged; - merged.merge("tu0", include_id, header_idx, {}); - merged.merge("tu1", include_id, header_idx, {}); - - // TU 0 moves on, TU 1 still holds the shared canonical contribution. - merged.merge("tu0", include_id, tu_index.main_file_index, {}); - - bool found = false; - merged.lookup(defined, RelationKind::Definition, [&](const index::Relation&) { - found = true; - return false; - }); - ASSERT_TRUE(found); -} - -TEST_CASE(CompactionDropsMasked) { - build_index(R"( - int §(target)foo() { return 42; } - )"); - - // Merge as compilation context, then remove: the rows are masked. - index::MergedIndex merged; - merged.merge("tu0", tu_index.built_at, {}, tu_index.main_file_index, {}); - merged.remove("tu0"); - - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - merged.serialize(os); - - // Serialized shards are served through buffer-only lookups that never - // consult the removed bitmap — masked rows must not reach disk at all. - auto restored = index::MergedIndex(buf); - auto offset = point("target"); - bool found = false; - restored.lookup(offset, [&](const index::Occurrence&) { - found = true; - return false; - }); - ASSERT_FALSE(found); -} - -TEST_CASE(SerializeCompactsInPlace) { - // Two contributions with distinct content, so each gets its own - // canonical id; only one is removed. - index::FileIndex live_idx; - live_idx.occurrences.emplace_back(index::Range{0, 3}, 100); - index::FileIndex dead_idx; - dead_idx.occurrences.emplace_back(index::Range{10, 13}, 200); - - index::MergedIndex merged; - merged.merge("tu0", std::uint32_t(0), live_idx, "synthetic"); - merged.merge("tu1", std::uint32_t(0), dead_idx, "synthetic"); - merged.remove("tu1"); - - llvm::SmallString<1024> buf; - llvm::raw_svector_ostream os(buf); - merged.serialize(os); - - // The save flip is conditional: when it does not happen, the in-memory - // impl — now compacted by serialize() — keeps serving queries. Surviving - // rows must still resolve and removed ones stay gone. - auto hits_at = [&](std::uint32_t offset) { - std::size_t hits = 0; - merged.lookup(offset, [&](const index::Occurrence&) { - hits += 1; - return true; - }); - return hits; - }; - ASSERT_EQ(hits_at(1), 1u); - ASSERT_EQ(hits_at(11), 0u); - ASSERT_TRUE(merged.has_contribution("tu0")); - ASSERT_FALSE(merged.has_contribution("tu1")); - - // A second serialize of the compacted impl round-trips identically. - llvm::SmallString<1024> again; - llvm::raw_svector_ostream os2(again); - merged.serialize(os2); - ASSERT_EQ(llvm::StringRef(buf), llvm::StringRef(again)); -} - -TEST_CASE(HasContributionTracking) { - add_file("header.h", R"( - #pragma once - inline int shared() { return 1; } - )"); - add_main("main.cpp", R"( - #include "header.h" - int use() { return shared(); } - )"); - ASSERT_TRUE(compile()); - tu_index = index::TUIndex::build(*unit); - - auto header_fid = unit->file_id("header.h"); - auto& header_idx = tu_index.file_indices[header_fid]; - auto include_id = tu_index.graph.include_location_id(header_fid); - - index::MergedIndex merged; - merged.merge("tu0", include_id, header_idx, {}); - merged.merge("tu1", include_id, header_idx, {}); - - ASSERT_TRUE(merged.has_contribution("tu0")); - ASSERT_TRUE(merged.has_contribution("tu1")); - ASSERT_FALSE(merged.has_contribution("tu2")); - - // The buffer path must answer without deserializing the shard. - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - merged.serialize(os); - auto restored = index::MergedIndex(buf); - ASSERT_TRUE(restored.has_contribution("tu0")); - ASSERT_FALSE(restored.has_contribution("tu2")); - - merged.remove("tu0"); - ASSERT_FALSE(merged.has_contribution("tu0")); - ASSERT_TRUE(merged.has_contribution("tu1")); -} - -TEST_CASE(LookupFiltersRemoved) { - build_index(R"( - int §(target)foo() { return 42; } - )"); - - // Merge as compilation context. - index::MergedIndex merged; - merged.merge("tu0", tu_index.built_at, {}, tu_index.main_file_index, {}); - - // Verify lookup finds something before removal. - auto offset = point("target"); - bool found_before = false; - merged.lookup(offset, [&](const index::Occurrence& occ) { - if(occ.range.contains(offset)) - found_before = true; - return true; - }); - ASSERT_TRUE(found_before); - - // Remove the compilation context. - merged.remove("tu0"); - - // Verify lookup finds nothing after removal. - bool found_after = false; - merged.lookup(offset, [&](const index::Occurrence& occ) { - if(occ.range.contains(offset)) - found_after = true; - return true; - }); - ASSERT_FALSE(found_after); -} - -TEST_CASE(CacheInvalidatedAfterMerge) { - build_index(R"( - int §(first)foo() { return 42; } - )"); - - // Merge first TU as header context. - index::MergedIndex merged; - auto fid = unit->interested_file(); - merged.merge("tu0", tu_index.graph.include_location_id(fid), tu_index.main_file_index, {}); - - // Trigger cache build by doing a lookup. - auto first_offset = point("first"); - bool found_first = false; - merged.lookup(first_offset, [&](const index::Occurrence& occ) { - if(occ.range.contains(first_offset)) - found_first = true; - return true; - }); - ASSERT_TRUE(found_first); - - // Build a second TU with different content. - build_index(R"( - int §(second)bar() { return 99; } - )"); - - // Merge second TU. - auto fid2 = unit->interested_file(); - merged.merge("tu1", tu_index.graph.include_location_id(fid2), tu_index.main_file_index, {}); - - // Verify lookup finds the new occurrence (cache was invalidated). - auto second_offset = point("second"); - bool found_second = false; - merged.lookup(second_offset, [&](const index::Occurrence& occ) { - if(occ.range.contains(second_offset)) - found_second = true; - return true; - }); - ASSERT_TRUE(found_second); -} - -TEST_CASE(LocalSymbolTable) { - build_index(R"( - void foo() { int local = 42; } - int global = 0; - )"); - - index::MergedIndex merged; - auto main_path_id = static_cast(tu_index.graph.paths.size() - 1); - merged.merge("tu0", tu_index.built_at, {}, tu_index.main_file_index, ""); - - // Collect non-External symbols from the TU that appear in the FileIndex. - index::SymbolTable local_syms; - for(auto& occ: tu_index.main_file_index.occurrences) { - auto it = tu_index.symbols.find(occ.target); - if(it != tu_index.symbols.end() && it->second.scope != index::SymbolScope::External) { - local_syms.try_emplace(occ.target, it->second); - } - } - ASSERT_FALSE(local_syms.empty()); - merged.merge_symbols(local_syms); - - // FileLocal symbols should be findable in the shard. - std::string name; - SymbolKind kind; - bool found_local = false; - for(auto& [hash, symbol]: local_syms) { - if(symbol.name == "local") { - ASSERT_TRUE(merged.find_symbol(hash, name, kind)); - ASSERT_EQ(name, "local"); - found_local = true; - } - } - ASSERT_TRUE(found_local); - - // External symbol should NOT be in the shard's local table. - for(auto& [hash, symbol]: tu_index.symbols) { - if(symbol.name == "global") { - ASSERT_FALSE(merged.find_symbol(hash, name, kind)); - } - } -} - -TEST_CASE(LocalSymbolSerialization) { - build_index(R"( - static int static_var = 0; - void foo() { int local = 1; } - )"); - - index::MergedIndex merged; - auto main_path_id = static_cast(tu_index.graph.paths.size() - 1); - merged.merge("tu0", tu_index.built_at, {}, tu_index.main_file_index, ""); - - index::SymbolTable local_syms; - for(auto& occ: tu_index.main_file_index.occurrences) { - auto it = tu_index.symbols.find(occ.target); - if(it != tu_index.symbols.end() && it->second.scope != index::SymbolScope::External) { - local_syms.try_emplace(occ.target, it->second); - } - } - ASSERT_FALSE(local_syms.empty()); - merged.merge_symbols(local_syms); - - // Serialize and deserialize. - llvm::SmallString<4096> buf; - { - llvm::raw_svector_ostream os(buf); - merged.serialize(os); - } - auto restored = index::MergedIndex(llvm::StringRef(buf.data(), buf.size())); - - // Symbols should survive round-trip (via buffer path). - std::string name; - SymbolKind kind; - for(auto& [hash, symbol]: local_syms) { - ASSERT_TRUE(restored.find_symbol(hash, name, kind)); - ASSERT_EQ(name, symbol.name); - } -} - -// The dep is backdated an hour so the merge (build_at = one minute ago) -// records its baseline hash; a later write bumps the mtime past build_at, -// so staleness reaches the Layer 2 content-hash check — exactly the branch -// these tests exercise. A dep newer than build_at gets no baseline at all -// (its content may postdate the indexed snapshot). -index::MergedIndex build_ctx_shard(llvm::StringRef dep_path) { - namespace stdfs = std::filesystem; - stdfs::last_write_time(dep_path.str(), - stdfs::file_time_type::clock::now() - std::chrono::hours(1)); - auto build_at = std::chrono::duration_cast( - std::chrono::system_clock::now().time_since_epoch() - std::chrono::minutes(1)); - - index::MergedIndex merged; - index::FileIndex file_idx; - index::DepLocation deps[] = { - {.path = dep_path, .line = 1} - }; - merged.merge("tu0", build_at, deps, file_idx, ""); - return merged; -} - -TEST_CASE(TouchNoUpdate) { - TempDir dir; - auto dep = dir.path("dep.h"); - dir.touch("dep.h", "int shared = 1;"); - - auto merged = build_ctx_shard(dep); - - // Same content, newer mtime — a pure touch must not trigger a reindex. - ASSERT_FALSE(merged.need_update()); - - // The buffer path (serialized shard) must reach the same conclusion. - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - merged.serialize(os); - auto restored = index::MergedIndex(llvm::StringRef(buf.data(), buf.size())); - ASSERT_FALSE(restored.need_update()); -} - -TEST_CASE(ContentChangeUpdate) { - TempDir dir; - auto dep = dir.path("dep.h"); - dir.touch("dep.h", "int shared = 1;"); - - auto merged = build_ctx_shard(dep); - - // Real edit: content hash diverges from the stored baseline. - dir.touch("dep.h", "int shared = 2;"); - ASSERT_TRUE(merged.need_update()); -} - -TEST_CASE(OldShardDiscarded) { - TempDir dir; - - // A current shard round-trips through disk and loads normally. - { - index::MergedIndex merged; - index::FileIndex file_idx; - merged.merge("tu0", std::chrono::milliseconds(1), {}, file_idx, "valid-shard"); - auto path = dir.path("valid.idx"); - std::error_code ec; - llvm::raw_fd_ostream os(path, ec); - merged.serialize(os); - os.flush(); - ASSERT_TRUE(index::MergedIndex::load(path).content() == "valid-shard"); - } - - // A version-less (format_version=0) shard from an older build is silently - // discarded — load returns an empty index, as if nothing were on disk. - // Only the version slot is written: every other field reads back absent, - // which is structurally valid — rejection must come from the version - // check. - { - struct VersionOnly { - std::uint32_t format_version = 0; - }; - - auto blob = kota::codec::fbs::to_bytes(VersionOnly{}); - ASSERT_TRUE(blob.has_value()); - - auto path = dir.path("stale.idx"); - std::error_code ec; - llvm::raw_fd_ostream os(path, ec); - os.write(reinterpret_cast(blob->data()), blob->size()); - os.flush(); - - auto loaded = index::MergedIndex::load(path); - ASSERT_TRUE(loaded.content().empty()); - ASSERT_TRUE(loaded.need_update()); - } - - // Positive control: the same single-slot shape carrying the CURRENT - // version is kept — slot 0 really is the version slot and the rejection - // above comes from its value, not from the blob's shape. - { - struct VersionOnly { - std::uint32_t format_version = 0; - }; - - auto blob = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); - ASSERT_TRUE(blob.has_value()); - - auto path = dir.path("current.idx"); - std::error_code ec; - llvm::raw_fd_ostream os(path, ec); - os.write(reinterpret_cast(blob->data()), blob->size()); - os.flush(); - - ASSERT_TRUE(index::MergedIndex::load(path).loaded()); - } -} - -TEST_CASE(GarbageLoadRejected) { - TempDir dir; - dir.touch("garbage.idx", "not a flatbuffer"); - - auto loaded = index::MergedIndex::load(dir.path("garbage.idx")); - ASSERT_FALSE(loaded.loaded()); - ASSERT_TRUE(loaded.content().empty()); - ASSERT_TRUE(loaded.need_update()); - - // Queries on the rejected shard answer with silence, not UB. - bool visited = false; - loaded.lookup(0, [&](const index::Occurrence&) { - visited = true; - return true; - }); - ASSERT_FALSE(visited); - - loaded.lookup(index::SymbolHash(1), RelationKind::Definition, [&](const index::Relation&) { - visited = true; - return true; - }); - ASSERT_FALSE(visited); - - std::string name; - SymbolKind kind; - ASSERT_FALSE(loaded.find_symbol(1, name, kind)); -} - -TEST_CASE(CorruptShardRejected) { - TempDir dir; - - index::MergedIndex merged; - index::FileIndex file_idx; - merged.merge("tu0", std::chrono::milliseconds(1), {}, file_idx, "corrupt-me"); - - llvm::SmallString<1024> blob; - llvm::raw_svector_ostream os(blob); - merged.serialize(os); - ASSERT_TRUE(blob.size() > 8); - - auto write = [&](llvm::StringRef name, llvm::StringRef bytes) { - dir.touch(name, bytes); - return dir.path(name); - }; - - // Sanity: the intact bytes load, so the rejections below are earned. - ASSERT_TRUE(index::MergedIndex::load(write("valid.idx", blob)).loaded()); - - llvm::StringRef bytes(blob.data(), blob.size()); - ASSERT_FALSE( - index::MergedIndex::load(write("half.idx", bytes.take_front(bytes.size() / 2))).loaded()); - ASSERT_FALSE(index::MergedIndex::load(write("minus1.idx", bytes.drop_back(1))).loaded()); - - // Bytes 4-7 carry the buffer identifier; a blob from another format - // must be rejected up front. - std::string clobbered = bytes.str(); - for(std::size_t i = 4; i < 8; ++i) { - clobbered[i] = 'X'; - } - ASSERT_FALSE(index::MergedIndex::load(write("clobbered.idx", clobbered)).loaded()); -} - -TEST_CASE(OutOfRangeCanonicalIdRejected) { - TempDir dir; - - // Field order MUST mirror the persisted shapes in merged_index.cpp - // (MergedIndex::Impl prefix — skip-annotated fields occupy no slot — - // HeaderContext, IncludeContext, CompilationContext prefix); the - // trailing fields read back absent, which is structurally valid. - struct IncludeContextMirror { - std::uint32_t include_id = 0; - std::uint32_t canonical_id = 0; - }; - - struct HeaderContextMirror { - std::uint32_t version = 0; - llvm::SmallVector includes; - }; - - // The vector matters beyond field parity: without one the mirror would - // be trivially copyable and encode as an inline struct, while the real - // CompilationContext encodes as a table — the verifier tells them apart. - struct CompilationContextMirror { - std::uint32_t version = 0; - std::uint32_t canonical_id = 0; - std::uint64_t build_at = 0; - std::vector include_locations; - }; - - struct ReprMirror { - std::uint32_t format_version = 0; - std::vector paths; - std::string content; - std::vector line_starts; - llvm::SmallDenseMap header_contexts; - llvm::SmallDenseMap compilation_contexts; - std::vector> canonical_cache; - std::uint32_t max_canonical_id = 0; - }; - - // A consistent base: one path, one canonical id, one header context - // referencing it. - auto base = [] { - ReprMirror mirror; - mirror.format_version = index::index_format_version; - mirror.max_canonical_id = 1; - mirror.paths = {"/proj/tu.cpp"}; - mirror.canonical_cache.emplace_back("hash", 0); - mirror.header_contexts[0].includes.push_back({.include_id = 0, .canonical_id = 0}); - return mirror; - }; - - // Structure and version pass, so load() accepts the blob off disk; the - // first mutation materializes it in memory, where the id values face - // the range check. Nullopt = the blob never reached that check. - auto materialized_contribution = [&](llvm::StringRef name, - const ReprMirror& mirror) -> std::optional { - auto blob = kota::codec::fbs::to_bytes(mirror); - if(!blob) { - return std::nullopt; - } - dir.touch(name, llvm::StringRef(reinterpret_cast(blob->data()), blob->size())); - auto shard = index::MergedIndex::load(dir.path(name)); - if(!shard.loaded()) { - return std::nullopt; - } - shard.remove("/proj/never-indexed.cpp"); - return shard.has_contribution("/proj/tu.cpp"); - }; - - // Positive control first: the base materializes intact, so the - // rejections below come from the hostile ids, not the blob's shape. - auto good = materialized_contribution("good.idx", base()); - ASSERT_TRUE(good.has_value() && *good); - - // Structural verification does not constrain field values: each blob - // carries one canonical id at or past max_canonical_id, which would - // index canonical_ref_counts out of bounds if the in-memory load - // accepted it. The blob is dropped and the shard reads as empty. - auto bad_cache = base(); - bad_cache.canonical_cache.front().second = 5; - auto cache_verdict = materialized_contribution("bad-cache.idx", bad_cache); - ASSERT_TRUE(cache_verdict.has_value() && !*cache_verdict); - - auto bad_include = base(); - bad_include.header_contexts[0].includes.front().canonical_id = 5; - auto include_verdict = materialized_contribution("bad-include.idx", bad_include); - ASSERT_TRUE(include_verdict.has_value() && !*include_verdict); - - auto bad_compilation = base(); - bad_compilation.compilation_contexts[0].canonical_id = 5; - auto compilation_verdict = materialized_contribution("bad-compilation.idx", bad_compilation); - ASSERT_TRUE(compilation_verdict.has_value() && !*compilation_verdict); -} - -TEST_CASE(BufferPathLookupParity) { - build_index(R"( - void §(a)alpha_func() {} - int §(b)beta_var = 1; - )"); - - index::MergedIndex merged; - auto fid = unit->interested_file(); - merged.merge("tu0", - tu_index.graph.include_location_id(fid), - tu_index.main_file_index, - unit->interested_content()); - - auto hash_at = [&](llvm::StringRef pos) { - auto offset = point(pos); - index::SymbolHash hash = 0; - merged.lookup(offset, [&](const index::Occurrence& occ) { - hash = occ.target; - return false; - }); - return hash; - }; - - using Row = std::tuple; - auto definitions = [](index::MergedIndex& index, index::SymbolHash hash) { - std::vector rows; - index.lookup(hash, RelationKind::Definition, [&](const index::Relation& relation) { - rows.emplace_back(relation.range.begin, relation.range.end, relation.target_symbol); - return true; - }); - std::ranges::sort(rows); - return rows; - }; - - index::SymbolHash hashes[2] = {hash_at("a"), hash_at("b")}; - std::vector expected[2]; - for(std::size_t i = 0; i < 2; ++i) { - ASSERT_TRUE(hashes[i] != 0); - expected[i] = definitions(merged, hashes[i]); - ASSERT_FALSE(expected[i].empty()); - } - ASSERT_TRUE(hashes[0] != hashes[1]); - - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - merged.serialize(os); - auto restored = index::MergedIndex(buf); - - // The zero-copy view is the production read path: it must agree BEFORE - // anything materializes the impl (operator== would, so it comes last). - for(std::size_t i = 0; i < 2; ++i) { - ASSERT_TRUE(definitions(restored, hashes[i]) == expected[i]); - } - - ASSERT_FALSE(restored.line_starts().empty()); - ASSERT_TRUE(std::ranges::equal(restored.line_starts(), merged.line_starts())); -} - -TEST_CASE(BufferPathMultiOccurrenceLookup) { - // Synthesized occurrences sorted by (begin, end, target), spaced so each - // probe hits exactly one range (contains() is inclusive at both ends). - index::FileIndex file_idx; - index::SymbolHash target = 100; - for(std::uint32_t begin = 0; begin < 60; begin += 10) { - file_idx.occurrences.emplace_back(index::Range{begin, begin + 3}, target++); - } - - index::MergedIndex merged; - merged.merge("tu0", std::uint32_t(0), file_idx, "synthetic"); - - llvm::SmallString<1024> buf; - llvm::raw_svector_ostream os(buf); - merged.serialize(os); - auto restored = index::MergedIndex(buf); - - // The buffer path binary-searches the serialized rows: every probe must - // land on exactly its own range. - for(auto& occurrence: file_idx.occurrences) { - std::vector hits; - restored.lookup(occurrence.range.begin + 1, [&](const index::Occurrence& hit) { - hits.push_back(hit); - return true; - }); - ASSERT_EQ(hits.size(), 1u); - ASSERT_TRUE(hits.front() == occurrence); - } -} - -std::uint64_t file_hash(llvm::StringRef path) { - auto buf = llvm::MemoryBuffer::getFile(path); - return buf ? llvm::xxh3_64bits((*buf)->getBuffer()) : 0; -} - -/// A build_at far enough in the future that every existing file clears the -/// mtime guard and earns a stat fast path at merge. -std::chrono::milliseconds generous_build_at() { - return std::chrono::duration_cast( - std::chrono::system_clock::now().time_since_epoch()) + - std::chrono::milliseconds(10'000); -} - -TEST_CASE(NeedUpdateChecksAllContexts) { - TempDir tmp; - tmp.touch("a.h", "int a();\n"); - tmp.touch("b.h", "int b();\n"); - auto a = tmp.path("a.h"); - auto b = tmp.path("b.h"); - - index::MergedIndex shard; - index::FileIndex fi_a, fi_b; - auto dep_of = [&](const std::string& path) { - return llvm::SmallVector{ - {path, 1, 0, file_hash(path)} - }; - }; - shard.merge("tuA", generous_build_at(), dep_of(a), fi_a, "int a();\n"); - shard.merge("tuB", generous_build_at(), dep_of(b), fi_b, "int b();\n"); - - ASSERT_FALSE(shard.need_update()); - - // Only one contribution's dependency goes stale at a time; a check that - // stops at a single context would miss whichever the iteration order - // hides, so exercise both. - tmp.touch("b.h", "int b2();\n"); - ASSERT_TRUE(shard.need_update()); - - shard.merge("tuB", generous_build_at(), dep_of(b), fi_b, "int b2();\n"); - ASSERT_FALSE(shard.need_update()); - tmp.touch("a.h", "int a2();\n"); - ASSERT_TRUE(shard.need_update()); - - // The serialized reader shares the loop: both contexts again through a - // reloaded view. - shard.merge("tuA", generous_build_at(), dep_of(a), fi_a, "int a2();\n"); - llvm::SmallString<1024> s; - llvm::raw_svector_ostream os(s); - shard.serialize(os); - auto view = index::MergedIndex(s); - ASSERT_FALSE(view.need_update()); - - // Same-size rewrites move the mtime explicitly: Windows file times - // advance in ~16ms ticks, so a rewrite landing in the stamp's tick - // reproduces size AND mtime exactly and the stat fast path rightly - // trusts it. A real edit arrives long after the stamp; the bump - // models that and pins these verdicts on the hash layer. - tmp.touch("b.h", "int b3();\n"); - set_file_mtime(b, file_mtime_ns(b) + 5'000'000'000); - ASSERT_TRUE(view.need_update()); - - // Restore b (fresh again via the hash layer), then break a: the verdict - // now hinges on the second context alone. - tmp.touch("b.h", "int b2();\n"); - ASSERT_FALSE(view.need_update()); - tmp.touch("a.h", "int a3();\n"); - set_file_mtime(a, file_mtime_ns(a) + 5'000'000'000); - ASSERT_TRUE(view.need_update()); -} - -TEST_CASE(NeedUpdateBackdatedEdit) { - TempDir tmp; - tmp.touch("dep.h", "int old_name();\n"); - auto dep = tmp.path("dep.h"); - - index::MergedIndex shard; - index::FileIndex fi; - llvm::SmallVector deps{ - {dep, 1, 0, file_hash(dep)} - }; - shard.merge("tu", generous_build_at(), deps, fi, "content"); - ASSERT_FALSE(shard.need_update()); - - // Same length, mtime rolled back: a watermark would call this fresh; - // stamp equality sends it to the hash layer. - auto recorded = file_mtime_ns(dep); - tmp.touch("dep.h", "int new_name();\n"); - set_file_mtime(dep, recorded - 5'000'000'000); - ASSERT_TRUE(shard.need_update()); -} - -TEST_CASE(SerializedStampsValidate) { - TempDir tmp; - tmp.touch("dep.h", "int f();\n"); - auto dep = tmp.path("dep.h"); - - index::MergedIndex shard; - index::FileIndex fi; - llvm::SmallVector deps{ - {dep, 1, 0, file_hash(dep)} - }; - shard.merge("tu", generous_build_at(), deps, fi, "content"); - - llvm::SmallString<1024> s; - llvm::raw_svector_ostream os(s); - shard.serialize(os); - auto view = index::MergedIndex(s); - - ASSERT_FALSE(view.need_update()); - - // Touched, not modified: the immutable stamp mismatches, the hash - // proves the content unchanged. - set_file_mtime(dep, file_mtime_ns(dep) + 5'000'000'000); - ASSERT_FALSE(view.need_update()); - - // A real edit is caught by the hash layer. The explicit mtime bump - // keeps the same-size rewrite out of the stamp's Windows time tick - // (see NeedUpdateChecksAllContexts). - tmp.touch("dep.h", "int g();\n"); - set_file_mtime(dep, file_mtime_ns(dep) + 5'000'000'000); - ASSERT_TRUE(view.need_update()); -} - -}; // TEST_SUITE(MergedIndex) -} // namespace -} // namespace clice::testing diff --git a/tests/unit/index/persisted_index_tests.cpp b/tests/unit/index/persisted_index_tests.cpp index 341c1c718..9790153ab 100644 --- a/tests/unit/index/persisted_index_tests.cpp +++ b/tests/unit/index/persisted_index_tests.cpp @@ -1,4 +1,5 @@ #include "test/test.h" +#include "index/manifest.h" #include "index/project_index.h" #include "index/serialization.h" @@ -9,85 +10,164 @@ namespace { TEST_SUITE(PersistedIndex) { -/// A ProjectIndex whose only symbol references `path` through `pool`. -index::ProjectIndex reference_one(clice::PathPool& pool, llvm::StringRef path) { - index::ProjectIndex project; - auto& symbol = project.symbols[42]; - symbol.name = "sym"; - symbol.reference_files.add(pool.intern(path)); - return project; +llvm::StringRef bytes_of(const std::vector& blob) { + return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); } -TEST_CASE(SerializeCollectsGarbage) { - clice::PathPool pool; - auto project = reference_one(pool, "/proj/used.cpp"); - // Interned but referenced by nothing — must not reach disk. - pool.intern("/proj/garbage.cpp"); +TEST_CASE(ManifestRoundTrip) { + index::TUManifest manifest; + manifest.global_gen = 7; + manifest.built_at = 1234567; + manifest.tu_fv = 300; + // A root node, a multi-byte-varint line, and a parent that FOLLOWS its + // child (the include graph resolves parent chains after appending). + manifest.nodes = { + {300, ~0u, 1 }, + {301, 2, 70000}, + {302, 0, 12 }, + }; + manifest.contributions = { + {300, 0xdeadbeefdeadbeefull}, + {302, 42 }, + }; - llvm::SmallString<1024> buf; + llvm::SmallString<256> buf; llvm::raw_svector_ostream os(buf); - project.serialize(os, pool, {}); + index::serialize_manifest(manifest, os); - clice::PathPool fresh; - llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); + auto loaded = index::deserialize_manifest(buf.str()); ASSERT_TRUE(loaded.has_value()); - ASSERT_TRUE(fresh.find("/proj/used.cpp").has_value()); - ASSERT_FALSE(fresh.find("/proj/garbage.cpp").has_value()); + ASSERT_TRUE(*loaded == manifest); +} + +TEST_CASE(ManifestJunkRejected) { + ASSERT_FALSE(index::deserialize_manifest("not a flatbuffer").has_value()); +} + +/// Field order MUST mirror ManifestBlob (manifest.cpp). +struct ManifestBlobMirror { + std::uint32_t format_version = 0; + std::uint64_t global_gen = 0; + std::uint64_t built_at = 0; + std::uint32_t tu_fv = 0; + std::uint32_t node_count = 0; + std::uint32_t contribution_count = 0; + std::vector nodes; + std::vector contributions; +}; + +TEST_CASE(ManifestCountMismatchRejected) { + // A node count claiming more nodes than the payload holds must not + // decode. + ManifestBlobMirror mirror; + mirror.format_version = index::index_format_version; + mirror.node_count = 2; + mirror.nodes = {1, 0, 5}; // one node's worth of varints + + auto blob = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(blob.has_value()); + ASSERT_FALSE(index::deserialize_manifest(bytes_of(*blob)).has_value()); +} + +TEST_CASE(ManifestVarintOverflowRejected) { + // A ten-byte varint whose last byte carries more than value bit 63 + // would silently shift the excess out and decode to an unrelated small + // id, redirecting contributions to another file. + ManifestBlobMirror mirror; + mirror.format_version = index::index_format_version; + mirror.node_count = 1; + mirror.nodes = {0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x80, 0x02, 0x00, 0x00}; + + auto blob = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(blob.has_value()); + ASSERT_FALSE(index::deserialize_manifest(bytes_of(*blob)).has_value()); } -TEST_CASE(RemapAcrossSessions) { +/// A project whose FileVersion for `path` is referenced by one manifest of +/// `tu` (so garbage collection keeps it) and whose only symbol references +/// `path` through `pool`. +index::ProjectIndex build_project(clice::PathPool& pool, llvm::StringRef path, llvm::StringRef tu) { + index::ProjectIndex project; + auto path_id = pool.intern(path); + auto fv = project.intern_file_version(path_id, 0xabcd); + project.file_versions.find(fv)->second.size = 100; + project.file_versions.find(fv)->second.mtime_ns = 5555; + + index::TUManifest manifest; + manifest.tu_fv = project.intern_file_version(pool.intern(tu), 0x1111); + manifest.nodes = { + {fv, ~0u, 3} + }; + manifest.contributions = { + {fv, 777} + }; + project.apply_manifest(pool.intern(tu), std::move(manifest)); + + auto& symbol = project.symbols[42]; + symbol.name = "sym"; + symbol.reference_files.add(path_id); + return project; +} + +TEST_CASE(GlobalRoundTripRemap) { clice::PathPool pool; - auto project = reference_one(pool, "/proj/used.cpp"); + auto project = build_project(pool, "/proj/used.h", "/proj/tu.cpp"); + project.global_generation = 9; + auto& manifest = project.manifests.find(pool.intern("/proj/tu.cpp"))->second; + manifest.global_gen = 9; llvm::SmallString<1024> buf; llvm::raw_svector_ostream os(buf); - project.serialize(os, pool, {}); + project.serialize_global(os, pool); // The next session interns other paths first, so the same file gets a - // different pool id; the loaded bitmap must follow the path, not the id. + // different pool id; both the FileVersion table and the loaded bitmap + // must follow the path, not the id. clice::PathPool fresh; fresh.intern("/proj/opened-first.cpp"); - llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); - ASSERT_TRUE(loaded.has_value()); + index::ProjectIndex loaded; + llvm::DenseMap pins; + ASSERT_TRUE(loaded.load_global(buf.str(), fresh, pins)); - auto id = fresh.find("/proj/used.cpp"); + auto id = fresh.find("/proj/used.h"); ASSERT_TRUE(id.has_value()); - ASSERT_TRUE(loaded->symbols[42].reference_files.contains(*id)); + ASSERT_TRUE(loaded.symbols[42].reference_files.contains(*id)); + ASSERT_EQ(loaded.next_fv_id, project.next_fv_id); + ASSERT_EQ(loaded.global_generation, 9u); + + // The blob pins the TU's manifest at the stamp it was saved under. + ASSERT_EQ(pins.size(), std::size_t(1)); + ASSERT_EQ(pins.find(manifest.tu_fv)->second, 9u); + + auto fv_it = loaded.fv_ids.find({*id, std::uint64_t(0xabcd)}); + ASSERT_TRUE(fv_it != loaded.fv_ids.end()); + auto& record = loaded.file_versions.find(fv_it->second)->second; + ASSERT_EQ(record.size, 100u); + ASSERT_EQ(record.mtime_ns, 5555); } -TEST_CASE(ShardManifestRoundTrip) { +TEST_CASE(GlobalCollectsGarbage) { clice::PathPool pool; - auto project = reference_one(pool, "/proj/used.cpp"); - auto shard_owner = pool.intern("/proj/tu.cpp"); + auto project = build_project(pool, "/proj/used.h", "/proj/tu.cpp"); + // Interned but referenced by no manifest — must not reach disk, and + // must be dropped from memory by the write. + auto dead_id = pool.intern("/proj/dead.h"); + project.intern_file_version(dead_id, 0xdead); llvm::SmallString<1024> buf; llvm::raw_svector_ostream os(buf); - project.serialize(os, pool, {shard_owner}); + project.serialize_global(os, pool); + ASSERT_FALSE(project.fv_ids.contains({dead_id, std::uint64_t(0xdead)})); clice::PathPool fresh; - llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); - ASSERT_TRUE(loaded.has_value()); - ASSERT_EQ(shards.size(), 1u); - ASSERT_EQ(fresh.resolve(shards.front()), "/proj/tu.cpp"); -} - -TEST_CASE(OldBlobDiscarded) { - clice::PathPool pool; - llvm::SmallVector shards; - // Arbitrary bytes are rejected by verification, not misread. - const char junk[] = "not a flatbuffer"; - ASSERT_FALSE( - index::ProjectIndex::from(llvm::StringRef(junk, sizeof(junk)), pool, shards).has_value()); + index::ProjectIndex loaded; + llvm::DenseMap pins; + ASSERT_TRUE(loaded.load_global(buf.str(), fresh, pins)); + ASSERT_FALSE(fresh.find("/proj/dead.h").has_value()); + ASSERT_TRUE(fresh.find("/proj/used.h").has_value()); } -llvm::StringRef bytes_of(const std::vector& blob) { - return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); -} - -TEST_CASE(VersionGate) { +TEST_CASE(GlobalVersionGate) { // Only the version slot is written: every other field reads back absent, // which is structurally valid — the verdict must hinge on the value. struct VersionOnly { @@ -95,54 +175,171 @@ TEST_CASE(VersionGate) { }; clice::PathPool pool; - llvm::SmallVector shards; + index::ProjectIndex loaded; + llvm::DenseMap pins; auto stale = kota::codec::fbs::to_bytes(VersionOnly{}); ASSERT_TRUE(stale.has_value()); - ASSERT_FALSE(index::ProjectIndex::from(bytes_of(*stale), pool, shards).has_value()); + ASSERT_FALSE(loaded.load_global(bytes_of(*stale), pool, pins)); auto current = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); ASSERT_TRUE(current.has_value()); - auto loaded = index::ProjectIndex::from(bytes_of(*current), pool, shards); - ASSERT_TRUE(loaded.has_value()); - ASSERT_TRUE(loaded->symbols.empty()); - ASSERT_TRUE(shards.empty()); + ASSERT_TRUE(loaded.load_global(bytes_of(*current), pool, pins)); + ASSERT_TRUE(loaded.symbols.empty()); + ASSERT_TRUE(pins.empty()); + + ASSERT_FALSE(loaded.load_global("not a flatbuffer", pool, pins)); } -TEST_CASE(OutOfRangeLocalIdsDropped) { - // Field order MUST mirror ProjectIndex (project_index.h): - // format_version, paths, symbols, shards. - struct ProjectIndexMirror { - std::uint32_t format_version = 0; - std::vector> paths; - index::SymbolTable symbols; - std::vector shards; +/// Field order MUST mirror GlobalBlob (project_index.cpp). +struct GlobalBlobMirror { + std::uint32_t format_version = 0; + std::uint64_t generation = 0; + std::uint32_t next_fv_id = 0; + std::vector fv_ids; + std::vector fv_paths; + std::vector fv_hashes; + std::vector fv_sizes; + std::vector fv_mtimes; + std::vector sym_hashes; + std::vector sym_names; + std::vector sym_kinds; + std::vector> sym_bitmaps; + std::vector manifest_fvs; + std::vector manifest_gens; + std::vector> sym_paths; +}; + +TEST_CASE(GlobalBitmapPayloadGate) { + // A malformed reference bitmap must fail the whole load: normalized to + // empty it would silently lose the symbol's reference files, with + // nothing ever rebuilding them. + GlobalBlobMirror mirror; + mirror.format_version = index::index_format_version; + mirror.sym_hashes = {42}; + mirror.sym_names = {"sym"}; + mirror.sym_kinds = {0}; + + clice::Bitmap bits; + bits.add(3); + mirror.sym_bitmaps = {index::write_bitmap(bits)}; + mirror.sym_paths = { + {3, "/proj/ref.h"} }; - ProjectIndexMirror mirror; + clice::PathPool pool; + llvm::DenseMap pins; + auto valid = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(valid.has_value()); + index::ProjectIndex loaded; + ASSERT_TRUE(loaded.load_global(bytes_of(*valid), pool, pins)); + ASSERT_TRUE(loaded.symbols.contains(42)); + + // A malformed image after columns that decoded fine: the reject must + // leave no partial state — file versions or symbols — that later + // merges would build on and the next save persist. + mirror.fv_ids = {7}; + mirror.fv_paths = {"/proj/partial.h"}; + mirror.fv_hashes = {0x1}; + mirror.fv_sizes = {10}; + mirror.fv_mtimes = {10}; + mirror.sym_hashes = {42, 43}; + mirror.sym_names = {"sym", "other"}; + mirror.sym_kinds = {0, 0}; + mirror.sym_bitmaps = { + index::write_bitmap(bits), + {std::byte{0xff}, std::byte{0xff}, std::byte{0xff}} + }; + auto corrupt = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(corrupt.has_value()); + index::ProjectIndex rejecting; + clice::PathPool untouched; + ASSERT_FALSE(rejecting.load_global(bytes_of(*corrupt), untouched, pins)); + ASSERT_TRUE(rejecting.symbols.empty()); + ASSERT_TRUE(rejecting.file_versions.empty()); + ASSERT_FALSE(untouched.find("/proj/partial.h").has_value()); +} + +TEST_CASE(UncoveredBitmapIdRejected) { + // The writer emits a path-table entry for every id its bitmaps + // reference; dropping an uncovered id would silently lose the symbol's + // reference files while every manifest stays fresh. + GlobalBlobMirror mirror; mirror.format_version = index::index_format_version; - mirror.paths = { - {0, "/proj/used.cpp"} + mirror.sym_hashes = {42}; + mirror.sym_names = {"sym"}; + mirror.sym_kinds = {0}; + clice::Bitmap bits; + bits.add(3); + mirror.sym_bitmaps = {index::write_bitmap(bits)}; + + clice::PathPool pool; + llvm::DenseMap pins; + auto uncovered = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(uncovered.has_value()); + index::ProjectIndex loaded; + ASSERT_FALSE(loaded.load_global(bytes_of(*uncovered), pool, pins)); + ASSERT_TRUE(loaded.symbols.empty()); + + mirror.sym_paths = { + {3, "/proj/ref.h"} }; - auto& symbol = mirror.symbols[42]; - symbol.name = "sym"; - symbol.reference_files.add(7); // Only pool id 0 is in the table. - mirror.shards = {9}; + auto covered = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(covered.has_value()); + ASSERT_TRUE(loaded.load_global(bytes_of(*covered), pool, pins)); + ASSERT_TRUE(loaded.symbols.contains(42)); +} - auto blob = kota::codec::fbs::to_bytes(mirror); - ASSERT_TRUE(blob.has_value()); +TEST_CASE(GlobalDuplicateVersionsRejected) { + // Version-table ids and (path, hash) pairs are both map keys in the + // writer; a repeated id in particular would intern the earlier pair to + // an id whose record names the later path, attributing contributions + // to the wrong file. + GlobalBlobMirror mirror; + mirror.format_version = index::index_format_version; + mirror.fv_ids = {7, 7}; + mirror.fv_paths = {"/proj/a.h", "/proj/b.h"}; + mirror.fv_hashes = {0x1, 0x2}; + mirror.fv_sizes = {1, 2}; + mirror.fv_mtimes = {1, 2}; - // Dangling local ids are dropped, not misresolved into the pool: the - // blob still loads, the symbol survives with an empty bitmap, and no - // shard is fetched. clice::PathPool pool; - llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(bytes_of(*blob), pool, shards); - ASSERT_TRUE(loaded.has_value()); - ASSERT_TRUE(loaded->symbols.contains(42)); - ASSERT_EQ(loaded->symbols[42].name, "sym"); - ASSERT_EQ(loaded->symbols[42].reference_files.cardinality(), 0u); - ASSERT_TRUE(shards.empty()); + llvm::DenseMap pins; + auto dup_id = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(dup_id.has_value()); + index::ProjectIndex loaded; + ASSERT_FALSE(loaded.load_global(bytes_of(*dup_id), pool, pins)); + ASSERT_TRUE(loaded.file_versions.empty()); + + mirror.fv_ids = {7, 8}; + mirror.fv_paths = {"/proj/a.h", "/proj/a.h"}; + mirror.fv_hashes = {0x1, 0x1}; + auto dup_pair = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(dup_pair.has_value()); + ASSERT_FALSE(loaded.load_global(bytes_of(*dup_pair), pool, pins)); + + // The same path under two content hashes is the legitimate shape: two + // observed versions of one file. + mirror.fv_hashes = {0x1, 0x2}; + auto distinct = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(distinct.has_value()); + ASSERT_TRUE(loaded.load_global(bytes_of(*distinct), pool, pins)); + ASSERT_EQ(loaded.file_versions.size(), std::size_t(2)); +} + +TEST_CASE(UnknownFileVersionsDetected) { + index::ProjectIndex project; + auto known = project.intern_file_version(0, 0x1); + + index::TUManifest manifest; + manifest.tu_fv = known; + manifest.nodes = { + {known, ~0u, 1} + }; + ASSERT_TRUE(project.knows_file_versions(manifest)); + + manifest.nodes.push_back({known + 1, ~0u, 2}); + ASSERT_FALSE(project.knows_file_versions(manifest)); } }; // TEST_SUITE(PersistedIndex) diff --git a/tests/unit/index/project_index_tests.cpp b/tests/unit/index/project_index_tests.cpp index 5e795bddd..9a5253141 100644 --- a/tests/unit/index/project_index_tests.cpp +++ b/tests/unit/index/project_index_tests.cpp @@ -1,294 +1,254 @@ +#include +#include +#include +#include +#include + #include "test/test.h" #include "test/tester.h" #include "index/project_index.h" #include "index/serialization.h" +#include "llvm/Support/raw_ostream.h" + namespace clice::testing { namespace { TEST_SUITE(ProjectIndex, Tester) { -bool build_and_index(llvm::StringRef code, index::TUIndex& out) { - add_main("main.cpp", code); - if(!compile()) { - return false; - } - out = index::TUIndex::build(*unit); - return true; -} - -TEST_CASE(MergeSingleTU) { - index::TUIndex tu; - ASSERT_TRUE(build_and_index(R"( - int foo() { return 42; } - int bar() { return foo() + 1; } - )", - tu)); - - index::ProjectIndex project; - clice::PathPool pool; - auto file_ids_map = project.merge(tu, pool); - - // The shared pool should have entries for the TU's files. - ASSERT_FALSE(pool.paths.empty()); +std::string wire; - // Symbols from the TU should be merged into the project. - ASSERT_FALSE(project.symbols.empty()); +/// Build the current unit's TUIndex and return the zero-copy view the +/// merge path consumes; `wire` keeps the bytes alive. +std::optional build_view() { + auto tu_index = index::TUIndex::build(*unit); + wire.clear(); + llvm::raw_string_ostream os(wire); + tu_index.serialize(os); + return index::TUIndexView::from(wire); +} - // Only External symbols should be in the project. - for(auto& [hash, symbol]: tu.symbols) { - if(symbol.scope == index::SymbolScope::External) { - ASSERT_TRUE(project.symbols.contains(hash)); +index::SymbolHash find_symbol(const index::ProjectIndex& project, llvm::StringRef name) { + for(auto& [hash, symbol]: project.symbols) { + if(symbol.name == name) { + return hash; } } + return 0; } -TEST_CASE(MergeMultipleTUs) { - index::TUIndex tu1; - ASSERT_TRUE(build_and_index(R"( - int foo() { return 42; } - )", - tu1)); - - index::TUIndex tu2; - ASSERT_TRUE(build_and_index(R"( - int bar() { return 99; } - )", - tu2)); - - index::ProjectIndex project; - clice::PathPool pool; - project.merge(tu1, pool); - project.merge(tu2, pool); - - // All symbols from both TUs should be present. - for(auto& [hash, symbol]: tu1.symbols) { - ASSERT_TRUE(project.symbols.contains(hash)); - } - for(auto& [hash, symbol]: tu2.symbols) { - ASSERT_TRUE(project.symbols.contains(hash)); +/// The TU-local id -> pool id mapping merge() consumes, as Indexer::merge +/// computes it. +llvm::SmallVector intern_paths(const index::TUIndexView& view, + clice::PathPool& pool) { + llvm::SmallVector ids; + for(std::uint32_t i = 0; i < view.path_count(); i += 1) { + ids.push_back(pool.intern(view.path(i))); } + return ids; } -TEST_CASE(MergeDuplicateSymbol) { - // Build two TUs that both define/reference the same function via header. - add_file("shared.h", R"( - #pragma once - inline int shared_func() { return 1; } - )"); - add_main("a.cpp", R"( - #include "shared.h" - int use_a() { return shared_func(); } - )"); - ASSERT_TRUE(compile()); - auto tu_a = index::TUIndex::build(*unit); - - add_file("shared.h", R"( - #pragma once - inline int shared_func() { return 1; } - )"); - add_main("b.cpp", R"( - #include "shared.h" - int use_b() { return shared_func(); } - )"); - ASSERT_TRUE(compile()); - auto tu_b = index::TUIndex::build(*unit); - - index::ProjectIndex project; - clice::PathPool pool; - project.merge(tu_a, pool); - project.merge(tu_b, pool); - - // Find the shared_func symbol hash from TU A's symbol table. - index::SymbolHash shared_hash = 0; - for(auto& [hash, symbol]: tu_a.symbols) { - if(symbol.name == "shared_func") { - shared_hash = hash; - break; - } - } - ASSERT_TRUE(shared_hash != 0); - - // The same hash should exist in project symbols. - ASSERT_TRUE(project.symbols.contains(shared_hash)); - - // reference_files bitmap should contain entries from both TUs. - auto& proj_sym = project.symbols[shared_hash]; - ASSERT_TRUE(proj_sym.reference_files.cardinality() >= 2U); +llvm::StringRef bytes_of(const std::vector& blob) { + return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); } -TEST_CASE(SerializationRoundTrip) { - index::TUIndex tu; - ASSERT_TRUE(build_and_index(R"( - struct Foo { int x; }; - void bar(Foo f) { f.x = 42; } - )", - tu)); +TEST_CASE(MergeCollectsExternalSymbols) { + add_file("header.h", R"( + int external_fn(); + )"); + add_main("main.cpp", R"( + #include "header.h" + static int local_fn() { return 1; } + int use() { return external_fn() + local_fn(); } + )"); + ASSERT_TRUE(compile()); - index::ProjectIndex project; clice::PathPool pool; - project.merge(tu, pool); - - // Serialize. - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - project.serialize(os, pool, {}); - - // Deserialize into a fresh pool, as a new session would. - clice::PathPool fresh; - llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); - ASSERT_TRUE(loaded.has_value()); - auto& restored = *loaded; + index::ProjectIndex project; + auto view = build_view(); + ASSERT_TRUE(view.has_value()); + ASSERT_TRUE(project.merge(*view, intern_paths(*view, pool))); - // Symbol tables should have same size. - ASSERT_EQ(project.symbols.size(), restored.symbols.size()); + auto external = find_symbol(project, "external_fn"); + ASSERT_TRUE(external != 0); + // Referenced from both the header (declaration) and the main file. + ASSERT_TRUE(project.symbols[external].reference_files.cardinality() >= 2); - // Each symbol should be present in restored with same reference count. - for(auto& [hash, symbol]: project.symbols) { - ASSERT_TRUE(restored.symbols.contains(hash)); - auto& restored_sym = restored.symbols[hash]; - ASSERT_EQ(symbol.reference_files.cardinality(), restored_sym.reference_files.cardinality()); - } + // Non-External symbols never reach the project table. + ASSERT_EQ(find_symbol(project, "local_fn"), 0u); } -TEST_CASE(FileIdsMapCorrectness) { - index::TUIndex tu; - ASSERT_TRUE(build_and_index(R"( - int x = 1; - )", - tu)); - - index::ProjectIndex project; +TEST_CASE(MergeRejectsBadBitmap) { + // Field order MUST mirror TUIndex up to `symbols` (the skip-annotated + // file_indices holds no slot): serialize() always writes valid bitmap + // images, so a malformed one has to be planted by hand. + struct SymbolMirror { + std::string name; + std::uint8_t kind = 0; + std::uint8_t scope = 0; + std::vector reference_files; + }; + + struct TUIndexPrefixMirror { + std::uint32_t format_version = 0; + std::int64_t built_at = 0; + index::IncludeGraph graph; + llvm::DenseMap symbols{}; + }; + + TUIndexPrefixMirror mirror; + mirror.format_version = index::index_format_version; + mirror.graph.paths = {"/proj/main.cpp"}; + clice::Bitmap bits; + bits.add(0); + mirror.symbols[42] = {.name = "good_sym", .reference_files = index::write_bitmap(bits)}; + + // Control: the mirror layout matches — the view sees the symbol and a + // valid image merges. + auto valid = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(valid.has_value()); + auto valid_view = index::TUIndexView::from(bytes_of(*valid)); + ASSERT_TRUE(valid_view.has_value()); clice::PathPool pool; - auto file_ids_map = project.merge(tu, pool); - - // file_ids_map should have same size as TU's include graph paths. - ASSERT_EQ(file_ids_map.size(), tu.graph.paths.size()); - - // Each mapped ID should be valid in the shared pool. - for(auto mapped_id: file_ids_map) { - ASSERT_TRUE(mapped_id < pool.paths.size()); - } + index::ProjectIndex accepting; + ASSERT_TRUE(accepting.merge(*valid_view, intern_paths(*valid_view, pool))); + ASSERT_EQ(find_symbol(accepting, "good_sym"), 42u); + + // One malformed image rejects the whole result: merged bits would + // persist behind versions that match the disk, with the lost ones + // never rebuilt. The symbols that decoded fine must not stay behind. + mirror.symbols[43] = { + .name = "bad_sym", + .reference_files = {std::byte{0xff}, std::byte{0xff}, std::byte{0xff}}, + }; + auto corrupt = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(corrupt.has_value()); + auto corrupt_view = index::TUIndexView::from(bytes_of(*corrupt)); + ASSERT_TRUE(corrupt_view.has_value()); + index::ProjectIndex rejecting; + ASSERT_FALSE(rejecting.merge(*corrupt_view, intern_paths(*corrupt_view, pool))); + ASSERT_TRUE(rejecting.symbols.empty()); + + // An id past the path table is the same corruption in a decodable + // coat: silently dropped, the symbol's relations would sit in a shard + // its fan-out never visits — reject like the full TUIndex::from does. + clice::Bitmap stray; + stray.add(7); + mirror.symbols[43] = {.name = "bad_sym", .reference_files = index::write_bitmap(stray)}; + auto out_of_range = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(out_of_range.has_value()); + auto stray_view = index::TUIndexView::from(bytes_of(*out_of_range)); + ASSERT_TRUE(stray_view.has_value()); + index::ProjectIndex bounding; + ASSERT_FALSE(bounding.merge(*stray_view, intern_paths(*stray_view, pool))); + ASSERT_TRUE(bounding.symbols.empty()); } -TEST_CASE(NameSurvivesRoundTrip) { - index::TUIndex tu; - ASSERT_TRUE(build_and_index(R"( - int my_variable = 42; - void my_function() {} - )", - tu)); - +TEST_CASE(FileVersionInterning) { index::ProjectIndex project; - clice::PathPool pool; - project.merge(tu, pool); + auto a = project.intern_file_version(7, 0x1111); + ASSERT_EQ(project.intern_file_version(7, 0x1111), a); - // Verify names are populated after merge. - bool found_var = false; - bool found_func = false; - for(auto& [hash, symbol]: project.symbols) { - if(symbol.name == "my_variable") - found_var = true; - if(symbol.name == "my_function") - found_func = true; - } - ASSERT_TRUE(found_var); - ASSERT_TRUE(found_func); - - // Serialize and deserialize. - llvm::SmallString<4096> buf; - llvm::raw_svector_ostream os(buf); - project.serialize(os, pool, {}); - clice::PathPool fresh; - llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); - ASSERT_TRUE(loaded.has_value()); - auto& restored = *loaded; - - // Verify names survive round-trip. - for(auto& [hash, symbol]: project.symbols) { - ASSERT_TRUE(restored.symbols.contains(hash)); - ASSERT_EQ(restored.symbols[hash].name, symbol.name); - ASSERT_EQ(restored.symbols[hash].kind.value(), symbol.kind.value()); - } + auto b = project.intern_file_version(7, 0x2222); + ASSERT_TRUE(b != a); + ASSERT_EQ(project.file_versions.find(b)->second.path_id, 7u); + ASSERT_EQ(project.file_versions.find(b)->second.content_hash, 0x2222u); } -TEST_CASE(LocalSymbolsExcluded) { - index::TUIndex tu; - ASSERT_TRUE(build_and_index(R"( - int global = 0; - static int file_static = 1; - void foo() { int local = 2; } - )", - tu)); - +TEST_CASE(ManifestContributions) { index::ProjectIndex project; - clice::PathPool pool; - project.merge(tu, pool); - - // global (External) should be in ProjectIndex. - bool found_global = false; - bool found_static = false; - bool found_local = false; - for(auto& [hash, symbol]: project.symbols) { - if(symbol.name == "global") - found_global = true; - if(symbol.name == "file_static") - found_static = true; - if(symbol.name == "local") - found_local = true; - } - ASSERT_TRUE(found_global); - ASSERT_FALSE(found_static); - ASSERT_FALSE(found_local); + auto fv_a = project.intern_file_version(1, 0xa); + auto fv_b = project.intern_file_version(2, 0xb); + + auto manifest_for = [&](std::uint32_t tu_fv, + std::initializer_list> rows) { + index::TUManifest manifest; + manifest.tu_fv = tu_fv; + manifest.contributions = rows; + return manifest; + }; + + auto tu1_fv = project.intern_file_version(10, 0x1); + auto tu2_fv = project.intern_file_version(11, 0x2); + + // TU 1 contributes h1 to file 1 and h2 to file 2. + auto affected = project.apply_manifest(10, + manifest_for(tu1_fv, + { + {fv_a, 100}, + {fv_b, 200} + })); + ASSERT_EQ(affected.size(), std::size_t(2)); + ASSERT_EQ(project.live_variants(1).size(), std::size_t(1)); + + // TU 2 shares file 1's variant: the live set does not grow. + project.apply_manifest(11, + manifest_for(tu2_fv, + { + {fv_a, 100} + })); + ASSERT_EQ(project.live_variants(1).size(), std::size_t(1)); + + // TU 1 re-indexes with a new variant for file 1 and drops file 2: both + // hashes stay live on file 1 (TU 2 still holds the old one), file 2 + // loses its only contribution. + project.apply_manifest(10, + manifest_for(tu1_fv, + { + {fv_a, 300} + })); + ASSERT_EQ(project.live_variants(1).size(), std::size_t(2)); + ASSERT_TRUE(project.live_variants(2).empty()); + + project.remove_manifest(11); + auto live = project.live_variants(1); + ASSERT_EQ(live.size(), std::size_t(1)); + ASSERT_EQ(live.front(), 300u); + + project.remove_manifest(10); + ASSERT_TRUE(project.contributions.empty()); } -TEST_CASE(EmptyPathRejected) { - // A codec-valid blob whose path table carries an empty entry is corrupt: - // the writer only emits interned (never empty) paths. - index::ProjectIndex corrupt; - corrupt.format_version = index::index_format_version; - corrupt.paths.emplace_back(0, ""); - - llvm::SmallString<128> buf; - llvm::raw_svector_ostream os(buf); - index::serialize_blob(corrupt, os); +TEST_CASE(GlobalRoundTripWithRealMerge) { + add_main("main.cpp", R"( + int global_value = 42; + int reader() { return global_value; } + )"); + ASSERT_TRUE(compile()); clice::PathPool pool; - llvm::SmallVector shards; - ASSERT_FALSE(index::ProjectIndex::from(buf.str(), pool, shards).has_value()); - ASSERT_TRUE(pool.paths.empty()); -} - -TEST_CASE(ScopeRoundTrip) { - index::TUIndex tu; - ASSERT_TRUE(build_and_index(R"( - int external_var = 0; - static int tu_local_var = 1; - void foo() { int file_local_var = 2; } - )", - tu)); - index::ProjectIndex project; - clice::PathPool pool; - project.merge(tu, pool); + auto view = build_view(); + ASSERT_TRUE(view.has_value()); + auto file_ids_map = intern_paths(*view, pool); + ASSERT_TRUE(project.merge(*view, file_ids_map)); + + // A manifest referencing the main file keeps its FileVersion alive + // through the write's garbage collection. + auto main_fv = project.intern_file_version(file_ids_map[view->path_count() - 1], + view->path_hash(view->path_count() - 1)); + index::TUManifest manifest; + manifest.tu_fv = main_fv; + project.apply_manifest(file_ids_map[view->path_count() - 1], std::move(manifest)); llvm::SmallString<4096> buf; llvm::raw_svector_ostream os(buf); - project.serialize(os, pool, {}); - clice::PathPool fresh; - llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); - ASSERT_TRUE(loaded.has_value()); - auto& restored = *loaded; + project.serialize_global(os, pool); - for(auto& [hash, symbol]: project.symbols) { - ASSERT_TRUE(restored.symbols.contains(hash)); - ASSERT_EQ(static_cast(restored.symbols[hash].scope), static_cast(symbol.scope)); - } + clice::PathPool fresh; + index::ProjectIndex loaded; + llvm::DenseMap pins; + ASSERT_TRUE(loaded.load_global(buf.str(), fresh, pins)); + + auto symbol = find_symbol(loaded, "global_value"); + ASSERT_TRUE(symbol != 0); + auto main_path = pool.resolve(file_ids_map[view->path_count() - 1]); + auto fresh_id = fresh.find(main_path); + ASSERT_TRUE(fresh_id.has_value()); + ASSERT_TRUE(loaded.symbols[symbol].reference_files.contains(*fresh_id)); } }; // TEST_SUITE(ProjectIndex) + } // namespace } // namespace clice::testing diff --git a/tests/unit/index/shard_tests.cpp b/tests/unit/index/shard_tests.cpp new file mode 100644 index 000000000..2efe49178 --- /dev/null +++ b/tests/unit/index/shard_tests.cpp @@ -0,0 +1,756 @@ +#include +#include +#include +#include + +#include "test/test.h" +#include "test/tester.h" +#include "index/serialization.h" +#include "index/shard.h" +#include "index/tu_index.h" + +#include "llvm/Support/MemoryBuffer.h" +#include "llvm/Support/raw_ostream.h" +#include "llvm/Support/xxhash.h" + +namespace clice::testing { +namespace { + +TEST_SUITE(Shard, Tester) { + +index::TUIndex tu_index; + +void build_index(llvm::StringRef code, + std::source_location location = std::source_location::current()) { + add_main("main.cpp", code); + ASSERT_TRUE(compile()); + tu_index = index::TUIndex::build(*unit); +} + +std::optional lookup_symbol(index::SymbolHash hash) { + auto it = tu_index.symbols.find(hash); + if(it == tu_index.symbols.end()) { + return std::nullopt; + } + return index::SymbolIdentity{it->second.name, it->second.kind, it->second.scope}; +} + +std::string write_fresh(const index::FileIndex& rows, + index::RowsHash hash, + llvm::StringRef content, + bool with_symbols = false) { + auto resolve = [this](index::SymbolHash symbol) { + return lookup_symbol(symbol); + }; + index::VariantInput fresh{hash, &rows, {}}; + if(with_symbols) { + fresh.symbols = resolve; + } + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard(index::Shard(), {}, fresh, content, llvm::xxh3_64bits(content), os); + return bytes; +} + +std::string append_variant(const index::Shard& old, + const index::FileIndex& rows, + index::RowsHash hash) { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard(old, + old.variants(), + {hash, &rows, {}}, + old.content(), + old.content_hash(), + os); + return bytes; +} + +/// Owning wrap: from_bytes borrows, and every builder here returns a +/// temporary string. +index::Shard make_shard(llvm::StringRef bytes) { + return index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); +} + +index::SymbolHash hash_at(const index::Shard& shard, std::uint32_t offset) { + index::SymbolHash result = 0; + shard.lookup(offset, [&](const index::Occurrence& o) { + result = o.target; + return false; + }); + return result; +} + +TEST_CASE(RoundtripLookups) { + build_index(R"( + int §(def)⟦§(def)foo⟧() { return 42; } + int bar() { return §(ref)⟦§(ref)foo⟧(); } + )"); + + auto content = sources.all_files.find("main.cpp")->second.content; + auto bytes = + write_fresh(tu_index.main_file_index, tu_index.main_file_index.rows_hash(), content); + auto shard = make_shard(bytes); + ASSERT_TRUE(shard.loaded()); + ASSERT_EQ(shard.content(), llvm::StringRef(content)); + ASSERT_FALSE(shard.line_starts().empty()); + + auto expected = range("ref"); + bool found = false; + shard.lookup(point("ref"), [&](const index::Occurrence& o) { + found = true; + EXPECT_EQ(o.range.begin, expected.begin); + return false; + }); + ASSERT_TRUE(found); + + // The definition relation of the symbol under the reference resolves to + // the definition site, with the full extent in the payload. + auto symbol = hash_at(shard, point("ref")); + ASSERT_TRUE(symbol != 0); + bool has_definition = false; + shard.lookup(symbol, RelationKind::Definition, [&](const index::Relation& r) { + has_definition = true; + EXPECT_EQ(r.range.begin, range("def").begin); + return false; + }); + ASSERT_TRUE(has_definition); +} + +index::FileIndex simple_rows(std::initializer_list occurrences) { + index::FileIndex rows; + rows.occurrences = occurrences; + return rows; +} + +TEST_CASE(VariantMaskFiltering) { + auto a = simple_rows({ + {{0, 3}, 111} + }); + auto b = simple_rows({ + {{0, 3}, 111}, + {{10, 13}, 222} + }); + + auto first = make_shard(write_fresh(a, 1, "aaa bbb ccc ddd")); + auto shard = make_shard(append_variant(first, b, 2)); + ASSERT_TRUE(shard.has_variant(1)); + ASSERT_TRUE(shard.has_variant(2)); + + // All variants live by default: both rows serve. + ASSERT_EQ(hash_at(shard, 1), 111u); + ASSERT_EQ(hash_at(shard, 11), 222u); + + // Restricting to variant 1 hides the row only variant 2 holds, while + // the shared row keeps serving. + shard.set_live({1}); + ASSERT_TRUE(shard.has_dead_variants()); + ASSERT_EQ(hash_at(shard, 1), 111u); + ASSERT_EQ(hash_at(shard, 11), 0u); + + shard.set_live({1, 2}); + ASSERT_FALSE(shard.has_dead_variants()); + ASSERT_EQ(hash_at(shard, 11), 222u); + + shard.set_live({}); + ASSERT_EQ(hash_at(shard, 1), 0u); +} + +TEST_CASE(CompactionDropsVariant) { + auto a = simple_rows({ + {{0, 3}, 111} + }); + auto b = simple_rows({ + {{0, 3}, 111}, + {{10, 13}, 222} + }); + auto first = make_shard(write_fresh(a, 1, "aaa bbb ccc ddd")); + auto both = make_shard(append_variant(first, b, 2)); + + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard(both, {1}, {}, both.content(), both.content_hash(), os); + auto compacted = make_shard(bytes); + ASSERT_TRUE(compacted.has_variant(1)); + ASSERT_FALSE(compacted.has_variant(2)); + ASSERT_EQ(hash_at(compacted, 1), 111u); + ASSERT_EQ(hash_at(compacted, 11), 0u); +} + +/// Grow a shard to `count` variants: variant i holds the shared occurrence +/// and relation plus a unique one of each at offset i * 16. +index::Shard grow_variants(std::uint32_t count) { + std::string content(16 * (count + 2), 'x'); + index::Shard shard; + for(std::uint32_t i = 1; i <= count; i += 1) { + auto rows = simple_rows({ + {{0, 3}, 111 }, + {{i * 16, i * 16 + 3}, 1000 + i} + }); + rows.relations[999] = { + {.kind = RelationKind::Reference, .range = {0, 3}, .target_symbol = 0}, + {.kind = RelationKind::Reference, .range = {i * 16, i * 16 + 3}, .target_symbol = 0}, + }; + shard = make_shard(shard.loaded() ? append_variant(shard, rows, i) + : write_fresh(rows, i, content)); + } + return shard; +} + +std::size_t reference_count(const index::Shard& shard, index::SymbolHash symbol) { + std::size_t count = 0; + shard.lookup(symbol, RelationKind::Reference, [&](const index::Relation&) { + count += 1; + return true; + }); + return count; +} + +void expect_tier_behavior(std::uint32_t count) { + auto shard = grow_variants(count); + ASSERT_EQ(shard.variants().size(), std::size_t(count)); + + // Every variant's unique row serves under the full live set, and the + // shared relation collapsed to one row across all variants. + for(std::uint32_t i = 1; i <= count; i += 1) { + ASSERT_EQ(hash_at(shard, i * 16 + 1), 1000u + i); + } + ASSERT_EQ(reference_count(shard, 999), std::size_t(count) + 1); + + // One live variant: its unique rows and the shared rows serve, another + // variant's do not — on the occurrence and the relation side alike. + shard.set_live({3}); + ASSERT_EQ(hash_at(shard, 1), 111u); + ASSERT_EQ(hash_at(shard, 3 * 16 + 1), 1003u); + ASSERT_EQ(hash_at(shard, 5 * 16 + 1), 0u); + ASSERT_EQ(reference_count(shard, 999), std::size_t(2)); +} + +TEST_CASE(MaskTier32) { + expect_tier_behavior(5); +} + +TEST_CASE(MaskTier64) { + expect_tier_behavior(40); +} + +TEST_CASE(MaskTierRoaring) { + expect_tier_behavior(70); +} + +TEST_CASE(LongTokenEscape) { + auto rows = simple_rows({ + {{0, 300}, 111}, + {{400, 404}, 222} + }); + std::string content(500, 'y'); + auto shard = make_shard(write_fresh(rows, 1, content)); + + bool found = false; + shard.lookup(299, [&](const index::Occurrence& o) { + found = true; + EXPECT_EQ(o.range.end, 300u); + return false; + }); + ASSERT_TRUE(found); + ASSERT_EQ(hash_at(shard, 402), 222u); +} + +TEST_CASE(RelationPayloadRoundtrip) { + index::FileIndex rows; + index::Relation definition{ + .kind = RelationKind::Definition, + .range = {0, 3} + }; + definition.set_definition_range({0, 50}); + rows.relations[111] = { + definition, + {.kind = RelationKind::Reference, .range = {10, 13}, .target_symbol = 0}, + }; + rows.relations[333] = { + {.kind = RelationKind::Base, .range = {20, 23}, .target_symbol = 444}, + }; + + auto shard = make_shard(write_fresh(rows, 1, std::string(60, 'z'))); + + bool checked_definition = false; + shard.lookup(111, RelationKind::Definition, [&](const index::Relation& r) { + checked_definition = true; + auto extent = index::Relation(r).definition_range(); + EXPECT_EQ(extent.begin, 0u); + EXPECT_EQ(extent.end, 50u); + return false; + }); + ASSERT_TRUE(checked_definition); + + bool checked_reference = false; + shard.lookup(111, RelationKind::Reference, [&](const index::Relation& r) { + checked_reference = true; + EXPECT_EQ(r.target_symbol, 0u); + return false; + }); + ASSERT_TRUE(checked_reference); + + bool checked_pair = false; + shard.lookup(333, RelationKind::Base, [&](const index::Relation& r) { + checked_pair = true; + EXPECT_EQ(r.target_symbol, 444u); + return false; + }); + ASSERT_TRUE(checked_pair); +} + +TEST_CASE(LocalSymbolNames) { + build_index(R"( + static int §(local)⟦§(local)helper⟧() { return 1; } + int visible() { return §(use)⟦§(use)helper⟧(); } + )"); + + auto content = sources.all_files.find("main.cpp")->second.content; + auto shard = make_shard(write_fresh(tu_index.main_file_index, + tu_index.main_file_index.rows_hash(), + content, + /*with_symbols=*/true)); + + auto local = hash_at(shard, point("use")); + ASSERT_TRUE(local != 0); + std::string name; + SymbolKind kind; + ASSERT_TRUE(shard.find_symbol(local, name, kind)); + ASSERT_EQ(name, "helper"); + + // External names live in the ProjectIndex, never in the blob. + auto external = [&] { + for(auto& [hash, symbol]: tu_index.symbols) { + if(symbol.name == "visible") { + return hash; + } + } + return index::SymbolHash(0); + }(); + ASSERT_TRUE(external != 0); + ASSERT_FALSE(shard.find_symbol(external, name, kind)); +} + +TEST_CASE(WideSymbolIds) { + // Past 65535 distinct symbols the id columns must widen to u32; a + // truncating writer corrupts resolution only on indexes this large. + index::FileIndex rows; + constexpr std::uint32_t count = 70000; + rows.occurrences.reserve(count); + for(std::uint32_t i = 0; i < count; i += 1) { + rows.occurrences.push_back({ + {i * 8, i * 8 + 3}, + 0x100000u + i + }); + } + std::string content(count * 8 + 16, 'w'); + auto shard = make_shard(write_fresh(rows, 1, content)); + ASSERT_EQ(hash_at(shard, 69999 * 8 + 1), 0x100000u + 69999); + ASSERT_EQ(hash_at(shard, 3 * 8 + 1), 0x100000u + 3); +} + +TEST_CASE(UnloadedShardNoops) { + index::Shard shard; + shard.lookup(0, [&](const index::Occurrence&) { return true; }); + shard.lookup(1, RelationKind::Reference, [&](const index::Relation&) { return true; }); + std::string name; + SymbolKind kind; + ASSERT_FALSE(shard.find_symbol(1, name, kind)); + ASSERT_TRUE(shard.content().empty()); + ASSERT_TRUE(shard.line_starts().empty()); +} + +TEST_CASE(CorruptBlobRejected) { + ASSERT_FALSE(index::Shard::from_bytes("not a flatbuffer").loaded()); + + // A valid blob cut mid-structure must fail verification, not be + // misread. (One trailing byte can be alignment padding, so the cut + // must reach real data.) + auto rows = simple_rows({ + {{0, 3}, 111} + }); + auto bytes = write_fresh(rows, 1, "aaaa"); + ASSERT_FALSE( + index::Shard::from_bytes(llvm::StringRef(bytes).take_front(bytes.size() / 2)).loaded()); + + // A structurally valid blob of the current version but with no variants + // is impossible output of the writer, and must not load either. + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto stale = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); + ASSERT_TRUE(stale.has_value()); + auto data = llvm::StringRef(reinterpret_cast(stale->data()), stale->size()); + ASSERT_FALSE(index::Shard::from_bytes(data).loaded()); +} + +TEST_CASE(ContentHashMismatchRejected) { + // Every freshness decision compares the advertised content hash, so + // content bytes corrupted under an intact structure would keep loading + // as fresh while position mapping reads the wrong text. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content = "aaaa"; + blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.content = "aaab"; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(MisorderedRowsRejected) { + // Occurrence lookup binary-searches decoded row ends; a corrupt blob + // whose rows lost their order must load as "not on disk" and be + // rebuilt, not keep misresolving queries on every restart. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content = "aaaaaaaaaaaaaaaa"; + blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + blob.occ_begins = {0, 8}; + blob.occ_lengths = {3, 0xff}; // 0xff escapes to (row, end) + blob.occ_long_rows = {1}; + blob.occ_long_ends = {12}; + blob.occ_syms16 = {0, 0}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + // Begins out of order. + blob.occ_begins = {8, 0}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + // Begins sorted, but the escaped end regresses below the row before. + blob.occ_begins = {0, 8}; + blob.occ_lengths = {0xff, 3}; + blob.occ_long_rows = {0}; + blob.occ_long_ends = {14}; // ends decode to {14, 11} + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + // An escaped end before its own begin. + blob.occ_lengths = {3, 0xff}; + blob.occ_long_rows = {1}; + blob.occ_long_ends = {5}; // row 1: begin 8, end 5 + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(EscapeTableMismatchRejected) { + // A sentinel length without its sparse entry decodes as begin + 255 + // (end_of's fallback) and a stray entry is silently ignored: with + // content long enough both pass every range bound and would serve + // wrong ranges forever, so only the pairing check can reject them. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content = std::string(300, 'a'); + blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + blob.occ_begins = {0}; + blob.occ_lengths = {0xff}; + blob.occ_long_rows = {0}; + blob.occ_long_ends = {260}; + blob.occ_syms16 = {0}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + // A sentinel without its sparse entry. + blob.occ_long_rows = {}; + blob.occ_long_ends = {}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + // A sparse entry pointing at an unescaped row. + blob.occ_lengths = {3}; + blob.occ_long_rows = {0}; + blob.occ_long_ends = {260}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + // The relation escape table is validated alike. + blob.occ_lengths = {0xff}; + blob.sym_rel_offsets = {0, 1}; + blob.rel_kinds = {static_cast(RelationKind::Reference)}; + blob.rel_begins = {0}; + blob.rel_lengths = {0xff}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(RangesBeyondContentRejected) { + // Every decoded range is served as a source range into the stored + // content; an end past it would map positions through text that does + // not exist — forever, since the blob's content hash still matches the + // disk and nothing rebuilds it. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content = "aaaaaaaaaaaaaaaa"; + blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 1}; + blob.occ_begins = {0}; + blob.occ_lengths = {3}; + blob.occ_syms16 = {0}; + blob.rel_kinds = {static_cast(RelationKind::Reference)}; + blob.rel_begins = {0}; + blob.rel_lengths = {3}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + // A plain length overruns the 16-byte content. + blob.occ_lengths = {100}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + // An escaped end does too. + blob.occ_lengths = {0xff}; + blob.occ_long_rows = {0}; + blob.occ_long_ends = {600}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + blob.occ_lengths = {3}; + blob.occ_long_rows = {}; + blob.occ_long_ends = {}; + + // Relation ranges are bounded alike. + blob.rel_lengths = {100}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + // Except the no-range sentinel a pair relation legitimately carries — + // on a source-located kind the same sentinel is corruption. + blob.rel_kinds = {static_cast(RelationKind::Base)}; + blob.rel_begins = {0xffffffff}; + blob.rel_lengths = {0}; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + blob.rel_kinds = {static_cast(RelationKind::Reference)}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + blob.rel_begins = {0}; + blob.rel_lengths = {3}; + + // And definition-range payloads. + blob.rel_def_rows = {0}; + blob.rel_def_begins = {0}; + blob.rel_def_ends = {600}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(DuplicateSymbolHashRejected) { + // Symbol lookups lower-bound the hash column and read only the first + // match's slices: a duplicated hash strands the later id's relations + // unreachably while the blob keeps loading as fresh. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content = "aaaa"; + blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + blob.variants = {1}; + blob.sym_hashes = {111, 222}; + blob.sym_rel_offsets = {0, 0, 0}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.sym_hashes = {111, 111}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(DuplicateVariantRejected) { + // Liveness and compaction select variants by rows hash; a duplicated + // entry would make every copy live at once, and rows masked only to the + // extra id would serve and survive with no contribution owning them. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content = "aaaa"; + blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + blob.variants = {1, 2}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.variants = {1, 1}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(StraySymbolIdRejected) { + // Lookups dereference symbol ids straight into the hash table; an id + // past it would previously read as "no symbol", missing the occurrence + // or dropping the relation's target forever with no reindex triggered. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content = "aaaaaaaaaaaaaaaa"; + blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + blob.variants = {1}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 1}; + blob.occ_begins = {0}; + blob.occ_lengths = {3}; + blob.occ_syms16 = {0}; + blob.rel_kinds = {static_cast(RelationKind::Base)}; + blob.rel_begins = {4}; + blob.rel_lengths = {3}; + blob.rel_sym_rows = {0}; + blob.rel_sym16 = {0}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.occ_syms16 = {5}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + blob.occ_syms16 = {0}; + + blob.rel_sym16 = {5}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(OwnerlessMaskRejected) { + // A mask owning no stored variant serves its row unconditionally while + // every variant is live (row_live's live.all fast path never consults + // it), vanishes once any variant dies, and the next compaction erases + // it for real — so it must reject the blob at load. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content = "aaaa"; + blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + blob.variants = {1, 2}; + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + blob.occ_begins = {0}; + blob.occ_lengths = {3}; + blob.occ_syms16 = {0}; + blob.occ_masks32 = {0b01}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + // An empty mask, then one whose only bit lies past the variant table. + blob.occ_masks32 = {0}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + blob.occ_masks32 = {0b100}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + // The u64 tier is bounded alike. + for(std::uint32_t i = 3; i <= 40; i += 1) { + blob.variants.push_back(i); + } + blob.occ_masks32 = {}; + blob.occ_masks64 = {1}; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + blob.occ_masks64 = {std::uint64_t(1) << 45}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + + // And roaring masks: decodable but empty, or holding only dropped ids. + for(std::uint32_t i = 41; i <= 70; i += 1) { + blob.variants.push_back(i); + } + blob.occ_masks64 = {}; + auto set_mask = [&](const clice::Bitmap& mask) { + blob.occ_roaring.clear(); + for(auto byte: index::write_bitmap(mask)) { + blob.occ_roaring.push_back(static_cast(byte)); + } + blob.occ_roaring_offsets = {0, static_cast(blob.occ_roaring.size())}; + blob.rel_roaring_offsets = {0}; + }; + clice::Bitmap in_range; + in_range.add(69); + set_mask(in_range); + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + set_mask({}); + ASSERT_FALSE(make_shard(bytes_of()).loaded()); + clice::Bitmap stray; + stray.add(70); + set_mask(stray); + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +TEST_CASE(CorruptRoaringMaskRejected) { + // Roaring row masks gate liveness and are rewritten by compaction; a + // slice failing decode would read the row as dead and the next + // compaction would erase it for real, every manifest still fresh — so + // an undecodable slice must reject the blob at load. + index::ShardBlob blob; + blob.format_version = index::index_format_version; + blob.content = "aaaa"; + blob.content_hash = llvm::xxh3_64bits(llvm::StringRef(blob.content)); + for(std::uint32_t i = 1; i <= 65; i += 1) { + blob.variants.push_back(i); + } + blob.sym_hashes = {111}; + blob.sym_rel_offsets = {0, 0}; + blob.occ_begins = {0}; + blob.occ_lengths = {3}; + blob.occ_syms16 = {0}; + + clice::Bitmap mask; + mask.add(2); + for(auto byte: index::write_bitmap(mask)) { + blob.occ_roaring.push_back(static_cast(byte)); + } + blob.occ_roaring_offsets = {0, static_cast(blob.occ_roaring.size())}; + blob.rel_roaring_offsets = {0}; + + auto bytes_of = [&] { + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_blob(blob, os); + return bytes; + }; + ASSERT_TRUE(make_shard(bytes_of()).loaded()); + + blob.occ_roaring = {0xff, 0xff, 0xff}; + blob.occ_roaring_offsets = {0, 3}; + ASSERT_FALSE(make_shard(bytes_of()).loaded()); +} + +}; // TEST_SUITE(Shard) + +} // namespace +} // namespace clice::testing diff --git a/tests/unit/index/tu_index_tests.cpp b/tests/unit/index/tu_index_tests.cpp index c60554830..ec3a7de71 100644 --- a/tests/unit/index/tu_index_tests.cpp +++ b/tests/unit/index/tu_index_tests.cpp @@ -1210,24 +1210,34 @@ TEST_CASE(SerializeRoundTrip) { ASSERT_TRUE(loaded->graph.locations == tu_index.graph.locations); ASSERT_TRUE(loaded->graph.path_hashes == tu_index.graph.path_hashes); - // The persisted per-file rows are keyed by path id; recompute the - // expected conversion from the build-time FileID-keyed state. + // The persisted per-file rows travel as wire sections keyed by path id; + // recompute the expected conversion from the build-time FileID-keyed + // state. Empty rows get no section. llvm::DenseMap> expected; for(auto& [fid, file_index]: tu_index.file_indices) { - expected[tu_index.graph.path_id(fid)] = {file_index.occurrences.size(), - file_index.relations.size()}; + if(!file_index.empty()) { + expected[tu_index.graph.path_id(fid)] = {file_index.occurrences.size(), + file_index.relations.size()}; + } } ASSERT_FALSE(expected.empty()); - ASSERT_EQ(loaded->path_file_indices.size(), expected.size()); + // The interested file's rows travel as the last section. + ASSERT_EQ(loaded->sections.size(), expected.size() + 1); for(auto& [path_id, counts]: expected) { - auto it = loaded->path_file_indices.find(path_id); - ASSERT_TRUE(it != loaded->path_file_indices.end()); - ASSERT_EQ(it->second.occurrences.size(), counts.first); - ASSERT_EQ(it->second.relations.size(), counts.second); + auto it = std::ranges::find(loaded->sections, path_id, &index::FileSection::path_id); + ASSERT_TRUE(it != loaded->sections.end()); + auto rows = index::TUIndex::decode_rows(*it); + ASSERT_TRUE(rows.has_value()); + ASSERT_EQ(rows->occurrences.size(), counts.first); + ASSERT_EQ(rows->relations.size(), counts.second); } - ASSERT_TRUE(loaded->main_file_index.occurrences == tu_index.main_file_index.occurrences); - ASSERT_EQ(loaded->main_file_index.relations.size(), tu_index.main_file_index.relations.size()); + auto* main_sec = loaded->main_section(); + ASSERT_TRUE(main_sec != nullptr); + auto main_rows = index::TUIndex::decode_rows(*main_sec); + ASSERT_TRUE(main_rows.has_value()); + ASSERT_TRUE(main_rows->occurrences == tu_index.main_file_index.occurrences); + ASSERT_EQ(main_rows->relations.size(), tu_index.main_file_index.relations.size()); ASSERT_EQ(loaded->symbols.size(), tu_index.symbols.size()); for(auto& [hash, symbol]: tu_index.symbols) { @@ -1310,7 +1320,7 @@ TEST_CASE(FromRejectsOutOfRangePathIds) { honest.built_at = std::chrono::milliseconds(0); honest.graph.paths = {"/proj/main.cpp"}; honest.graph.locations.push_back({.path_id = 0, .line = 1, .include = 0}); - honest.path_file_indices.try_emplace(0); + honest.sections.push_back({.path_id = 0}); honest.symbols[42].reference_files.add(0); ASSERT_TRUE(index::TUIndex::from(serialized(honest)).has_value()); @@ -1325,7 +1335,7 @@ TEST_CASE(FromRejectsOutOfRangePathIds) { index::TUIndex hostile; hostile.built_at = std::chrono::milliseconds(0); hostile.graph.paths = {"/proj/main.cpp"}; - hostile.path_file_indices.try_emplace(7); // Only path id 0 exists. + hostile.sections.push_back({.path_id = 7}); // Only path id 0 exists. ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); } { @@ -1358,7 +1368,7 @@ TEST_CASE(FromNormalizesPathHashes) { } } -TEST_CASE(ReserializeKeepsPathIndices) { +TEST_CASE(ReserializeKeepsSections) { add_file("header.h", R"( #pragma once inline int helper() { return 1; } @@ -1377,19 +1387,20 @@ TEST_CASE(ReserializeKeepsPathIndices) { auto loaded = index::TUIndex::from(buf); ASSERT_TRUE(loaded.has_value()); ASSERT_TRUE(loaded->file_indices.empty()); - ASSERT_FALSE(loaded->path_file_indices.empty()); + ASSERT_FALSE(loaded->sections.empty()); // A deserialized index has no FileID-keyed state; re-serializing must - // keep the path-keyed rows instead of wiping them from an empty map. + // keep the wire sections instead of wiping them from the empty maps. llvm::SmallString<4096> again; llvm::raw_svector_ostream os2(again); loaded->serialize(os2); auto reloaded = index::TUIndex::from(again); ASSERT_TRUE(reloaded.has_value()); - ASSERT_EQ(reloaded->path_file_indices.size(), loaded->path_file_indices.size()); - for(auto& [path_id, file_index]: loaded->path_file_indices) { - ASSERT_TRUE(reloaded->path_file_indices.contains(path_id)); + ASSERT_EQ(reloaded->sections.size(), loaded->sections.size()); + for(std::size_t i = 0; i < loaded->sections.size(); i += 1) { + ASSERT_EQ(reloaded->sections[i].path_id, loaded->sections[i].path_id); + ASSERT_EQ(reloaded->sections[i].rows_hash, loaded->sections[i].rows_hash); } } diff --git a/tests/unit/server/indexer_tests.cpp b/tests/unit/server/indexer_tests.cpp index 380251a0f..cbbecfbad 100644 --- a/tests/unit/server/indexer_tests.cpp +++ b/tests/unit/server/indexer_tests.cpp @@ -1,9 +1,14 @@ +#include #include +#include #include "test/temp_dir.h" #include "test/test.h" #include "command/argument_parser.h" #include "compile/compilation.h" +#include "index/manifest.h" +#include "index/shard.h" +#include "index/storage.h" #include "index/tu_index.h" #include "server/compiler/context_resolver.h" #include "server/compiler/indexer.h" @@ -12,6 +17,7 @@ #include "server/worker/worker_pool.h" #include "llvm/Support/raw_ostream.h" +#include "llvm/Support/xxhash.h" namespace clice::testing { @@ -58,18 +64,40 @@ struct IndexerFixture { void consume(std::uint32_t id) { indexer.pending_ids.erase(id); } -}; -namespace { + /// Start a fresh staleness round: per-round FileVersion verdicts are + /// cleared by run_background_indexing, which these tests bypass. + void clear_verdicts() { + indexer.fv_verdicts.clear(); + } -TEST_SUITE(IndexerMerge) { + /// Judge staleness inside the current round (see clear_verdicts). + bool need_update(llvm::StringRef path) { + return indexer.need_update(path); + } -kota::event_loop loop; -Workspace workspace; -SessionStore store; -WorkerPool pool{loop}; -ContextResolver resolver{workspace}; -Indexer indexer{loop, workspace, pool, resolver, store}; + bool global_dirty() { + return indexer.global_dirty; + } + + /// Drop the merge's own dirty mark so a later assertion isolates the + /// stamp-repair path. + void reset_global_dirty() { + indexer.global_dirty = false; + } + + /// Run one save() to completion on the fixture's loop. + void save() { + auto body = [this]() -> kota::task<> { + co_await indexer.save(); + }; + auto task = body(); + loop.schedule(task); + loop.run(); + } +}; + +namespace { struct IndexedTU { std::string data; ///< Serialized TUIndex, as a worker would ship it. @@ -77,10 +105,11 @@ struct IndexedTU { }; /// Index a real on-disk file in-process and serialize its TUIndex. -IndexedTU index_file(TempDir& tmp, llvm::StringRef file) { +IndexedTU index_file(TempDir& tmp, llvm::StringRef file, std::vector extra_args = {}) { std::string resource = std::string(resource_dir()); std::vector args = {"clang++", "-fsyntax-only", "-resource-dir", resource, "-c", std::string(file)}; + args.insert(args.end(), extra_args.begin(), extra_args.end()); CompilationParams cp; cp.kind = CompilationKind::Indexing; @@ -101,16 +130,37 @@ IndexedTU index_file(TempDir& tmp, llvm::StringRef file) { return result; } +void open_store(TempDir& tmp, Workspace& workspace) { + auto store = CacheStore::open(tmp.path("cache"), 1); + ASSERT_TRUE(store.has_value()); + workspace.store.emplace(std::move(*store)); + workspace.index_storage = index::make_fs_index_storage(*workspace.store); +} + +/// The storage key of a file's shard or manifest blob (Indexer's naming). +std::string blob_key(llvm::StringRef path) { + return std::format("{:016x}", llvm::xxh3_64bits(path)); +} + +TEST_SUITE(IndexerMerge) { + +kota::event_loop loop; +Workspace workspace; +SessionStore store; +WorkerPool pool{loop}; +ContextResolver resolver{workspace}; +Indexer indexer{loop, workspace, pool, resolver, store}; + TEST_CASE(MergeRejectsGarbage) { // A worker shipping corrupted bytes (torn write, stale format) must not // crash the master or leave partial state behind. - ASSERT_TRUE(workspace.merged_indices.empty()); + ASSERT_TRUE(workspace.shards.empty()); ASSERT_TRUE(workspace.project_index.symbols.empty()); std::string garbage = "definitely not a flatbuffer, but long enough to try"; indexer.merge(garbage.data(), garbage.size()); - ASSERT_TRUE(workspace.merged_indices.empty()); + ASSERT_TRUE(workspace.shards.empty()); ASSERT_TRUE(workspace.project_index.symbols.empty()); } @@ -124,8 +174,8 @@ TEST_CASE(MergeSkipsMovedDisk) { indexer.merge(indexed.data.data(), indexed.data.size()); auto path_id = workspace.path_pool.intern(indexed.tu_path); - auto it = workspace.merged_indices.find(path_id); - ASSERT_TRUE(it != workspace.merged_indices.end()); + auto it = workspace.shards.find(path_id); + ASSERT_TRUE(it != workspace.shards.end()); ASSERT_EQ(it->second.content(), "int value() { return 1; }\n"); // The disk moved on since the rows were indexed: merging them would @@ -134,7 +184,7 @@ TEST_CASE(MergeSkipsMovedDisk) { tmp.touch("main.cpp", "int renamed() { return 2; }\n"); indexer.merge(indexed.data.data(), indexed.data.size()); ASSERT_EQ(it->second.content(), "int value() { return 1; }\n"); - ASSERT_TRUE(it->second.has_contribution(indexed.tu_path)); + ASSERT_TRUE(workspace.project_index.contributions.lookup(path_id).contains(path_id)); // Once the rows describe the settled content again, the merge lands. auto fresh = index_file(tmp, src); @@ -143,15 +193,7 @@ TEST_CASE(MergeSkipsMovedDisk) { ASSERT_EQ(it->second.content(), "int renamed() { return 2; }\n"); } -void open_store(TempDir& tmp, Workspace& workspace) { - auto store = CacheStore::open(tmp.path("cache"), 1); - ASSERT_TRUE(store.has_value()); - store->register_namespace( - {.name = "index", .extension = ".idx", .policy = CachePolicy::Persistent}); - workspace.store.emplace(std::move(*store)); -} - -TEST_CASE(SaveFlipsShards) { +TEST_CASE(SaveCommitsDirtyShard) { TempDir tmp; tmp.touch("main.cpp", "int flip_value() { return 1; }\n"); auto src = tmp.path("main.cpp"); @@ -162,7 +204,7 @@ TEST_CASE(SaveFlipsShards) { indexer.merge(indexed.data.data(), indexed.data.size()); auto path_id = workspace.path_pool.intern(indexed.tu_path); - ASSERT_TRUE(workspace.merged_indices.find(path_id)->second.need_rewrite()); + ASSERT_EQ(indexer.pending_shard_writes(), 1u); // Named body: a temporary lambda's captures die with the statement // while the coroutine frame still references them. @@ -173,13 +215,14 @@ TEST_CASE(SaveFlipsShards) { loop.schedule(task); loop.run(); - // Committed and flipped back to the buffer-backed blob; the shard - // still answers identically. - auto it = workspace.merged_indices.find(path_id); - ASSERT_TRUE(it != workspace.merged_indices.end()); - ASSERT_FALSE(it->second.need_rewrite()); + // Committed: the dirty state is drained and the shard still answers + // identically. + auto it = workspace.shards.find(path_id); + ASSERT_TRUE(it != workspace.shards.end()); + ASSERT_EQ(indexer.pending_shard_writes(), 0u); + ASSERT_EQ(indexer.last_save_shards(), 1u); ASSERT_EQ(it->second.content(), "int flip_value() { return 1; }\n"); - ASSERT_TRUE(it->second.has_contribution(indexed.tu_path)); + ASSERT_TRUE(workspace.project_index.contributions.lookup(path_id).contains(path_id)); } TEST_CASE(MidSaveMergeKept) { @@ -199,13 +242,15 @@ TEST_CASE(MidSaveMergeKept) { auto fresh = index_file(tmp, src); ASSERT_FALSE(fresh.data.empty()); - // The merge task runs when save() suspends at its first commit await: - // it lands between the serialize snapshot and the flip check, exactly - // the window the revision guard exists for. + // The merge task runs when save() suspends at its write await: it + // lands after the dirty snapshot was taken and cleared, exactly the + // window re-dirtying exists for. auto save_body = [&]() -> kota::task<> { co_await indexer.save(); }; + std::size_t mid_save_pending = 0; auto merge_body = [&]() -> kota::task<> { + mid_save_pending = indexer.pending_shard_writes(); indexer.merge(fresh.data.data(), fresh.data.size()); co_return; }; @@ -215,11 +260,16 @@ TEST_CASE(MidSaveMergeKept) { loop.schedule(merge_task); loop.run(); - // The stale committed blob must not overwrite the newer merge: the - // shard keeps the new content and stays dirty for the next save. - auto it = workspace.merged_indices.find(path_id); - ASSERT_TRUE(it != workspace.merged_indices.end()); - ASSERT_TRUE(it->second.need_rewrite()); + // Sampled while save() awaited its commit: the settle gauge must keep + // covering the in-flight batch, or a stats poll in that window reads + // "settled" with last_save_shards still holding its reset. + ASSERT_TRUE(mid_save_pending >= 1); + + // The save committed the pre-merge snapshot: the shard keeps the new + // content and stays dirty so the next save commits it. + auto it = workspace.shards.find(path_id); + ASSERT_TRUE(it != workspace.shards.end()); + ASSERT_EQ(indexer.pending_shard_writes(), 1u); ASSERT_EQ(it->second.content(), "int second_value() { return 2; }\n"); auto again_body = [&]() -> kota::task<> { @@ -229,13 +279,856 @@ TEST_CASE(MidSaveMergeKept) { loop.schedule(task); loop.run(); - it = workspace.merged_indices.find(path_id); - ASSERT_FALSE(it->second.need_rewrite()); + it = workspace.shards.find(path_id); + ASSERT_EQ(indexer.pending_shard_writes(), 0u); ASSERT_EQ(it->second.content(), "int second_value() { return 2; }\n"); } +TEST_CASE(MergeHitWritesNothing) { + TempDir tmp; + tmp.touch("main.cpp", "int steady() { return 1; }\n"); + auto src = tmp.path("main.cpp"); + open_store(tmp, workspace); + + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + indexer.merge(indexed.data.data(), indexed.data.size()); + auto save_body = [&]() -> kota::task<> { + co_await indexer.save(); + }; + auto task = save_body(); + loop.schedule(task); + loop.run(); + ASSERT_EQ(indexer.pending_shard_writes(), 0u); + + // A re-merge whose rows the shard already stores is the steady state of + // every background round: it must record contributions and touch no + // blob at all. + indexer.merge(indexed.data.data(), indexed.data.size()); + ASSERT_EQ(indexer.pending_shard_writes(), 0u); +} + +TEST_CASE(SharedHeaderVariants) { + TempDir tmp; + tmp.touch("shared.h", + "#pragma once\n#ifdef MODE\nint mode_fn();\n#endif\n" + "inline int shared_fn() { return 1; }\n"); + tmp.touch("a.cpp", "#include \"shared.h\"\nint a() { return shared_fn(); }\n"); + tmp.touch("b.cpp", "#include \"shared.h\"\nint b() { return shared_fn(); }\n"); + + auto a = index_file(tmp, tmp.path("a.cpp")); + auto b = index_file(tmp, tmp.path("b.cpp"), {"-DMODE"}); + ASSERT_FALSE(a.data.empty()); + ASSERT_FALSE(b.data.empty()); + + // Two TUs preprocess the header differently: both variants coexist in + // one blob, each TU's contribution live. + indexer.merge(a.data.data(), a.data.size()); + indexer.merge(b.data.data(), b.data.size()); + auto header_id = workspace.path_pool.intern(tmp.path("shared.h")); + auto& shard = workspace.shards[header_id]; + ASSERT_EQ(shard.variants().size(), std::size_t(2)); + ASSERT_EQ(workspace.project_index.contributions.lookup(header_id).size(), std::size_t(2)); + + // A third TU sharing a's preprocessing hits the stored variant: the + // set does not grow, and neither existing contribution is disturbed. + tmp.touch("c.cpp", "#include \"shared.h\"\nint c() { return shared_fn(); }\n"); + auto c = index_file(tmp, tmp.path("c.cpp")); + ASSERT_FALSE(c.data.empty()); + indexer.merge(c.data.data(), c.data.size()); + ASSERT_EQ(shard.variants().size(), std::size_t(2)); + ASSERT_EQ(workspace.project_index.contributions.lookup(header_id).size(), std::size_t(3)); + + // Re-indexing a TU whose header rows are unchanged must not disturb + // the other TUs' variants either. + tmp.touch("a.cpp", "#include \"shared.h\"\nint a2() { return shared_fn(); }\n"); + auto fresh = index_file(tmp, tmp.path("a.cpp")); + ASSERT_FALSE(fresh.data.empty()); + indexer.merge(fresh.data.data(), fresh.data.size()); + ASSERT_EQ(shard.variants().size(), std::size_t(2)); + ASSERT_EQ(workspace.project_index.contributions.lookup(header_id).size(), std::size_t(3)); +} + +TEST_CASE(HeaderSkipCarriesContribution) { + TempDir tmp; + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 1; }\n"); + tmp.touch("main.cpp", "#include \"dep.h\"\nint use() { return dep(); }\n"); + auto src = tmp.path("main.cpp"); + + auto v1 = index_file(tmp, src); + ASSERT_FALSE(v1.data.empty()); + indexer.merge(v1.data.data(), v1.data.size()); + auto header_id = workspace.path_pool.intern(tmp.path("dep.h")); + auto tu_id = workspace.path_pool.intern(v1.tu_path); + auto old_hash = workspace.project_index.contributions.lookup(header_id).lookup(tu_id); + ASSERT_TRUE(old_hash != 0); + + // The header changes, a reindex captures it — and the header changes + // AGAIN before the result merges. The stale section must not land, but + // the previous contribution keeps serving (its rows still match the + // shard) until a follow-up pass settles. + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 2; }\n"); + auto v2 = index_file(tmp, src); + ASSERT_FALSE(v2.data.empty()); + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 3; }\n"); + + indexer.merge(v2.data.data(), v2.data.size()); + ASSERT_EQ(workspace.project_index.contributions.lookup(header_id).lookup(tu_id), old_hash); + ASSERT_TRUE(workspace.shards[header_id].has_variant(old_hash)); +} + +TEST_CASE(SaveCompactsAndRetires) { + TempDir tmp; + tmp.touch("shared.h", + "#pragma once\n#ifdef MODE\nint mode_fn();\n#endif\n" + "inline int shared_fn() { return 1; }\n"); + tmp.touch("a.cpp", "#include \"shared.h\"\nint a() { return shared_fn(); }\n"); + tmp.touch("b.cpp", "#include \"shared.h\"\nint b() { return shared_fn(); }\n"); + open_store(tmp, workspace); + + auto a = index_file(tmp, tmp.path("a.cpp")); + auto b = index_file(tmp, tmp.path("b.cpp"), {"-DMODE"}); + ASSERT_FALSE(a.data.empty()); + ASSERT_FALSE(b.data.empty()); + indexer.merge(a.data.data(), a.data.size()); + indexer.merge(b.data.data(), b.data.size()); + auto header_id = workspace.path_pool.intern(tmp.path("shared.h")); + ASSERT_EQ(workspace.shards[header_id].variants().size(), std::size_t(2)); + + auto save = [&] { + auto body = [&]() -> kota::task<> { + co_await indexer.save(); + }; + auto task = body(); + loop.schedule(task); + loop.run(); + }; + save(); + + // b stops including the header: its variant dies, and the next save + // erases the dead rows for real. + tmp.touch("b.cpp", "int b() { return 2; }\n"); + auto b2 = index_file(tmp, tmp.path("b.cpp")); + ASSERT_FALSE(b2.data.empty()); + indexer.merge(b2.data.data(), b2.data.size()); + ASSERT_TRUE(workspace.shards[header_id].has_dead_variants()); + save(); + ASSERT_EQ(workspace.shards[header_id].variants().size(), std::size_t(1)); + + // a drops it too: no contribution is left, so the shard retires from + // memory and from storage — with no owner left to re-enqueue. + tmp.touch("a.cpp", "int a() { return 3; }\n"); + auto a2 = index_file(tmp, tmp.path("a.cpp")); + ASSERT_FALSE(a2.data.empty()); + indexer.merge(a2.data.data(), a2.data.size()); + save(); + ASSERT_FALSE(workspace.shards.contains(header_id)); + ASSERT_FALSE(indexer.pending_reason(workspace.path_pool.intern(a2.tu_path)).has_value()); + bool on_disk = false; + auto key = blob_key(workspace.path_pool.resolve(header_id)); + workspace.index_storage->for_each_key(index::IndexBlobKind::Shard, + [&](llvm::StringRef k) { on_disk |= k == key; }); + ASSERT_FALSE(on_disk); +} + +TEST_CASE(SaveRetiresPinnedShard) { + TempDir tmp; + tmp.touch("pinned.h", + "#pragma once\n#ifdef MODE\nint pin_mode();\n#endif\n" + "inline int pin_fn() { return 1; }\n"); + tmp.touch("pa.cpp", "#include \"pinned.h\"\nint pa() { return pin_fn(); }\n"); + tmp.touch("pb.cpp", "#include \"pinned.h\"\nint pb() { return pin_fn(); }\n"); + open_store(tmp, workspace); + + auto a = index_file(tmp, tmp.path("pa.cpp")); + auto b = index_file(tmp, tmp.path("pb.cpp"), {"-DMODE"}); + ASSERT_FALSE(a.data.empty()); + ASSERT_FALSE(b.data.empty()); + indexer.merge(a.data.data(), a.data.size()); + indexer.merge(b.data.data(), b.data.size()); + auto header_id = workspace.path_pool.intern(tmp.path("pinned.h")); + ASSERT_EQ(workspace.shards[header_id].variants().size(), std::size_t(2)); + + // The header moves to a new content generation and only pa catches up: + // the blob starts over with pa's variant, while pb's manifest still + // pins a hash the blob no longer stores. + tmp.touch("pinned.h", + "#pragma once\n#ifdef MODE\nint pin_mode();\n#endif\n" + "inline int pin_fn() { return 2; }\n"); + auto a2 = index_file(tmp, tmp.path("pa.cpp")); + ASSERT_FALSE(a2.data.empty()); + indexer.merge(a2.data.data(), a2.data.size()); + ASSERT_EQ(workspace.shards[header_id].variants().size(), std::size_t(1)); + + // pa's index drops before pb reindexes: every stored variant is dead, + // but pb's pinned hash keeps the live set nonempty. The save must + // retire the shard rather than compact to an empty variant set. + indexer.drop_index(workspace.path_pool.intern(a2.tu_path)); + auto body = [&]() -> kota::task<> { + co_await indexer.save(); + }; + auto task = body(); + loop.schedule(task); + loop.run(); + + ASSERT_FALSE(workspace.shards.contains(header_id)); + bool on_disk = false; + auto key = blob_key(workspace.path_pool.resolve(header_id)); + workspace.index_storage->for_each_key(index::IndexBlobKind::Shard, + [&](llvm::StringRef k) { on_disk |= k == key; }); + ASSERT_FALSE(on_disk); + + // pb's manifest survives, still pinning rows the retirement made + // unservable; nothing else in this process would rebuild them (a + // reverted header even reads fresh by hash), so the retirement must + // re-enqueue pb itself. + ASSERT_TRUE(indexer.pending_reason(workspace.path_pool.intern(b.tu_path)) == + ReindexReason::ContentChanged); +} + +TEST_CASE(RejectsCorruptSection) { + TempDir tmp; + tmp.touch("cor.h", "#pragma once\ninline int cor() { return 1; }\n"); + tmp.touch("cor_main.cpp", "#include \"cor.h\"\nint use_cor() { return cor(); }\n"); + + auto indexed = index_file(tmp, tmp.path("cor_main.cpp")); + ASSERT_FALSE(indexed.data.empty()); + + // Corrupt the main file's nested rows section: the outer wire still + // verifies (sections are opaque bytes to it), only the nested decode + // fails. + auto tampered = index::TUIndex::from(indexed.data); + ASSERT_TRUE(tampered.has_value()); + auto main_id = static_cast(tampered->graph.paths.size() - 1); + for(auto& section: tampered->sections) { + if(section.path_id == main_id) { + section.rows = {0, 1, 2, 3}; + } + } + std::string corrupt; + llvm::raw_string_ostream os(corrupt); + tampered->serialize(os); + + // The header section decodes fine and is staged before the main + // section's decode fails; the reject must discard the whole result — a + // manifest whose recorded versions all match the disk would otherwise + // be judged fresh forever with the main file's rows missing. + indexer.merge(corrupt.data(), corrupt.size()); + auto tu_id = workspace.path_pool.intern(indexed.tu_path); + auto header_id = workspace.path_pool.intern(tmp.path("cor.h")); + ASSERT_FALSE(workspace.project_index.manifests.contains(tu_id)); + ASSERT_FALSE(workspace.shards.contains(header_id)); + // No global trace either: symbol identities from an untrusted result + // would stay canonical for their hashes forever (later merges only + // fill empty names), and stray FileVersions would persist with the + // next save. + ASSERT_TRUE(workspace.project_index.symbols.empty()); + ASSERT_TRUE(workspace.project_index.file_versions.empty()); + + // The intact result still lands afterwards. + indexer.merge(indexed.data.data(), indexed.data.size()); + ASSERT_TRUE(workspace.project_index.manifests.contains(tu_id)); + ASSERT_TRUE(workspace.shards.contains(header_id)); +} + +TEST_CASE(UnverifiedPairingSkipped) { + TempDir tmp; + tmp.touch("unv.cpp", "int unverified_fn() { return 123456; }\n"); + auto src = tmp.path("unv.cpp"); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + + // Strip the consumed-content hashes (a file behind a PCM ships none) + // and shrink the file: the content arbitration cannot see the disk + // moving on, but the rows overrun the shorter content and the fresh + // blob's own range bounds must catch it — a skip, not a blob serving + // ranges past its content. + auto tampered = index::TUIndex::from(indexed.data); + ASSERT_TRUE(tampered.has_value()); + for(auto& hash: tampered->graph.path_hashes) { + hash = 0; + } + std::string wire; + llvm::raw_string_ostream os(wire); + tampered->serialize(os); + tmp.touch("unv.cpp", "int f;\n"); + + indexer.merge(wire.data(), wire.size()); + ASSERT_FALSE(workspace.shards.contains(workspace.path_pool.intern(src))); +} + +TEST_CASE(HashlessRemergeHits) { + TempDir tmp; + tmp.touch("pcm.cpp", "int hashless_fn() { return 7; }\n"); + auto src = tmp.path("pcm.cpp"); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + + // A file behind a PCM ships no consumed-content hash, so the no-IO + // fast path cannot vouch for a stored variant; only the disk read can. + auto tampered = index::TUIndex::from(indexed.data); + ASSERT_TRUE(tampered.has_value()); + for(auto& hash: tampered->graph.path_hashes) { + hash = 0; + } + std::string wire; + llvm::raw_string_ostream os(wire); + tampered->serialize(os); + + indexer.merge(wire.data(), wire.size()); + auto path_id = workspace.path_pool.intern(src); + ASSERT_EQ(workspace.shards[path_id].variants().size(), std::size_t(1)); + + // Re-merging the same rows must register as a hit, not append the + // stored variant to the blob a second time. + indexer.merge(wire.data(), wire.size()); + ASSERT_EQ(workspace.shards[path_id].variants().size(), std::size_t(1)); +} + +TEST_CASE(FailedWriteNotCounted) { + TempDir tmp; + tmp.touch("main.cpp", "int uncommitted() { return 1; }\n"); + auto src = tmp.path("main.cpp"); + + // A storage whose commits never land (disk full, permissions): the + // gauge must report what was durably committed, not what the save + // attempted. + struct FailingStorage final : index::IndexStorage { + std::unique_ptr read(index::IndexBlobKind, llvm::StringRef) override { + return nullptr; + } + + bool contains(index::IndexBlobKind, llvm::StringRef) override { + return false; + } + + llvm::SmallVector write(llvm::ArrayRef batch) override { + llvm::SmallVector failed; + for(std::size_t i = 0; i < batch.size(); i += 1) { + failed.push_back(i); + } + return failed; + } + + void remove(index::IndexBlobKind, llvm::StringRef) override {} + + void for_each_key(index::IndexBlobKind, + llvm::function_ref) override {} + }; + + workspace.index_storage = std::make_unique(); + + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + indexer.merge(indexed.data.data(), indexed.data.size()); + ASSERT_EQ(indexer.pending_shard_writes(), 1u); + + auto save = [&] { + auto body = [&]() -> kota::task<> { + co_await indexer.save(); + }; + auto task = body(); + loop.schedule(task); + loop.run(); + }; + save(); + ASSERT_EQ(indexer.last_save_shards(), 0u); + // The failed batch is re-dirtied rather than discarded, so a later + // save has it to retry and the cache converges once the storage + // recovers. + ASSERT_EQ(indexer.pending_shard_writes(), 1u); + + open_store(tmp, workspace); + save(); + ASSERT_EQ(indexer.last_save_shards(), 1u); + ASSERT_EQ(indexer.pending_shard_writes(), 0u); +} + }; // TEST_SUITE(IndexerMerge) +TEST_SUITE(IndexerStaleness) { + +/// A merged TU with one header dependency, ready for staleness probing. +struct Indexed { + TempDir tmp; + IndexerFixture f; + std::string src; + std::string header; + + bool setup() { + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 1; }\n"); + tmp.touch("main.cpp", "#include \"dep.h\"\nint use() { return dep(); }\n"); + src = tmp.path("main.cpp"); + header = tmp.path("dep.h"); + auto indexed = index_file(tmp, src); + if(indexed.data.empty()) { + return false; + } + f.indexer.merge(indexed.data.data(), indexed.data.size()); + return true; + } +}; + +TEST_CASE(TouchRepairsStamp) { + Indexed x; + ASSERT_TRUE(x.setup()); + ASSERT_FALSE(x.f.need_update(x.src)); + + // Same bytes, new mtime: the stat fast path misses, the hash proves a + // mere touch, and the stamp is repaired in place (dirtying the global + // blob so the repair persists). + ASSERT_TRUE(set_file_mtime(x.header, file_mtime_ns(x.header) + 5'000'000'000)); + x.f.reset_global_dirty(); + x.f.clear_verdicts(); + ASSERT_FALSE(x.f.need_update(x.src)); + ASSERT_TRUE(x.f.global_dirty()); +} + +TEST_CASE(PreservedMtimeEditStale) { + Indexed x; + ASSERT_TRUE(x.setup()); + auto recorded = file_mtime_ns(x.header); + + // Different content restored to the recorded mtime (rsync -t, git + // restore-mtime): equality of the stat is not enough — the size moved, + // and the hash check must catch the edit. + x.tmp.touch("dep.h", "#pragma once\ninline int dep() { return 12345; }\n"); + ASSERT_TRUE(set_file_mtime(x.header, recorded)); + x.f.clear_verdicts(); + ASSERT_TRUE(x.f.need_update(x.src)); +} + +TEST_CASE(AllDepsChecked) { + TempDir tmp; + tmp.touch("first.h", "#pragma once\ninline int first() { return 1; }\n"); + tmp.touch("second.h", "#pragma once\ninline int second() { return 2; }\n"); + tmp.touch("main.cpp", + "#include \"first.h\"\n#include \"second.h\"\n" + "int use() { return first() + second(); }\n"); + IndexerFixture f; + auto indexed = index_file(tmp, tmp.path("main.cpp")); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + ASSERT_FALSE(f.need_update(tmp.path("main.cpp"))); + + // Only the second dependency changes; a partial iteration would call + // the TU fresh. + auto recorded = file_mtime_ns(tmp.path("second.h")); + tmp.touch("second.h", "#pragma once\ninline int second() { return 22222; }\n"); + ASSERT_TRUE(set_file_mtime(tmp.path("second.h"), recorded)); + f.clear_verdicts(); + ASSERT_TRUE(f.need_update(tmp.path("main.cpp"))); +} + +}; // TEST_SUITE(IndexerStaleness) + +TEST_SUITE(IndexerLoad) { + +TEST_CASE(LoadRestoresIndex) { + TempDir tmp; + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 1; }\n"); + tmp.touch("main.cpp", "#include \"dep.h\"\nint use() { return dep(); }\n"); + auto src = tmp.path("main.cpp"); + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + f.save(); + } + + IndexerFixture f; + open_store(tmp, f.workspace); + f.indexer.load(); + + auto tu_id = f.workspace.path_pool.intern(src); + auto header_id = f.workspace.path_pool.intern(tmp.path("dep.h")); + ASSERT_TRUE(f.workspace.shards.contains(tu_id)); + ASSERT_TRUE(f.workspace.shards.contains(header_id)); + ASSERT_TRUE(f.workspace.project_index.contributions.lookup(header_id).contains(tu_id)); + // The persisted FileVersion stamps make the untouched TU judge fresh + // without any reindex. + ASSERT_FALSE(f.need_update(src)); +} + +TEST_CASE(LoadHealsBrokenShard) { + TempDir tmp; + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 1; }\n"); + tmp.touch("extra.h", "#pragma once\ninline int extra() { return 2; }\n"); + tmp.touch("main.cpp", + "#include \"dep.h\"\n#include \"extra.h\"\n" + "int use() { return dep() + extra(); }\n"); + auto src = tmp.path("main.cpp"); + std::string header_key; + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + f.save(); + header_key = blob_key( + f.workspace.path_pool.resolve(f.workspace.path_pool.intern(tmp.path("dep.h")))); + } + + // Corrupt the header's blob and plant an orphan nothing references + // (the store's layout is {root}/cache/v{N}, under the "cache" root). + tmp.touch("cache/cache/v1/index/" + header_key + ".idx", "corrupted beyond verification"); + tmp.touch("cache/cache/v1/index/deadbeefdeadbeef.idx", "orphan"); + + IndexerFixture f; + open_store(tmp, f.workspace); + f.indexer.load(); + + // The header's rows are unservable, so its contributing TU's manifest + // is dropped and the TU re-enqueued — no CDB entry would ever re-index + // a header otherwise. The orphan is swept. + auto tu_id = f.workspace.path_pool.intern(src); + ASSERT_TRUE(f.workspace.project_index.manifests.empty()); + ASSERT_TRUE(f.indexer.pending_reason(tu_id).has_value()); + + // The dropped manifest also retired the TU's contribution to the + // OTHER header: its loaded shard's live mask must follow, or it keeps + // serving a variant nothing contributes any more. + auto extra_id = f.workspace.path_pool.intern(tmp.path("extra.h")); + auto extra_it = f.workspace.shards.find(extra_id); + ASSERT_TRUE(extra_it != f.workspace.shards.end()); + ASSERT_TRUE(extra_it->second.has_dead_variants()); + bool orphan_alive = false; + f.workspace.index_storage->for_each_key(index::IndexBlobKind::Shard, [&](llvm::StringRef key) { + orphan_alive |= key == "deadbeefdeadbeef"; + }); + ASSERT_FALSE(orphan_alive); +} + +TEST_CASE(LoadHealsMissingVariant) { + TempDir tmp; + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 1; }\n"); + tmp.touch("main.cpp", "#include \"dep.h\"\nint use() { return dep(); }\n"); + auto src = tmp.path("main.cpp"); + std::string header_key; + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + f.save(); + header_key = blob_key( + f.workspace.path_pool.resolve(f.workspace.path_pool.intern(tmp.path("dep.h")))); + } + + // Replace the header's blob with one that verifies but stores a variant + // no manifest contributed — the residue of a crash or failed write that + // landed the manifest without its shard. + index::FileIndex no_rows; + index::VariantInput stranger{.hash = 0x1234, .rows = &no_rows}; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard({}, {}, stranger, "", 0, os); + tmp.touch("cache/cache/v1/index/" + header_key + ".idx", bytes); + + IndexerFixture f; + open_store(tmp, f.workspace); + f.indexer.load(); + + // set_live would silently drop the missing rows, so the shard is as + // unservable as an unreadable one: the TU's manifest goes and the TU + // re-enqueues. + auto tu_id = f.workspace.path_pool.intern(src); + ASSERT_TRUE(f.workspace.project_index.manifests.empty()); + ASSERT_TRUE(f.indexer.pending_reason(tu_id).has_value()); +} + +TEST_CASE(LoadHealsWrongGeneration) { + TempDir tmp; + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 1; }\n"); + tmp.touch("main.cpp", "#include \"dep.h\"\nint use() { return dep(); }\n"); + auto src = tmp.path("main.cpp"); + std::string header_key; + std::uint64_t rows_hash = 0; + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + f.save(); + auto header_id = f.workspace.path_pool.intern(tmp.path("dep.h")); + auto tu_id = f.workspace.path_pool.intern(src); + rows_hash = f.workspace.project_index.contributions.lookup(header_id).lookup(tu_id); + ASSERT_TRUE(rows_hash != 0); + header_key = blob_key(f.workspace.path_pool.resolve(header_id)); + } + + // Replace the header's blob with one from ANOTHER content generation + // that still stores the contributed variant — the residue of a failed + // shard write when an edit past every indexed row keeps the rows hash + // identical. Every recorded FileVersion matches the disk, so only the + // generation pin can tell that positions would map through stale text. + index::FileIndex no_rows; + index::VariantInput same_rows{.hash = rows_hash, .rows = &no_rows}; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard({}, {}, same_rows, "stale text", llvm::xxh3_64bits("stale text"), os); + tmp.touch("cache/cache/v1/index/" + header_key + ".idx", bytes); + + IndexerFixture f; + open_store(tmp, f.workspace); + f.indexer.load(); + + auto tu_id = f.workspace.path_pool.intern(src); + ASSERT_TRUE(f.workspace.project_index.manifests.empty()); + ASSERT_TRUE(f.indexer.pending_reason(tu_id).has_value()); +} + +TEST_CASE(LoadDropsNewerManifest) { + TempDir tmp; + tmp.touch("main.cpp", "int lone() { return 1; }\n"); + auto src = tmp.path("main.cpp"); + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + f.save(); + + // Plant what a lost global write leaves behind: a manifest stamped + // with a generation the persisted global never reached. Every + // FileVersion it references is known and its shard variant stored + // (a rows-only reindex), so only the stamp can tell that the + // global's symbols never landed. + auto tu_id = f.workspace.path_pool.intern(src); + auto raced = f.workspace.project_index.manifests.find(tu_id)->second; + raced.global_gen = f.workspace.project_index.global_generation + 1; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_manifest(raced, os); + // Keyed by the interned (canonical) spelling, like save() itself: + // on Windows the raw TempDir spelling hashes to a different key. + f.workspace.index_storage->write({ + {index::IndexBlobKind::Manifest, + blob_key(f.workspace.path_pool.resolve(tu_id)), + std::move(bytes)} + }); + } + + IndexerFixture f; + open_store(tmp, f.workspace); + f.indexer.load(); + + // The raced manifest is dropped and its TU re-enqueued; the reindex + // rewrites the manifest and the global together. + auto tu_id = f.workspace.path_pool.intern(src); + ASSERT_TRUE(f.workspace.project_index.manifests.empty()); + ASSERT_TRUE(f.indexer.pending_reason(tu_id) == ReindexReason::ContentChanged); +} + +TEST_CASE(LoadDropsLostManifest) { + TempDir tmp; + tmp.touch("main.cpp", "int lone() { return 1; }\n"); + auto src = tmp.path("main.cpp"); + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + f.save(); + + // Plant what a failed manifest write under a landed global leaves + // behind: the previous manifest, older-stamped, with every + // FileVersion still resolvable and its shard variant stored (a + // reindex that changed rows or the include tree only). Only the + // global's pin can tell it is not the manifest the save meant. + auto tu_id = f.workspace.path_pool.intern(src); + auto lost = f.workspace.project_index.manifests.find(tu_id)->second; + lost.global_gen = f.workspace.project_index.global_generation - 1; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_manifest(lost, os); + f.workspace.index_storage->write({ + {index::IndexBlobKind::Manifest, + blob_key(f.workspace.path_pool.resolve(tu_id)), + std::move(bytes)} + }); + } + + IndexerFixture f; + open_store(tmp, f.workspace); + f.indexer.load(); + + // The mistamped manifest is dropped and the TU re-enqueued instead of + // the previous reindex's dependency set and rows serving as current. + auto tu_id = f.workspace.path_pool.intern(src); + ASSERT_TRUE(f.workspace.project_index.manifests.empty()); + ASSERT_TRUE(f.indexer.pending_reason(tu_id) == ReindexReason::ContentChanged); +} + +TEST_CASE(LoadRequeuesStaleManifest) { + TempDir tmp; + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 1; }\n"); + tmp.touch("main.cpp", "#include \"dep.h\"\nint use() { return dep(); }\n"); + auto src = tmp.path("main.cpp"); + auto header = tmp.path("dep.h"); + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + f.save(); + + // Plant what a crash between save phases leaves: a manifest whose + // dependency FileVersion the persisted global table never learned, + // while the TU's own version is known — here for the header, whose + // standalone index no CDB sweep would ever rebuild. + auto header_id = f.workspace.path_pool.intern(header); + std::uint32_t header_fv = ~0u; + for(auto& [fv, record]: f.workspace.project_index.file_versions) { + if(record.path_id == header_id) { + header_fv = fv; + } + } + ASSERT_TRUE(header_fv != ~0u); + index::TUManifest stale; + stale.tu_fv = header_fv; + stale.nodes.push_back({.fv = 9999}); + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::serialize_manifest(stale, os); + f.workspace.index_storage->write({ + {index::IndexBlobKind::Manifest, blob_key(header), std::move(bytes)} + }); + } + + IndexerFixture f; + open_store(tmp, f.workspace); + f.indexer.load(); + + // The unresolvable manifest is dropped from storage and its TU + // re-enqueued instead of losing its persisted index forever. + auto header_id = f.workspace.path_pool.intern(header); + ASSERT_TRUE(f.indexer.pending_reason(header_id) == ReindexReason::ContentChanged); + ASSERT_FALSE(f.workspace.project_index.manifests.contains(header_id)); + bool stale_alive = false; + f.workspace.index_storage->for_each_key( + index::IndexBlobKind::Manifest, + [&](llvm::StringRef key) { stale_alive |= key == blob_key(header); }); + ASSERT_FALSE(stale_alive); + + // The TU whose manifest resolved is untouched. + auto tu_id = f.workspace.path_pool.intern(src); + ASSERT_TRUE(f.workspace.project_index.manifests.contains(tu_id)); + ASSERT_FALSE(f.indexer.pending_reason(tu_id).has_value()); +} + +TEST_CASE(UnreadableGlobalPreserved) { + TempDir tmp; + tmp.touch("main.cpp", "int keep() { return 1; }\n"); + auto src = tmp.path("main.cpp"); + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + f.save(); + } + + // The global blob exists but fails to open — a transient IO error at + // startup, not absence. Sweeping would destroy the intact index; a + // fresh lineage saved over the unread one could alias its fv ids and + // generation stamps. The session must run memory-only and leave every + // blob for the next start. + struct UnreadableGlobal final : index::IndexStorage { + std::unique_ptr real; + + std::unique_ptr read(index::IndexBlobKind kind, + llvm::StringRef key) override { + return kind == index::IndexBlobKind::Global ? nullptr : real->read(kind, key); + } + + bool contains(index::IndexBlobKind kind, llvm::StringRef key) override { + return real->contains(kind, key); + } + + llvm::SmallVector write(llvm::ArrayRef batch) override { + return real->write(batch); + } + + void remove(index::IndexBlobKind kind, llvm::StringRef key) override { + real->remove(kind, key); + } + + void for_each_key(index::IndexBlobKind kind, + llvm::function_ref fn) override { + real->for_each_key(kind, fn); + } + }; + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto wrapper = std::make_unique(); + wrapper->real = std::move(f.workspace.index_storage); + f.workspace.index_storage = std::move(wrapper); + f.indexer.load(); + ASSERT_TRUE(f.workspace.project_index.manifests.empty()); + ASSERT_TRUE(f.workspace.index_storage == nullptr); + } + + IndexerFixture f; + open_store(tmp, f.workspace); + f.indexer.load(); + ASSERT_FALSE(f.workspace.project_index.manifests.empty()); + ASSERT_FALSE(f.workspace.shards.empty()); +} + +TEST_CASE(DropIndexEvictsPersisted) { + TempDir tmp; + tmp.touch("dep.h", "#pragma once\ninline int dep() { return 1; }\n"); + tmp.touch("main.cpp", "#include \"dep.h\"\nint use() { return dep(); }\n"); + auto src = tmp.path("main.cpp"); + + { + IndexerFixture f; + open_store(tmp, f.workspace); + auto indexed = index_file(tmp, src); + ASSERT_FALSE(indexed.data.empty()); + f.indexer.merge(indexed.data.data(), indexed.data.size()); + f.save(); + + // The compile command changed: content freshness cannot see it, so + // the TU's index is dropped wholesale and staleness flips at once. + f.indexer.drop_index(f.workspace.path_pool.intern(src)); + ASSERT_TRUE(f.workspace.project_index.manifests.empty()); + ASSERT_TRUE(f.need_update(src)); + f.save(); + } + + // The drop survives a restart: nothing on disk resurrects the + // old-command rows as fresh. + IndexerFixture f; + open_store(tmp, f.workspace); + f.indexer.load(); + ASSERT_TRUE(f.workspace.project_index.manifests.empty()); + ASSERT_TRUE(f.workspace.shards.empty()); + ASSERT_TRUE(f.need_update(src)); +} + +}; // TEST_SUITE(IndexerLoad) + TEST_SUITE(IndexerRequeue) { TEST_CASE(PreemptionKeepsBudget) { diff --git a/tests/unit/server/invalidator_tests.cpp b/tests/unit/server/invalidator_tests.cpp index c46d4642e..73a017241 100644 --- a/tests/unit/server/invalidator_tests.cpp +++ b/tests/unit/server/invalidator_tests.cpp @@ -226,7 +226,7 @@ TEST_CASE(CloseCurrentShardDepsOnly) { Workspace workspace; SessionStore store; auto closed = workspace.path_pool.intern("/proj/a.cpp"); - workspace.merged_indices[closed]; + workspace.shards[closed]; ContextResolver resolver(workspace); // Disk matches the shard's stored content: a browse-and-close must not @@ -244,7 +244,7 @@ TEST_CASE(CloseDivergentShardContentChanged) { Workspace workspace; SessionStore store; auto closed = workspace.path_pool.intern("/proj/a.cpp"); - workspace.merged_indices[closed]; + workspace.shards[closed]; ContextResolver resolver(workspace); // Disk holds edits the shard never saw (saved while open): the shard's @@ -454,6 +454,33 @@ TEST_CASE(RemoveRecreateBatchOrder) { } } +TEST_CASE(EntryChangeThenRemoval) { + TempDir tmp; + tmp.touch("a.cpp", R"(int a;)"); + + Workspace workspace; + SessionStore store; + auto json = build_cdb_json({ + {tmp.root, tmp.path("a.cpp"), {}} + }); + write_cdb(tmp, workspace.cdb, json); + auto file = workspace.path_pool.intern(tmp.path("a.cpp")); + + ContextResolver resolver(workspace); + Invalidator invalidator(workspace, store, resolver); + FileEvent::CDBDelta delta; + delta.changed = {file}; + FileEvent events[] = {FileEvent::cdb_changed(std::move(delta)), FileEvent::disk_removed(file)}; + auto dirty = invalidator.apply(events); + + // The removal is the later fact: the file keeps its last-known index + // serving, so the entry change's drop and enqueue must not survive — a + // surviving drop would mask the shard and let the next save retire it. + ASSERT_TRUE(dirty.drop_index.empty()); + ASSERT_TRUE(dirty.reindex_content_changed.empty()); + ASSERT_EQ(dirty.clear_reindex, llvm::SmallVector{file}); +} + TEST_CASE(CloseOfDeletedFile) { Workspace workspace; SessionStore store; @@ -498,6 +525,7 @@ TEST_CASE(CDBAddedScansAndEnqueues) { // A command change rewrites rows as thoroughly as an edit. ASSERT_EQ(workspace.dep_graph.get_includers(header_id), llvm::ArrayRef{main_id}); ASSERT_EQ(dirty.reindex_content_changed, llvm::SmallVector{main_id}); + ASSERT_EQ(dirty.drop_index, llvm::SmallVector{main_id}); ASSERT_TRUE(dirty.reindex_deps_only.empty()); ASSERT_TRUE(dirty.recheck_contexts); ASSERT_TRUE(dirty.ensure_compile_graph); @@ -518,8 +546,8 @@ TEST_CASE(CDBChangedSplitsOpenClosed) { auto open_id = workspace.path_pool.intern(tmp.path("a.cpp")); auto closed_id = workspace.path_pool.intern(tmp.path("b.cpp")); store.open(open_id); - workspace.merged_indices[open_id]; - workspace.merged_indices[closed_id]; + workspace.shards[open_id]; + workspace.shards[closed_id]; ContextResolver resolver(workspace); Invalidator invalidator(workspace, store, resolver); @@ -538,12 +566,15 @@ TEST_CASE(CDBChangedSplitsOpenClosed) { ASSERT_TRUE(dirty.reindex_deps_only.empty()); ASSERT_TRUE(dirty.recheck_contexts); - // Both shards were built under the old command and look fresh to - // content-only validation: evict them so the queued reindexes are not - // filtered out. The open file's slot is skipped while open-file - // indexing is off; its next compile owns the session-side refresh. - ASSERT_EQ(workspace.merged_indices.count(closed_id), 0u); - ASSERT_EQ(workspace.merged_indices.count(open_id), 0u); + // Both indexes were built under the old command and look fresh to + // content-only validation: drop them so the queued reindexes are not + // filtered out, here or after a restart. The shards themselves stay + // with the indexer, which masks and retires them off the manifests. + auto dropped = dirty.drop_index; + llvm::sort(dropped); + ASSERT_EQ(dropped, reindexed); + ASSERT_EQ(workspace.shards.count(closed_id), 1u); + ASSERT_EQ(workspace.shards.count(open_id), 1u); } TEST_CASE(CDBAddedOpenMarksDirty) { @@ -563,6 +594,7 @@ TEST_CASE(CDBAddedOpenMarksDirty) { // under the real command once open-file indexing is on. ASSERT_EQ(dirty.mark_ast_dirty, llvm::SmallVector{file}); ASSERT_EQ(dirty.reindex_content_changed, llvm::SmallVector{file}); + ASSERT_EQ(dirty.drop_index, llvm::SmallVector{file}); ASSERT_TRUE(dirty.reindex_deps_only.empty()); } @@ -574,7 +606,7 @@ TEST_CASE(CDBChangedDropsHostedContext) { auto closed_header = workspace.path_pool.intern("/proj/closed.h"); auto other_header = workspace.path_pool.intern("/proj/other.h"); store.open(open_header); - workspace.merged_indices[closed_header]; + workspace.shards[closed_header]; ContextResolver resolver(workspace); resolver.header_contexts[open_header].host_path_id = host; @@ -586,14 +618,20 @@ TEST_CASE(CDBChangedDropsHostedContext) { auto dirty = invalidator.apply(FileEvent::cdb_changed(std::move(delta))); // Headers borrowing the changed entry re-resolve their context; the - // open one recompiles, the closed one loses its stale shard and - // reindexes. Unrelated contexts are untouched. + // open one recompiles, the closed one reindexes. Any standalone index + // of theirs borrowed the changed command too, so it is dropped along + // with the host's. Unrelated contexts are untouched. llvm::SmallVector dropped{open_header, closed_header}; llvm::sort(dropped); ASSERT_EQ(dirty.drop_context, dropped); + llvm::SmallVector evicted{host, open_header, closed_header}; + llvm::sort(evicted); + auto drop = dirty.drop_index; + llvm::sort(drop); + ASSERT_EQ(drop, evicted); ASSERT_TRUE(llvm::is_contained(dirty.mark_ast_dirty, open_header)); ASSERT_TRUE(llvm::is_contained(dirty.reindex_content_changed, closed_header)); - ASSERT_EQ(workspace.merged_indices.count(closed_header), 0u); + ASSERT_EQ(workspace.shards.count(closed_header), 1u); } TEST_CASE(CDBChangedCascadesModule) { @@ -634,6 +672,7 @@ TEST_CASE(CDBChangedCascadesModule) { // resolves the overlap to ContentChanged. EXPECT_EQ(dirty.mark_ast_dirty, llvm::SmallVector{open_user}); EXPECT_EQ(dirty.reindex_content_changed, llvm::SmallVector{mod}); + EXPECT_EQ(dirty.drop_index, llvm::SmallVector{mod}); llvm::SmallVector deps{mod, closed_user}; llvm::sort(deps); EXPECT_EQ(dirty.reindex_deps_only, deps); @@ -693,9 +732,11 @@ TEST_CASE(CDBRemovedDropsSourceRole) { delta.removed = {gone_id}; auto dirty = invalidator.apply(FileEvent::cdb_changed(std::move(delta))); - // The rebuild resolves includes from the surviving entries only. + // The rebuild resolves includes from the surviving entries only. A + // removed entry keeps its index — the last-known rows still serve. ASSERT_TRUE(workspace.dep_graph.get_all_includes(gone_id).empty()); ASSERT_EQ(workspace.dep_graph.get_includers(header_id), llvm::ArrayRef{kept_id}); + ASSERT_TRUE(dirty.drop_index.empty()); ASSERT_TRUE(dirty.recheck_contexts); } diff --git a/tests/unit/server/query_freshness_tests.cpp b/tests/unit/server/query_freshness_tests.cpp index 6f7560112..449809808 100644 --- a/tests/unit/server/query_freshness_tests.cpp +++ b/tests/unit/server/query_freshness_tests.cpp @@ -5,6 +5,7 @@ #include "test/test.h" #include "test/tester.h" +#include "index/shard.h" #include "index/tu_index.h" #include "server/compiler/context_resolver.h" #include "server/compiler/indexer.h" @@ -12,7 +13,9 @@ #include "server/state/session_store.h" #include "server/worker/worker_pool.h" +#include "llvm/ADT/SmallVector.h" #include "llvm/Support/Path.h" +#include "llvm/Support/xxhash.h" namespace clice::testing { namespace { @@ -35,38 +38,41 @@ std::uint32_t header_id = 0; /// with real contents, so shards can map their rows to positions. void merge_into_workspace() { auto tu_index = index::TUIndex::build(*unit); - auto file_ids_map = workspace.project_index.merge(tu_index, workspace.path_pool); + std::string wire; + llvm::raw_string_ostream wos(wire); + tu_index.serialize(wos); + auto view = index::TUIndexView::from(wire); + ASSERT_TRUE(view.has_value()); + + llvm::SmallVector file_ids_map; + for(std::uint32_t i = 0; i < view->path_count(); i += 1) { + file_ids_map.push_back(workspace.path_pool.intern(view->path(i))); + } + ASSERT_TRUE(workspace.project_index.merge(*view, file_ids_map)); + main_id = file_ids_map[view->path_count() - 1]; auto content_of = [&](llvm::StringRef path) -> llvm::StringRef { auto it = sources.all_files.find(llvm::sys::path::filename(path)); return it != sources.all_files.end() ? llvm::StringRef(it->second.content) : llvm::StringRef(); }; + auto lookup_symbol = [&](index::SymbolHash hash) { + return view->find_symbol(hash); + }; - auto main_tu_path_id = static_cast(tu_index.graph.paths.size() - 1); - llvm::StringRef main_tu_path = tu_index.graph.paths[main_tu_path_id]; - main_id = file_ids_map[main_tu_path_id]; - - llvm::SmallVector deps; - for(auto& loc: tu_index.graph.locations) { - deps.push_back({tu_index.graph.paths[loc.path_id], loc.line, loc.include}); - } - workspace.merged_indices[main_id].merge(main_tu_path, - tu_index.built_at, - deps, - tu_index.main_file_index, - content_of(main_tu_path)); - - for(auto& [fid, file_idx]: tu_index.file_indices) { - auto tu_pid = tu_index.graph.path_id(fid); - auto global_pid = file_ids_map[tu_pid]; - auto include_id = tu_index.graph.include_location_id(fid); - workspace.merged_indices[global_pid].merge(main_tu_path, - include_id, - file_idx, - content_of(tu_index.graph.paths[tu_pid])); - if(llvm::sys::path::filename(tu_index.graph.paths[tu_pid]) == "header.h") { - header_id = global_pid; + for(std::uint32_t section = 0; section < view->section_count(); section += 1) { + auto local_id = view->section_path(section); + auto rows = view->decode_section_rows(section); + ASSERT_TRUE(rows.has_value()); + auto content = content_of(view->path(local_id)); + index::VariantInput fresh{view->section_rows_hash(section), &*rows, lookup_symbol}; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard(index::Shard(), {}, fresh, content, llvm::xxh3_64bits(content), os); + workspace.shards[file_ids_map[local_id]] = + index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); + if(llvm::sys::path::filename(view->path(local_id)) == "header.h") { + header_id = file_ids_map[local_id]; } } } @@ -74,7 +80,7 @@ void merge_into_workspace() { /// The symbol hash at an offset in a file's merged shard. index::SymbolHash symbol_at(std::uint32_t path_id, std::uint32_t offset) { index::SymbolHash result = 0; - workspace.merged_indices[path_id].lookup(offset, [&](const index::Occurrence& o) { + workspace.shards[path_id].lookup(offset, [&](const index::Occurrence& o) { result = o.target; return false; }); diff --git a/tests/unit/server/query_overlay_tests.cpp b/tests/unit/server/query_overlay_tests.cpp index 22cb9825c..565fdc1ff 100644 --- a/tests/unit/server/query_overlay_tests.cpp +++ b/tests/unit/server/query_overlay_tests.cpp @@ -5,6 +5,7 @@ #include "test/test.h" #include "test/tester.h" #include "index/preamble_state.h" +#include "index/shard.h" #include "server/compiler/context_resolver.h" #include "server/compiler/indexer.h" #include "server/service/query.h" @@ -12,8 +13,10 @@ #include "server/worker/worker_pool.h" #include "kota/ipc/lsp/text.h" +#include "llvm/ADT/SmallVector.h" #include "llvm/Support/Path.h" #include "llvm/Support/raw_ostream.h" +#include "llvm/Support/xxhash.h" namespace clice::testing { namespace { @@ -96,34 +99,38 @@ std::string header_path(llvm::StringRef basename) { /// Merge the full TUIndex into the workspace's disk index with real /// contents, as background indexing would. void merge_disk_index() { - auto file_ids_map = workspace.project_index.merge(full_index, workspace.path_pool); + std::string wire; + llvm::raw_string_ostream wos(wire); + full_index.serialize(wos); + auto view = index::TUIndexView::from(wire); + ASSERT_TRUE(view.has_value()); + + llvm::SmallVector file_ids_map; + for(std::uint32_t i = 0; i < view->path_count(); i += 1) { + file_ids_map.push_back(workspace.path_pool.intern(view->path(i))); + } + ASSERT_TRUE(workspace.project_index.merge(*view, file_ids_map)); auto content_of = [&](llvm::StringRef path) -> llvm::StringRef { auto it = sources.all_files.find(llvm::sys::path::filename(path)); return it != sources.all_files.end() ? llvm::StringRef(it->second.content) : llvm::StringRef(); }; + auto lookup_symbol = [&](index::SymbolHash hash) { + return view->find_symbol(hash); + }; - auto main_tu_path_id = static_cast(full_index.graph.paths.size() - 1); - llvm::StringRef main_tu_path = full_index.graph.paths[main_tu_path_id]; - - llvm::SmallVector deps; - for(auto& loc: full_index.graph.locations) { - deps.push_back({full_index.graph.paths[loc.path_id], loc.line, loc.include}); - } - workspace.merged_indices[file_ids_map[main_tu_path_id]].merge(main_tu_path, - full_index.built_at, - deps, - full_index.main_file_index, - content_of(main_tu_path)); - - for(auto& [fid, file_idx]: full_index.file_indices) { - auto tu_pid = full_index.graph.path_id(fid); - workspace.merged_indices[file_ids_map[tu_pid]].merge( - main_tu_path, - full_index.graph.include_location_id(fid), - file_idx, - content_of(full_index.graph.paths[tu_pid])); + for(std::uint32_t section = 0; section < view->section_count(); section += 1) { + auto local_id = view->section_path(section); + auto rows = view->decode_section_rows(section); + ASSERT_TRUE(rows.has_value()); + auto content = content_of(view->path(local_id)); + index::VariantInput fresh{view->section_rows_hash(section), &*rows, lookup_symbol}; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard(index::Shard(), {}, fresh, content, llvm::xxh3_64bits(content), os); + workspace.shards[file_ids_map[local_id]] = + index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); } } @@ -417,7 +424,12 @@ int main() { §(ref)⟦foo⟧(); return 0; } relation.set_definition_range({0, 3}); fake.relations[foo].push_back(relation); auto header_id = workspace.path_pool.intern(header_path("foo.h")); - workspace.merged_indices[header_id].merge("other_tu", 0, fake, "xxx\n"); + index::VariantInput fresh{.hash = fake.rows_hash(), .rows = &fake}; + std::string bytes; + llvm::raw_string_ostream os(bytes); + index::write_shard(index::Shard(), {}, fresh, "xxx\n", llvm::xxh3_64bits("xxx\n"), os); + workspace.shards[header_id] = + index::Shard::from_buffer(llvm::MemoryBuffer::getMemBufferCopy(bytes)); workspace.project_index.symbols[foo].reference_files.add(header_id); auto def_loc = index_query.find_definition_location(foo); diff --git a/tests/unit/test/temp_dir.h b/tests/unit/test/temp_dir.h index 0dd8f661f..6a439490a 100644 --- a/tests/unit/test/temp_dir.h +++ b/tests/unit/test/temp_dir.h @@ -28,6 +28,15 @@ struct TempDir { TempDir(llvm::StringRef prefix = "clice-test") { llvm::sys::fs::createUniqueDirectory(prefix, root); + // Canonicalize the root so path() spellings match what the compiler + // reports (CompilationUnit::file_path realpaths every file): the + // macOS temp dir lives behind the /var -> /private/var symlink and + // Windows runners hand out 8.3 short names — either would make + // every path-keyed index lookup in tests miss. + llvm::SmallString<128> real; + if(!llvm::sys::fs::real_path(root, real)) { + root = real; + } } ~TempDir() { diff --git a/tools/client/workspace.ts b/tools/client/workspace.ts index b0c5a3366..7edf1864c 100644 --- a/tools/client/workspace.ts +++ b/tools/client/workspace.ts @@ -10,7 +10,7 @@ import { generateCDB } from "../compile_commands.ts"; /// Versioned root of the unified cache store; bump together with /// cache_format_version in src/server/state/workspace.h. -const CACHE_ROOT = path.join(".clice", "cache", "v5"); +const CACHE_ROOT = path.join(".clice", "cache", "v6"); /// The harness-wide canonical URI spelling: percent-decoded. vscode-uri /// encodes the drive colon (file:///c%3A/...) while the server emits it