From 48f19ee771df8b0cc9d3d03c59f5c64375031b48 Mon Sep 17 00:00:00 2001 From: ykiko Date: Wed, 12 Aug 2026 20:25:39 +0800 Subject: [PATCH 01/12] feat(index): automatic flatbuffers serialization via kotatsu codec::fbs, drop flatc Index blobs are now serialized by reflecting the index types directly (kota::codec::fbs::to_bytes) and read back either eagerly (verified from_bytes: TUIndex, ProjectIndex) or through zero-copy views verified once at load (table_view: MergedIndex, PreambleState). schema.fbs, the flatc toolchain and all hand-written builder/GetRoot code are gone. - Relation::kind stores the raw RelationKind::Kind enum so Relation keeps its reflected memcpy image (struct vectors, map keys) - build-time-only fields (FileID-keyed maps) carry skip annotations and never reach the schema - repr specializations map Bitmap to its roaring image, SymbolKind to its underlying value, milliseconds to its count - TUIndex::from/ProjectIndex::from take sized buffers and verify before reading (closes the unverified GetRoot holes) - wire format break: cache_format_version 4 -> 5, index_format_version 1 -> 2, preamble_format_version 3 -> 4; old cache directories are swept and rebuilt --- CMakeLists.txt | 20 +- cmake/package.cmake | 2 +- pixi.lock | 92 ----- pixi.toml | 1 - src/index/include_graph.h | 9 +- src/index/merged_index.cpp | 460 ++++++++------------- src/index/preamble_state.cpp | 338 +++++++-------- src/index/preamble_state.h | 2 +- src/index/project_index.cpp | 131 +++--- src/index/project_index.h | 3 +- src/index/schema.fbs | 295 ------------- src/index/serialization.h | 141 ++++--- src/index/tu_index.cpp | 145 ++----- src/index/tu_index.h | 31 +- src/server/compiler/compiler.cpp | 10 +- src/server/compiler/indexer.cpp | 13 +- src/server/service/query.cpp | 2 +- src/server/state/workspace.h | 2 +- tests/unit/index/merged_index_tests.cpp | 42 +- tests/unit/index/persisted_index_tests.cpp | 9 +- tests/unit/index/preamble_state_tests.cpp | 22 +- tests/unit/index/project_index_tests.cpp | 6 +- tests/unit/index/tu_index_tests.cpp | 76 ++-- 23 files changed, 583 insertions(+), 1269 deletions(-) delete mode 100644 src/index/schema.fbs diff --git a/CMakeLists.txt b/CMakeLists.txt index 02fbb8e2d..7a2f35448 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -152,24 +152,6 @@ if(CLICE_CI_ENVIRONMENT) target_compile_definitions(clice_options INTERFACE CLICE_CI_ENVIRONMENT=1) endif() -set(FBS_SCHEMA_FILE "${PROJECT_SOURCE_DIR}/src/index/schema.fbs") -set(GENERATED_HEADER "${PROJECT_BINARY_DIR}/generated/schema_generated.h") - -# kotatsu builds the flatbuffers runtime but not its schema compiler, so flatc -# comes from the pixi environment (conda-forge flatbuffers). Its version must -# match the headers kotatsu fetches — the generated header static_asserts on -# that, so a mismatch surfaces as a compile error in schema_generated.h. -find_program(FLATC_EXECUTABLE flatc REQUIRED) - -add_custom_command( - OUTPUT "${GENERATED_HEADER}" - COMMAND ${FLATC_EXECUTABLE} --cpp -o "${PROJECT_BINARY_DIR}/generated" "${FBS_SCHEMA_FILE}" - DEPENDS "${FBS_SCHEMA_FILE}" "${FLATC_EXECUTABLE}" - COMMENT "Generating C++ header from ${FBS_SCHEMA_FILE}" -) - -add_custom_target(generate_flatbuffers_schema DEPENDS "${GENERATED_HEADER}") - # Version header, regenerated on every build so the embedded git describe # tracks the checked-out commit (content-unchanged writes are elided). set(CLICE_VERSION_HEADER "${PROJECT_BINARY_DIR}/generated/version.h") @@ -192,7 +174,7 @@ add_custom_target(generate_version_header file(GLOB_RECURSE CLICE_CORE_SOURCES CONFIGURE_DEPENDS "${PROJECT_SOURCE_DIR}/src/*.cpp") add_library(clice-core STATIC ${CLICE_CORE_SOURCES}) add_library(clice::core ALIAS clice-core) -add_dependencies(clice-core generate_flatbuffers_schema generate_version_header) +add_dependencies(clice-core generate_version_header) target_include_directories(clice-core PUBLIC "${PROJECT_SOURCE_DIR}/src" diff --git a/cmake/package.cmake b/cmake/package.cmake index 1f649512a..aeb337a04 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG af2b6d1c2bf19d5b9cd643e7dfa43739bfffb361 + GIT_TAG e38c704c0d43486b019a8a15531a56269b605b7c ) set(KOTA_ENABLE_ZEST ON) diff --git a/pixi.lock b/pixi.lock index 103c2ac7d..a509bc259 100644 --- a/pixi.lock +++ b/pixi.lock @@ -57,7 +57,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt-22.1.8-ha770c72_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt22-22.1.8-hb700be7_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/conda-gcc-specs-15.2.0-h53410ce_20.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-h8ddf172_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-aarch64-15.2.0-he60e42a_20.conda @@ -146,7 +145,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/cmake-4.4.0-hc9d863e_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt-22.1.8-h8af1aa0_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt22-22.1.8-hfefdfc9_1.conda - - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/icu-78.3-hcab7f73_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/keyutils-1.6.3-h86ecc28_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/krb5-1.22.2-h2fb54aa_1.conda @@ -224,7 +222,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/cmake-4.4.0-h2426fb6_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt-22.1.8-h694c41f_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt22-22.1.8-h1637cdf_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/krb5-1.22.2-h3ddfcb2_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/ld64-956.6-llvm22_1_hc399b6d_4.conda @@ -294,7 +291,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/cmake-4.4.0-h8cb302d_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt-22.1.8-hce30654_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt22-22.1.8-hd34ed20_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-78.3-hef89b57_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/krb5-1.22.2-hfd3d5f3_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/ld64-956.6-llvm22_1_h5b97f1b_4.conda @@ -356,7 +352,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/cmake-4.4.0-hdcbee5b_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt-22.1.8-h57928b3_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt22-22.1.8-h49e36cd_1.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/icu-78.3-h637d24d_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/krb5-1.22.2-h719d79b_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libclang13-22.1.8-default_hf735972_3.conda @@ -415,7 +410,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt-22.1.8-ha770c72_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt22-22.1.8-hb700be7_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/conda-gcc-specs-15.2.0-h53410ce_20.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-h8ddf172_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gxx-15.2.0-h834e499_7.conda @@ -494,7 +488,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/cmake-4.3.1-hc9d863e_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt-22.1.8-h8af1aa0_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt22-22.1.8-hfefdfc9_1.conda - - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/keyutils-1.6.3-h86ecc28_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/krb5-1.22.2-hfd895c2_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/ld_impl_linux-aarch64-2.45.1-default_h1979696_102.conda @@ -570,7 +563,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/cmake-4.3.1-h2426fb6_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt-22.1.8-h694c41f_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt22-22.1.8-h1637cdf_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/krb5-1.22.2-h207b36a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/ld64-956.6-llvm22_1_hc399b6d_4.conda @@ -639,7 +631,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/cmake-4.3.1-h8cb302d_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt-22.1.8-hce30654_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt22-22.1.8-hd34ed20_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-78.3-hef89b57_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/krb5-1.22.2-h385eeb1_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/ld64-956.6-llvm22_1_h5b97f1b_4.conda @@ -700,7 +691,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/cmake-4.3.1-hdcbee5b_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt-22.1.8-h57928b3_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt22-22.1.8-h49e36cd_1.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/krb5-1.22.2-h0ea6238_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libclang13-22.1.8-default_hf735972_3.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libcompiler-rt-22.1.8-h49e36cd_1.conda @@ -758,7 +748,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt-22.1.8-ha770c72_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt22-22.1.8-hb700be7_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/conda-gcc-specs-15.2.0-h53410ce_20.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-h8ddf172_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gxx-15.2.0-h834e499_7.conda @@ -843,7 +832,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/cmake-4.3.1-hc9d863e_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt-22.1.8-h8af1aa0_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt22-22.1.8-hfefdfc9_1.conda - - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/icu-78.3-h7ac5ae9_2.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/keyutils-1.6.3-h86ecc28_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/krb5-1.22.2-hfd895c2_0.conda @@ -925,7 +913,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/cmake-4.3.1-h2426fb6_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt-22.1.8-h694c41f_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt22-22.1.8-h1637cdf_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/krb5-1.22.2-h207b36a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/ld64-956.6-llvm22_1_hc399b6d_4.conda @@ -999,7 +986,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/cmake-4.2.1-h54ad630_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt-22.1.8-hce30654_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt22-22.1.8-hd34ed20_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-78.3-hc7cc350_2.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/krb5-1.21.3-h237132a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/ld64-956.6-llvm22_1_h5b97f1b_4.conda @@ -1065,7 +1051,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/cmake-4.2.1-hdcbee5b_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt-22.1.8-h57928b3_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt22-22.1.8-h49e36cd_1.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/icu-78.1-h637d24d_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/krb5-1.21.3-hdf4eb48_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libclang13-22.1.8-default_hf735972_3.conda @@ -1615,7 +1600,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt-22.1.8-ha770c72_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt22-22.1.8-hb700be7_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/conda-gcc-specs-15.2.0-h53410ce_20.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-h8ddf172_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gxx-15.2.0-h834e499_7.conda @@ -1700,7 +1684,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/cmake-4.3.1-hc9d863e_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt-22.1.8-h8af1aa0_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt22-22.1.8-hfefdfc9_1.conda - - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/icu-78.3-h7ac5ae9_2.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/keyutils-1.6.3-h86ecc28_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/krb5-1.22.2-hfd895c2_0.conda @@ -1782,7 +1765,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/cmake-4.3.1-h2426fb6_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt-22.1.8-h694c41f_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt22-22.1.8-h1637cdf_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/krb5-1.22.2-h207b36a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/ld64-956.6-llvm22_1_hc399b6d_4.conda @@ -1856,7 +1838,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/cmake-4.2.1-h54ad630_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt-22.1.8-hce30654_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt22-22.1.8-hd34ed20_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-78.3-hc7cc350_2.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/krb5-1.21.3-h237132a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/ld64-956.6-llvm22_1_h5b97f1b_4.conda @@ -1922,7 +1903,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/cmake-4.2.1-hdcbee5b_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt-22.1.8-h57928b3_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt22-22.1.8-h49e36cd_1.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/krb5-1.21.3-hdf4eb48_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libclang13-22.1.8-default_hf735972_3.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libcompiler-rt-22.1.8-h49e36cd_1.conda @@ -2707,21 +2687,6 @@ packages: license_family: MIT size: 1209421 timestamp: 1757336717570 -- conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - sha256: 0e58114d0e16bc89b94ef9068558e304d2eccae5dbaa55b955274ea60da81dfd - md5: 279ba9719d1afc81538d8260f31e42a0 - depends: - - __glibc >=2.17,<3.0.a0 - - libgcc >=13 - - libstdcxx >=13 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1539958 - timestamp: 1747130572350 - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda sha256: 559aebbc0b6754681c7eec0b6ef8924a13eaa4c110510e99b8224ef3020fed69 md5: 07a50d62feaa7d8656ee442bdc2c0e8d @@ -4856,20 +4821,6 @@ packages: license_family: MIT size: 1113801 timestamp: 1773352768018 -- conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - sha256: 0d802dd9a8b804521a25ee21423a674d73d5ac6cecc2faae4264b5286f9d2deb - md5: 2093f2029d159ec0dc522f42990c0bd2 - depends: - - libgcc >=13 - - libstdcxx >=13 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1380724 - timestamp: 1747130553663 - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/icu-78.3-h7ac5ae9_2.conda sha256: 424a4d54715d11f7fdf033c2ba2fd1b19d6ebd069a15e007ea467b719af197b3 md5: 9a2f5f714c8962bfa551cd9c887b1029 @@ -6765,20 +6716,6 @@ packages: license_family: MIT size: 1133353 timestamp: 1773353161332 -- conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - sha256: eb6be3a3db53cb53f9300f08cfd6579549787e6ec45007d589f4629fec1b9a42 - md5: 109d4025e003f228844a06f246503177 - depends: - - __osx >=10.13 - - libcxx >=18 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1337567 - timestamp: 1747130405020 - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda sha256: 1294117122d55246bb83ad5b589e2a031aacdf2d0b1f99fd338aa4394f881735 md5: 627eca44e62e2b665eeec57a984a7f00 @@ -8182,20 +8119,6 @@ packages: license_family: MIT size: 1050638 timestamp: 1757337263602 -- conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - sha256: d339e7b15c6a927b6ecdb27513d001ab037e3d4bb146fa498e330cbec0cdf9fe - md5: 87c66c4a31165b25b9f56da755197a64 - depends: - - __osx >=11.0 - - libcxx >=18 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1286290 - timestamp: 1747130536643 - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-75.1-hfee45f7_0.conda sha256: 9ba12c93406f3df5ab0a43db8a4b4ef67a5871dfd401010fbe29b218b2cbe620 md5: 5eb22c1d7b3fc4abb50d92d621583137 @@ -9850,21 +9773,6 @@ packages: license_family: MIT size: 1196708 timestamp: 1757337405047 -- conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - sha256: 8c26cca2271d99e8b723847c3a3a7e7de3f5f1908dbd1d2413e6b0b154b97d47 - md5: 29353e2ac55f6192b1a5bb0244021128 - depends: - - ucrt >=10.0.20348.0 - - vc >=14.2,<15 - - vc14_runtime >=14.29.30139 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1753609 - timestamp: 1747130826577 - conda: https://conda.anaconda.org/conda-forge/win-64/icu-78.1-h637d24d_0.conda sha256: bee083d5a0f05c380fcec1f30a71ef5518b23563aeb0a21f6b60b792645f9689 md5: cb8048bed35ef01431184d6a88e46b3e diff --git a/pixi.toml b/pixi.toml index f1b001b20..f20f5107e 100644 --- a/pixi.toml +++ b/pixi.toml @@ -43,7 +43,6 @@ llvm-tools = "==22.1.8" clang-tools = "==22.1.8" compiler-rt = "==22.1.8" # Must match the headers kotatsu fetches — generated code static_asserts on it. -flatbuffers = "==25.2.10" # cmake/archive.cmake pipes tar through xz for the symbol archives. xz = ">=5.8.1,<6" diff --git a/src/index/include_graph.h b/src/index/include_graph.h index 98746888f..f83a75f9e 100644 --- a/src/index/include_graph.h +++ b/src/index/include_graph.h @@ -7,6 +7,8 @@ #include "syntax/token.h" +#include "kota/codec/macro.h" +#include "kota/meta/annotation.h" #include "llvm/ADT/ArrayRef.h" #include "llvm/ADT/DenseMap.h" @@ -51,8 +53,11 @@ struct IncludeGraph { /// Each `FileID` represents a new header context and is introduced /// by a new include directive. So a include directive is a new header - /// context. A map between FileID and its include location. - llvm::DenseMap file_table; + /// context. A map between FileID and its include location. Build-time + /// only: FileIDs mean nothing outside the compilation, so the field is + /// excluded from serialization. + KOTATSU_ANNOTATE(skip = true) + > file_table; /// Build the graph for `unit`. The file table covers the union of /// every file included in this parse (from the replayed directives) diff --git a/src/index/merged_index.cpp b/src/index/merged_index.cpp index f64c07a1e..71a50d797 100644 --- a/src/index/merged_index.cpp +++ b/src/index/merged_index.cpp @@ -1,7 +1,9 @@ #include "index/merged_index.h" #include +#include #include +#include #include #include "compile/dep_file.h" @@ -49,7 +51,7 @@ struct DenseMapInfo { inline static R getEmptyKey() { return R{ - .kind = clice::RelationKind(), + .kind = clice::RelationKind::Invalid, .range = clice::LocalSourceRange(-1, 0), .target_symbol = 0, }; @@ -57,7 +59,7 @@ struct DenseMapInfo { inline static R getTombstoneKey() { return R{ - .kind = clice::RelationKind(), + .kind = clice::RelationKind::Invalid, .range = clice::LocalSourceRange(-2, 0), .target_symbol = 0, }; @@ -65,7 +67,7 @@ struct DenseMapInfo { /// Contextual doesn’t take part in hashing and equality. static auto getHashValue(const R& relation) { - return dense_hash(relation.kind.value(), + return dense_hash(static_cast(relation.kind), relation.range.begin, relation.range.end, relation.target_symbol); @@ -82,7 +84,6 @@ struct DenseMapInfo { namespace clice::index { /// (path_id, content_hash) captured for one dependency at index-build time. -/// Layout must mirror `binary::DepHash` so `safe_cast` can alias it. struct DepHash { std::uint32_t path_id; std::uint64_t content_hash; @@ -92,7 +93,6 @@ struct DepHash { /// Stat fast path for one dependency, recorded at merge time only when the /// file provably did not change since before the indexed build started. -/// Layout must mirror `binary::DepStamp` so `safe_cast` can alias it. /// The server's mutable, in-place-repairing form of the same fast path is /// `DepState` (server/state/workspace.h). struct DepStamp { @@ -121,7 +121,7 @@ std::uint64_t hash_file(llvm::StringRef path) { /// cannot masquerade as fresh; Layer 2 re-hashes the disk against the /// consumed-content hash and treats a match as a mere touch, not an edit. bool dep_stale(llvm::StringRef path, - const DepStamp* stamp, + const std::optional& stamp, std::optional stored_hash) { fs::file_status status; if(auto err = fs::status(path, status)) { @@ -276,6 +276,34 @@ struct MergedIndex::Impl { friend bool operator==(const Impl&, const Impl&) = default; }; +namespace { + +/// The persisted shape of a shard. Serialization reflects an instance of +/// this (built from Impl with the compaction applied); the buffer-backed +/// query paths read it through a zero-copy view. +struct MergedIndexRepr { + std::uint32_t format_version = 0; + std::uint32_t max_canonical_id = 0; + std::vector paths; + std::map canonical_cache; + llvm::SmallDenseMap header_contexts; + llvm::SmallDenseMap compilation_contexts; + llvm::DenseMap occurrences; + llvm::DenseMap> relations; + std::string content; + std::vector line_starts; + SymbolTable symbols; +}; + +using ShardView = kota::codec::fbs::table_view; + +/// The blob was fully verified at load(); per-query views skip that cost. +ShardView root_of(const llvm::MemoryBuffer& buffer) { + return ShardView::from_verified_bytes(blob_bytes(buffer.getBuffer())); +} + +} // namespace + MergedIndex::MergedIndex(std::unique_ptr buffer, std::unique_ptr impl) : buffer(std::move(buffer)), impl(std::move(impl)) {} @@ -311,103 +339,46 @@ void MergedIndex::load_in_memory(this Self& self) { } auto& index = *self.impl; - auto root = fbs::GetRoot(self.buffer->getBufferStart()); + // The buffer was verified at load(); decode straight into the repr and + // move its containers into place. Out-parameter overload: the DenseMap + // members' explicit default constructors make the repr fail the + // std::default_initializable constraint of the value-returning one. + MergedIndexRepr repr; + auto decoded = kota::codec::fbs::from_bytes(blob_bytes(self.buffer->getBuffer()), repr); + assert(decoded.has_value()); - index.max_canonical_id = root->max_canonical_id(); + index.max_canonical_id = repr.max_canonical_id; - if(root->paths()) { - for(auto path: *root->paths()) { - index.paths.path_id(path->string_view()); - } + // Interning in order reproduces the shard-local ids: path_id assigns + // sequentially from zero. + for(auto& path: repr.paths) { + index.paths.path_id(path); } - for(auto entry: *root->canonical_cache()) { - index.canonical_cache.try_emplace(entry->sha256()->string_view(), entry->canonical_id()); + for(auto& [hash, canonical_id]: repr.canonical_cache) { + index.canonical_cache.try_emplace(hash, canonical_id); } index.canonical_ref_counts.resize(index.max_canonical_id, 0); - for(auto entry: *root->header_contexts()) { - HeaderContext context; - auto path = entry->path_id(); - context.version = entry->version(); - for(auto include: *entry->includes()) { - index.canonical_ref_counts[include->canonical_id()] += 1; - context.includes.emplace_back(*safe_cast(include)); + for(auto& [path, context]: repr.header_contexts) { + for(auto& include: context.includes) { + index.canonical_ref_counts[include.canonical_id] += 1; } - index.header_contexts.try_emplace(path, std::move(context)); } - - for(auto entry: *root->compilation_contexts()) { - CompilationContext context; - auto path = entry->path_id(); - context.version = entry->version(); - context.canonical_id = entry->canonical_id(); - context.build_at = entry->build_at(); - for(auto include: *entry->include_locations()) { - context.include_locations.emplace_back(*safe_cast(include)); - } - if(entry->dep_hashes()) { - for(auto dep: *entry->dep_hashes()) { - context.dep_hashes.emplace_back(*safe_cast(dep)); - } - } - // Absent from shards written before the stamp field existed: their - // deps simply have no fast path and validate by hash. - if(entry->dep_stamps()) { - for(auto stamp: *entry->dep_stamps()) { - context.dep_stamps.emplace_back(*safe_cast(stamp)); - } - } - index.compilation_contexts.try_emplace(path, std::move(context)); + for(auto& [path, context]: repr.compilation_contexts) { + index.canonical_ref_counts[context.canonical_id] += 1; } - // Count ref counts from compilation contexts. - for(auto entry: *root->compilation_contexts()) { - index.canonical_ref_counts[entry->canonical_id()] += 1; - } - - // Deserialize removed bitmap. - if(root->removed() && root->removed()->size() > 0) { - index.removed = read_bitmap(root->removed()); - } - - for(auto entry: *root->occurrences()) { - index.occurrences.try_emplace(*safe_cast(entry->occurrence()), - read_bitmap(entry->context())); - } - - for(auto entry: *root->relations()) { - auto& relations = index.relations[entry->symbol()]; - for(auto relation_entry: *entry->relations()) { - relations.try_emplace(*safe_cast(relation_entry->relation()), - read_bitmap(relation_entry->context())); - } - } - - if(root->content()) { - index.content = root->content()->str(); - } - - if(root->line_starts() && root->line_starts()->size() > 0) { - auto* ls = root->line_starts(); - index.line_starts.assign(ls->begin(), ls->end()); - } else if(!index.content.empty()) { - index.line_starts = kota::ipc::lsp::build_line_starts(index.content); - } - - if(root->symbols()) { - for(auto entry: *root->symbols()) { - auto& symbol = index.symbols[entry->symbol_id()]; - if(auto* s = entry->symbol()) { - if(s->name()) - symbol.name = s->name()->str(); - symbol.kind = SymbolKind(static_cast(s->kind())); - symbol.scope = static_cast(s->scope()); - symbol.reference_files = read_bitmap(s->refs()); - } - } - } + index.header_contexts = std::move(repr.header_contexts); + index.compilation_contexts = std::move(repr.compilation_contexts); + // The persisted removed bitmap is always empty (compaction drops masked + // rows before they reach disk), so nothing restores it here. + index.occurrences = std::move(repr.occurrences); + index.relations = std::move(repr.relations); + index.content = std::move(repr.content); + index.line_starts = std::move(repr.line_starts); + index.symbols = std::move(repr.symbols); self.buffer.reset(); } @@ -418,19 +389,14 @@ MergedIndex MergedIndex::load(llvm::StringRef path) { return MergedIndex(); } - // A stale cache directory from an older build must never crash the server - // or be misread. Verify the blob is a structurally valid flatbuffer, then - // discard any shard whose format version differs (including version-less - // shards, which report 0). A discarded shard is treated as "not on disk" - // and the background indexer rebuilds it. - auto data = reinterpret_cast((*buffer)->getBufferStart()); - fbs::Verifier verifier(data, (*buffer)->getBufferSize()); - if(!verifier.VerifyBuffer(nullptr)) { - return MergedIndex(); - } - - auto root = fbs::GetRoot((*buffer)->getBufferStart()); - if(root->format_version() != index_format_version) { + // A stale cache directory from an older build must never crash the + // server or be misread. from_bytes deep-verifies every offset, string, + // vector and table the views can reach; shards whose format version + // differs are discarded (version-less shards read back 0). A discarded + // shard is treated as "not on disk" and the background indexer rebuilds + // it. + auto root = ShardView::from_bytes(blob_bytes((*buffer)->getBuffer())); + if(!root.valid() || root[&MergedIndexRepr::format_version] != index_format_version) { return MergedIndex(); } @@ -449,148 +415,65 @@ void MergedIndex::serialize(this const Self& self, llvm::raw_ostream& out) { auto& index = self.impl; - fbs::FlatBufferBuilder builder(1024); - - llvm::SmallVector buffer; - // Compaction: rows whose every canonical was released are masked at // runtime by the removed bitmap, but the serialized shard is served // through buffer-only lookups that never consult it — so masked state // must not reach disk at all. Dead rows are dropped, live bitmaps are // written pre-subtracted, dead cache entries go with them (a later // re-merge of identical content mints a fresh canonical), and the - // persisted removed bitmap is always empty. + // persisted shape carries no removed bitmap at all. auto& removed = index->removed; auto live = [&](const roaring::Roaring& bitmap) { return removed.isEmpty() ? bitmap : bitmap - removed; }; - llvm::SmallVector> canonical_cache; + MergedIndexRepr repr; + repr.format_version = index_format_version; + repr.max_canonical_id = index->max_canonical_id; + + repr.paths.reserve(index->paths.paths.size()); + for(llvm::StringRef path: index->paths.paths) { + repr.paths.emplace_back(path); + } + for(auto& [hash, canonical_id]: index->canonical_cache) { if(removed.contains(canonical_id)) { continue; } - canonical_cache.push_back( - binary::CreateCacheEntry(builder, CreateString(builder, hash.str()), canonical_id)); + repr.canonical_cache.emplace(hash.str(), canonical_id); } - auto header_contexts = transform(index->header_contexts, [&](auto&& value) { - auto& [path_id, context] = value; - return binary::CreateHeaderContextEntry( - builder, - path_id, - context.version, - CreateStructVector(builder, context.includes)); - }); - - auto compilation_contexts = transform(index->compilation_contexts, [&](auto&& value) { - auto& [path_id, context] = value; - return binary::CreateCompilationContextEntry( - builder, - path_id, - context.version, - context.canonical_id, - context.build_at, - CreateStructVector(builder, context.include_locations), - CreateStructVector(builder, context.dep_hashes), - CreateStructVector(builder, context.dep_stamps)); - }); + repr.header_contexts = index->header_contexts; + repr.compilation_contexts = index->compilation_contexts; - llvm::SmallVector occurrence_keys; - llvm::SmallVector> occurrences; - occurrence_keys.reserve(index->occurrences.size()); - occurrences.reserve(index->occurrences.size()); for(auto& [occurrence, bitmap]: index->occurrences) { auto masked = live(bitmap); if(masked.isEmpty()) { continue; } - buffer.clear(); - buffer.resize_for_overwrite(masked.getSizeInBytes(false)); - masked.write(buffer.data(), false); - occurrence_keys.emplace_back(&occurrence); - occurrences.push_back( - binary::CreateOccurrenceEntry(builder, - safe_cast(&occurrence), - CreateVector(builder, buffer))); + repr.occurrences.try_emplace(occurrence, std::move(masked)); } - std::ranges::sort(std::views::zip(occurrence_keys, occurrences), [](auto lhs, auto rhs) { - const auto& lo = *std::get<0>(lhs); - const auto& ro = *std::get<0>(rhs); - return std::tuple(lo.range.begin, lo.range.end, lo.target) < - std::tuple(ro.range.begin, ro.range.end, ro.target); - }); - llvm::SmallVector relation_keys; - llvm::SmallVector> relations; - relation_keys.reserve(index->relations.size()); - relations.reserve(index->relations.size()); for(auto& [symbol_id, symbol_relations]: index->relations) { - llvm::SmallVector> entries; + llvm::DenseMap entries; for(auto& [relation, bitmap]: symbol_relations) { auto masked = live(bitmap); if(masked.isEmpty()) { continue; } - buffer.clear(); - buffer.resize_for_overwrite(masked.getSizeInBytes(false)); - masked.write(buffer.data(), false); - entries.push_back(binary::CreateRelationEntry(builder, - safe_cast(&relation), - CreateVector(builder, buffer))); + entries.try_emplace(relation, std::move(masked)); } if(entries.empty()) { continue; } - relation_keys.emplace_back(symbol_id); - relations.push_back( - binary::CreateSymbolRelationsEntry(builder, symbol_id, CreateVector(builder, entries))); + repr.relations.try_emplace(symbol_id, std::move(entries)); } - std::ranges::sort(std::views::zip(relation_keys, relations), {}, [](auto e) { - return std::get<0>(e); - }); - // Post-compaction nothing on disk is masked; the persisted removed - // bitmap is always empty. - buffer.clear(); - auto removed_offset = CreateVector(builder, buffer); - - auto content_offset = CreateString(builder, index->content); - auto line_starts_offset = builder.CreateVector(index->line_starts); - - auto symbols = transform(index->symbols, [&](auto&& value) { - auto& [symbol_id, symbol] = value; - buffer.clear(); - buffer.resize_for_overwrite(symbol.reference_files.getSizeInBytes(false)); - symbol.reference_files.write(buffer.data(), false); - return binary::CreateSymbolEntry(builder, - symbol_id, - binary::CreateSymbol(builder, - CreateString(builder, symbol.name), - symbol.kind.value(), - CreateVector(builder, buffer), - static_cast(symbol.scope))); - }); + repr.content = index->content; + repr.line_starts = index->line_starts; + repr.symbols = index->symbols; - auto paths = transform(index->paths.paths, - [&](llvm::StringRef path) { return CreateString(builder, path); }); - - auto merged_index = binary::CreateMergedIndex(builder, - index->max_canonical_id, - CreateVector(builder, paths), - CreateVector(builder, canonical_cache), - CreateVector(builder, header_contexts), - CreateVector(builder, compilation_contexts), - CreateVector(builder, occurrences), - CreateVector(builder, relations), - removed_offset, - content_offset, - line_starts_offset, - CreateVector(builder, symbols), - index_format_version); - builder.Finish(merged_index); - - out.write(safe_cast(builder.GetBufferPointer()), builder.GetSize()); + serialize_blob(repr, out); } void MergedIndex::lookup(this const Self& self, @@ -638,25 +521,29 @@ void MergedIndex::lookup(this const Self& self, break; } } else if(self.buffer) { - auto index = fbs::GetRoot(self.buffer->getBufferStart()); - auto& occurrences = *index->occurrences(); - - auto it = std::ranges::lower_bound(occurrences, offset, {}, [](auto o) { - return o->occurrence()->range().end(); - }); - - while(it != occurrences.end()) { - auto o = safe_cast(it->occurrence()); - if(o->range.contains(offset)) { - if(!callback(*o)) { - break; - } - - it++; - continue; + auto occurrences = root_of(*self.buffer)[&MergedIndexRepr::occurrences]; + + // First entry whose range ends at or past the offset; entries are + // sorted by (range.begin, range.end, target). + std::size_t lo = 0; + std::size_t hi = occurrences.size(); + while(lo < hi) { + auto mid = lo + (hi - lo) / 2; + if(occurrences.at(mid).get<0>().range.end < offset) { + lo = mid + 1; + } else { + hi = mid; } + } - break; + for(; lo < occurrences.size(); ++lo) { + Occurrence occurrence = occurrences.at(lo).get<0>(); + if(!occurrence.range.contains(offset)) { + break; + } + if(!callback(occurrence)) { + break; + } } } } @@ -673,7 +560,7 @@ void MergedIndex::lookup(this const Self& self, auto& relations = it->second; for(auto& [relation, bitmap]: relations) { - if(relation.kind & kind) { + if(RelationKind(relation.kind) & kind) { // Skip relations whose canonical_ids are all removed. if(!self.impl->removed.isEmpty()) { auto remaining = bitmap - self.impl->removed; @@ -688,18 +575,16 @@ void MergedIndex::lookup(this const Self& self, } } } else if(self.buffer) { - auto index = fbs::GetRoot(self.buffer->getBufferStart()); - auto& entries = *index->relations(); - - auto it = std::ranges::lower_bound(entries, symbol, {}, [](auto e) { return e->symbol(); }); - if(it == entries.end() || it->symbol() != symbol) [[unlikely]] { + auto found = root_of(*self.buffer)[&MergedIndexRepr::relations].find(symbol); + if(!found) [[unlikely]] { return; } - for(auto entry: *it->relations()) { - auto r = safe_cast(entry->relation()); - if(r->kind & kind) { - if(!callback(*r)) { + auto entries = found->get<1>(); + for(std::size_t i = 0; i < entries.size(); ++i) { + Relation relation = entries.at(i).get<0>(); + if(RelationKind(relation.kind) & kind) { + if(!callback(relation)) { break; } } @@ -725,9 +610,9 @@ bool MergedIndex::need_update(this const Self& self) { for(auto& dep: context.dep_hashes) { hashes.try_emplace(dep.path_id, dep.content_hash); } - llvm::DenseMap stamps; + llvm::DenseMap stamps; for(auto& stamp: context.dep_stamps) { - stamps.try_emplace(stamp.path_id, &stamp); + stamps.try_emplace(stamp.path_id, stamp); } llvm::DenseSet deps; @@ -742,7 +627,8 @@ bool MergedIndex::need_update(this const Self& self) { auto stamp_it = stamps.find(location.path_id); auto it = hashes.find(location.path_id); if(dep_stale(paths[location.path_id], - stamp_it != stamps.end() ? stamp_it->second : nullptr, + stamp_it != stamps.end() ? std::optional(stamp_it->second) + : std::nullopt, it != hashes.end() ? std::optional(it->second) : std::nullopt)) { return true; } @@ -751,41 +637,46 @@ bool MergedIndex::need_update(this const Self& self) { return false; } else if(self.buffer) { - auto index = fbs::GetRoot(self.buffer->getBufferStart()); - if(index->compilation_contexts()->empty()) { + auto root = root_of(*self.buffer); + auto contexts = root[&MergedIndexRepr::compilation_contexts]; + if(contexts.empty()) { return true; } - auto* paths = index->paths(); + auto paths = root[&MergedIndexRepr::paths]; + + for(std::size_t c = 0; c < contexts.size(); ++c) { + auto context = contexts.at(c).get<1>(); - for(auto context: *index->compilation_contexts()) { llvm::DenseMap hashes; - if(context->dep_hashes()) { - for(auto dep: *context->dep_hashes()) { - hashes.try_emplace(dep->path_id(), dep->content_hash()); - } + auto dep_hashes = context[&CompilationContext::dep_hashes]; + for(std::size_t i = 0; i < dep_hashes.size(); ++i) { + DepHash dep = dep_hashes[i]; + hashes.try_emplace(dep.path_id, dep.content_hash); } - llvm::DenseMap stamps; - if(context->dep_stamps()) { - for(auto stamp: *context->dep_stamps()) { - stamps.try_emplace(stamp->path_id(), stamp); - } + llvm::DenseMap stamps; + auto dep_stamps = context[&CompilationContext::dep_stamps]; + for(std::size_t i = 0; i < dep_stamps.size(); ++i) { + DepStamp stamp = dep_stamps[i]; + stamps.try_emplace(stamp.path_id, stamp); } llvm::DenseSet deps; - for(auto location: *context->include_locations()) { - if(!deps.insert(location->path_id()).second) { + auto locations = context[&CompilationContext::include_locations]; + for(std::size_t i = 0; i < locations.size(); ++i) { + IncludeLocation location = locations[i]; + if(!deps.insert(location.path_id).second) { continue; } // A dep the table does not cover cannot be validated: rebuild. - if(!paths || location->path_id() >= paths->size()) { + if(location.path_id >= paths.size()) { return true; } - auto stamp_it = stamps.find(location->path_id()); - auto it = hashes.find(location->path_id()); - if(dep_stale(paths->Get(location->path_id())->string_view(), - stamp_it != stamps.end() ? safe_cast(stamp_it->second) - : nullptr, + auto stamp_it = stamps.find(location.path_id); + auto it = hashes.find(location.path_id); + if(dep_stale(to_ref(paths[location.path_id]), + stamp_it != stamps.end() ? std::optional(stamp_it->second) + : std::nullopt, it != hashes.end() ? std::optional(it->second) : std::nullopt)) { return true; } @@ -817,13 +708,11 @@ bool MergedIndex::has_contribution(this const Self& self, llvm::StringRef contex } if(self.buffer) { - auto root = fbs::GetRoot(self.buffer->getBufferStart()); - if(!root->paths()) { - return false; - } + auto root = root_of(*self.buffer); + auto paths = root[&MergedIndexRepr::paths]; std::optional local; - for(std::uint32_t i = 0; i < root->paths()->size(); ++i) { - if(llvm::StringRef(root->paths()->Get(i)->string_view()) == context_path) { + for(std::uint32_t i = 0; i < paths.size(); ++i) { + if(to_ref(paths[i]) == context_path) { local = i; break; } @@ -831,16 +720,8 @@ bool MergedIndex::has_contribution(this const Self& self, llvm::StringRef contex if(!local) { return false; } - for(auto entry: *root->header_contexts()) { - if(entry->path_id() == *local) { - return true; - } - } - for(auto entry: *root->compilation_contexts()) { - if(entry->path_id() == *local) { - return true; - } - } + return root[&MergedIndexRepr::header_contexts].contains(*local) || + root[&MergedIndexRepr::compilation_contexts].contains(*local); } return false; @@ -889,18 +770,12 @@ bool MergedIndex::find_symbol(this const Self& self, return true; } } else if(self.buffer) { - auto root = fbs::GetRoot(self.buffer->getBufferStart()); - if(root->symbols()) { - for(auto entry: *root->symbols()) { - if(entry->symbol_id() == hash) { - if(auto* s = entry->symbol()) { - if(s->name()) - name = s->name()->str(); - kind = SymbolKind(static_cast(s->kind())); - } - return true; - } - } + auto found = root_of(*self.buffer)[&MergedIndexRepr::symbols].find(hash); + if(found) { + auto symbol = found->get<1>(); + name = std::string(symbol[&Symbol::name]); + kind = SymbolKind(symbol[&Symbol::kind]); + return true; } } return false; @@ -1022,10 +897,7 @@ llvm::StringRef MergedIndex::content(this const Self& self) { if(self.impl) { return self.impl->content; } else if(self.buffer) { - auto root = fbs::GetRoot(self.buffer->getBufferStart()); - if(root->content()) { - return root->content()->string_view(); - } + return to_ref(root_of(*self.buffer)[&MergedIndexRepr::content]); } return {}; } @@ -1034,10 +906,8 @@ std::span MergedIndex::line_starts(this const Self& self) { if(self.impl) { return self.impl->line_starts; } else if(self.buffer) { - auto root = fbs::GetRoot(self.buffer->getBufferStart()); - if(root->line_starts() && root->line_starts()->size() > 0) { - return {root->line_starts()->data(), root->line_starts()->size()}; - } + auto starts = to_array_ref(root_of(*self.buffer)[&MergedIndexRepr::line_starts]); + return {starts.data(), starts.size()}; } return {}; } diff --git a/src/index/preamble_state.cpp b/src/index/preamble_state.cpp index 81c885394..100c8d413 100644 --- a/src/index/preamble_state.cpp +++ b/src/index/preamble_state.cpp @@ -1,6 +1,6 @@ #include "index/preamble_state.h" -#include +#include #include #include #include @@ -9,42 +9,55 @@ #include "index/serialization.h" #include "kota/ipc/lsp/text.h" -#include "llvm/ADT/SmallVector.h" namespace clice::index { namespace { -/// Serialize one FileIndex into a PreambleFileEntry, with relations sorted -/// by symbol hash so the read path can binary-search them. -fbs::Offset serialize_entry(fbs::FlatBufferBuilder& builder, - std::uint32_t path_id, - const FileIndex& index, - llvm::StringRef content, - llvm::ArrayRef line_starts) { - auto occs = CreateStructVector(builder, index.occurrences); - - llvm::SmallVector*>, 0> sorted; - sorted.reserve(index.relations.size()); - for(auto& [symbol_id, relations]: index.relations) { - sorted.emplace_back(symbol_id, &relations); - } - std::ranges::sort(sorted, {}, [](const auto& entry) { return entry.first; }); - - auto rels = transform(sorted, [&](const auto& entry) { - return binary::CreateTUFileRelationsEntry( - builder, - entry.first, - CreateStructVector(builder, *entry.second)); - }); - - return binary::CreatePreambleFileEntry( - builder, - path_id, - occs, - CreateVector(builder, rels), - content.empty() ? 0 : CreateString(builder, content), - line_starts.empty() ? 0 : CreateVector(builder, line_starts)); +/// One file covered by the preamble compilation: its rows plus content and +/// line starts for position mapping. +struct PreambleFileEntryRepr { + std::uint32_t path_id = 0; + FileIndex index; + std::string content; + std::vector line_starts; +}; + +struct PreambleSymbolRepr { + std::string name; + SymbolKind kind; +}; + +/// The persisted shape of a `.pch.idx` blob. Queries run on a zero-copy +/// view of this layout; nothing is deserialized up front. +struct PreambleStateRepr { + std::uint32_t format_version = 0; + std::vector paths; + std::vector files; + PreambleFileEntryRepr preamble; + llvm::DenseMap symbols; + std::vector links; + std::vector inactive_regions; + std::vector open_conditionals; +}; + +using StateView = kota::codec::fbs::table_view; +using FileEntryView = kota::codec::fbs::table_view; + +/// The blob was fully verified at load(); per-query views skip that cost. +StateView root_of(const llvm::MemoryBuffer& buffer) { + return StateView::from_verified_bytes(blob_bytes(buffer.getBuffer())); +} + +PreambleState::File file_of(StateView root, FileEntryView entry) { + auto paths = root[&PreambleStateRepr::paths]; + auto path_id = entry[&PreambleFileEntryRepr::path_id]; + auto line_starts = to_array_ref(entry[&PreambleFileEntryRepr::line_starts]); + return PreambleState::File{ + .path = to_ref(paths[path_id]), + .content = to_ref(entry[&PreambleFileEntryRepr::content]), + .line_starts = std::span(line_starts.data(), line_starts.size()), + }; } } // namespace @@ -55,13 +68,11 @@ void PreambleState::serialize(CompilationUnitRef unit, llvm::ArrayRef inactive_regions, llvm::ArrayRef open_conditionals, llvm::raw_ostream& os) { - fbs::FlatBufferBuilder builder(4096); - - auto paths = - transform(index.graph.paths, [&](const std::string& p) { return builder.CreateString(p); }); + PreambleStateRepr repr; + repr.format_version = preamble_format_version; + repr.paths = index.graph.paths; - Offsets files; - files.reserve(index.file_indices.size()); + repr.files.reserve(index.file_indices.size()); for(auto& [fid, file_index]: index.file_indices) { // A file with no include edge is a synthetic buffer (predefines, // ): it has no real path to attribute rows to, and @@ -73,10 +84,12 @@ void PreambleState::serialize(CompilationUnitRef unit, continue; } auto content = unit.file_content(fid); - auto line_starts = + auto& entry = repr.files.emplace_back(); + entry.path_id = index.graph.path_id(fid); + entry.index = file_index; + entry.content = content; + entry.line_starts = kota::ipc::lsp::build_line_starts(std::string_view(content.data(), content.size())); - files.push_back( - serialize_entry(builder, index.graph.path_id(fid), file_index, content, line_starts)); } // The source file is the last path in graph.paths (convention from @@ -85,46 +98,21 @@ void PreambleState::serialize(CompilationUnitRef unit, // PCH was built from — stored so consumers can compare it against the // live buffer's prefix before serving these rows. auto preamble_text = unit.interested_content(); - auto preamble_starts = kota::ipc::lsp::build_line_starts( + repr.preamble.path_id = static_cast(index.graph.paths.size() - 1); + repr.preamble.index = index.main_file_index; + repr.preamble.content = preamble_text; + repr.preamble.line_starts = kota::ipc::lsp::build_line_starts( std::string_view(preamble_text.data(), preamble_text.size())); - auto preamble_entry = serialize_entry(builder, - static_cast(index.graph.paths.size() - 1), - index.main_file_index, - preamble_text, - preamble_starts); - - llvm::SmallVector, 0> sorted_symbols; - sorted_symbols.reserve(index.symbols.size()); + for(auto& [symbol_id, symbol]: index.symbols) { - sorted_symbols.emplace_back(symbol_id, &symbol); + repr.symbols[symbol_id] = PreambleSymbolRepr{.name = symbol.name, .kind = symbol.kind}; } - std::ranges::sort(sorted_symbols, {}, [](const auto& entry) { return entry.first; }); - - auto syms = transform(sorted_symbols, [&](const auto& entry) { - return binary::CreatePreambleSymbolEntry(builder, - entry.first, - CreateString(builder, entry.second->name), - entry.second->kind.value()); - }); - - auto link_entries = transform(links, [&](const feature::DocumentLink& link) { - binary::Range range(link.range.begin, link.range.end); - return binary::CreatePreambleDocumentLink(builder, - &range, - CreateString(builder, link.target)); - }); - - auto root = binary::CreatePreambleState(builder, - preamble_format_version, - CreateVector(builder, paths), - CreateVector(builder, files), - preamble_entry, - CreateVector(builder, syms), - CreateVector(builder, link_entries), - CreateVector(builder, inactive_regions), - CreateVector(builder, open_conditionals)); - builder.Finish(root); - os.write(safe_cast(builder.GetBufferPointer()), builder.GetSize()); + + repr.links.assign(links.begin(), links.end()); + repr.inactive_regions.assign(inactive_regions.begin(), inactive_regions.end()); + repr.open_conditionals.assign(open_conditionals.begin(), open_conditionals.end()); + + serialize_blob(repr, os); } PreambleState::PreambleState(std::unique_ptr buffer) : @@ -136,23 +124,13 @@ std::shared_ptr PreambleState::load(llvm::StringRef path) { return nullptr; } - // A stale or truncated blob must never crash the server. Verify it is - // a structurally valid flatbuffer, then discard any blob whose format - // version differs (version-less blobs read back 0). The table budget - // is far above the default: a large preamble's symbol table alone can - // exceed a million entries, and a verification failure here would - // otherwise send every compile into a rebuild loop. - auto data = reinterpret_cast((*buffer)->getBufferStart()); - fbs::Verifier verifier(data, - (*buffer)->getBufferSize(), - /*max_depth=*/64, - /*max_tables=*/1u << 26); - if(!verifier.VerifyBuffer(nullptr)) { - return nullptr; - } - - auto root = fbs::GetRoot((*buffer)->getBufferStart()); - if(root->format_version() != preamble_format_version) { + // A stale or truncated blob must never crash the server. from_bytes + // deep-verifies every offset, string, vector and table the views can + // reach — queries then run unchecked (root_of). Blobs of a different + // format version load as "missing" (version-less blobs read back 0) + // and the PCH pair is rebuilt. + auto root = StateView::from_bytes(blob_bytes((*buffer)->getBuffer())); + if(!root.valid() || root[&PreambleStateRepr::format_version] != preamble_format_version) { return nullptr; } @@ -162,40 +140,30 @@ std::shared_ptr PreambleState::load(llvm::StringRef path) { void PreambleState::lookup(SymbolHash symbol, RelationKind kind, llvm::function_ref callback) const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto paths = root->paths(); - - for(auto entry: *root->files()) { - auto rels = entry->relations(); - if(!rels) { - continue; - } - - auto it = std::ranges::lower_bound(*rels, symbol, {}, [](auto e) { return e->symbol(); }); - if(it == rels->end() || it->symbol() != symbol || !it->relations()) { + auto root = root_of(*buffer); + auto paths = root[&PreambleStateRepr::paths]; + auto files = root[&PreambleStateRepr::files]; + + for(std::size_t i = 0; i < files.size(); ++i) { + auto entry = files[i]; + auto relations = entry[&PreambleFileEntryRepr::index][&FileIndex::relations]; + auto found = relations.find(symbol); + if(!found) { continue; } // The verifier checks structure, not cross-references: a corrupt - // path_id would index out of bounds on the mapped file. - if(entry->path_id() >= paths->size()) { + // path_id must not attribute rows to an arbitrary path. + if(entry[&PreambleFileEntryRepr::path_id] >= paths.size()) { continue; } - auto path = paths->Get(entry->path_id()); - File file{ - .path = llvm::StringRef(path->c_str(), path->size()), - .content = entry->content() - ? llvm::StringRef(entry->content()->c_str(), entry->content()->size()) - : llvm::StringRef(), - .line_starts = entry->line_starts() ? std::span(entry->line_starts()->data(), - entry->line_starts()->size()) - : std::span(), - }; - - for(auto rel: *it->relations()) { - auto r = safe_cast(rel); - if(r->kind & kind) { - if(!callback(file, *r)) { + auto file = file_of(root, entry); + + auto rels = found->get<1>(); + for(std::size_t j = 0; j < rels.size(); ++j) { + Relation relation = rels[j]; + if(RelationKind(relation.kind) & kind) { + if(!callback(file, relation)) { return; } } @@ -204,68 +172,66 @@ void PreambleState::lookup(SymbolHash symbol, } llvm::StringRef PreambleState::source_path() const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto paths = root->paths(); - if(paths->size() == 0) { + auto root = root_of(*buffer); + auto paths = root[&PreambleStateRepr::paths]; + if(paths.empty()) { return {}; } // The source file is the last path, by IncludeGraph convention. - auto path = paths->Get(paths->size() - 1); - return llvm::StringRef(path->c_str(), path->size()); + return to_ref(paths[paths.size() - 1]); } llvm::StringRef PreambleState::preamble_content() const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto preamble = root->preamble(); - if(!preamble || !preamble->content()) { - return {}; - } - return llvm::StringRef(preamble->content()->c_str(), preamble->content()->size()); + auto root = root_of(*buffer); + return to_ref(root[&PreambleStateRepr::preamble][&PreambleFileEntryRepr::content]); } void PreambleState::lookup_preamble(std::uint32_t offset, llvm::function_ref callback) const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto preamble = root->preamble(); - if(!preamble || !preamble->occurrences()) { - return; + auto root = root_of(*buffer); + auto occurrences = + root[&PreambleStateRepr::preamble][&PreambleFileEntryRepr::index][&FileIndex::occurrences]; + + // First occurrence whose range ends at or past the offset; entries are + // sorted by (range.begin, range.end, target). + std::size_t lo = 0; + std::size_t hi = occurrences.size(); + while(lo < hi) { + auto mid = lo + (hi - lo) / 2; + if(occurrences[mid].range.end < offset) { + lo = mid + 1; + } else { + hi = mid; + } } - auto& occurrences = *preamble->occurrences(); - auto it = - std::ranges::lower_bound(occurrences, offset, {}, [](auto o) { return o->range().end(); }); - - while(it != occurrences.end()) { - auto o = safe_cast(*it); - if(!o->range.contains(offset)) { + for(; lo < occurrences.size(); ++lo) { + Occurrence occurrence = occurrences[lo]; + if(!occurrence.range.contains(offset)) { break; } - if(!callback(*o)) { + if(!callback(occurrence)) { break; } - ++it; } } void PreambleState::lookup_preamble(SymbolHash symbol, RelationKind kind, llvm::function_ref callback) const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto preamble = root->preamble(); - if(!preamble || !preamble->relations()) { + auto root = root_of(*buffer); + auto relations = + root[&PreambleStateRepr::preamble][&PreambleFileEntryRepr::index][&FileIndex::relations]; + auto found = relations.find(symbol); + if(!found) { return; } - auto& rels = *preamble->relations(); - auto it = std::ranges::lower_bound(rels, symbol, {}, [](auto e) { return e->symbol(); }); - if(it == rels.end() || it->symbol() != symbol || !it->relations()) { - return; - } - - for(auto rel: *it->relations()) { - auto r = safe_cast(rel); - if(r->kind & kind) { - if(!callback(*r)) { + auto rels = found->get<1>(); + for(std::size_t i = 0; i < rels.size(); ++i) { + Relation relation = rels[i]; + if(RelationKind(relation.kind) & kind) { + if(!callback(relation)) { return; } } @@ -273,54 +239,42 @@ void PreambleState::lookup_preamble(SymbolHash symbol, } bool PreambleState::find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto& syms = *root->symbols(); - - auto it = std::ranges::lower_bound(syms, hash, {}, [](auto e) { return e->symbol_id(); }); - if(it == syms.end() || it->symbol_id() != hash) { + auto root = root_of(*buffer); + auto found = root[&PreambleStateRepr::symbols].find(hash); + if(!found) { return false; } - name = it->name() ? it->name()->str() : std::string(); - kind = SymbolKind(static_cast(it->kind())); + auto symbol = found->get<1>(); + name = std::string(symbol[&PreambleSymbolRepr::name]); + kind = SymbolKind(symbol[&PreambleSymbolRepr::kind]); return true; } std::vector PreambleState::links() const { + auto root = root_of(*buffer); + auto entries = root[&PreambleStateRepr::links]; + std::vector links; - auto root = fbs::GetRoot(buffer->getBufferStart()); - if(auto ls = root->links()) { - links.reserve(ls->size()); - for(auto entry: *ls) { - feature::DocumentLink link; - if(auto range = entry->range()) { - link.range = *safe_cast(range); - } - if(auto target = entry->target()) { - link.target = target->str(); - } - links.push_back(std::move(link)); - } + links.reserve(entries.size()); + for(std::size_t i = 0; i < entries.size(); ++i) { + auto entry = entries[i]; + links.push_back(feature::DocumentLink{ + .range = entry[&feature::DocumentLink::range], + .target = std::string(entry[&feature::DocumentLink::target]), + }); } return links; } llvm::ArrayRef PreambleState::inactive_regions() const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto regions = root->inactive_regions(); - if(!regions) { - return {}; - } - return llvm::ArrayRef(regions->data(), regions->size()); + auto root = root_of(*buffer); + return to_array_ref(root[&PreambleStateRepr::inactive_regions]); } llvm::ArrayRef PreambleState::open_conditionals() const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto conditionals = root->open_conditionals(); - if(!conditionals) { - return {}; - } - return llvm::ArrayRef(conditionals->data(), conditionals->size()); + auto root = root_of(*buffer); + return to_array_ref(root[&PreambleStateRepr::open_conditionals]); } } // namespace clice::index diff --git a/src/index/preamble_state.h b/src/index/preamble_state.h index 7af03c0c4..ceeab2755 100644 --- a/src/index/preamble_state.h +++ b/src/index/preamble_state.h @@ -19,7 +19,7 @@ namespace clice::index { /// carrying a different value loads as "missing" and the PCH pair is /// rebuilt. cache.json records it so a version change is caught at load /// time instead of on the first overlay query. -constexpr inline std::uint32_t preamble_format_version = 3; +constexpr inline std::uint32_t preamble_format_version = 4; /// All master-visible state derived from one PCH build. /// diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index 2857af3b3..3fe245a7d 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -6,6 +6,21 @@ namespace clice::index { +namespace { + +/// The persisted shape of the project blob: the symbol table with pool ids +/// remapped into a compact local path table (only ids the blob references +/// are written — garbage paths are collected here), plus the shard list as +/// local ids. +struct ProjectIndexRepr { + std::uint32_t format_version = 0; + std::vector paths; + SymbolTable symbols; + std::vector shards; +}; + +} // namespace + llvm::SmallVector ProjectIndex::merge(this ProjectIndex& self, TUIndex& index, clice::PathPool& pool) { @@ -37,90 +52,57 @@ void ProjectIndex::serialize(this const ProjectIndex& self, llvm::raw_ostream& os, const clice::PathPool& pool, llvm::ArrayRef shards) { - fbs::FlatBufferBuilder builder(1024); + ProjectIndexRepr repr; + repr.format_version = index_format_version; - // Compact path table: only ids the blob actually references are written, - // in first-seen order. This is where garbage paths get collected — a - // path the pool accumulated but nothing references never reaches disk. llvm::DenseMap local_ids; - std::vector table; auto to_local = [&](std::uint32_t pool_id) -> std::uint32_t { - auto [it, inserted] = local_ids.try_emplace(pool_id, table.size()); + auto [it, inserted] = local_ids.try_emplace(pool_id, repr.paths.size()); if(inserted) { - table.push_back(pool.resolve(pool_id)); + repr.paths.emplace_back(pool.resolve(pool_id)); } return it->second; }; - llvm::SmallVector buffer; llvm::SmallVector remapped; - - auto symbols = transform(self.symbols, [&](auto&& value) { - auto& [symbol_id, symbol] = value; - + for(auto& [symbol_id, symbol]: self.symbols) { remapped.clear(); for(auto ref: symbol.reference_files) { remapped.push_back(to_local(ref)); } - roaring::Roaring local_refs(remapped.size(), remapped.data()); - - buffer.clear(); - buffer.resize_for_overwrite(local_refs.getSizeInBytes(false)); - local_refs.write(buffer.data(), false); - - return binary::CreateSymbolEntry(builder, - symbol_id, - binary::CreateSymbol(builder, - CreateString(builder, symbol.name), - symbol.kind.value(), - CreateVector(builder, buffer), - static_cast(symbol.scope))); - }); - - llvm::SmallVector shard_locals; - shard_locals.reserve(shards.size()); - for(auto shard: shards) { - shard_locals.push_back(to_local(shard)); - } - auto paths = - transform(table, [&](llvm::StringRef path) { return CreateString(builder, path); }); + auto& target = repr.symbols[symbol_id]; + target.name = symbol.name; + target.kind = symbol.kind; + target.scope = symbol.scope; + target.reference_files = Bitmap(remapped.size(), remapped.data()); + } - auto project_index = - binary::CreateProjectIndex(builder, - CreateVector(builder, paths), - CreateVector(builder, symbols), - builder.CreateVector(shard_locals.data(), shard_locals.size()), - index_format_version); + for(auto shard: shards) { + repr.shards.push_back(to_local(shard)); + } - builder.Finish(project_index); - os.write(safe_cast(builder.GetBufferPointer()), builder.GetSize()); + serialize_blob(repr, os); } -std::optional ProjectIndex::from(const void* data, - std::size_t size, +std::optional ProjectIndex::from(llvm::StringRef data, clice::PathPool& pool, llvm::SmallVectorImpl& shards) { - fbs::Verifier verifier(static_cast(data), size); - if(!verifier.VerifyBuffer()) { - return std::nullopt; - } - - auto root = fbs::GetRoot(data); - if(root->format_version() != index_format_version) { + // Out-parameter overload: SymbolTable is an llvm::DenseMap whose explicit + // default constructor makes the repr fail the std::default_initializable + // constraint of the value-returning from_bytes. + ProjectIndexRepr repr; + auto decoded = kota::codec::fbs::from_bytes(blob_bytes(data), repr); + if(!decoded || repr.format_version != index_format_version) { return std::nullopt; } - ProjectIndex loaded; - // Intern the blob's compact path table into the running pool; every id // in the blob is an index into it. llvm::SmallVector pool_ids; - if(root->paths()) { - pool_ids.reserve(root->paths()->size()); - for(auto path: *root->paths()) { - pool_ids.push_back(pool.intern(path->string_view())); - } + pool_ids.reserve(repr.paths.size()); + for(auto& path: repr.paths) { + pool_ids.push_back(pool.intern(path)); } auto to_pool = [&](std::uint32_t local) -> std::optional { @@ -130,31 +112,22 @@ std::optional ProjectIndex::from(const void* data, return pool_ids[local]; }; - if(root->symbols()) { - for(auto entry: *root->symbols()) { - auto* fb_symbol = entry->symbol(); - if(!fb_symbol) { - continue; - } - auto& symbol = loaded.symbols[entry->symbol_id()]; - if(auto* name = fb_symbol->name()) { - symbol.name = name->str(); - } - symbol.kind = SymbolKind(static_cast(fb_symbol->kind())); - symbol.scope = static_cast(fb_symbol->scope()); - for(auto local: read_bitmap(fb_symbol->refs())) { - if(auto id = to_pool(local)) { - symbol.reference_files.add(*id); - } + ProjectIndex loaded; + for(auto& [symbol_id, symbol]: repr.symbols) { + auto& target = loaded.symbols[symbol_id]; + target.name = std::move(symbol.name); + target.kind = symbol.kind; + target.scope = symbol.scope; + for(auto local: symbol.reference_files) { + if(auto id = to_pool(local)) { + target.reference_files.add(*id); } } } - if(root->shards()) { - for(auto local: *root->shards()) { - if(auto id = to_pool(local)) { - shards.push_back(*id); - } + for(auto local: repr.shards) { + if(auto id = to_pool(local)) { + shards.push_back(*id); } } diff --git a/src/index/project_index.h b/src/index/project_index.h index 50f0034a8..97c11eab7 100644 --- a/src/index/project_index.h +++ b/src/index/project_index.h @@ -42,8 +42,7 @@ struct ProjectIndex { /// the loader should fetch. Returns nullopt for an unreadable or /// old-format blob — the caller treats that as "no index on disk" and /// rebuilds in the background. - static std::optional from(const void* data, - std::size_t size, + static std::optional from(llvm::StringRef data, clice::PathPool& pool, llvm::SmallVectorImpl& shards); }; diff --git a/src/index/schema.fbs b/src/index/schema.fbs deleted file mode 100644 index b9534da52..000000000 --- a/src/index/schema.fbs +++ /dev/null @@ -1,295 +0,0 @@ -namespace clice.index.binary; - -struct Range { - begin : uint; - end : uint; -} - -struct Occurrence { - range : Range; - target : ulong; -} - -struct Relation { - kind : uint; - padding : uint; - range : Range; - target_symbol : ulong; -} - -table CacheEntry { -sha256: - string; -canonical_id: - uint; -} - -struct IncludeContext { - include_id : uint; - canonical_id : uint; -} - -table HeaderContextEntry { -path_id: - uint; -version: - uint; -includes: - [IncludeContext]; -} - -struct IncludeLocation { - path_id : uint; - line : uint; - include_id : uint; -} - -struct DepHash { - path_id : uint; - content_hash : ulong; -} - -// Stat fast path for one dependency: recorded only when the file provably -// did not change since before the indexed build started, so an equal stat -// proves the disk still holds the content the rows were built from. -struct DepStamp { - path_id : uint; - size : ulong; - mtime_ns : long; -} - -table CompilationContextEntry { -path_id: - uint; -version: - uint; -canonical_id: - uint; -build_at: - ulong; -include_locations: - [IncludeLocation]; -dep_hashes: - [DepHash]; - -// Added fields read back as absent from older shards — validation then has -// no fast path and goes by dep_hashes alone, which stays correct. That is -// why this addition does not bump index_format_version. -dep_stamps: - [DepStamp]; -} - -table OccurrenceEntry { -occurrence: - Occurrence; -context: - [ubyte]; -} - -table RelationEntry { -relation: - Relation; -context: - [ubyte]; -} - -table SymbolRelationsEntry { -symbol: - ulong; -relations: - [RelationEntry]; -} - -table Symbol { -name: - string; -kind: - ubyte; -refs: - [ubyte] (required); -scope: - ubyte; -} - -table SymbolEntry { -symbol_id: - ulong; -symbol: - Symbol; -} - -table MergedIndex { -max_canonical_id: - uint; - -// Shard-local path table: every path id inside this shard (dependency -// locations, dependency hashes, context keys) indexes into it. Shards are -// fully self-contained — runtime pool ids are per-session and never persist. -paths: - [string] (required); - -canonical_cache: - [CacheEntry] (required); - -header_contexts: - [HeaderContextEntry] (required); - -compilation_contexts: - [CompilationContextEntry] (required); - -occurrences: - [OccurrenceEntry] (required); - -relations: - [SymbolRelationsEntry] (required); - -removed: - [ubyte]; - -content: - string; - -line_starts: - [uint]; - -symbols: - [SymbolEntry]; - -format_version: - uint; -} - -table TUFileRelationsEntry { -symbol: - ulong; -relations: - [Relation]; -} - -table PreambleDocumentLink { -range: - Range; -target: - string; -} - -// One file covered by a preamble compilation. Content and line starts are -// stored (like MergedIndex shards) so the master converts byte offsets to -// LSP positions without touching the filesystem. -table PreambleFileEntry { -path_id: - uint; - -// Sorted by (range.begin, range.end, target). -occurrences: - [Occurrence]; - -// Sorted by symbol hash for binary search. -relations: - [TUFileRelationsEntry]; - -content: - string; - -line_starts: - [uint]; -} - -table PreambleSymbolEntry { -symbol_id: - ulong; -name: - string; -kind: - ubyte; -} - -// Everything the master needs from one PCH build, written by the stateless -// worker next to the PCH blob and opened as a memory-mapped FlatBuffer: -// the preamble's symbol index plus the PCH-derived feature state that the -// master splices into main-file results (document links, inactive regions, -// open conditional stack). -table PreambleState { -format_version: - uint; - -// Blob-local path table; every path_id indexes into it. -paths: - [string] (required); - -// Header entries covered by the preamble. -files: - [PreambleFileEntry] (required); - -// The source file's own preamble region, in buffer offsets. Its content -// is the exact preamble text the PCH was built from — consumers compare -// it against the live buffer's prefix before serving these rows. -preamble: - PreambleFileEntry; - -// Sorted by symbol_id for binary search. -symbols: - [PreambleSymbolEntry] (required); - -links: - [PreambleDocumentLink]; - -// Flat begin/end offset pairs within the preamble region. -inactive_regions: - [uint]; - -// Conditional stack still open at the preamble bound (see -// feature::InactiveScan::open_stack encoding). -open_conditionals: - [ubyte]; -} - -table TUFileIndexEntry { -file_id: - uint; -occurrences: - [Occurrence]; -relations: - [TUFileRelationsEntry]; -} - -table TUIndex { -built_at: - ulong; -paths: - [string]; -locations: - [IncludeLocation]; -symbols: - [SymbolEntry]; -file_indices: - [TUFileIndexEntry]; -main_file_index: - TUFileIndexEntry; - -// Parallel to `paths`: hash of the bytes the indexing compile actually -// consumed per file (0 = unavailable). The master validates its own disk -// reads against these before pairing rows with content, and stores them -// as the shards' staleness baselines. -path_hashes: - [ulong]; -} - -// The blob's path table is compact: only paths referenced by the symbol -// bitmaps or the shard list are written (garbage paths are collected at -// save time), and every id in the blob is an index into it. Runtime pool -// ids are per-session and never persist. -table ProjectIndex { -paths: - [string] (required); -symbols: - [SymbolEntry] (required); - -// Path-table indices of the files owning a MergedIndex shard blob; the -// loader fetches exactly these (keyed by path hash) and sweeps the rest. -shards: - [uint] (required); - -format_version: - uint; -} diff --git a/src/index/serialization.h b/src/index/serialization.h index bf3e34f66..34bb96ac0 100644 --- a/src/index/serialization.h +++ b/src/index/serialization.h @@ -1,85 +1,104 @@ +#pragma once + +#include +#include +#include #include -#include -#include +#include +#include +#include -#include "schema_generated.h" +#include "semantic/symbol.h" #include "support/bitmap.h" -#include "llvm/ADT/SmallVector.h" +#include "kota/codec/fbs/fbs.h" +#include "llvm/ADT/ArrayRef.h" #include "llvm/ADT/StringRef.h" +#include "llvm/Support/raw_ostream.h" -namespace clice::index { +namespace kota::meta { -namespace fbs = flatbuffers; +/// Roaring bitmaps travel as their own serialized image (the non-portable +/// format, matching every in-process reader). +template <> +struct repr { + using type = std::vector; -/// On-disk MergedIndex shard schema version. Bump this whenever `schema.fbs` -/// changes the shard layout so that `MergedIndex::load` silently discards any -/// shard carrying a different value — including version-less shards written by -/// older builds, which read back the field's default of 0. -constexpr inline std::uint32_t index_format_version = 1; + static type to(const clice::Bitmap& bitmap) { + type buffer(bitmap.getSizeInBytes(false)); + bitmap.write(reinterpret_cast(buffer.data()), false); + return buffer; + } -namespace { + static clice::Bitmap from(const type& buffer) { + return clice::Bitmap::read(reinterpret_cast(buffer.data()), false); + } +}; -template -concept sequence_range = std::ranges::input_range && - !requires { typename Range::key_type; } && requires(const Range& r) { - r.data(); - r.size(); - }; +/// SymbolKind hides its enum behind constructors, which keeps it out of +/// reflection; persist the underlying value. +template <> +struct repr { + using type = std::uint8_t; -template -using Offsets = llvm::SmallVector, 0>; - -template -const U* safe_cast(const V* v) { - static_assert(sizeof(U) == sizeof(V), "size mismatch"); - static_assert(alignof(U) == alignof(V), "alignment mismatch"); - static_assert(std::is_trivially_copyable_v && std::is_trivially_copyable_v, - "requires trivially copyable"); - /// If aliasing issues arise, prefer copying into a temporary SmallVector. - return reinterpret_cast(v); -} + static type to(clice::SymbolKind kind) { + return kind.value(); + } -auto CreateString(fbs::FlatBufferBuilder& builder, llvm::StringRef string) { - return builder.CreateString(string.data(), string.size()); -} + static clice::SymbolKind from(type value) { + return clice::SymbolKind(value); + } +}; -template -auto CreateVector(fbs::FlatBufferBuilder& builder, const Range& range) { - return builder.CreateVector(range.data(), range.size()); -} +template <> +struct repr { + using type = std::int64_t; -auto CreateVector(fbs::FlatBufferBuilder& builder, const llvm::SmallVector& range) { - return builder.CreateVector(reinterpret_cast(range.data()), range.size()); -} + static type to(std::chrono::milliseconds ms) { + return ms.count(); + } -template -auto CreateStructVector(fbs::FlatBufferBuilder& builder, const Range& range) { - using V = std::ranges::range_value_t; - (void)sizeof(V); - return builder.CreateVectorOfStructs(safe_cast(range.data()), range.size()); -} + static std::chrono::milliseconds from(type count) { + return std::chrono::milliseconds(count); + } +}; + +} // namespace kota::meta -template -auto transform(const Range& range, const Functor& functor) { - using V = std::ranges::range_value_t; - using R = std::invoke_result_t; +namespace clice::index { - llvm::SmallVector result; - result.resize_for_overwrite(std::ranges::size(range)); +/// On-disk index blob schema version. Every persisted blob carries it as a +/// regular field and every loader discards blobs with a different value — +/// including version-less blobs from older builds, which read back as 0. +/// Bump it whenever a persisted type's reflected layout changes. +constexpr inline std::uint32_t index_format_version = 2; - auto i = 0; - for(auto&& v: range) { - result[i] = functor(v); - i += 1; - } - return result; +/// Serialize a reflected index blob to `os` as a verified-readable +/// flatbuffer. Encoding only fails on structural impossibilities (e.g. more +/// fields than slots), which the persisted index types cannot hit. +template +void serialize_blob(const T& value, llvm::raw_ostream& os) { + auto encoded = kota::codec::fbs::to_bytes(value); + assert(encoded.has_value()); + os.write(reinterpret_cast(encoded->data()), encoded->size()); } -Bitmap read_bitmap(const fbs::Vector* buffer) { - return Bitmap::read(reinterpret_cast(buffer->data()), false); +/// The bytes of `data` as the span every kota fbs entry point takes. +inline std::span blob_bytes(llvm::StringRef data) { + return {reinterpret_cast(data.data()), data.size()}; } -} // namespace +inline llvm::StringRef to_ref(std::string_view text) { + return {text.data(), text.size()}; +} + +/// A scalar array view as an ArrayRef borrowing the mapped blob. +template +llvm::ArrayRef to_array_ref(kota::codec::fbs::array_view view) { + if(!view.valid()) { + return {}; + } + return {view.raw()->data(), view.raw()->size()}; +} } // namespace clice::index diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index 4b03c9f3d..9df91dabb 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -530,13 +530,14 @@ class Projector { for(auto& [fid, index]: result.file_indices) { for(auto& [symbol_id, relations]: index.relations) { std::ranges::sort(relations, [](const Relation& lhs, const Relation& rhs) { - return std::tuple(lhs.kind.value(), + return std::tuple(static_cast(lhs.kind), lhs.range.begin, lhs.range.end, - lhs.target_symbol) < std::tuple(rhs.kind.value(), - rhs.range.begin, - rhs.range.end, - rhs.target_symbol); + lhs.target_symbol) < + std::tuple(static_cast(rhs.kind), + rhs.range.begin, + rhs.range.end, + rhs.target_symbol); }); auto range = std::ranges::unique(relations, [](const Relation& lhs, const Relation& rhs) { @@ -594,7 +595,7 @@ void FileIndex::lookup(SymbolHash symbol, if(it == relations.end()) return; for(auto& r: it->second) { - if(r.kind & kind) { + if(RelationKind(r.kind) & kind) { if(!callback(r)) return; } @@ -640,127 +641,27 @@ TUIndex TUIndex::build(CompilationUnitRef unit, bool interested_only) { return index; } -void TUIndex::serialize(llvm::raw_ostream& os) const { - fbs::FlatBufferBuilder builder(4096); - - llvm::SmallVector buffer; - - auto paths = - transform(graph.paths, [&](const std::string& p) { return builder.CreateString(p); }); - - auto syms = transform(symbols, [&](auto&& value) { - auto& [symbol_id, symbol] = value; - buffer.clear(); - buffer.resize_for_overwrite(symbol.reference_files.getSizeInBytes(false)); - symbol.reference_files.write(buffer.data(), false); - return binary::CreateSymbolEntry(builder, - symbol_id, - binary::CreateSymbol(builder, - CreateString(builder, symbol.name), - symbol.kind.value(), - CreateVector(builder, buffer), - static_cast(symbol.scope))); - }); - - /// Serialize a single FileIndex into a TUFileIndexEntry. - auto serialize_file_index = [&](std::uint32_t fid, const FileIndex& index) { - auto occs = CreateStructVector(builder, index.occurrences); - auto rels = transform(index.relations, [&](auto&& value) { - auto& [symbol_id, relations] = value; - return binary::CreateTUFileRelationsEntry( - builder, - symbol_id, - CreateStructVector(builder, relations)); - }); - return binary::CreateTUFileIndexEntry(builder, fid, occs, CreateVector(builder, rels)); - }; - - /// Convert FileID-keyed file_indices to path_id-keyed entries. - llvm::SmallVector> file_idx_vec; - for(auto& [fid, index]: file_indices) { - auto pid = graph.path_id(fid); - file_idx_vec.push_back(serialize_file_index(pid, index)); +void TUIndex::serialize(llvm::raw_ostream& os) { + /// Convert the FileID-keyed working state into the persisted + /// path_id-keyed form. Multiple FileIDs can share a path id (repeated + /// header contexts); last-wins matches what deserialization of the old + /// per-FileID entries did. + path_file_indices.clear(); + for(auto& [fid, file_index]: file_indices) { + path_file_indices[graph.path_id(fid)] = file_index; } - /// Main file is the last path in graph.paths (convention from IncludeGraph). - auto main_idx = - serialize_file_index(static_cast(graph.paths.size() - 1), main_file_index); - - auto tu_index = - binary::CreateTUIndex(builder, - static_cast(built_at.count()), - CreateVector(builder, paths), - CreateStructVector(builder, graph.locations), - CreateVector(builder, syms), - builder.CreateVector(file_idx_vec.data(), file_idx_vec.size()), - main_idx, - CreateVector(builder, graph.path_hashes)); - - builder.Finish(tu_index); - os.write(safe_cast(builder.GetBufferPointer()), builder.GetSize()); + serialize_blob(*this, os); } -TUIndex TUIndex::from(const void* data) { - auto root = fbs::GetRoot(data); - - TUIndex index; - index.built_at = std::chrono::milliseconds(root->built_at()); - - for(auto p: *root->paths()) { - index.graph.paths.emplace_back(p->str()); - } - - for(auto loc: *root->locations()) { - index.graph.locations.emplace_back(*safe_cast(loc)); - } - - if(root->path_hashes()) { - index.graph.path_hashes.assign(root->path_hashes()->begin(), root->path_hashes()->end()); +std::optional TUIndex::from(llvm::StringRef data) { + // The out-parameter overload: TUIndex holds llvm::DenseMap members whose + // explicit default constructors make it fail std::default_initializable, + // which the value-returning from_bytes requires. + std::optional index{std::in_place}; + if(!kota::codec::fbs::from_bytes(blob_bytes(data), *index)) { + return std::nullopt; } - index.graph.path_hashes.resize(index.graph.paths.size(), 0); - - for(auto entry: *root->symbols()) { - auto& symbol = index.symbols[entry->symbol_id()]; - symbol.name = entry->symbol()->name()->str(); - symbol.kind = SymbolKind(static_cast(entry->symbol()->kind())); - symbol.scope = static_cast(entry->symbol()->scope()); - symbol.reference_files = read_bitmap(entry->symbol()->refs()); - } - - /// Helper to deserialize a TUFileIndexEntry into a FileIndex. - auto deserialize_file_index = [](const binary::TUFileIndexEntry* entry) -> FileIndex { - FileIndex fi; - if(entry->occurrences()) { - fi.occurrences.reserve(entry->occurrences()->size()); - for(auto o: *entry->occurrences()) { - fi.occurrences.emplace_back(*safe_cast(o)); - } - } - if(entry->relations()) { - for(auto rel_entry: *entry->relations()) { - auto& rels = fi.relations[rel_entry->symbol()]; - if(rel_entry->relations()) { - rels.reserve(rel_entry->relations()->size()); - for(auto r: *rel_entry->relations()) { - rels.emplace_back(*safe_cast(r)); - } - } - } - } - return fi; - }; - - /// Populate path_file_indices keyed by path_id (no clang::FileID needed). - if(root->file_indices()) { - for(auto entry: *root->file_indices()) { - index.path_file_indices[entry->file_id()] = deserialize_file_index(entry); - } - } - - if(root->main_file_index()) { - index.main_file_index = deserialize_file_index(root->main_file_index()); - } - return index; } diff --git a/src/index/tu_index.h b/src/index/tu_index.h index fc91aac93..50e3d0836 100644 --- a/src/index/tu_index.h +++ b/src/index/tu_index.h @@ -4,6 +4,7 @@ #include #include #include +#include #include #include @@ -11,6 +12,8 @@ #include "semantic/symbol.h" #include "support/bitmap.h" +#include "kota/codec/macro.h" +#include "kota/meta/annotation.h" #include "llvm/ADT/STLFunctionalExtras.h" #include "llvm/Support/raw_ostream.h" @@ -34,7 +37,10 @@ enum class SymbolScope : std::uint8_t { }; struct Relation { - RelationKind kind; + /// The raw enum rather than the RelationKind wrapper: the wrapper's + /// constructors hide it from reflection, and reflection is what lets a + /// relation vector persist as one contiguous struct vector. + RelationKind::Kind kind = RelationKind::Invalid; std::uint32_t padding = 0; @@ -62,7 +68,11 @@ struct Occurrence { }; struct FileIndex { - llvm::DenseMap> relations; + /// The braces matter: fbs decode value-constructs map entries with + /// `FileIndex{}`, and without an initializer this member would be + /// copy-initialized from an empty list, which DenseMap's explicit + /// default constructor rejects. + llvm::DenseMap> relations{}; std::vector occurrences; @@ -99,10 +109,14 @@ struct TUIndex { SymbolTable symbols; - llvm::DenseMap file_indices; + /// Build-time working state keyed by FileID — clang::FileID means nothing + /// outside the compilation, so it never persists; serialize() converts it + /// through graph.path_id. + KOTATSU_ANNOTATE(skip = true) + > file_indices; - /// File indices keyed by path_id, populated by from() for deserialized data. - /// When built from AST, this is empty and file_indices (keyed by FileID) is used. + /// File indices keyed by path_id: populated from file_indices by + /// serialize(), and directly by from() for deserialized data. llvm::DenseMap path_file_indices; FileIndex main_file_index; @@ -115,9 +129,12 @@ struct TUIndex { /// under the main path id. static TUIndex build(CompilationUnitRef unit, bool interested_only = false); - void serialize(llvm::raw_ostream& os) const; + /// Serialization reflects this object directly (path_file_indices is + /// populated from file_indices first — hence non-const). + void serialize(llvm::raw_ostream& os); - static TUIndex from(const void* data); + /// Deserialize a verified buffer; nullopt when verification fails. + static std::optional from(llvm::StringRef data); }; } // namespace clice::index diff --git a/src/server/compiler/compiler.cpp b/src/server/compiler/compiler.cpp index af8ea3a47..b93d4af98 100644 --- a/src/server/compiler/compiler.cpp +++ b/src/server/compiler/compiler.cpp @@ -1189,10 +1189,12 @@ kota::task<> Compiler::run_compile(std::shared_ptr session) { pc->succeeded = true; record_deps(*session, result.value().deps, result.value().build_at); - if(!result.value().tu_index_data.empty()) { - auto tu_index = index::TUIndex::from(result.value().tu_index_data.data()); - session->file_index = std::move(tu_index.main_file_index); - session->symbols = std::move(tu_index.symbols); + auto tu_index = result.value().tu_index_data.empty() + ? std::nullopt + : index::TUIndex::from(result.value().tu_index_data); + if(tu_index) { + session->file_index = std::move(tu_index->main_file_index); + session->symbols = std::move(tu_index->symbols); } else { // The AST and the file index settle together — that pairing is // what lets navigation trust the index after ensure_compiled. A diff --git a/src/server/compiler/indexer.cpp b/src/server/compiler/indexer.cpp index 4d90d8226..d9642b3dd 100644 --- a/src/server/compiler/indexer.cpp +++ b/src/server/compiler/indexer.cpp @@ -27,7 +27,13 @@ namespace clice { void Indexer::merge(const void* tu_index_data, std::size_t size) { - auto tu_index = index::TUIndex::from(tu_index_data); + auto loaded = + index::TUIndex::from(llvm::StringRef(static_cast(tu_index_data), size)); + if(!loaded) { + LOG_WARN("Ignoring TUIndex that failed verification"); + return; + } + auto& tu_index = *loaded; if(tu_index.graph.paths.empty()) { LOG_WARN("Ignoring TUIndex with empty path graph"); return; @@ -345,10 +351,7 @@ void Indexer::load() { // An unreadable or old-format blob loads as "no index on disk": // everything is swept and rebuilt once in the background. llvm::SmallVector manifest; - auto loaded = index::ProjectIndex::from((*buf)->getBufferStart(), - (*buf)->getBufferSize(), - workspace.path_pool, - manifest); + auto loaded = index::ProjectIndex::from((*buf)->getBuffer(), workspace.path_pool, manifest); if(loaded) { workspace.project_index = std::move(*loaded); has_project = true; diff --git a/src/server/service/query.cpp b/src/server/service/query.cpp index a72cea1f4..492a301af 100644 --- a/src/server/service/query.cpp +++ b/src/server/service/query.cpp @@ -605,7 +605,7 @@ void IndexQuery::collect_unique_targets(index::SymbolHash hash, if(rel_it == session.file_index->relations.end()) return true; for(auto& r: rel_it->second) { - if(r.kind & kind) { + if(RelationKind(r.kind) & kind) { if(seen.insert(r.target_symbol).second) { targets.push_back(r.target_symbol); } diff --git a/src/server/state/workspace.h b/src/server/state/workspace.h index 2278189bd..035a32a89 100644 --- a/src/server/state/workspace.h +++ b/src/server/state/workspace.h @@ -35,7 +35,7 @@ class ContextResolver; /// On-disk cache layout version (CacheStore root `cache/v{N}`). /// Bump to discard all cached artifacts after incompatible format changes. -constexpr inline std::uint32_t cache_format_version = 4; +constexpr inline std::uint32_t cache_format_version = 5; /// Sentinel for "no path": path pool ids start at 0, so 0 is a real file. constexpr inline std::uint32_t no_path_id = ~0u; diff --git a/tests/unit/index/merged_index_tests.cpp b/tests/unit/index/merged_index_tests.cpp index a97a30581..7bd70a417 100644 --- a/tests/unit/index/merged_index_tests.cpp +++ b/tests/unit/index/merged_index_tests.cpp @@ -1,10 +1,10 @@ #include -#include "schema_generated.h" #include "test/temp_dir.h" #include "test/test.h" #include "test/tester.h" #include "index/merged_index.h" +#include "index/serialization.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Support/xxhash.h" @@ -321,7 +321,7 @@ TEST_CASE(RemergeReplacesContribution) { index::SymbolHash defined{}; for(auto& [symbol, relations]: header_idx.relations) { for(auto& relation: relations) { - if(relation.kind & RelationKind(RelationKind::Definition)) { + if(RelationKind(relation.kind) & RelationKind(RelationKind::Definition)) { defined = symbol; } } @@ -369,7 +369,7 @@ TEST_CASE(RemergePreservesOtherTus) { index::SymbolHash defined{}; for(auto& [symbol, relations]: header_idx.relations) { for(auto& relation: relations) { - if(relation.kind & RelationKind(RelationKind::Definition)) { + if(RelationKind(relation.kind) & RelationKind(RelationKind::Definition)) { defined = symbol; } } @@ -673,37 +673,21 @@ TEST_CASE(OldShardDiscarded) { // A version-less (format_version=0) shard from an older build is silently // discarded — load returns an empty index, as if nothing were on disk. + // Only the version slot is written: every other field reads back absent, + // which is structurally valid — rejection must come from the version + // check. { - namespace binary = clice::index::binary; - flatbuffers::FlatBufferBuilder builder; - auto content = builder.CreateString("stale-shard"); - auto paths = builder.CreateVector>({}); - auto cache = builder.CreateVector>({}); - auto headers = builder.CreateVector>({}); - auto compilations = - builder.CreateVector>({}); - auto occurrences = builder.CreateVector>({}); - auto relations = - builder.CreateVector>({}); - auto root = binary::CreateMergedIndex(builder, - 0, - paths, - cache, - headers, - compilations, - occurrences, - relations, - 0, - content, - 0, - 0, - /*format_version=*/0); - builder.Finish(root); + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto blob = kota::codec::fbs::to_bytes(VersionOnly{}); + ASSERT_TRUE(blob.has_value()); auto path = dir.path("stale.idx"); std::error_code ec; llvm::raw_fd_ostream os(path, ec); - os.write(reinterpret_cast(builder.GetBufferPointer()), builder.GetSize()); + os.write(reinterpret_cast(blob->data()), blob->size()); os.flush(); auto loaded = index::MergedIndex::load(path); diff --git a/tests/unit/index/persisted_index_tests.cpp b/tests/unit/index/persisted_index_tests.cpp index 867d6b820..fa07f8520 100644 --- a/tests/unit/index/persisted_index_tests.cpp +++ b/tests/unit/index/persisted_index_tests.cpp @@ -29,7 +29,7 @@ TEST_CASE(SerializeCollectsGarbage) { clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); ASSERT_TRUE(fresh.find("/proj/used.cpp").has_value()); ASSERT_FALSE(fresh.find("/proj/garbage.cpp").has_value()); @@ -48,7 +48,7 @@ TEST_CASE(RemapAcrossSessions) { clice::PathPool fresh; fresh.intern("/proj/opened-first.cpp"); llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); auto id = fresh.find("/proj/used.cpp"); @@ -67,7 +67,7 @@ TEST_CASE(ShardManifestRoundTrip) { clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); ASSERT_EQ(shards.size(), 1u); ASSERT_EQ(fresh.resolve(shards.front()), "/proj/tu.cpp"); @@ -78,7 +78,8 @@ TEST_CASE(OldBlobDiscarded) { llvm::SmallVector shards; // Arbitrary bytes are rejected by verification, not misread. const char junk[] = "not a flatbuffer"; - ASSERT_FALSE(index::ProjectIndex::from(junk, sizeof(junk), pool, shards).has_value()); + ASSERT_FALSE( + index::ProjectIndex::from(llvm::StringRef(junk, sizeof(junk)), pool, shards).has_value()); } }; // TEST_SUITE(PersistedIndex) diff --git a/tests/unit/index/preamble_state_tests.cpp b/tests/unit/index/preamble_state_tests.cpp index fa167c0b3..276459da9 100644 --- a/tests/unit/index/preamble_state_tests.cpp +++ b/tests/unit/index/preamble_state_tests.cpp @@ -1,8 +1,8 @@ -#include "schema_generated.h" #include "test/temp_dir.h" #include "test/test.h" #include "test/tester.h" #include "index/preamble_state.h" +#include "index/serialization.h" #include "llvm/Support/raw_ostream.h" @@ -225,19 +225,19 @@ TEST_CASE(RejectBadBlob) { TEST_CASE(RejectVersionMismatch) { // A structurally valid blob written by a different format version (0 is // what a version-less blob reads back) must load as missing, so the - // PCH pair rebuilds instead of serving a stale layout. - flatbuffers::FlatBufferBuilder builder(64); - auto paths = builder.CreateVector(std::vector>{}); - auto files = - builder.CreateVector(std::vector>{}); - auto symbols = builder.CreateVector( - std::vector>{}); - builder.Finish(index::binary::CreatePreambleState(builder, 0, paths, files, 0, symbols)); + // PCH pair rebuilds instead of serving a stale layout. The blob only + // needs the version slot: every other field reads back absent, which is + // structurally valid — rejection must come from the version check. + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto blob = kota::codec::fbs::to_bytes(VersionOnly{}); + ASSERT_TRUE(blob.has_value()); auto blob_path = dir.path("stale.pch.idx"); dir.touch("stale.pch.idx", - llvm::StringRef(reinterpret_cast(builder.GetBufferPointer()), - builder.GetSize())); + llvm::StringRef(reinterpret_cast(blob->data()), blob->size())); EXPECT_TRUE(index::PreambleState::load(blob_path) == nullptr); } diff --git a/tests/unit/index/project_index_tests.cpp b/tests/unit/index/project_index_tests.cpp index d12951e56..3adca5f21 100644 --- a/tests/unit/index/project_index_tests.cpp +++ b/tests/unit/index/project_index_tests.cpp @@ -136,7 +136,7 @@ TEST_CASE(SerializationRoundTrip) { // Deserialize into a fresh pool, as a new session would. clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); auto& restored = *loaded; @@ -201,7 +201,7 @@ TEST_CASE(NameSurvivesRoundTrip) { project.serialize(os, pool, {}); clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); auto& restored = *loaded; @@ -261,7 +261,7 @@ TEST_CASE(ScopeRoundTrip) { project.serialize(os, pool, {}); clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); auto& restored = *loaded; diff --git a/tests/unit/index/tu_index_tests.cpp b/tests/unit/index/tu_index_tests.cpp index 947dad1c5..9e5617383 100644 --- a/tests/unit/index/tu_index_tests.cpp +++ b/tests/unit/index/tu_index_tests.cpp @@ -91,7 +91,7 @@ void GO_TO_DEFINITION(llvm::StringRef pos, auto& relations = it->second; auto target = std::ranges::find_if(relations, [](const index::Relation& relation) { - return relation.kind.value() == static_cast(RelationKind::Definition); + return relation.kind == RelationKind::Definition; }); ASSERT_TRUE(target != relations.end()); @@ -273,7 +273,7 @@ TEST_CASE(Reference) { auto& relations = it->second; auto ref = std::ranges::find_if(relations, [](const index::Relation& r) { - return r.kind.value() == static_cast(RelationKind::Reference); + return r.kind == RelationKind::Reference; }); ASSERT_TRUE(ref != relations.end()); } @@ -297,18 +297,19 @@ TEST_CASE(BaseAndDerived) { auto base_hash = base_occs.front().target; auto derived_hash = derived_occs.front().target; - auto has_pair = [&](index::SymbolHash source, RelationKind kind, index::SymbolHash target) { - auto it = index.relations.find(source); - if(it == index.relations.end()) { - return false; - } - for(auto& r: it->second) { - if(r.kind.value() == static_cast(kind) && r.target_symbol == target) { - return true; + auto has_pair = + [&](index::SymbolHash source, RelationKind::Kind kind, index::SymbolHash target) { + auto it = index.relations.find(source); + if(it == index.relations.end()) { + return false; } - } - return false; - }; + for(auto& r: it->second) { + if(r.kind == kind && r.target_symbol == target) { + return true; + } + } + return false; + }; ASSERT_TRUE(has_pair(derived_hash, RelationKind::Base, base_hash)); ASSERT_TRUE(has_pair(base_hash, RelationKind::Derived, derived_hash)); @@ -335,7 +336,7 @@ TEST_CASE(CallerAndCallee) { bool found_callee = false; for(auto& r: caller_it->second) { - if(r.kind.value() == static_cast(RelationKind::Callee)) { + if(r.kind == RelationKind::Callee) { found_callee = true; break; } @@ -352,7 +353,7 @@ TEST_CASE(CallerAndCallee) { bool found_caller = false; for(auto& r: callee_it->second) { - if(r.kind.value() == static_cast(RelationKind::Caller)) { + if(r.kind == RelationKind::Caller) { found_caller = true; break; } @@ -388,8 +389,7 @@ TEST_CASE(MethodCallerCallee) { bool found_callee = false; for(auto& r: method_it->second) { - if(r.kind.value() == static_cast(RelationKind::Callee) && - r.target_symbol == callee_hash) { + if(r.kind == RelationKind::Callee && r.target_symbol == callee_hash) { found_callee = true; break; } @@ -401,8 +401,7 @@ TEST_CASE(MethodCallerCallee) { bool found_caller = false; for(auto& r: callee_it->second) { - if(r.kind.value() == static_cast(RelationKind::Caller) && - r.target_symbol == method_hash) { + if(r.kind == RelationKind::Caller && r.target_symbol == method_hash) { found_caller = true; break; } @@ -430,8 +429,7 @@ TEST_CASE(UsingRelationKey) { bool found_use = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::WeakReference) && - r.range == range("use")) { + if(r.kind == RelationKind::WeakReference && r.range == range("use")) { found_use = true; break; } @@ -515,8 +513,7 @@ TEST_CASE(DependentWeakReference) { bool found_weak = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::WeakReference) && - r.range == range("use")) { + if(r.kind == RelationKind::WeakReference && r.range == range("use")) { found_weak = true; break; } @@ -552,8 +549,7 @@ TEST_CASE(TypeDefinitionRelations) { return false; } for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::TypeDefinition) && - r.target_symbol == target_hash) { + if(r.kind == RelationKind::TypeDefinition && r.target_symbol == target_hash) { return true; } } @@ -587,11 +583,10 @@ TEST_CASE(ConstructorDestructorRelations) { bool found_ctor = false; bool found_dtor = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Constructor) && - r.target_symbol == ctor_hash) { + if(r.kind == RelationKind::Constructor && r.target_symbol == ctor_hash) { found_ctor = true; } - if(r.kind.value() == static_cast(RelationKind::Destructor)) { + if(r.kind == RelationKind::Destructor) { found_dtor = true; } } @@ -604,8 +599,7 @@ TEST_CASE(ConstructorDestructorRelations) { bool found_type = false; for(auto& r: ctor_it->second) { - if(r.kind.value() == static_cast(RelationKind::TypeDefinition) && - r.target_symbol == class_hash) { + if(r.kind == RelationKind::TypeDefinition && r.target_symbol == class_hash) { found_type = true; } } @@ -631,12 +625,10 @@ TEST_CASE(MacroRelations) { bool found_definition = false; bool found_reference = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Definition) && - r.range == range("def")) { + if(r.kind == RelationKind::Definition && r.range == range("def")) { found_definition = true; } - if(r.kind.value() == static_cast(RelationKind::Reference) && - r.range == range("use")) { + if(r.kind == RelationKind::Reference && r.range == range("use")) { found_reference = true; } } @@ -656,7 +648,7 @@ TEST_CASE(ModuleName) { bool found_definition = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Definition)) { + if(r.kind == RelationKind::Definition) { found_definition = true; } } @@ -677,7 +669,7 @@ TEST_CASE(ModulePartitionName) { bool found_definition = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Definition)) { + if(r.kind == RelationKind::Definition) { found_definition = true; } } @@ -707,10 +699,10 @@ module §(m)⟦§(m)foo⟧; bool found_reference = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Definition)) { + if(r.kind == RelationKind::Definition) { ASSERT_TRUE(false); } - if(r.kind.value() == static_cast(RelationKind::Reference)) { + if(r.kind == RelationKind::Reference) { found_reference = true; } } @@ -738,9 +730,9 @@ TEST_CASE(OverrideRelation) { auto check_relations = [&](index::FileIndex& idx) { for(auto& [hash, rels]: idx.relations) { for(auto& r: rels) { - if(r.kind.value() == RelationKind::Interface) + if(r.kind == RelationKind::Interface) found_interface = true; - if(r.kind.value() == RelationKind::Implementation) + if(r.kind == RelationKind::Implementation) found_implementation = true; } } @@ -775,10 +767,10 @@ TEST_CASE(DeclarationAndDefinition) { bool found_decl = false; bool found_def = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Declaration)) { + if(r.kind == RelationKind::Declaration) { found_decl = true; } - if(r.kind.value() == static_cast(RelationKind::Definition)) { + if(r.kind == RelationKind::Definition) { found_def = true; } } From e5bc91c4c7d5cce892ba56730bf6543afbf0c8d8 Mon Sep 17 00:00:00 2001 From: ykiko Date: Wed, 12 Aug 2026 21:08:07 +0800 Subject: [PATCH 02/12] review: fold self-review findings, close coverage gaps Code folds: TUIndex::from normalizes path_hashes to the path table's length (the verifier checks structure, not cross-field invariants); serialize() no longer wipes a deserialized index's path-keyed rows; shared deserialize_blob/scan_occurrences_at helpers replace triplicated decode calls and duplicated binary searches; integration tooling's cache root caught up to v5; stale flatc-era comments trimmed. Tests: TUIndex direct round-trip (multi-file, built_at, symbols, path_file_indices), hostile-input rejection for every loader (garbage, truncation, identifier clobber), version-gate positive controls pinning slot alignment, buffer-path relation lookup parity with the impl path, multi-occurrence binary search, ProjectIndex out-of-range local-id guard, Indexer::merge garbage rejection. --- cmake/package.cmake | 8 +- package-lock.json | 15 -- pixi.toml | 1 - src/index/merged_index.cpp | 41 ++--- src/index/preamble_state.cpp | 35 +---- src/index/preamble_state.h | 8 +- src/index/project_index.cpp | 6 +- src/index/serialization.h | 41 +++++ src/index/tu_index.cpp | 34 ++-- src/index/tu_index.h | 2 +- tests/unit/index/merged_index_tests.cpp | 172 +++++++++++++++++++++ tests/unit/index/persisted_index_tests.cpp | 61 ++++++++ tests/unit/index/preamble_state_tests.cpp | 51 ++++++ tests/unit/index/tu_index_tests.cpp | 137 ++++++++++++++++ tests/unit/server/indexer_tests.cpp | 13 ++ tools/client/workspace.ts | 2 +- 16 files changed, 520 insertions(+), 107 deletions(-) diff --git a/cmake/package.cmake b/cmake/package.cmake index aeb337a04..5fe288e8b 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG e38c704c0d43486b019a8a15531a56269b605b7c + GIT_TAG fbae31ca01005959cdc9f0a7368886f8878e40d4 ) set(KOTA_ENABLE_ZEST ON) @@ -38,10 +38,8 @@ set(KOTA_ENABLE_TEST OFF) set(KOTA_CODEC_ENABLE_SIMDJSON ON) set(KOTA_CODEC_ENABLE_YYJSON ON) set(KOTA_CODEC_ENABLE_TOML ON) -# kotatsu already fetches flatbuffers (v25.2.10) for its own codec and links the -# runtime lib into anything that uses kota::codec, so clice rides on that copy -# instead of fetching a second one. kotatsu does not build flatc, though, so the -# schema compiler comes from pixi instead (see CMakeLists.txt). +# kotatsu fetches the flatbuffers runtime (v25.2.10) for its codec and links it +# into anything that uses kota::codec; index serialization rides on that copy. set(KOTA_CODEC_ENABLE_FLATBUFFERS ON) set(KOTA_ENABLE_EXCEPTIONS OFF) set(KOTA_ENABLE_RTTI OFF) diff --git a/package-lock.json b/package-lock.json index 467a7a7a7..e9a87f67d 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1679,7 +1679,6 @@ "integrity": "sha512-zWW5KPngR/yvakJgGOmZ5vTBemDoSqF3AcV/LrO5u5wTWyEAVVh+IT39G4gtyAkh3CtTZs8aX/yRM82OfzHJRg==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "undici-types": "~7.16.0" } @@ -1711,7 +1710,6 @@ "integrity": "sha512-p088eaGrzYz1s+7cov0aMOCkNGTJlVxF4jgubf28c8L0Cv9Rloj8YBHnv4hXLq6IIEE1AsjNWavO+k+8kP2Y0A==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@eslint-community/regexpp": "^4.12.2", "@typescript-eslint/scope-manager": "8.66.0", @@ -1751,7 +1749,6 @@ "integrity": "sha512-X6ypGChaWYk6PBtUg2BwuTZEFFcHJAtGTVJ9/lCTOufhZ4i9fNolQNnktq+kkMCwMj7V8Svsq7+TxSDslmhE0g==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@typescript-eslint/scope-manager": "8.66.0", "@typescript-eslint/types": "8.66.0", @@ -2572,7 +2569,6 @@ "integrity": "sha512-lGq+9yr1/GuAWaVYIHRjvvySG5/4VfKIvC8EWxStPdcDh/Ka7FG3twP6v4d5BkravUilhIAsG4Qj83t02LWUPQ==", "dev": true, "license": "MIT", - "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -2606,7 +2602,6 @@ "integrity": "sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", @@ -2896,7 +2891,6 @@ } ], "license": "MIT", - "peer": true, "dependencies": { "baseline-browser-mapping": "^2.10.44", "caniuse-lite": "^1.0.30001806", @@ -3976,7 +3970,6 @@ "integrity": "sha512-nuKKvN+oIBO0koN7Tm7dlkmnkc21mtt0QJLwAKzjLq14y6lRTdVG36MZHJ8eQHwdJMwZbQNMlPOYedMq/oVJvQ==", "dev": true, "license": "MIT", - "peer": true, "workspaces": [ "packages/*" ], @@ -7870,7 +7863,6 @@ "integrity": "sha512-7A2xlQ5EnGT8KPA92dUh6RbRYTVw8hEaEN9L1K68l4UOXFuV511NnAqObGoRqGOQofQcMypisu1s3xawCEHrvA==", "dev": true, "license": "BSD-2-Clause", - "peer": true, "dependencies": { "@jridgewell/source-map": "^0.3.3", "acorn": "^8.15.0", @@ -7984,7 +7976,6 @@ "integrity": "sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==", "dev": true, "license": "MIT", - "peer": true, "engines": { "node": ">=12" }, @@ -8171,7 +8162,6 @@ "integrity": "sha512-Mtq29sKDAEYP7aljRgtPOpTvOfbwRWlS6dPRzwjdE+C0R4brX/GUyhHSecbHMFLNBLcJIPt9nl9yG5TZ1weH+Q==", "dev": true, "license": "Apache-2.0", - "peer": true, "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" @@ -8358,7 +8348,6 @@ "integrity": "sha512-4XP60spRGjSZFf1qYH+dJIkK2znL3zQfl9KkOV9MkkRR/3Dls0dxaBsQPTloEc5BLXWPL9vsOxopxyKoMmDueg==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "esbuild": "^0.27.0 || ^0.28.0", "fdir": "^6.5.0", @@ -8475,7 +8464,6 @@ "integrity": "sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==", "dev": true, "license": "MIT", - "peer": true, "engines": { "node": ">=12" }, @@ -8489,7 +8477,6 @@ "integrity": "sha512-KrxIJ62Fd89gfysR4WotlgZABiz2dqFPgqGzX7s+CwsqLFomRH7777ZcrOD6+WVAh7khPQP41A+BKbpcJFrdEg==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@types/chai": "^5.2.2", "@vitest/expect": "3.2.7", @@ -8695,7 +8682,6 @@ "integrity": "sha512-U9/cvLzxObKNEZ9+TtdqrHM5/9z3lgl2c+c4BzbqGxFQvQvBAq87yql5A8pQ+rrMbS496MZJeF5enVBndIy2hw==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@types/estree": "^1.0.8", "@types/json-schema": "^7.0.15", @@ -8740,7 +8726,6 @@ "integrity": "sha512-MfwFQ6SfwinsUVi0rNJm7rHZ31GyTcpVE5pgVA3hwFRb7COD4TzjUUwhGWKfO50+xdc2MQPuEBBJoqIMGt3JDw==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@discoveryjs/json-ext": "^0.6.1", "@webpack-cli/configtest": "^3.0.1", diff --git a/pixi.toml b/pixi.toml index f20f5107e..284973a8a 100644 --- a/pixi.toml +++ b/pixi.toml @@ -42,7 +42,6 @@ lld = "==22.1.8" llvm-tools = "==22.1.8" clang-tools = "==22.1.8" compiler-rt = "==22.1.8" -# Must match the headers kotatsu fetches — generated code static_asserts on it. # cmake/archive.cmake pipes tar through xz for the symbol archives. xz = ">=5.8.1,<6" diff --git a/src/index/merged_index.cpp b/src/index/merged_index.cpp index 71a50d797..44dd188ec 100644 --- a/src/index/merged_index.cpp +++ b/src/index/merged_index.cpp @@ -13,6 +13,7 @@ #include "kota/ipc/lsp/position.h" #include "llvm/ADT/DenseSet.h" +#include "llvm/ADT/STLExtras.h" #include "llvm/Support/raw_os_ostream.h" #include "llvm/Support/xxhash.h" @@ -340,12 +341,10 @@ void MergedIndex::load_in_memory(this Self& self) { auto& index = *self.impl; // The buffer was verified at load(); decode straight into the repr and - // move its containers into place. Out-parameter overload: the DenseMap - // members' explicit default constructors make the repr fail the - // std::default_initializable constraint of the value-returning one. + // move its containers into place. MergedIndexRepr repr; - auto decoded = kota::codec::fbs::from_bytes(blob_bytes(self.buffer->getBuffer()), repr); - assert(decoded.has_value()); + [[maybe_unused]] bool decoded = deserialize_blob(self.buffer->getBuffer(), repr); + assert(decoded); index.max_canonical_id = repr.max_canonical_id; @@ -361,12 +360,12 @@ void MergedIndex::load_in_memory(this Self& self) { index.canonical_ref_counts.resize(index.max_canonical_id, 0); - for(auto& [path, context]: repr.header_contexts) { + for(auto& context: llvm::make_second_range(repr.header_contexts)) { for(auto& include: context.includes) { index.canonical_ref_counts[include.canonical_id] += 1; } } - for(auto& [path, context]: repr.compilation_contexts) { + for(auto& context: llvm::make_second_range(repr.compilation_contexts)) { index.canonical_ref_counts[context.canonical_id] += 1; } @@ -522,29 +521,11 @@ void MergedIndex::lookup(this const Self& self, } } else if(self.buffer) { auto occurrences = root_of(*self.buffer)[&MergedIndexRepr::occurrences]; - - // First entry whose range ends at or past the offset; entries are - // sorted by (range.begin, range.end, target). - std::size_t lo = 0; - std::size_t hi = occurrences.size(); - while(lo < hi) { - auto mid = lo + (hi - lo) / 2; - if(occurrences.at(mid).get<0>().range.end < offset) { - lo = mid + 1; - } else { - hi = mid; - } - } - - for(; lo < occurrences.size(); ++lo) { - Occurrence occurrence = occurrences.at(lo).get<0>(); - if(!occurrence.range.contains(offset)) { - break; - } - if(!callback(occurrence)) { - break; - } - } + scan_occurrences_at( + occurrences.size(), + offset, + [&](std::size_t i) { return occurrences.at(i).get<0>(); }, + callback); } } diff --git a/src/index/preamble_state.cpp b/src/index/preamble_state.cpp index 100c8d413..b8150ce45 100644 --- a/src/index/preamble_state.cpp +++ b/src/index/preamble_state.cpp @@ -49,12 +49,10 @@ StateView root_of(const llvm::MemoryBuffer& buffer) { return StateView::from_verified_bytes(blob_bytes(buffer.getBuffer())); } -PreambleState::File file_of(StateView root, FileEntryView entry) { - auto paths = root[&PreambleStateRepr::paths]; - auto path_id = entry[&PreambleFileEntryRepr::path_id]; +PreambleState::File file_of(kota::codec::fbs::array_view paths, FileEntryView entry) { auto line_starts = to_array_ref(entry[&PreambleFileEntryRepr::line_starts]); return PreambleState::File{ - .path = to_ref(paths[path_id]), + .path = to_ref(paths[entry[&PreambleFileEntryRepr::path_id]]), .content = to_ref(entry[&PreambleFileEntryRepr::content]), .line_starts = std::span(line_starts.data(), line_starts.size()), }; @@ -157,7 +155,7 @@ void PreambleState::lookup(SymbolHash symbol, if(entry[&PreambleFileEntryRepr::path_id] >= paths.size()) { continue; } - auto file = file_of(root, entry); + auto file = file_of(paths, entry); auto rels = found->get<1>(); for(std::size_t j = 0; j < rels.size(); ++j) { @@ -192,28 +190,11 @@ void PreambleState::lookup_preamble(std::uint32_t offset, auto occurrences = root[&PreambleStateRepr::preamble][&PreambleFileEntryRepr::index][&FileIndex::occurrences]; - // First occurrence whose range ends at or past the offset; entries are - // sorted by (range.begin, range.end, target). - std::size_t lo = 0; - std::size_t hi = occurrences.size(); - while(lo < hi) { - auto mid = lo + (hi - lo) / 2; - if(occurrences[mid].range.end < offset) { - lo = mid + 1; - } else { - hi = mid; - } - } - - for(; lo < occurrences.size(); ++lo) { - Occurrence occurrence = occurrences[lo]; - if(!occurrence.range.contains(offset)) { - break; - } - if(!callback(occurrence)) { - break; - } - } + scan_occurrences_at( + occurrences.size(), + offset, + [&](std::size_t i) { return occurrences[i]; }, + callback); } void PreambleState::lookup_preamble(SymbolHash symbol, diff --git a/src/index/preamble_state.h b/src/index/preamble_state.h index ceeab2755..7e3e7c07b 100644 --- a/src/index/preamble_state.h +++ b/src/index/preamble_state.h @@ -15,10 +15,10 @@ namespace clice::index { /// On-disk PreambleState blob schema version (the PCH's `.pch.idx` pair). -/// Bump whenever `schema.fbs` changes the PreambleState layout; a blob -/// carrying a different value loads as "missing" and the PCH pair is -/// rebuilt. cache.json records it so a version change is caught at load -/// time instead of on the first overlay query. +/// Bump whenever the persisted PreambleState layout (its reflected repr) +/// changes; a blob carrying a different value loads as "missing" and the +/// PCH pair is rebuilt. cache.json records it so a version change is +/// caught at load time instead of on the first overlay query. constexpr inline std::uint32_t preamble_format_version = 4; /// All master-visible state derived from one PCH build. diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index 3fe245a7d..479b3d315 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -88,12 +88,8 @@ void ProjectIndex::serialize(this const ProjectIndex& self, std::optional ProjectIndex::from(llvm::StringRef data, clice::PathPool& pool, llvm::SmallVectorImpl& shards) { - // Out-parameter overload: SymbolTable is an llvm::DenseMap whose explicit - // default constructor makes the repr fail the std::default_initializable - // constraint of the value-returning from_bytes. ProjectIndexRepr repr; - auto decoded = kota::codec::fbs::from_bytes(blob_bytes(data), repr); - if(!decoded || repr.format_version != index_format_version) { + if(!deserialize_blob(data, repr) || repr.format_version != index_format_version) { return std::nullopt; } diff --git a/src/index/serialization.h b/src/index/serialization.h index 34bb96ac0..9373c8ef8 100644 --- a/src/index/serialization.h +++ b/src/index/serialization.h @@ -8,6 +8,7 @@ #include #include +#include "index/tu_index.h" #include "semantic/symbol.h" #include "support/bitmap.h" @@ -88,6 +89,46 @@ inline std::span blob_bytes(llvm::StringRef data) { return {reinterpret_cast(data.data()), data.size()}; } +/// Verify and deserialize a blob into `out`. The out-parameter overload of +/// from_bytes is deliberate: index types hold llvm::DenseMap members whose +/// explicit default constructors fail std::default_initializable, which the +/// value-returning overload requires. +template +bool deserialize_blob(llvm::StringRef data, T& out) { + return kota::codec::fbs::from_bytes(blob_bytes(data), out).has_value(); +} + +/// Scan the occurrences containing `offset` in a sequence sorted by +/// (range.begin, range.end, target): binary-search the first entry whose +/// range ends at or past the offset, then walk while ranges contain it. +/// `get(i)` yields the i-th Occurrence. +template +void scan_occurrences_at(std::size_t size, + std::uint32_t offset, + GetOccurrence&& get, + llvm::function_ref callback) { + std::size_t lo = 0; + std::size_t hi = size; + while(lo < hi) { + auto mid = lo + (hi - lo) / 2; + if(get(mid).range.end < offset) { + lo = mid + 1; + } else { + hi = mid; + } + } + + for(; lo < size; ++lo) { + Occurrence occurrence = get(lo); + if(!occurrence.range.contains(offset)) { + break; + } + if(!callback(occurrence)) { + break; + } + } +} + inline llvm::StringRef to_ref(std::string_view text) { return {text.data(), text.size()}; } diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index 9df91dabb..2f9be2d5c 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -530,14 +530,8 @@ class Projector { for(auto& [fid, index]: result.file_indices) { for(auto& [symbol_id, relations]: index.relations) { std::ranges::sort(relations, [](const Relation& lhs, const Relation& rhs) { - return std::tuple(static_cast(lhs.kind), - lhs.range.begin, - lhs.range.end, - lhs.target_symbol) < - std::tuple(static_cast(rhs.kind), - rhs.range.begin, - rhs.range.end, - rhs.target_symbol); + return std::tuple(lhs.kind, lhs.range.begin, lhs.range.end, lhs.target_symbol) < + std::tuple(rhs.kind, rhs.range.begin, rhs.range.end, rhs.target_symbol); }); auto range = std::ranges::unique(relations, [](const Relation& lhs, const Relation& rhs) { @@ -643,25 +637,29 @@ TUIndex TUIndex::build(CompilationUnitRef unit, bool interested_only) { void TUIndex::serialize(llvm::raw_ostream& os) { /// Convert the FileID-keyed working state into the persisted - /// path_id-keyed form. Multiple FileIDs can share a path id (repeated - /// header contexts); last-wins matches what deserialization of the old - /// per-FileID entries did. - path_file_indices.clear(); - for(auto& [fid, file_index]: file_indices) { - path_file_indices[graph.path_id(fid)] = file_index; + /// path_id-keyed form; multiple FileIDs can share a path id (repeated + /// header contexts), last-wins. A deserialized index has no FileID-keyed + /// state at all — its path-keyed rows already are the persisted form, so + /// re-serializing must not wipe them. + if(!file_indices.empty()) { + path_file_indices.clear(); + for(auto& [fid, file_index]: file_indices) { + path_file_indices[graph.path_id(fid)] = file_index; + } } serialize_blob(*this, os); } std::optional TUIndex::from(llvm::StringRef data) { - // The out-parameter overload: TUIndex holds llvm::DenseMap members whose - // explicit default constructors make it fail std::default_initializable, - // which the value-returning from_bytes requires. std::optional index{std::in_place}; - if(!kota::codec::fbs::from_bytes(blob_bytes(data), *index)) { + if(!deserialize_blob(data, *index)) { return std::nullopt; } + // The verifier checks structure, not cross-field consistency: consumers + // index path_hashes by path id, so normalize its length to the path + // table's (absent hashes read as 0 = "unavailable"). + index->graph.path_hashes.resize(index->graph.paths.size(), 0); return index; } diff --git a/src/index/tu_index.h b/src/index/tu_index.h index 50e3d0836..f5196e128 100644 --- a/src/index/tu_index.h +++ b/src/index/tu_index.h @@ -133,7 +133,7 @@ struct TUIndex { /// populated from file_indices first — hence non-const). void serialize(llvm::raw_ostream& os); - /// Deserialize a verified buffer; nullopt when verification fails. + /// Verify and deserialize a buffer; nullopt when verification fails. static std::optional from(llvm::StringRef data); }; diff --git a/tests/unit/index/merged_index_tests.cpp b/tests/unit/index/merged_index_tests.cpp index 7bd70a417..405fb2913 100644 --- a/tests/unit/index/merged_index_tests.cpp +++ b/tests/unit/index/merged_index_tests.cpp @@ -1,4 +1,6 @@ +#include #include +#include #include "test/temp_dir.h" #include "test/test.h" @@ -694,6 +696,176 @@ TEST_CASE(OldShardDiscarded) { ASSERT_TRUE(loaded.content().empty()); ASSERT_TRUE(loaded.need_update()); } + + // Positive control: the same single-slot shape carrying the CURRENT + // version is kept — slot 0 really is the version slot and the rejection + // above comes from its value, not from the blob's shape. + { + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto blob = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); + ASSERT_TRUE(blob.has_value()); + + auto path = dir.path("current.idx"); + std::error_code ec; + llvm::raw_fd_ostream os(path, ec); + os.write(reinterpret_cast(blob->data()), blob->size()); + os.flush(); + + ASSERT_TRUE(index::MergedIndex::load(path).loaded()); + } +} + +TEST_CASE(GarbageLoadRejected) { + TempDir dir; + dir.touch("garbage.idx", "not a flatbuffer"); + + auto loaded = index::MergedIndex::load(dir.path("garbage.idx")); + ASSERT_FALSE(loaded.loaded()); + ASSERT_TRUE(loaded.content().empty()); + ASSERT_TRUE(loaded.need_update()); + + // Queries on the rejected shard answer with silence, not UB. + bool visited = false; + loaded.lookup(0, [&](const index::Occurrence&) { + visited = true; + return true; + }); + ASSERT_FALSE(visited); + + loaded.lookup(index::SymbolHash(1), RelationKind::Definition, [&](const index::Relation&) { + visited = true; + return true; + }); + ASSERT_FALSE(visited); + + std::string name; + SymbolKind kind; + ASSERT_FALSE(loaded.find_symbol(1, name, kind)); +} + +TEST_CASE(CorruptShardRejected) { + TempDir dir; + + index::MergedIndex merged; + index::FileIndex file_idx; + merged.merge("tu0", std::chrono::milliseconds(1), {}, file_idx, "corrupt-me"); + + llvm::SmallString<1024> blob; + llvm::raw_svector_ostream os(blob); + merged.serialize(os); + ASSERT_TRUE(blob.size() > 8); + + auto write = [&](llvm::StringRef name, llvm::StringRef bytes) { + dir.touch(name, bytes); + return dir.path(name); + }; + + // Sanity: the intact bytes load, so the rejections below are earned. + ASSERT_TRUE(index::MergedIndex::load(write("valid.idx", blob)).loaded()); + + llvm::StringRef bytes(blob.data(), blob.size()); + ASSERT_FALSE( + index::MergedIndex::load(write("half.idx", bytes.take_front(bytes.size() / 2))).loaded()); + ASSERT_FALSE(index::MergedIndex::load(write("minus1.idx", bytes.drop_back(1))).loaded()); + + // Bytes 4-7 carry the buffer identifier; a blob from another format + // must be rejected up front. + std::string clobbered = bytes.str(); + for(std::size_t i = 4; i < 8; ++i) { + clobbered[i] = 'X'; + } + ASSERT_FALSE(index::MergedIndex::load(write("clobbered.idx", clobbered)).loaded()); +} + +TEST_CASE(BufferPathLookupParity) { + build_index(R"( + void §(a)alpha_func() {} + int §(b)beta_var = 1; + )"); + + index::MergedIndex merged; + auto fid = unit->interested_file(); + merged.merge("tu0", + tu_index.graph.include_location_id(fid), + tu_index.main_file_index, + unit->interested_content()); + + auto hash_at = [&](llvm::StringRef pos) { + auto offset = point(pos); + index::SymbolHash hash = 0; + merged.lookup(offset, [&](const index::Occurrence& occ) { + hash = occ.target; + return false; + }); + return hash; + }; + + using Row = std::tuple; + auto definitions = [](index::MergedIndex& index, index::SymbolHash hash) { + std::vector rows; + index.lookup(hash, RelationKind::Definition, [&](const index::Relation& relation) { + rows.emplace_back(relation.range.begin, relation.range.end, relation.target_symbol); + return true; + }); + std::ranges::sort(rows); + return rows; + }; + + index::SymbolHash hashes[2] = {hash_at("a"), hash_at("b")}; + std::vector expected[2]; + for(std::size_t i = 0; i < 2; ++i) { + ASSERT_TRUE(hashes[i] != 0); + expected[i] = definitions(merged, hashes[i]); + ASSERT_FALSE(expected[i].empty()); + } + ASSERT_TRUE(hashes[0] != hashes[1]); + + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + merged.serialize(os); + auto restored = index::MergedIndex(buf); + + // The zero-copy view is the production read path: it must agree BEFORE + // anything materializes the impl (operator== would, so it comes last). + for(std::size_t i = 0; i < 2; ++i) { + ASSERT_TRUE(definitions(restored, hashes[i]) == expected[i]); + } + + ASSERT_FALSE(restored.line_starts().empty()); + ASSERT_TRUE(std::ranges::equal(restored.line_starts(), merged.line_starts())); +} + +TEST_CASE(BufferPathMultiOccurrenceLookup) { + // Synthesized occurrences sorted by (begin, end, target), spaced so each + // probe hits exactly one range (contains() is inclusive at both ends). + index::FileIndex file_idx; + index::SymbolHash target = 100; + for(std::uint32_t begin = 0; begin < 60; begin += 10) { + file_idx.occurrences.emplace_back(index::Range{begin, begin + 3}, target++); + } + + index::MergedIndex merged; + merged.merge("tu0", std::uint32_t(0), file_idx, "synthetic"); + + llvm::SmallString<1024> buf; + llvm::raw_svector_ostream os(buf); + merged.serialize(os); + auto restored = index::MergedIndex(buf); + + // The buffer path binary-searches the serialized rows: every probe must + // land on exactly its own range. + for(auto& occurrence: file_idx.occurrences) { + std::vector hits; + restored.lookup(occurrence.range.begin + 1, [&](const index::Occurrence& hit) { + hits.push_back(hit); + return true; + }); + ASSERT_EQ(hits.size(), 1u); + ASSERT_TRUE(hits.front() == occurrence); + } } std::uint64_t file_hash(llvm::StringRef path) { diff --git a/tests/unit/index/persisted_index_tests.cpp b/tests/unit/index/persisted_index_tests.cpp index fa07f8520..54384bde4 100644 --- a/tests/unit/index/persisted_index_tests.cpp +++ b/tests/unit/index/persisted_index_tests.cpp @@ -1,5 +1,6 @@ #include "test/test.h" #include "index/project_index.h" +#include "index/serialization.h" #include "llvm/Support/raw_ostream.h" @@ -82,6 +83,66 @@ TEST_CASE(OldBlobDiscarded) { index::ProjectIndex::from(llvm::StringRef(junk, sizeof(junk)), pool, shards).has_value()); } +llvm::StringRef bytes_of(const std::vector& blob) { + return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); +} + +TEST_CASE(VersionGate) { + // Only the version slot is written: every other field reads back absent, + // which is structurally valid — the verdict must hinge on the value. + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + clice::PathPool pool; + llvm::SmallVector shards; + + auto stale = kota::codec::fbs::to_bytes(VersionOnly{}); + ASSERT_TRUE(stale.has_value()); + ASSERT_FALSE(index::ProjectIndex::from(bytes_of(*stale), pool, shards).has_value()); + + auto current = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); + ASSERT_TRUE(current.has_value()); + auto loaded = index::ProjectIndex::from(bytes_of(*current), pool, shards); + ASSERT_TRUE(loaded.has_value()); + ASSERT_TRUE(loaded->symbols.empty()); + ASSERT_TRUE(shards.empty()); +} + +TEST_CASE(OutOfRangeLocalIdsDropped) { + // Field order MUST mirror ProjectIndexRepr (project_index.cpp): + // format_version, paths, symbols, shards. + struct ProjectIndexMirror { + std::uint32_t format_version = 0; + std::vector paths; + index::SymbolTable symbols; + std::vector shards; + }; + + ProjectIndexMirror mirror; + mirror.format_version = index::index_format_version; + mirror.paths = {"/proj/used.cpp"}; + auto& symbol = mirror.symbols[42]; + symbol.name = "sym"; + symbol.reference_files.add(7); // Only local id 0 exists. + mirror.shards = {9}; + + auto blob = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(blob.has_value()); + + // Dangling local ids are dropped, not misresolved into the pool: the + // blob still loads, the symbol survives with an empty bitmap, and no + // shard is fetched. + clice::PathPool pool; + llvm::SmallVector shards; + auto loaded = index::ProjectIndex::from(bytes_of(*blob), pool, shards); + ASSERT_TRUE(loaded.has_value()); + ASSERT_TRUE(loaded->symbols.contains(42)); + ASSERT_EQ(loaded->symbols[42].name, "sym"); + ASSERT_EQ(loaded->symbols[42].reference_files.cardinality(), 0u); + ASSERT_TRUE(shards.empty()); +} + }; // TEST_SUITE(PersistedIndex) } // namespace diff --git a/tests/unit/index/preamble_state_tests.cpp b/tests/unit/index/preamble_state_tests.cpp index 276459da9..6efbf848b 100644 --- a/tests/unit/index/preamble_state_tests.cpp +++ b/tests/unit/index/preamble_state_tests.cpp @@ -241,6 +241,57 @@ TEST_CASE(RejectVersionMismatch) { EXPECT_TRUE(index::PreambleState::load(blob_path) == nullptr); } +TEST_CASE(AcceptCurrentVersionBlob) { + // Positive control for RejectVersionMismatch: the same single-slot shape + // carrying the CURRENT version loads — slot 0 really is the version slot + // and the rejection comes from its value, not from the blob's shape. + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto blob = kota::codec::fbs::to_bytes(VersionOnly{index::preamble_format_version}); + ASSERT_TRUE(blob.has_value()); + + dir.touch("current.pch.idx", + llvm::StringRef(reinterpret_cast(blob->data()), blob->size())); + EXPECT_TRUE(index::PreambleState::load(dir.path("current.pch.idx")) != nullptr); +} + +TEST_CASE(RejectCorruptBlob) { + add_main("main.cpp", R"( +int main() { return 0; } +)"); + build_state(); + + auto buffer = llvm::MemoryBuffer::getFile(dir.path("state.pch.idx")); + ASSERT_TRUE(bool(buffer)); + auto bytes = (*buffer)->getBuffer(); + ASSERT_TRUE(bytes.size() > 8); + + dir.touch("truncated.pch.idx", bytes.take_front(bytes.size() / 2)); + EXPECT_TRUE(index::PreambleState::load(dir.path("truncated.pch.idx")) == nullptr); + + // Bytes 4-7 carry the buffer identifier; a blob from another format + // must be rejected up front. + std::string clobbered = bytes.str(); + for(std::size_t i = 4; i < 8; ++i) { + clobbered[i] = 'X'; + } + dir.touch("clobbered.pch.idx", clobbered); + EXPECT_TRUE(index::PreambleState::load(dir.path("clobbered.pch.idx")) == nullptr); +} + +TEST_CASE(SourcePathAndContent) { + add_main("main.cpp", R"( +int value = 42; +int other = 1; +)"); + build_state(); + + EXPECT_TRUE(state->source_path().ends_with("main.cpp")); + EXPECT_EQ(state->preamble_content(), unit->interested_content()); +} + }; // TEST_SUITE(PreambleState) } // namespace diff --git a/tests/unit/index/tu_index_tests.cpp b/tests/unit/index/tu_index_tests.cpp index 9e5617383..02da153af 100644 --- a/tests/unit/index/tu_index_tests.cpp +++ b/tests/unit/index/tu_index_tests.cpp @@ -1184,6 +1184,143 @@ TEST_CASE(SuperQualifierRef) { GO_TO_DEFINITION("use", "def"); } +TEST_CASE(SerializeRoundTrip) { + add_file("header.h", R"( + #pragma once + inline int §(hdr)helper() { return 1; } + )"); + add_main("main.cpp", R"( + #include "header.h" + int main() { return §(use)helper(); } + )"); + ASSERT_TRUE(compile()); + tu_index = index::TUIndex::build(*unit); + ASSERT_FALSE(tu_index.file_indices.empty()); + + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + tu_index.serialize(os); + + auto loaded = index::TUIndex::from(buf); + ASSERT_TRUE(loaded.has_value()); + + ASSERT_EQ(loaded->built_at.count(), tu_index.built_at.count()); + ASSERT_TRUE(loaded->graph.paths == tu_index.graph.paths); + ASSERT_TRUE(loaded->graph.locations == tu_index.graph.locations); + ASSERT_TRUE(loaded->graph.path_hashes == tu_index.graph.path_hashes); + + // The persisted per-file rows are keyed by path id; recompute the + // expected conversion from the build-time FileID-keyed state. + llvm::DenseMap> expected; + for(auto& [fid, file_index]: tu_index.file_indices) { + expected[tu_index.graph.path_id(fid)] = {file_index.occurrences.size(), + file_index.relations.size()}; + } + ASSERT_FALSE(expected.empty()); + ASSERT_EQ(loaded->path_file_indices.size(), expected.size()); + for(auto& [path_id, counts]: expected) { + auto it = loaded->path_file_indices.find(path_id); + ASSERT_TRUE(it != loaded->path_file_indices.end()); + ASSERT_EQ(it->second.occurrences.size(), counts.first); + ASSERT_EQ(it->second.relations.size(), counts.second); + } + + ASSERT_TRUE(loaded->main_file_index.occurrences == tu_index.main_file_index.occurrences); + ASSERT_EQ(loaded->main_file_index.relations.size(), tu_index.main_file_index.relations.size()); + + ASSERT_EQ(loaded->symbols.size(), tu_index.symbols.size()); + for(auto& [hash, symbol]: tu_index.symbols) { + auto it = loaded->symbols.find(hash); + ASSERT_TRUE(it != loaded->symbols.end()); + ASSERT_EQ(it->second.name, symbol.name); + ASSERT_EQ(it->second.kind.value(), symbol.kind.value()); + ASSERT_EQ(static_cast(it->second.scope), static_cast(symbol.scope)); + ASSERT_TRUE(it->second.reference_files == symbol.reference_files); + } +} + +TEST_CASE(FromRejectsHostileInput) { + ASSERT_FALSE(index::TUIndex::from("not a flatbuffer at all").has_value()); + + build_index(R"( + int foo() { return 42; } + )"); + + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + tu_index.serialize(os); + + // Sanity: the intact blob loads, so the rejections below are earned. + ASSERT_TRUE(index::TUIndex::from(buf).has_value()); + + ASSERT_FALSE(index::TUIndex::from(llvm::StringRef(buf.data(), buf.size() / 2)).has_value()); + + // Bytes 4-7 carry the buffer identifier; a blob from another format + // must be rejected up front. + ASSERT_TRUE(buf.size() > 8); + std::string clobbered(buf.data(), buf.size()); + for(std::size_t i = 4; i < 8; ++i) { + clobbered[i] = 'X'; + } + ASSERT_FALSE(index::TUIndex::from(clobbered).has_value()); +} + +TEST_CASE(FromNormalizesPathHashes) { + build_index(R"( + int foo() { return 42; } + )"); + ASSERT_FALSE(tu_index.graph.paths.empty()); + + // A blob without path hashes (structurally valid: the field reads back + // empty) must come back resized to the path table, all "unavailable". + tu_index.graph.path_hashes.clear(); + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + tu_index.serialize(os); + + auto loaded = index::TUIndex::from(buf); + ASSERT_TRUE(loaded.has_value()); + ASSERT_EQ(loaded->graph.path_hashes.size(), loaded->graph.paths.size()); + for(auto hash: loaded->graph.path_hashes) { + ASSERT_EQ(hash, 0u); + } +} + +TEST_CASE(ReserializeKeepsPathIndices) { + add_file("header.h", R"( + #pragma once + inline int helper() { return 1; } + )"); + add_main("main.cpp", R"( + #include "header.h" + int main() { return helper(); } + )"); + ASSERT_TRUE(compile()); + tu_index = index::TUIndex::build(*unit); + + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + tu_index.serialize(os); + + auto loaded = index::TUIndex::from(buf); + ASSERT_TRUE(loaded.has_value()); + ASSERT_TRUE(loaded->file_indices.empty()); + ASSERT_FALSE(loaded->path_file_indices.empty()); + + // A deserialized index has no FileID-keyed state; re-serializing must + // keep the path-keyed rows instead of wiping them from an empty map. + llvm::SmallString<4096> again; + llvm::raw_svector_ostream os2(again); + loaded->serialize(os2); + + auto reloaded = index::TUIndex::from(again); + ASSERT_TRUE(reloaded.has_value()); + ASSERT_EQ(reloaded->path_file_indices.size(), loaded->path_file_indices.size()); + for(auto& [path_id, file_index]: loaded->path_file_indices) { + ASSERT_TRUE(reloaded->path_file_indices.contains(path_id)); + } +} + }; // TEST_SUITE(tu_index) } // namespace diff --git a/tests/unit/server/indexer_tests.cpp b/tests/unit/server/indexer_tests.cpp index 75d74e3b2..380251a0f 100644 --- a/tests/unit/server/indexer_tests.cpp +++ b/tests/unit/server/indexer_tests.cpp @@ -101,6 +101,19 @@ IndexedTU index_file(TempDir& tmp, llvm::StringRef file) { return result; } +TEST_CASE(MergeRejectsGarbage) { + // A worker shipping corrupted bytes (torn write, stale format) must not + // crash the master or leave partial state behind. + ASSERT_TRUE(workspace.merged_indices.empty()); + ASSERT_TRUE(workspace.project_index.symbols.empty()); + + std::string garbage = "definitely not a flatbuffer, but long enough to try"; + indexer.merge(garbage.data(), garbage.size()); + + ASSERT_TRUE(workspace.merged_indices.empty()); + ASSERT_TRUE(workspace.project_index.symbols.empty()); +} + TEST_CASE(MergeSkipsMovedDisk) { TempDir tmp; tmp.touch("main.cpp", "int value() { return 1; }\n"); diff --git a/tools/client/workspace.ts b/tools/client/workspace.ts index 9f1a2d69f..f6018636b 100644 --- a/tools/client/workspace.ts +++ b/tools/client/workspace.ts @@ -10,7 +10,7 @@ import { generateCDB } from "../compile_commands.ts"; /// Versioned root of the unified cache store; bump together with /// cache_format_version in src/server/state/workspace.h. -const CACHE_ROOT = path.join(".clice", "cache", "v4"); +const CACHE_ROOT = path.join(".clice", "cache", "v5"); /// The harness-wide canonical URI spelling: percent-decoded. vscode-uri /// encodes the drive colon (file:///c%3A/...) while the server emits it From c510edfebfe3d24495d5749191826fd3111f387a Mon Sep 17 00:00:00 2001 From: ykiko Date: Wed, 12 Aug 2026 21:28:33 +0800 Subject: [PATCH 03/12] chore: bump kotatsu pin to 24adcfc (zeroed inline-struct padding) --- cmake/package.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmake/package.cmake b/cmake/package.cmake index 5fe288e8b..cd76837a1 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG fbae31ca01005959cdc9f0a7368886f8878e40d4 + GIT_TAG 24adcfc75f675b46a0bf9389509e077e1714f5ed ) set(KOTA_ENABLE_ZEST ON) From 56c143b115108263261d4590675e562e2e67c9bc Mon Sep 17 00:00:00 2001 From: ykiko Date: Wed, 12 Aug 2026 22:20:06 +0800 Subject: [PATCH 04/12] chore: bump kotatsu pin to 9850507 (byte-storage wire images) --- cmake/package.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmake/package.cmake b/cmake/package.cmake index cd76837a1..6905d81ba 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG 24adcfc75f675b46a0bf9389509e077e1714f5ed + GIT_TAG 9850507984be208db957a24ebfe584f62f924da1 ) set(KOTA_ENABLE_ZEST ON) From f433008a21cdfd04a907cfa01a545d1b93d69940 Mon Sep 17 00:00:00 2001 From: ykiko Date: Wed, 12 Aug 2026 22:50:31 +0800 Subject: [PATCH 05/12] chore: bump kotatsu pin to latest feat/fbs-struct-map-keys --- cmake/package.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmake/package.cmake b/cmake/package.cmake index 6905d81ba..405769410 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG 9850507984be208db957a24ebfe584f62f924da1 + GIT_TAG 2b81f8e82b3efcbb10d54b84a3ccec6da6a62fb9 ) set(KOTA_ENABLE_ZEST ON) From 545091e6e4e062e9e7854e8de471b2f94c15d34f Mon Sep 17 00:00:00 2001 From: ykiko Date: Wed, 12 Aug 2026 23:50:08 +0800 Subject: [PATCH 06/12] chore: bump kotatsu pin (bool validation, long double exclusion) --- cmake/package.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmake/package.cmake b/cmake/package.cmake index 405769410..58f9cdf31 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG 2b81f8e82b3efcbb10d54b84a3ccec6da6a62fb9 + GIT_TAG bb79f3cd268a7bba9a30fdea0d84d9673987e646 ) set(KOTA_ENABLE_ZEST ON) From 00fdf893e0c2e6eea57be892959fe13bae0842e9 Mon Sep 17 00:00:00 2001 From: ykiko Date: Thu, 13 Aug 2026 00:46:30 +0800 Subject: [PATCH 07/12] chore: bump kotatsu pin (enum base gate, reflection-limit static rejection) --- cmake/package.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmake/package.cmake b/cmake/package.cmake index 58f9cdf31..07cc8ed93 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG bb79f3cd268a7bba9a30fdea0d84d9673987e646 + GIT_TAG cdca3d563c94043e87f5a9a34d38a5ef9087b91b ) set(KOTA_ENABLE_ZEST ON) From 836fa22fe01eac1f4c767188065fa6daf132edfc Mon Sep 17 00:00:00 2001 From: ykiko Date: Thu, 13 Aug 2026 02:27:23 +0800 Subject: [PATCH 08/12] chore: bump kotatsu pin (recursive annotation guard for inline structs) --- cmake/package.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmake/package.cmake b/cmake/package.cmake index 07cc8ed93..d42f36df1 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG cdca3d563c94043e87f5a9a34d38a5ef9087b91b + GIT_TAG 960a2d1c6cefbcafb666f64cdcf14381d9ab93ab ) set(KOTA_ENABLE_ZEST ON) From 771b7f8a5720cacd73594e36dcdb9f31e021162a Mon Sep 17 00:00:00 2001 From: ykiko Date: Thu, 13 Aug 2026 08:13:32 +0800 Subject: [PATCH 09/12] chore: pin kotatsu to main 5232e67 (#201 merged) --- cmake/package.cmake | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cmake/package.cmake b/cmake/package.cmake index d42f36df1..00f77d2fc 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG 960a2d1c6cefbcafb666f64cdcf14381d9ab93ab + GIT_TAG 5232e67c28fee3a85ad01341a675dd4b78b122b1 ) set(KOTA_ENABLE_ZEST ON) From c583b4bac280e15d14a19a5a129668ca2e653433 Mon Sep 17 00:00:00 2001 From: ykiko Date: Thu, 13 Aug 2026 09:03:44 +0800 Subject: [PATCH 10/12] review: version-gate TUIndex blobs, range-validate decoded ids MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit TUIndex now stamps and gates index_format_version: the blob is IPC-only, but a worker respawned after an on-disk binary upgrade can run one build ahead of the server, and a layout change need not be structurally detectable. from() also rejects blobs whose location path ids, file index keys or symbol bitmap values fall outside the blob's own path table, and MergedIndex::load_in_memory range-validates every decoded canonical id (cache entries, header includes, compilation contexts) instead of assert-only decode checking — the verifier checks structure, not cross-field invariants, and consumers index ref-count vectors with these values unchecked. Binary-searching occurrences on range.end is documented sound: name-token spans are pairwise disjoint or identical. --- src/index/merged_index.cpp | 40 +++++++++-- src/index/serialization.h | 3 + src/index/tu_index.cpp | 28 +++++++- src/index/tu_index.h | 11 ++- tests/unit/index/merged_index_tests.cpp | 95 +++++++++++++++++++++++++ tests/unit/index/tu_index_tests.cpp | 72 +++++++++++++++++++ 6 files changed, 243 insertions(+), 6 deletions(-) diff --git a/src/index/merged_index.cpp b/src/index/merged_index.cpp index 44dd188ec..3dbd9f5f9 100644 --- a/src/index/merged_index.cpp +++ b/src/index/merged_index.cpp @@ -340,11 +340,43 @@ void MergedIndex::load_in_memory(this Self& self) { } auto& index = *self.impl; - // The buffer was verified at load(); decode straight into the repr and - // move its containers into place. + // The buffer's structure was verified at load(), but structural + // verification does not constrain field values: a canonical id at or + // past max_canonical_id would index canonical_ref_counts out of bounds + // below (and in every later release_canonical). A blob carrying one — + // like a blob that fails to decode outright — is dropped, so the shard + // reads as empty and the background indexer rebuilds it. MergedIndexRepr repr; - [[maybe_unused]] bool decoded = deserialize_blob(self.buffer->getBuffer(), repr); - assert(decoded); + auto usable = [&] { + if(!deserialize_blob(self.buffer->getBuffer(), repr)) { + return false; + } + auto in_range = [&](std::uint32_t canonical_id) { + return canonical_id < repr.max_canonical_id; + }; + for(auto& [_, canonical_id]: repr.canonical_cache) { + if(!in_range(canonical_id)) { + return false; + } + } + for(auto& context: llvm::make_second_range(repr.header_contexts)) { + for(auto& include: context.includes) { + if(!in_range(include.canonical_id)) { + return false; + } + } + } + for(auto& context: llvm::make_second_range(repr.compilation_contexts)) { + if(!in_range(context.canonical_id)) { + return false; + } + } + return true; + }; + if(!usable()) { + self.buffer.reset(); + return; + } index.max_canonical_id = repr.max_canonical_id; diff --git a/src/index/serialization.h b/src/index/serialization.h index 9373c8ef8..4573fc5ad 100644 --- a/src/index/serialization.h +++ b/src/index/serialization.h @@ -101,6 +101,9 @@ bool deserialize_blob(llvm::StringRef data, T& out) { /// Scan the occurrences containing `offset` in a sequence sorted by /// (range.begin, range.end, target): binary-search the first entry whose /// range ends at or past the offset, then walk while ranges contain it. +/// Binary-searching on range.end is sound because occurrence ranges are +/// name-token spans, pairwise disjoint or identical — never partially +/// overlapping or nested — so under this order range.end is monotonic too. /// `get(i)` yields the i-th Occurrence. template void scan_occurrences_at(std::size_t size, diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index 2f9be2d5c..d88774891 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -636,6 +636,8 @@ TUIndex TUIndex::build(CompilationUnitRef unit, bool interested_only) { } void TUIndex::serialize(llvm::raw_ostream& os) { + format_version = index_format_version; + /// Convert the FileID-keyed working state into the persisted /// path_id-keyed form; multiple FileIDs can share a path id (repeated /// header contexts), last-wins. A deserialized index has no FileID-keyed @@ -653,13 +655,37 @@ void TUIndex::serialize(llvm::raw_ostream& os) { std::optional TUIndex::from(llvm::StringRef data) { std::optional index{std::in_place}; - if(!deserialize_blob(data, *index)) { + if(!deserialize_blob(data, *index) || index->format_version != index_format_version) { return std::nullopt; } // The verifier checks structure, not cross-field consistency: consumers // index path_hashes by path id, so normalize its length to the path // table's (absent hashes read as 0 = "unavailable"). index->graph.path_hashes.resize(index->graph.paths.size(), 0); + + // Nor does it constrain field values, and every decoded path id is + // dereferenced against the path table without further checks — graph + // locations and per-file rows in Indexer::merge, reference_files through + // ProjectIndex::merge's file_ids_map. A blob carrying an out-of-range + // one is rejected as a whole. + auto in_range = [count = index->graph.paths.size()](std::uint32_t path_id) { + return path_id < count; + }; + for(auto& location: index->graph.locations) { + if(!in_range(location.path_id)) { + return std::nullopt; + } + } + for(auto& [path_id, _]: index->path_file_indices) { + if(!in_range(path_id)) { + return std::nullopt; + } + } + for(auto& [_, symbol]: index->symbols) { + if(!symbol.reference_files.isEmpty() && !in_range(symbol.reference_files.maximum())) { + return std::nullopt; + } + } return index; } diff --git a/src/index/tu_index.h b/src/index/tu_index.h index f5196e128..c727d54c5 100644 --- a/src/index/tu_index.h +++ b/src/index/tu_index.h @@ -101,6 +101,13 @@ struct Symbol { using SymbolTable = llvm::DenseMap; struct TUIndex { + /// Persisted-blob schema version (index_format_version), stamped by + /// serialize() and gated by from(). These blobs never touch disk — they + /// travel worker→server over IPC — but a worker respawned after the + /// binary on disk changed can be one build ahead of the server, and a + /// layout change need not be structurally detectable. + std::uint32_t format_version = 0; + /// The building timestamp of this file. std::chrono::milliseconds built_at; @@ -133,7 +140,9 @@ struct TUIndex { /// populated from file_indices first — hence non-const). void serialize(llvm::raw_ostream& os); - /// Verify and deserialize a buffer; nullopt when verification fails. + /// Verify and deserialize a buffer; nullopt when structural + /// verification fails, the format version differs, or a decoded path id + /// falls outside the blob's own path table. static std::optional from(llvm::StringRef data); }; diff --git a/tests/unit/index/merged_index_tests.cpp b/tests/unit/index/merged_index_tests.cpp index 405fb2913..db90eb033 100644 --- a/tests/unit/index/merged_index_tests.cpp +++ b/tests/unit/index/merged_index_tests.cpp @@ -1,5 +1,7 @@ #include #include +#include +#include #include #include "test/temp_dir.h" @@ -8,6 +10,8 @@ #include "index/merged_index.h" #include "index/serialization.h" +#include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/SmallVector.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Support/xxhash.h" @@ -780,6 +784,97 @@ TEST_CASE(CorruptShardRejected) { ASSERT_FALSE(index::MergedIndex::load(write("clobbered.idx", clobbered)).loaded()); } +TEST_CASE(OutOfRangeCanonicalIdRejected) { + TempDir dir; + + // Field order MUST mirror the persisted shapes in merged_index.cpp + // (MergedIndexRepr prefix, HeaderContext, IncludeContext, + // CompilationContext prefix); the trailing fields read back absent, + // which is structurally valid. + struct IncludeContextMirror { + std::uint32_t include_id = 0; + std::uint32_t canonical_id = 0; + }; + + struct HeaderContextMirror { + std::uint32_t version = 0; + llvm::SmallVector includes; + }; + + // The vector matters beyond field parity: without one the mirror would + // be trivially copyable and encode as an inline struct, while the real + // CompilationContext encodes as a table — the verifier tells them apart. + struct CompilationContextMirror { + std::uint32_t version = 0; + std::uint32_t canonical_id = 0; + std::uint64_t build_at = 0; + std::vector include_locations; + }; + + struct ReprMirror { + std::uint32_t format_version = 0; + std::uint32_t max_canonical_id = 0; + std::vector paths; + std::map canonical_cache; + llvm::SmallDenseMap header_contexts; + llvm::SmallDenseMap compilation_contexts; + }; + + // A consistent base: one path, one canonical id, one header context + // referencing it. + auto base = [] { + ReprMirror mirror; + mirror.format_version = index::index_format_version; + mirror.max_canonical_id = 1; + mirror.paths = {"/proj/tu.cpp"}; + mirror.canonical_cache.emplace("hash", 0); + mirror.header_contexts[0].includes.push_back({.include_id = 0, .canonical_id = 0}); + return mirror; + }; + + // Structure and version pass, so load() accepts the blob off disk; the + // first mutation materializes it in memory, where the id values face + // the range check. Nullopt = the blob never reached that check. + auto materialized_contribution = [&](llvm::StringRef name, + const ReprMirror& mirror) -> std::optional { + auto blob = kota::codec::fbs::to_bytes(mirror); + if(!blob) { + return std::nullopt; + } + dir.touch(name, llvm::StringRef(reinterpret_cast(blob->data()), blob->size())); + auto shard = index::MergedIndex::load(dir.path(name)); + if(!shard.loaded()) { + return std::nullopt; + } + shard.remove("/proj/never-indexed.cpp"); + return shard.has_contribution("/proj/tu.cpp"); + }; + + // Positive control first: the base materializes intact, so the + // rejections below come from the hostile ids, not the blob's shape. + auto good = materialized_contribution("good.idx", base()); + ASSERT_TRUE(good.has_value() && *good); + + // Structural verification does not constrain field values: each blob + // carries one canonical id at or past max_canonical_id, which would + // index canonical_ref_counts out of bounds if the in-memory load + // accepted it. The blob is dropped and the shard reads as empty. + auto bad_cache = base(); + bad_cache.canonical_cache["hash"] = 5; + auto cache_verdict = materialized_contribution("bad-cache.idx", bad_cache); + ASSERT_TRUE(cache_verdict.has_value() && !*cache_verdict); + + auto bad_include = base(); + bad_include.header_contexts[0].includes.front().canonical_id = 5; + auto include_verdict = materialized_contribution("bad-include.idx", bad_include); + ASSERT_TRUE(include_verdict.has_value() && !*include_verdict); + + auto bad_compilation = base(); + bad_compilation.compilation_contexts[0].canonical_id = 5; + auto compilation_verdict = materialized_contribution("bad-compilation.idx", bad_compilation); + ASSERT_TRUE(compilation_verdict.has_value() && !*compilation_verdict); +} + TEST_CASE(BufferPathLookupParity) { build_index(R"( void §(a)alpha_func() {} diff --git a/tests/unit/index/tu_index_tests.cpp b/tests/unit/index/tu_index_tests.cpp index 02da153af..c60554830 100644 --- a/tests/unit/index/tu_index_tests.cpp +++ b/tests/unit/index/tu_index_tests.cpp @@ -5,6 +5,7 @@ #include "test/test.h" #include "test/tester.h" #include "feature/feature.h" +#include "index/serialization.h" #include "index/tu_index.h" #include "semantic/selection.h" @@ -1265,6 +1266,77 @@ TEST_CASE(FromRejectsHostileInput) { ASSERT_FALSE(index::TUIndex::from(clobbered).has_value()); } +TEST_CASE(FromRejectsStaleFormatVersion) { + // Only the version slot is written: every other field reads back + // absent, which is structurally valid — the verdict must hinge on the + // value. Field order MUST mirror TUIndex (tu_index.h): format_version + // is slot 0. + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto bytes_of = [](const std::vector& blob) { + return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); + }; + + auto stale = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version + 1}); + ASSERT_TRUE(stale.has_value()); + ASSERT_FALSE(index::TUIndex::from(bytes_of(*stale)).has_value()); + + // Positive control: the same shape carrying the current version loads, + // so the rejection above comes from the value, not the blob's shape. + auto current = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); + ASSERT_TRUE(current.has_value()); + ASSERT_TRUE(index::TUIndex::from(bytes_of(*current)).has_value()); +} + +TEST_CASE(FromRejectsOutOfRangePathIds) { + // Structural verification does not constrain field values, and the + // merge pipeline dereferences every decoded path id against the path + // table without further checks (Indexer::merge indexes paths and + // path_hashes, ProjectIndex::merge indexes file_ids_map with + // reference_files values) — a blob pointing outside its own table must + // be rejected as a whole. + auto serialized = [](index::TUIndex& index) { + std::string buf; + llvm::raw_string_ostream os(buf); + index.serialize(os); + return buf; + }; + + // Positive control first: the same shapes with in-range ids load, so + // the rejections below come from the hostile values. + index::TUIndex honest; + honest.built_at = std::chrono::milliseconds(0); + honest.graph.paths = {"/proj/main.cpp"}; + honest.graph.locations.push_back({.path_id = 0, .line = 1, .include = 0}); + honest.path_file_indices.try_emplace(0); + honest.symbols[42].reference_files.add(0); + ASSERT_TRUE(index::TUIndex::from(serialized(honest)).has_value()); + + { + index::TUIndex hostile; + hostile.built_at = std::chrono::milliseconds(0); + hostile.graph.paths = {"/proj/main.cpp"}; + hostile.graph.locations.push_back({.path_id = 7, .line = 1, .include = 0}); + ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); + } + { + index::TUIndex hostile; + hostile.built_at = std::chrono::milliseconds(0); + hostile.graph.paths = {"/proj/main.cpp"}; + hostile.path_file_indices.try_emplace(7); // Only path id 0 exists. + ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); + } + { + index::TUIndex hostile; + hostile.built_at = std::chrono::milliseconds(0); + hostile.graph.paths = {"/proj/main.cpp"}; + hostile.symbols[42].reference_files.add(7); + ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); + } +} + TEST_CASE(FromNormalizesPathHashes) { build_index(R"( int foo() { return 42; } From 0c6f2814a7b89808e5ab8e660b080a50e7f6b991 Mon Sep 17 00:00:00 2001 From: ykiko Date: Thu, 13 Aug 2026 12:01:11 +0800 Subject: [PATCH 11/12] refactor(index): serialize native structs directly, drop Repr mirrors The fbs blob schemas are now the native structs themselves instead of field-by-field mirror copies: - MergedIndex::Impl doubles as the shard schema: runtime-only fields are skip-annotated, serialize() compacts masked rows in place and reflects the impl straight onto the wire, load decodes into it directly. - PreambleState::serialize takes the TUIndex by value and assembles the blob by moving its rows; file contents and feature arrays are encoded as StringRef/ArrayRef borrows. Symbols persist as the full SymbolTable. - ProjectIndex persists its symbol bitmaps with raw pool ids plus an id-to-path table (still garbage-collected), remapped at load; the per-symbol bitmap rebuild on the encode side is gone. - PathPool repr drives the visitor imperatively (zero-copy encode, interning decode); new StringMap repr for the canonical cache. index_format_version 2 -> 3, preamble_format_version 4 -> 5. --- src/index/merged_index.cpp | 225 +++++++++------------ src/index/merged_index.h | 12 +- src/index/preamble_state.cpp | 115 +++++------ src/index/preamble_state.h | 6 +- src/index/project_index.cpp | 105 ++++------ src/index/project_index.h | 37 +++- src/index/serialization.h | 81 +++++++- src/server/worker/stateless_worker.cpp | 2 +- tests/unit/index/merged_index_tests.cpp | 57 +++++- tests/unit/index/persisted_index_tests.cpp | 10 +- tests/unit/index/preamble_state_tests.cpp | 42 ++++ 11 files changed, 407 insertions(+), 285 deletions(-) diff --git a/src/index/merged_index.cpp b/src/index/merged_index.cpp index 3dbd9f5f9..6fc7eddef 100644 --- a/src/index/merged_index.cpp +++ b/src/index/merged_index.cpp @@ -1,7 +1,6 @@ #include "index/merged_index.h" #include -#include #include #include #include @@ -188,6 +187,11 @@ struct CompilationContext { }; struct MergedIndex::Impl { + /// On-disk shard schema version (index_format_version), stamped by + /// serialize() and gated by load(); version-less blobs from older builds + /// read back as 0. + std::uint32_t format_version = 0; + /// Shard-local path table: every path id stored in this shard indexes /// into it, so shards are self-contained across sessions (runtime pool /// ids never persist). @@ -216,11 +220,15 @@ struct MergedIndex::Impl { /// The max canonical id we have allocated. std::uint32_t max_canonical_id = 0; - /// The reference count of each canonical id. - std::vector canonical_ref_counts; + /// The reference count of each canonical id. Derived state: rebuilt from + /// the context tables when a blob loads in memory. + KOTATSU_ANNOTATE(skip = true) + > canonical_ref_counts; - /// The canonical id set of removed index. - roaring::Roaring removed; + /// The canonical id set of removed index. Never persisted: compact() + /// erases the masked rows for real before a shard reaches disk. + KOTATSU_ANNOTATE(skip = true) + removed; /// All merged symbol occurrences. llvm::DenseMap occurrences; @@ -232,7 +240,8 @@ struct MergedIndex::Impl { SymbolTable symbols; /// Sorted occurrences cache for fast lookup. - std::vector occurrences_cache; + KOTATSU_ANNOTATE(skip = true) + > occurrences_cache; /// Drop one reference to a canonical index; the last reference masks its /// occurrences and relations via the removed bitmap. A later re-merge of @@ -274,29 +283,66 @@ struct MergedIndex::Impl { self.max_canonical_id += 1; } + /// Erase the rows masked by the removed bitmap for real. Queries are + /// unaffected (masked rows were already invisible), but a later re-merge + /// of identical content mints a fresh canonical instead of resurrecting + /// the id — the same behavior a save/load cycle produces. + void compact(this Impl& self) { + if(self.removed.isEmpty()) { + return; + } + + llvm::SmallVector dead_hashes; + for(const auto& entry: self.canonical_cache) { + if(self.removed.contains(entry.getValue())) { + dead_hashes.push_back(entry.getKey()); + } + } + for(auto hash: dead_hashes) { + self.canonical_cache.erase(hash); + } + + llvm::SmallVector dead_occurrences; + for(auto& [occurrence, bitmap]: self.occurrences) { + bitmap -= self.removed; + if(bitmap.isEmpty()) { + dead_occurrences.push_back(occurrence); + } + } + for(auto& occurrence: dead_occurrences) { + self.occurrences.erase(occurrence); + } + + llvm::SmallVector dead_symbols; + for(auto& [symbol, entries]: self.relations) { + llvm::SmallVector dead_relations; + for(auto& [relation, bitmap]: entries) { + bitmap -= self.removed; + if(bitmap.isEmpty()) { + dead_relations.push_back(relation); + } + } + for(auto& relation: dead_relations) { + entries.erase(relation); + } + if(entries.empty()) { + dead_symbols.push_back(symbol); + } + } + for(auto symbol: dead_symbols) { + self.relations.erase(symbol); + } + + self.removed = roaring::Roaring(); + self.occurrences_cache.clear(); + } + friend bool operator==(const Impl&, const Impl&) = default; }; namespace { -/// The persisted shape of a shard. Serialization reflects an instance of -/// this (built from Impl with the compaction applied); the buffer-backed -/// query paths read it through a zero-copy view. -struct MergedIndexRepr { - std::uint32_t format_version = 0; - std::uint32_t max_canonical_id = 0; - std::vector paths; - std::map canonical_cache; - llvm::SmallDenseMap header_contexts; - llvm::SmallDenseMap compilation_contexts; - llvm::DenseMap occurrences; - llvm::DenseMap> relations; - std::string content; - std::vector line_starts; - SymbolTable symbols; -}; - -using ShardView = kota::codec::fbs::table_view; +using ShardView = kota::codec::fbs::table_view; /// The blob was fully verified at load(); per-query views skip that cost. ShardView root_of(const llvm::MemoryBuffer& buffer) { @@ -346,27 +392,26 @@ void MergedIndex::load_in_memory(this Self& self) { // below (and in every later release_canonical). A blob carrying one — // like a blob that fails to decode outright — is dropped, so the shard // reads as empty and the background indexer rebuilds it. - MergedIndexRepr repr; auto usable = [&] { - if(!deserialize_blob(self.buffer->getBuffer(), repr)) { + if(!deserialize_blob(self.buffer->getBuffer(), index)) { return false; } auto in_range = [&](std::uint32_t canonical_id) { - return canonical_id < repr.max_canonical_id; + return canonical_id < index.max_canonical_id; }; - for(auto& [_, canonical_id]: repr.canonical_cache) { - if(!in_range(canonical_id)) { + for(const auto& entry: index.canonical_cache) { + if(!in_range(entry.getValue())) { return false; } } - for(auto& context: llvm::make_second_range(repr.header_contexts)) { + for(auto& context: llvm::make_second_range(index.header_contexts)) { for(auto& include: context.includes) { if(!in_range(include.canonical_id)) { return false; } } } - for(auto& context: llvm::make_second_range(repr.compilation_contexts)) { + for(auto& context: llvm::make_second_range(index.compilation_contexts)) { if(!in_range(context.canonical_id)) { return false; } @@ -374,43 +419,22 @@ void MergedIndex::load_in_memory(this Self& self) { return true; }; if(!usable()) { + self.impl = std::make_unique(); self.buffer.reset(); return; } - index.max_canonical_id = repr.max_canonical_id; - - // Interning in order reproduces the shard-local ids: path_id assigns - // sequentially from zero. - for(auto& path: repr.paths) { - index.paths.path_id(path); - } - - for(auto& [hash, canonical_id]: repr.canonical_cache) { - index.canonical_cache.try_emplace(hash, canonical_id); - } - index.canonical_ref_counts.resize(index.max_canonical_id, 0); - for(auto& context: llvm::make_second_range(repr.header_contexts)) { + for(auto& context: llvm::make_second_range(index.header_contexts)) { for(auto& include: context.includes) { index.canonical_ref_counts[include.canonical_id] += 1; } } - for(auto& context: llvm::make_second_range(repr.compilation_contexts)) { + for(auto& context: llvm::make_second_range(index.compilation_contexts)) { index.canonical_ref_counts[context.canonical_id] += 1; } - index.header_contexts = std::move(repr.header_contexts); - index.compilation_contexts = std::move(repr.compilation_contexts); - // The persisted removed bitmap is always empty (compaction drops masked - // rows before they reach disk), so nothing restores it here. - index.occurrences = std::move(repr.occurrences); - index.relations = std::move(repr.relations); - index.content = std::move(repr.content); - index.line_starts = std::move(repr.line_starts); - index.symbols = std::move(repr.symbols); - self.buffer.reset(); } @@ -427,14 +451,14 @@ MergedIndex MergedIndex::load(llvm::StringRef path) { // shard is treated as "not on disk" and the background indexer rebuilds // it. auto root = ShardView::from_bytes(blob_bytes((*buffer)->getBuffer())); - if(!root.valid() || root[&MergedIndexRepr::format_version] != index_format_version) { + if(!root.valid() || root[&Impl::format_version] != index_format_version) { return MergedIndex(); } return MergedIndex(std::move(*buffer), nullptr); } -void MergedIndex::serialize(this const Self& self, llvm::raw_ostream& out) { +void MergedIndex::serialize(this Self& self, llvm::raw_ostream& out) { if(self.buffer) { out.write(self.buffer->getBufferStart(), self.buffer->getBufferSize()); return; @@ -444,67 +468,12 @@ void MergedIndex::serialize(this const Self& self, llvm::raw_ostream& out) { return; } - auto& index = self.impl; - - // Compaction: rows whose every canonical was released are masked at - // runtime by the removed bitmap, but the serialized shard is served - // through buffer-only lookups that never consult it — so masked state - // must not reach disk at all. Dead rows are dropped, live bitmaps are - // written pre-subtracted, dead cache entries go with them (a later - // re-merge of identical content mints a fresh canonical), and the - // persisted shape carries no removed bitmap at all. - auto& removed = index->removed; - auto live = [&](const roaring::Roaring& bitmap) { - return removed.isEmpty() ? bitmap : bitmap - removed; - }; - - MergedIndexRepr repr; - repr.format_version = index_format_version; - repr.max_canonical_id = index->max_canonical_id; - - repr.paths.reserve(index->paths.paths.size()); - for(llvm::StringRef path: index->paths.paths) { - repr.paths.emplace_back(path); - } - - for(auto& [hash, canonical_id]: index->canonical_cache) { - if(removed.contains(canonical_id)) { - continue; - } - repr.canonical_cache.emplace(hash.str(), canonical_id); - } - - repr.header_contexts = index->header_contexts; - repr.compilation_contexts = index->compilation_contexts; - - for(auto& [occurrence, bitmap]: index->occurrences) { - auto masked = live(bitmap); - if(masked.isEmpty()) { - continue; - } - repr.occurrences.try_emplace(occurrence, std::move(masked)); - } - - for(auto& [symbol_id, symbol_relations]: index->relations) { - llvm::DenseMap entries; - for(auto& [relation, bitmap]: symbol_relations) { - auto masked = live(bitmap); - if(masked.isEmpty()) { - continue; - } - entries.try_emplace(relation, std::move(masked)); - } - if(entries.empty()) { - continue; - } - repr.relations.try_emplace(symbol_id, std::move(entries)); - } - - repr.content = index->content; - repr.line_starts = index->line_starts; - repr.symbols = index->symbols; - - serialize_blob(repr, out); + // The serialized shard is served through buffer-only lookups that never + // consult the removed bitmap, so masked state must not reach disk at + // all: compact first, then reflect the impl directly onto the wire. + self.impl->compact(); + self.impl->format_version = index_format_version; + serialize_blob(*self.impl, out); } void MergedIndex::lookup(this const Self& self, @@ -552,7 +521,7 @@ void MergedIndex::lookup(this const Self& self, break; } } else if(self.buffer) { - auto occurrences = root_of(*self.buffer)[&MergedIndexRepr::occurrences]; + auto occurrences = root_of(*self.buffer)[&Impl::occurrences]; scan_occurrences_at( occurrences.size(), offset, @@ -588,7 +557,7 @@ void MergedIndex::lookup(this const Self& self, } } } else if(self.buffer) { - auto found = root_of(*self.buffer)[&MergedIndexRepr::relations].find(symbol); + auto found = root_of(*self.buffer)[&Impl::relations].find(symbol); if(!found) [[unlikely]] { return; } @@ -651,12 +620,12 @@ bool MergedIndex::need_update(this const Self& self) { return false; } else if(self.buffer) { auto root = root_of(*self.buffer); - auto contexts = root[&MergedIndexRepr::compilation_contexts]; + auto contexts = root[&Impl::compilation_contexts]; if(contexts.empty()) { return true; } - auto paths = root[&MergedIndexRepr::paths]; + auto paths = root[&Impl::paths]; for(std::size_t c = 0; c < contexts.size(); ++c) { auto context = contexts.at(c).get<1>(); @@ -722,7 +691,7 @@ bool MergedIndex::has_contribution(this const Self& self, llvm::StringRef contex if(self.buffer) { auto root = root_of(*self.buffer); - auto paths = root[&MergedIndexRepr::paths]; + auto paths = root[&Impl::paths]; std::optional local; for(std::uint32_t i = 0; i < paths.size(); ++i) { if(to_ref(paths[i]) == context_path) { @@ -733,8 +702,8 @@ bool MergedIndex::has_contribution(this const Self& self, llvm::StringRef contex if(!local) { return false; } - return root[&MergedIndexRepr::header_contexts].contains(*local) || - root[&MergedIndexRepr::compilation_contexts].contains(*local); + return root[&Impl::header_contexts].contains(*local) || + root[&Impl::compilation_contexts].contains(*local); } return false; @@ -783,7 +752,7 @@ bool MergedIndex::find_symbol(this const Self& self, return true; } } else if(self.buffer) { - auto found = root_of(*self.buffer)[&MergedIndexRepr::symbols].find(hash); + auto found = root_of(*self.buffer)[&Impl::symbols].find(hash); if(found) { auto symbol = found->get<1>(); name = std::string(symbol[&Symbol::name]); @@ -910,7 +879,7 @@ llvm::StringRef MergedIndex::content(this const Self& self) { if(self.impl) { return self.impl->content; } else if(self.buffer) { - return to_ref(root_of(*self.buffer)[&MergedIndexRepr::content]); + return to_ref(root_of(*self.buffer)[&Impl::content]); } return {}; } @@ -919,7 +888,7 @@ std::span MergedIndex::line_starts(this const Self& self) { if(self.impl) { return self.impl->line_starts; } else if(self.buffer) { - auto starts = to_array_ref(root_of(*self.buffer)[&MergedIndexRepr::line_starts]); + auto starts = to_array_ref(root_of(*self.buffer)[&Impl::line_starts]); return {starts.data(), starts.size()}; } return {}; diff --git a/src/index/merged_index.h b/src/index/merged_index.h index 4b69e60a5..69d09d00f 100644 --- a/src/index/merged_index.h +++ b/src/index/merged_index.h @@ -25,9 +25,14 @@ struct DepLocation { }; class MergedIndex { -private: +public: + /// The in-memory shard state, defined in merged_index.cpp. Its reflected + /// layout doubles as the persisted shard schema: serialization reflects + /// an Impl directly and the buffer-backed query paths read the blob + /// through a zero-copy view of the same layout. struct Impl; +private: using Self = MergedIndex; MergedIndex(std::unique_ptr buffer, std::unique_ptr impl); @@ -52,8 +57,9 @@ class MergedIndex { /// Load merged index from disk static MergedIndex load(llvm::StringRef path); - /// Serialize it to binary format. - void serialize(this const Self& self, llvm::raw_ostream& out); + /// Serialize it to binary format. Compacts rows masked by removals in + /// place first, so the serialized blob is a direct reflection of Impl. + void serialize(this Self& self, llvm::raw_ostream& out); /// Lookup the occurrence in corresponding offset. void lookup(this const Self& self, diff --git a/src/index/preamble_state.cpp b/src/index/preamble_state.cpp index b8150ce45..0a5111513 100644 --- a/src/index/preamble_state.cpp +++ b/src/index/preamble_state.cpp @@ -15,34 +15,31 @@ namespace clice::index { namespace { /// One file covered by the preamble compilation: its rows plus content and -/// line starts for position mapping. -struct PreambleFileEntryRepr { +/// line starts for position mapping. The rows are moved out of the TUIndex +/// and the content borrows the compilation's buffers — entries are only +/// ever encoded, never decoded (queries run on the zero-copy view). +struct PreambleFileEntry { std::uint32_t path_id = 0; FileIndex index; - std::string content; + llvm::StringRef content; std::vector line_starts; }; -struct PreambleSymbolRepr { - std::string name; - SymbolKind kind; -}; - /// The persisted shape of a `.pch.idx` blob. Queries run on a zero-copy /// view of this layout; nothing is deserialized up front. -struct PreambleStateRepr { +struct PreambleBlob { std::uint32_t format_version = 0; std::vector paths; - std::vector files; - PreambleFileEntryRepr preamble; - llvm::DenseMap symbols; - std::vector links; - std::vector inactive_regions; - std::vector open_conditionals; + std::vector files; + PreambleFileEntry preamble; + SymbolTable symbols; + llvm::ArrayRef links; + llvm::ArrayRef inactive_regions; + llvm::ArrayRef open_conditionals; }; -using StateView = kota::codec::fbs::table_view; -using FileEntryView = kota::codec::fbs::table_view; +using StateView = kota::codec::fbs::table_view; +using FileEntryView = kota::codec::fbs::table_view; /// The blob was fully verified at load(); per-query views skip that cost. StateView root_of(const llvm::MemoryBuffer& buffer) { @@ -50,10 +47,10 @@ StateView root_of(const llvm::MemoryBuffer& buffer) { } PreambleState::File file_of(kota::codec::fbs::array_view paths, FileEntryView entry) { - auto line_starts = to_array_ref(entry[&PreambleFileEntryRepr::line_starts]); + auto line_starts = to_array_ref(entry[&PreambleFileEntry::line_starts]); return PreambleState::File{ - .path = to_ref(paths[entry[&PreambleFileEntryRepr::path_id]]), - .content = to_ref(entry[&PreambleFileEntryRepr::content]), + .path = to_ref(paths[entry[&PreambleFileEntry::path_id]]), + .content = to_ref(entry[&PreambleFileEntry::content]), .line_starts = std::span(line_starts.data(), line_starts.size()), }; } @@ -61,16 +58,15 @@ PreambleState::File file_of(kota::codec::fbs::array_view paths, Fil } // namespace void PreambleState::serialize(CompilationUnitRef unit, - const TUIndex& index, + TUIndex index, llvm::ArrayRef links, llvm::ArrayRef inactive_regions, llvm::ArrayRef open_conditionals, llvm::raw_ostream& os) { - PreambleStateRepr repr; - repr.format_version = preamble_format_version; - repr.paths = index.graph.paths; + PreambleBlob blob; + blob.format_version = preamble_format_version; - repr.files.reserve(index.file_indices.size()); + blob.files.reserve(index.file_indices.size()); for(auto& [fid, file_index]: index.file_indices) { // A file with no include edge is a synthetic buffer (predefines, // ): it has no real path to attribute rows to, and @@ -82,12 +78,13 @@ void PreambleState::serialize(CompilationUnitRef unit, continue; } auto content = unit.file_content(fid); - auto& entry = repr.files.emplace_back(); - entry.path_id = index.graph.path_id(fid); - entry.index = file_index; - entry.content = content; - entry.line_starts = - kota::ipc::lsp::build_line_starts(std::string_view(content.data(), content.size())); + blob.files.push_back({ + .path_id = index.graph.path_id(fid), + .index = std::move(file_index), + .content = content, + .line_starts = + kota::ipc::lsp::build_line_starts(std::string_view(content.data(), content.size())), + }); } // The source file is the last path in graph.paths (convention from @@ -96,21 +93,21 @@ void PreambleState::serialize(CompilationUnitRef unit, // PCH was built from — stored so consumers can compare it against the // live buffer's prefix before serving these rows. auto preamble_text = unit.interested_content(); - repr.preamble.path_id = static_cast(index.graph.paths.size() - 1); - repr.preamble.index = index.main_file_index; - repr.preamble.content = preamble_text; - repr.preamble.line_starts = kota::ipc::lsp::build_line_starts( - std::string_view(preamble_text.data(), preamble_text.size())); - - for(auto& [symbol_id, symbol]: index.symbols) { - repr.symbols[symbol_id] = PreambleSymbolRepr{.name = symbol.name, .kind = symbol.kind}; - } + blob.preamble = { + .path_id = static_cast(index.graph.paths.size() - 1), + .index = std::move(index.main_file_index), + .content = preamble_text, + .line_starts = kota::ipc::lsp::build_line_starts( + std::string_view(preamble_text.data(), preamble_text.size())), + }; - repr.links.assign(links.begin(), links.end()); - repr.inactive_regions.assign(inactive_regions.begin(), inactive_regions.end()); - repr.open_conditionals.assign(open_conditionals.begin(), open_conditionals.end()); + blob.symbols = std::move(index.symbols); + blob.paths = std::move(index.graph.paths); + blob.links = links; + blob.inactive_regions = inactive_regions; + blob.open_conditionals = open_conditionals; - serialize_blob(repr, os); + serialize_blob(blob, os); } PreambleState::PreambleState(std::unique_ptr buffer) : @@ -128,7 +125,7 @@ std::shared_ptr PreambleState::load(llvm::StringRef path) { // format version load as "missing" (version-less blobs read back 0) // and the PCH pair is rebuilt. auto root = StateView::from_bytes(blob_bytes((*buffer)->getBuffer())); - if(!root.valid() || root[&PreambleStateRepr::format_version] != preamble_format_version) { + if(!root.valid() || root[&PreambleBlob::format_version] != preamble_format_version) { return nullptr; } @@ -139,12 +136,12 @@ void PreambleState::lookup(SymbolHash symbol, RelationKind kind, llvm::function_ref callback) const { auto root = root_of(*buffer); - auto paths = root[&PreambleStateRepr::paths]; - auto files = root[&PreambleStateRepr::files]; + auto paths = root[&PreambleBlob::paths]; + auto files = root[&PreambleBlob::files]; for(std::size_t i = 0; i < files.size(); ++i) { auto entry = files[i]; - auto relations = entry[&PreambleFileEntryRepr::index][&FileIndex::relations]; + auto relations = entry[&PreambleFileEntry::index][&FileIndex::relations]; auto found = relations.find(symbol); if(!found) { continue; @@ -152,7 +149,7 @@ void PreambleState::lookup(SymbolHash symbol, // The verifier checks structure, not cross-references: a corrupt // path_id must not attribute rows to an arbitrary path. - if(entry[&PreambleFileEntryRepr::path_id] >= paths.size()) { + if(entry[&PreambleFileEntry::path_id] >= paths.size()) { continue; } auto file = file_of(paths, entry); @@ -171,7 +168,7 @@ void PreambleState::lookup(SymbolHash symbol, llvm::StringRef PreambleState::source_path() const { auto root = root_of(*buffer); - auto paths = root[&PreambleStateRepr::paths]; + auto paths = root[&PreambleBlob::paths]; if(paths.empty()) { return {}; } @@ -181,14 +178,14 @@ llvm::StringRef PreambleState::source_path() const { llvm::StringRef PreambleState::preamble_content() const { auto root = root_of(*buffer); - return to_ref(root[&PreambleStateRepr::preamble][&PreambleFileEntryRepr::content]); + return to_ref(root[&PreambleBlob::preamble][&PreambleFileEntry::content]); } void PreambleState::lookup_preamble(std::uint32_t offset, llvm::function_ref callback) const { auto root = root_of(*buffer); auto occurrences = - root[&PreambleStateRepr::preamble][&PreambleFileEntryRepr::index][&FileIndex::occurrences]; + root[&PreambleBlob::preamble][&PreambleFileEntry::index][&FileIndex::occurrences]; scan_occurrences_at( occurrences.size(), @@ -202,7 +199,7 @@ void PreambleState::lookup_preamble(SymbolHash symbol, llvm::function_ref callback) const { auto root = root_of(*buffer); auto relations = - root[&PreambleStateRepr::preamble][&PreambleFileEntryRepr::index][&FileIndex::relations]; + root[&PreambleBlob::preamble][&PreambleFileEntry::index][&FileIndex::relations]; auto found = relations.find(symbol); if(!found) { return; @@ -221,20 +218,20 @@ void PreambleState::lookup_preamble(SymbolHash symbol, bool PreambleState::find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const { auto root = root_of(*buffer); - auto found = root[&PreambleStateRepr::symbols].find(hash); + auto found = root[&PreambleBlob::symbols].find(hash); if(!found) { return false; } auto symbol = found->get<1>(); - name = std::string(symbol[&PreambleSymbolRepr::name]); - kind = SymbolKind(symbol[&PreambleSymbolRepr::kind]); + name = std::string(symbol[&Symbol::name]); + kind = SymbolKind(symbol[&Symbol::kind]); return true; } std::vector PreambleState::links() const { auto root = root_of(*buffer); - auto entries = root[&PreambleStateRepr::links]; + auto entries = root[&PreambleBlob::links]; std::vector links; links.reserve(entries.size()); @@ -250,12 +247,12 @@ std::vector PreambleState::links() const { llvm::ArrayRef PreambleState::inactive_regions() const { auto root = root_of(*buffer); - return to_array_ref(root[&PreambleStateRepr::inactive_regions]); + return to_array_ref(root[&PreambleBlob::inactive_regions]); } llvm::ArrayRef PreambleState::open_conditionals() const { auto root = root_of(*buffer); - return to_array_ref(root[&PreambleStateRepr::open_conditionals]); + return to_array_ref(root[&PreambleBlob::open_conditionals]); } } // namespace clice::index diff --git a/src/index/preamble_state.h b/src/index/preamble_state.h index 7e3e7c07b..d33f7a2fa 100644 --- a/src/index/preamble_state.h +++ b/src/index/preamble_state.h @@ -19,7 +19,7 @@ namespace clice::index { /// changes; a blob carrying a different value loads as "missing" and the /// PCH pair is rebuilt. cache.json records it so a version change is /// caught at load time instead of on the first overlay query. -constexpr inline std::uint32_t preamble_format_version = 4; +constexpr inline std::uint32_t preamble_format_version = 5; /// All master-visible state derived from one PCH build. /// @@ -57,8 +57,10 @@ class PreambleState { /// over the preamble unit with interested_only=false and its /// main_file_index intact (it holds the preamble region's own /// occurrences — macro definitions and references before the bound). + /// Taken by value and consumed: the blob is assembled by moving the + /// index's rows, never copying them. static void serialize(CompilationUnitRef unit, - const TUIndex& index, + TUIndex index, llvm::ArrayRef links, llvm::ArrayRef inactive_regions, llvm::ArrayRef open_conditionals, diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index 479b3d315..32a351d61 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -3,24 +3,10 @@ #include "index/serialization.h" #include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/STLExtras.h" namespace clice::index { -namespace { - -/// The persisted shape of the project blob: the symbol table with pool ids -/// remapped into a compact local path table (only ids the blob references -/// are written — garbage paths are collected here), plus the shard list as -/// local ids. -struct ProjectIndexRepr { - std::uint32_t format_version = 0; - std::vector paths; - SymbolTable symbols; - std::vector shards; -}; - -} // namespace - llvm::SmallVector ProjectIndex::merge(this ProjectIndex& self, TUIndex& index, clice::PathPool& pool) { @@ -48,86 +34,65 @@ llvm::SmallVector ProjectIndex::merge(this ProjectIndex& self, return file_ids_map; } -void ProjectIndex::serialize(this const ProjectIndex& self, +void ProjectIndex::serialize(this ProjectIndex& self, llvm::raw_ostream& os, const clice::PathPool& pool, llvm::ArrayRef shards) { - ProjectIndexRepr repr; - repr.format_version = index_format_version; - - llvm::DenseMap local_ids; - auto to_local = [&](std::uint32_t pool_id) -> std::uint32_t { - auto [it, inserted] = local_ids.try_emplace(pool_id, repr.paths.size()); - if(inserted) { - repr.paths.emplace_back(pool.resolve(pool_id)); - } - return it->second; - }; + self.format_version = index_format_version; + self.shards.assign(shards.begin(), shards.end()); - llvm::SmallVector remapped; - for(auto& [symbol_id, symbol]: self.symbols) { - remapped.clear(); - for(auto ref: symbol.reference_files) { - remapped.push_back(to_local(ref)); - } - - auto& target = repr.symbols[symbol_id]; - target.name = symbol.name; - target.kind = symbol.kind; - target.scope = symbol.scope; - target.reference_files = Bitmap(remapped.size(), remapped.data()); + Bitmap referenced(shards.size(), shards.data()); + for(auto& symbol: llvm::make_second_range(self.symbols)) { + referenced |= symbol.reference_files; } - for(auto shard: shards) { - repr.shards.push_back(to_local(shard)); + self.paths.clear(); + self.paths.reserve(referenced.cardinality()); + for(auto id: referenced) { + self.paths.emplace_back(id, pool.resolve(id).str()); } - serialize_blob(repr, os); + serialize_blob(self, os); } std::optional ProjectIndex::from(llvm::StringRef data, clice::PathPool& pool, llvm::SmallVectorImpl& shards) { - ProjectIndexRepr repr; - if(!deserialize_blob(data, repr) || repr.format_version != index_format_version) { + std::optional index{std::in_place}; + if(!deserialize_blob(data, *index) || index->format_version != index_format_version) { return std::nullopt; } - // Intern the blob's compact path table into the running pool; every id - // in the blob is an index into it. - llvm::SmallVector pool_ids; - pool_ids.reserve(repr.paths.size()); - for(auto& path: repr.paths) { - pool_ids.push_back(pool.intern(path)); + // The blob's ids are the writing session's pool ids: intern its path + // table and remap every decoded id into this session's pool. Ids the + // table does not cover are dropped, not misresolved. + llvm::DenseMap remap; + remap.reserve(index->paths.size()); + for(auto& [id, path]: index->paths) { + remap.try_emplace(id, pool.intern(path)); } - auto to_pool = [&](std::uint32_t local) -> std::optional { - if(local >= pool_ids.size()) { - return std::nullopt; - } - return pool_ids[local]; - }; - - ProjectIndex loaded; - for(auto& [symbol_id, symbol]: repr.symbols) { - auto& target = loaded.symbols[symbol_id]; - target.name = std::move(symbol.name); - target.kind = symbol.kind; - target.scope = symbol.scope; - for(auto local: symbol.reference_files) { - if(auto id = to_pool(local)) { - target.reference_files.add(*id); + for(auto& symbol: llvm::make_second_range(index->symbols)) { + Bitmap remapped; + for(auto id: symbol.reference_files) { + if(auto it = remap.find(id); it != remap.end()) { + remapped.add(it->second); } } + symbol.reference_files = std::move(remapped); } - for(auto local: repr.shards) { - if(auto id = to_pool(local)) { - shards.push_back(*id); + for(auto id: index->shards) { + if(auto it = remap.find(id); it != remap.end()) { + shards.push_back(it->second); } } - return loaded; + // The table and manifest were only the wire form; the runtime state is + // the pool and the caller's shard list. + index->paths.clear(); + index->shards.clear(); + return index; } } // namespace clice::index diff --git a/src/index/project_index.h b/src/index/project_index.h index 97c11eab7..7c0b29a0e 100644 --- a/src/index/project_index.h +++ b/src/index/project_index.h @@ -2,6 +2,9 @@ #include #include +#include +#include +#include #include "index/tu_index.h" #include "support/path_pool.h" @@ -15,24 +18,40 @@ namespace clice::index { /// /// There is a single path-id space at runtime: the server-wide /// clice::PathPool. Symbol reference bitmaps carry those ids directly, so -/// queries never translate between pools. Runtime ids are per-session and -/// never persist — serialization remaps every referenced id into a compact -/// self-contained path table (which is also the garbage collection: paths no -/// longer referenced by any symbol or shard are simply not written), and -/// loading interns the table back into the running pool. +/// queries never translate between pools, and they persist as-is — the blob +/// stays self-contained through a path table mapping every referenced id to +/// its path (which is also the garbage collection: paths no longer +/// referenced by any symbol or shard are simply not written). Loading +/// interns the table into the running pool and remaps every id. +/// +/// Serialization reflects this object directly; `format_version`, `paths` +/// and `shards` are serialize-time state populated by serialize() and +/// consumed by from(). struct ProjectIndex { + /// Persisted-blob schema version (index_format_version), stamped by + /// serialize() and gated by from(). + std::uint32_t format_version = 0; + + /// The blob's self-contained path table: pool id → path for every id + /// the symbol bitmaps and the shard manifest reference. + std::vector> paths; + SymbolTable symbols; + /// Pool ids of the files owning a MergedIndex shard blob, persisted so + /// the loader knows which blobs to fetch. + std::vector shards; + /// Merge a TU's external symbols, interning the TU's paths into `pool`. /// Returns the TU-local id → pool id mapping for the TU's path graph. llvm::SmallVector merge(this ProjectIndex& self, TUIndex& index, clice::PathPool& pool); - /// Serialize with a compact path table covering exactly the ids used by - /// the symbol bitmaps plus `shards`, the pool ids of the files owning a - /// MergedIndex shard blob (persisted so the loader knows what to fetch). - void serialize(this const ProjectIndex& self, + /// Serialize with a path table covering exactly the ids used by the + /// symbol bitmaps plus `shards`, the pool ids of the files owning a + /// MergedIndex shard blob. + void serialize(this ProjectIndex& self, llvm::raw_ostream& os, const clice::PathPool& pool, llvm::ArrayRef shards); diff --git a/src/index/serialization.h b/src/index/serialization.h index 4573fc5ad..b8b6dfb77 100644 --- a/src/index/serialization.h +++ b/src/index/serialization.h @@ -5,15 +5,20 @@ #include #include #include +#include #include +#include #include +#include "index/path_pool.h" #include "index/tu_index.h" #include "semantic/symbol.h" #include "support/bitmap.h" #include "kota/codec/fbs/fbs.h" #include "llvm/ADT/ArrayRef.h" +#include "llvm/ADT/STLExtras.h" +#include "llvm/ADT/StringMap.h" #include "llvm/ADT/StringRef.h" #include "llvm/Support/raw_ostream.h" @@ -51,6 +56,80 @@ struct repr { } }; +/// A PathPool persists as its path table; ids are the dense indices, so +/// interning the table back in order reproduces them. Both directions drive +/// the visitor: encoding writes the interned StringRefs straight to the +/// wire, decoding interns one path at a time. +template <> +struct repr { + using type = std::vector; + + template + static bool serialize(auto& vis, const clice::index::PathPool& pool) { + return codec::encode_value(vis, pool.paths); + } + + template + static bool deserialize(auto& vis, clice::index::PathPool& pool) { + type shape; + return vis.visit_seq(shape, [&](auto& sv) -> bool { + while(sv.has_element()) { + std::string path; + if(!sv.visit_element( + [&](auto& ev) -> bool { return codec::decode_value(ev, path); })) { + return false; + } + // The pool never interns an empty path, so a blob carrying + // one is corrupt; reject it instead of tripping the intern + // precondition. + if(path.empty()) { + return false; + } + pool.path_id(path); + } + return true; + }); + } +}; + +/// A StringMap iterates as StringMapEntry, which no codec understands; +/// persist the entries as key/value pairs, sorted by value for +/// deterministic blobs (values are unique canonical ids). Format-agnostic, +/// unlike the reprs above: the pair-list form is not fbs-specific, and the +/// schema layer classifies fields without a format tag — a format-scoped +/// repr would leave it staring at StringMapEntry, which it rejects. +template <> +struct repr> { + using type = std::vector>; + + template + static bool serialize(auto& vis, const llvm::StringMap& map) { + llvm::SmallVector> entries; + entries.reserve(map.size()); + for(const auto& entry: map) { + entries.emplace_back(entry.getKey(), entry.getValue()); + } + llvm::sort(entries, llvm::less_second{}); + return codec::encode_value(vis, entries); + } + + template + static bool deserialize(auto& vis, llvm::StringMap& map) { + type shape; + return vis.visit_seq(shape, [&](auto& sv) -> bool { + while(sv.has_element()) { + std::pair entry; + if(!sv.visit_element( + [&](auto& ev) -> bool { return codec::decode_value(ev, entry); })) { + return false; + } + map.try_emplace(entry.first, entry.second); + } + return true; + }); + } +}; + template <> struct repr { using type = std::int64_t; @@ -72,7 +151,7 @@ namespace clice::index { /// regular field and every loader discards blobs with a different value — /// including version-less blobs from older builds, which read back as 0. /// Bump it whenever a persisted type's reflected layout changes. -constexpr inline std::uint32_t index_format_version = 2; +constexpr inline std::uint32_t index_format_version = 3; /// Serialize a reflected index blob to `os` as a verified-readable /// flatbuffer. Encoding only fails on structural impossibilities (e.g. more diff --git a/src/server/worker/stateless_worker.cpp b/src/server/worker/stateless_worker.cpp index 032d2fd48..6676cd81a 100644 --- a/src/server/worker/stateless_worker.cpp +++ b/src/server/worker/stateless_worker.cpp @@ -67,7 +67,7 @@ static std::string serialize_preamble_state(CompilationUnit& unit, std::uint32_t std::string blob; llvm::raw_string_ostream os(blob); index::PreambleState::serialize(unit, - tu_index, + std::move(tu_index), links, inactive.regions, inactive.open_stack, diff --git a/tests/unit/index/merged_index_tests.cpp b/tests/unit/index/merged_index_tests.cpp index db90eb033..1052c33f9 100644 --- a/tests/unit/index/merged_index_tests.cpp +++ b/tests/unit/index/merged_index_tests.cpp @@ -1,6 +1,5 @@ #include #include -#include #include #include @@ -422,6 +421,46 @@ TEST_CASE(CompactionDropsMasked) { ASSERT_FALSE(found); } +TEST_CASE(SerializeCompactsInPlace) { + // Two contributions with distinct content, so each gets its own + // canonical id; only one is removed. + index::FileIndex live_idx; + live_idx.occurrences.emplace_back(index::Range{0, 3}, 100); + index::FileIndex dead_idx; + dead_idx.occurrences.emplace_back(index::Range{10, 13}, 200); + + index::MergedIndex merged; + merged.merge("tu0", std::uint32_t(0), live_idx, "synthetic"); + merged.merge("tu1", std::uint32_t(0), dead_idx, "synthetic"); + merged.remove("tu1"); + + llvm::SmallString<1024> buf; + llvm::raw_svector_ostream os(buf); + merged.serialize(os); + + // The save flip is conditional: when it does not happen, the in-memory + // impl — now compacted by serialize() — keeps serving queries. Surviving + // rows must still resolve and removed ones stay gone. + auto hits_at = [&](std::uint32_t offset) { + std::size_t hits = 0; + merged.lookup(offset, [&](const index::Occurrence&) { + hits += 1; + return true; + }); + return hits; + }; + ASSERT_EQ(hits_at(1), 1u); + ASSERT_EQ(hits_at(11), 0u); + ASSERT_TRUE(merged.has_contribution("tu0")); + ASSERT_FALSE(merged.has_contribution("tu1")); + + // A second serialize of the compacted impl round-trips identically. + llvm::SmallString<1024> again; + llvm::raw_svector_ostream os2(again); + merged.serialize(os2); + ASSERT_EQ(llvm::StringRef(buf), llvm::StringRef(again)); +} + TEST_CASE(HasContributionTracking) { add_file("header.h", R"( #pragma once @@ -788,9 +827,9 @@ TEST_CASE(OutOfRangeCanonicalIdRejected) { TempDir dir; // Field order MUST mirror the persisted shapes in merged_index.cpp - // (MergedIndexRepr prefix, HeaderContext, IncludeContext, - // CompilationContext prefix); the trailing fields read back absent, - // which is structurally valid. + // (MergedIndex::Impl prefix — skip-annotated fields occupy no slot — + // HeaderContext, IncludeContext, CompilationContext prefix); the + // trailing fields read back absent, which is structurally valid. struct IncludeContextMirror { std::uint32_t include_id = 0; std::uint32_t canonical_id = 0; @@ -813,11 +852,13 @@ TEST_CASE(OutOfRangeCanonicalIdRejected) { struct ReprMirror { std::uint32_t format_version = 0; - std::uint32_t max_canonical_id = 0; std::vector paths; - std::map canonical_cache; + std::string content; + std::vector line_starts; llvm::SmallDenseMap header_contexts; llvm::SmallDenseMap compilation_contexts; + std::vector> canonical_cache; + std::uint32_t max_canonical_id = 0; }; // A consistent base: one path, one canonical id, one header context @@ -827,7 +868,7 @@ TEST_CASE(OutOfRangeCanonicalIdRejected) { mirror.format_version = index::index_format_version; mirror.max_canonical_id = 1; mirror.paths = {"/proj/tu.cpp"}; - mirror.canonical_cache.emplace("hash", 0); + mirror.canonical_cache.emplace_back("hash", 0); mirror.header_contexts[0].includes.push_back({.include_id = 0, .canonical_id = 0}); return mirror; }; @@ -860,7 +901,7 @@ TEST_CASE(OutOfRangeCanonicalIdRejected) { // index canonical_ref_counts out of bounds if the in-memory load // accepted it. The blob is dropped and the shard reads as empty. auto bad_cache = base(); - bad_cache.canonical_cache["hash"] = 5; + bad_cache.canonical_cache.front().second = 5; auto cache_verdict = materialized_contribution("bad-cache.idx", bad_cache); ASSERT_TRUE(cache_verdict.has_value() && !*cache_verdict); diff --git a/tests/unit/index/persisted_index_tests.cpp b/tests/unit/index/persisted_index_tests.cpp index 54384bde4..341c1c718 100644 --- a/tests/unit/index/persisted_index_tests.cpp +++ b/tests/unit/index/persisted_index_tests.cpp @@ -110,21 +110,23 @@ TEST_CASE(VersionGate) { } TEST_CASE(OutOfRangeLocalIdsDropped) { - // Field order MUST mirror ProjectIndexRepr (project_index.cpp): + // Field order MUST mirror ProjectIndex (project_index.h): // format_version, paths, symbols, shards. struct ProjectIndexMirror { std::uint32_t format_version = 0; - std::vector paths; + std::vector> paths; index::SymbolTable symbols; std::vector shards; }; ProjectIndexMirror mirror; mirror.format_version = index::index_format_version; - mirror.paths = {"/proj/used.cpp"}; + mirror.paths = { + {0, "/proj/used.cpp"} + }; auto& symbol = mirror.symbols[42]; symbol.name = "sym"; - symbol.reference_files.add(7); // Only local id 0 exists. + symbol.reference_files.add(7); // Only pool id 0 is in the table. mirror.shards = {9}; auto blob = kota::codec::fbs::to_bytes(mirror); diff --git a/tests/unit/index/preamble_state_tests.cpp b/tests/unit/index/preamble_state_tests.cpp index 6efbf848b..df0004675 100644 --- a/tests/unit/index/preamble_state_tests.cpp +++ b/tests/unit/index/preamble_state_tests.cpp @@ -171,6 +171,48 @@ int main() { §(ref)⟦§(ref)foo⟧(); return 0; } EXPECT_TRUE(found_relation); } +TEST_CASE(MoveConsumedIndex) { + // The production path (stateless worker) moves the TUIndex into + // serialize; the blob must be complete even though the index is + // consumed rather than copied. + add_file("foo.h", R"( +inline void §(def)⟦foo⟧() {} +)"); + add_main("main.cpp", R"( +#include "foo.h" +int main() { §(ref)⟦foo⟧(); return 0; } +)"); + ASSERT_TRUE(compile()); + tu_index = index::TUIndex::build(*unit); + auto foo = hash_of("foo"); + + auto blob_path = dir.path("moved.pch.idx"); + std::error_code ec; + llvm::raw_fd_ostream os(blob_path, ec); + ASSERT_FALSE(bool(ec)); + index::PreambleState::serialize(*unit, std::move(tu_index), {}, {}, {}, os); + os.close(); + + state = index::PreambleState::load(blob_path); + ASSERT_TRUE(state != nullptr); + + bool found = false; + state->lookup(foo, + RelationKind::Definition, + [&](const index::PreambleState::File& file, const index::Relation& r) { + EXPECT_TRUE(file.path.ends_with("foo.h")); + EXPECT_EQ(dump(r.range), dump(range("def", "foo.h"))); + found = true; + return false; + }); + EXPECT_TRUE(found); + + std::string name; + SymbolKind kind; + EXPECT_TRUE(state->find_symbol(foo, name, kind)); + EXPECT_EQ(name, "foo"); +} + TEST_CASE(SymbolTableLookup) { add_file("foo.h", R"( inline void §(def)⟦foo⟧() {} From 94cb91853ce987dfd7076ba73570756a7c69956f Mon Sep 17 00:00:00 2001 From: ykiko Date: Thu, 13 Aug 2026 12:52:23 +0800 Subject: [PATCH 12/12] review: reduced preamble symbol entry, reject empty blob paths Review round: the preamble blob stores a borrowed {name, kind} entry per symbol instead of reflecting the full Symbol (scope and reference bitmaps were dead weight on the wire), and ProjectIndex::from rejects a corrupt blob whose path table carries an empty entry instead of interning it. --- src/index/preamble_state.cpp | 21 +++++++++++++++++---- src/index/project_index.cpp | 6 ++++++ tests/unit/index/project_index_tests.cpp | 18 ++++++++++++++++++ 3 files changed, 41 insertions(+), 4 deletions(-) diff --git a/src/index/preamble_state.cpp b/src/index/preamble_state.cpp index 0a5111513..0724d3b94 100644 --- a/src/index/preamble_state.cpp +++ b/src/index/preamble_state.cpp @@ -25,6 +25,16 @@ struct PreambleFileEntry { std::vector line_starts; }; +/// find_symbol serves only name and kind, so the blob stores this reduced +/// entry instead of the full Symbol — reflecting that would drag every +/// symbol's scope and reference bitmap into large SDK preamble blobs for +/// nothing. The name borrows the consumed TUIndex (encode-only, like +/// PreambleFileEntry). +struct PreambleSymbol { + llvm::StringRef name; + SymbolKind kind; +}; + /// The persisted shape of a `.pch.idx` blob. Queries run on a zero-copy /// view of this layout; nothing is deserialized up front. struct PreambleBlob { @@ -32,7 +42,7 @@ struct PreambleBlob { std::vector paths; std::vector files; PreambleFileEntry preamble; - SymbolTable symbols; + llvm::DenseMap symbols; llvm::ArrayRef links; llvm::ArrayRef inactive_regions; llvm::ArrayRef open_conditionals; @@ -101,7 +111,10 @@ void PreambleState::serialize(CompilationUnitRef unit, std::string_view(preamble_text.data(), preamble_text.size())), }; - blob.symbols = std::move(index.symbols); + blob.symbols.reserve(index.symbols.size()); + for(const auto& [hash, symbol]: index.symbols) { + blob.symbols.try_emplace(hash, PreambleSymbol{.name = symbol.name, .kind = symbol.kind}); + } blob.paths = std::move(index.graph.paths); blob.links = links; blob.inactive_regions = inactive_regions; @@ -224,8 +237,8 @@ bool PreambleState::find_symbol(SymbolHash hash, std::string& name, SymbolKind& } auto symbol = found->get<1>(); - name = std::string(symbol[&Symbol::name]); - kind = SymbolKind(symbol[&Symbol::kind]); + name = std::string(symbol[&PreambleSymbol::name]); + kind = SymbolKind(symbol[&PreambleSymbol::kind]); return true; } diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index 32a351d61..567f93e47 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -69,6 +69,12 @@ std::optional ProjectIndex::from(llvm::StringRef data, llvm::DenseMap remap; remap.reserve(index->paths.size()); for(auto& [id, path]: index->paths) { + // The writer only emits interned paths, which are never empty; an + // empty entry marks a corrupt blob and must not become a real pool + // entry. + if(path.empty()) { + return std::nullopt; + } remap.try_emplace(id, pool.intern(path)); } diff --git a/tests/unit/index/project_index_tests.cpp b/tests/unit/index/project_index_tests.cpp index 3adca5f21..5e795bddd 100644 --- a/tests/unit/index/project_index_tests.cpp +++ b/tests/unit/index/project_index_tests.cpp @@ -1,6 +1,7 @@ #include "test/test.h" #include "test/tester.h" #include "index/project_index.h" +#include "index/serialization.h" namespace clice::testing { namespace { @@ -243,6 +244,23 @@ TEST_CASE(LocalSymbolsExcluded) { ASSERT_FALSE(found_local); } +TEST_CASE(EmptyPathRejected) { + // A codec-valid blob whose path table carries an empty entry is corrupt: + // the writer only emits interned (never empty) paths. + index::ProjectIndex corrupt; + corrupt.format_version = index::index_format_version; + corrupt.paths.emplace_back(0, ""); + + llvm::SmallString<128> buf; + llvm::raw_svector_ostream os(buf); + index::serialize_blob(corrupt, os); + + clice::PathPool pool; + llvm::SmallVector shards; + ASSERT_FALSE(index::ProjectIndex::from(buf.str(), pool, shards).has_value()); + ASSERT_TRUE(pool.paths.empty()); +} + TEST_CASE(ScopeRoundTrip) { index::TUIndex tu; ASSERT_TRUE(build_and_index(R"(