diff --git a/CMakeLists.txt b/CMakeLists.txt index 02fbb8e2d..7a2f35448 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -152,24 +152,6 @@ if(CLICE_CI_ENVIRONMENT) target_compile_definitions(clice_options INTERFACE CLICE_CI_ENVIRONMENT=1) endif() -set(FBS_SCHEMA_FILE "${PROJECT_SOURCE_DIR}/src/index/schema.fbs") -set(GENERATED_HEADER "${PROJECT_BINARY_DIR}/generated/schema_generated.h") - -# kotatsu builds the flatbuffers runtime but not its schema compiler, so flatc -# comes from the pixi environment (conda-forge flatbuffers). Its version must -# match the headers kotatsu fetches — the generated header static_asserts on -# that, so a mismatch surfaces as a compile error in schema_generated.h. -find_program(FLATC_EXECUTABLE flatc REQUIRED) - -add_custom_command( - OUTPUT "${GENERATED_HEADER}" - COMMAND ${FLATC_EXECUTABLE} --cpp -o "${PROJECT_BINARY_DIR}/generated" "${FBS_SCHEMA_FILE}" - DEPENDS "${FBS_SCHEMA_FILE}" "${FLATC_EXECUTABLE}" - COMMENT "Generating C++ header from ${FBS_SCHEMA_FILE}" -) - -add_custom_target(generate_flatbuffers_schema DEPENDS "${GENERATED_HEADER}") - # Version header, regenerated on every build so the embedded git describe # tracks the checked-out commit (content-unchanged writes are elided). set(CLICE_VERSION_HEADER "${PROJECT_BINARY_DIR}/generated/version.h") @@ -192,7 +174,7 @@ add_custom_target(generate_version_header file(GLOB_RECURSE CLICE_CORE_SOURCES CONFIGURE_DEPENDS "${PROJECT_SOURCE_DIR}/src/*.cpp") add_library(clice-core STATIC ${CLICE_CORE_SOURCES}) add_library(clice::core ALIAS clice-core) -add_dependencies(clice-core generate_flatbuffers_schema generate_version_header) +add_dependencies(clice-core generate_version_header) target_include_directories(clice-core PUBLIC "${PROJECT_SOURCE_DIR}/src" diff --git a/cmake/package.cmake b/cmake/package.cmake index 1f649512a..00f77d2fc 100644 --- a/cmake/package.cmake +++ b/cmake/package.cmake @@ -30,7 +30,7 @@ set(ENABLE_ROARING_MICROBENCHMARKS OFF CACHE INTERNAL "" FORCE) FetchContent_Declare( kotatsu GIT_REPOSITORY https://github.com/clice-io/kotatsu - GIT_TAG af2b6d1c2bf19d5b9cd643e7dfa43739bfffb361 + GIT_TAG 5232e67c28fee3a85ad01341a675dd4b78b122b1 ) set(KOTA_ENABLE_ZEST ON) @@ -38,10 +38,8 @@ set(KOTA_ENABLE_TEST OFF) set(KOTA_CODEC_ENABLE_SIMDJSON ON) set(KOTA_CODEC_ENABLE_YYJSON ON) set(KOTA_CODEC_ENABLE_TOML ON) -# kotatsu already fetches flatbuffers (v25.2.10) for its own codec and links the -# runtime lib into anything that uses kota::codec, so clice rides on that copy -# instead of fetching a second one. kotatsu does not build flatc, though, so the -# schema compiler comes from pixi instead (see CMakeLists.txt). +# kotatsu fetches the flatbuffers runtime (v25.2.10) for its codec and links it +# into anything that uses kota::codec; index serialization rides on that copy. set(KOTA_CODEC_ENABLE_FLATBUFFERS ON) set(KOTA_ENABLE_EXCEPTIONS OFF) set(KOTA_ENABLE_RTTI OFF) diff --git a/package-lock.json b/package-lock.json index 467a7a7a7..e9a87f67d 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1679,7 +1679,6 @@ "integrity": "sha512-zWW5KPngR/yvakJgGOmZ5vTBemDoSqF3AcV/LrO5u5wTWyEAVVh+IT39G4gtyAkh3CtTZs8aX/yRM82OfzHJRg==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "undici-types": "~7.16.0" } @@ -1711,7 +1710,6 @@ "integrity": "sha512-p088eaGrzYz1s+7cov0aMOCkNGTJlVxF4jgubf28c8L0Cv9Rloj8YBHnv4hXLq6IIEE1AsjNWavO+k+8kP2Y0A==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@eslint-community/regexpp": "^4.12.2", "@typescript-eslint/scope-manager": "8.66.0", @@ -1751,7 +1749,6 @@ "integrity": "sha512-X6ypGChaWYk6PBtUg2BwuTZEFFcHJAtGTVJ9/lCTOufhZ4i9fNolQNnktq+kkMCwMj7V8Svsq7+TxSDslmhE0g==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@typescript-eslint/scope-manager": "8.66.0", "@typescript-eslint/types": "8.66.0", @@ -2572,7 +2569,6 @@ "integrity": "sha512-lGq+9yr1/GuAWaVYIHRjvvySG5/4VfKIvC8EWxStPdcDh/Ka7FG3twP6v4d5BkravUilhIAsG4Qj83t02LWUPQ==", "dev": true, "license": "MIT", - "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -2606,7 +2602,6 @@ "integrity": "sha512-Thbli+OlOj+iMPYFBVBfJ3OmCAnaSyNn4M1vz9T6Gka5Jt9ba/HIR56joy65tY6kx/FCF5VXNB819Y7/GUrBGA==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "fast-deep-equal": "^3.1.3", "fast-uri": "^3.0.1", @@ -2896,7 +2891,6 @@ } ], "license": "MIT", - "peer": true, "dependencies": { "baseline-browser-mapping": "^2.10.44", "caniuse-lite": "^1.0.30001806", @@ -3976,7 +3970,6 @@ "integrity": "sha512-nuKKvN+oIBO0koN7Tm7dlkmnkc21mtt0QJLwAKzjLq14y6lRTdVG36MZHJ8eQHwdJMwZbQNMlPOYedMq/oVJvQ==", "dev": true, "license": "MIT", - "peer": true, "workspaces": [ "packages/*" ], @@ -7870,7 +7863,6 @@ "integrity": "sha512-7A2xlQ5EnGT8KPA92dUh6RbRYTVw8hEaEN9L1K68l4UOXFuV511NnAqObGoRqGOQofQcMypisu1s3xawCEHrvA==", "dev": true, "license": "BSD-2-Clause", - "peer": true, "dependencies": { "@jridgewell/source-map": "^0.3.3", "acorn": "^8.15.0", @@ -7984,7 +7976,6 @@ "integrity": "sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==", "dev": true, "license": "MIT", - "peer": true, "engines": { "node": ">=12" }, @@ -8171,7 +8162,6 @@ "integrity": "sha512-Mtq29sKDAEYP7aljRgtPOpTvOfbwRWlS6dPRzwjdE+C0R4brX/GUyhHSecbHMFLNBLcJIPt9nl9yG5TZ1weH+Q==", "dev": true, "license": "Apache-2.0", - "peer": true, "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" @@ -8358,7 +8348,6 @@ "integrity": "sha512-4XP60spRGjSZFf1qYH+dJIkK2znL3zQfl9KkOV9MkkRR/3Dls0dxaBsQPTloEc5BLXWPL9vsOxopxyKoMmDueg==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "esbuild": "^0.27.0 || ^0.28.0", "fdir": "^6.5.0", @@ -8475,7 +8464,6 @@ "integrity": "sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==", "dev": true, "license": "MIT", - "peer": true, "engines": { "node": ">=12" }, @@ -8489,7 +8477,6 @@ "integrity": "sha512-KrxIJ62Fd89gfysR4WotlgZABiz2dqFPgqGzX7s+CwsqLFomRH7777ZcrOD6+WVAh7khPQP41A+BKbpcJFrdEg==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@types/chai": "^5.2.2", "@vitest/expect": "3.2.7", @@ -8695,7 +8682,6 @@ "integrity": "sha512-U9/cvLzxObKNEZ9+TtdqrHM5/9z3lgl2c+c4BzbqGxFQvQvBAq87yql5A8pQ+rrMbS496MZJeF5enVBndIy2hw==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@types/estree": "^1.0.8", "@types/json-schema": "^7.0.15", @@ -8740,7 +8726,6 @@ "integrity": "sha512-MfwFQ6SfwinsUVi0rNJm7rHZ31GyTcpVE5pgVA3hwFRb7COD4TzjUUwhGWKfO50+xdc2MQPuEBBJoqIMGt3JDw==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@discoveryjs/json-ext": "^0.6.1", "@webpack-cli/configtest": "^3.0.1", diff --git a/pixi.lock b/pixi.lock index 103c2ac7d..a509bc259 100644 --- a/pixi.lock +++ b/pixi.lock @@ -57,7 +57,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt-22.1.8-ha770c72_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt22-22.1.8-hb700be7_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/conda-gcc-specs-15.2.0-h53410ce_20.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-h8ddf172_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-aarch64-15.2.0-he60e42a_20.conda @@ -146,7 +145,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/cmake-4.4.0-hc9d863e_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt-22.1.8-h8af1aa0_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt22-22.1.8-hfefdfc9_1.conda - - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/icu-78.3-hcab7f73_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/keyutils-1.6.3-h86ecc28_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/krb5-1.22.2-h2fb54aa_1.conda @@ -224,7 +222,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/cmake-4.4.0-h2426fb6_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt-22.1.8-h694c41f_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt22-22.1.8-h1637cdf_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/krb5-1.22.2-h3ddfcb2_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/ld64-956.6-llvm22_1_hc399b6d_4.conda @@ -294,7 +291,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/cmake-4.4.0-h8cb302d_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt-22.1.8-hce30654_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt22-22.1.8-hd34ed20_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-78.3-hef89b57_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/krb5-1.22.2-hfd3d5f3_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/ld64-956.6-llvm22_1_h5b97f1b_4.conda @@ -356,7 +352,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/cmake-4.4.0-hdcbee5b_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt-22.1.8-h57928b3_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt22-22.1.8-h49e36cd_1.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/icu-78.3-h637d24d_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/krb5-1.22.2-h719d79b_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libclang13-22.1.8-default_hf735972_3.conda @@ -415,7 +410,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt-22.1.8-ha770c72_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt22-22.1.8-hb700be7_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/conda-gcc-specs-15.2.0-h53410ce_20.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-h8ddf172_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gxx-15.2.0-h834e499_7.conda @@ -494,7 +488,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/cmake-4.3.1-hc9d863e_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt-22.1.8-h8af1aa0_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt22-22.1.8-hfefdfc9_1.conda - - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/keyutils-1.6.3-h86ecc28_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/krb5-1.22.2-hfd895c2_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/ld_impl_linux-aarch64-2.45.1-default_h1979696_102.conda @@ -570,7 +563,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/cmake-4.3.1-h2426fb6_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt-22.1.8-h694c41f_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt22-22.1.8-h1637cdf_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/krb5-1.22.2-h207b36a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/ld64-956.6-llvm22_1_hc399b6d_4.conda @@ -639,7 +631,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/cmake-4.3.1-h8cb302d_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt-22.1.8-hce30654_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt22-22.1.8-hd34ed20_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-78.3-hef89b57_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/krb5-1.22.2-h385eeb1_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/ld64-956.6-llvm22_1_h5b97f1b_4.conda @@ -700,7 +691,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/cmake-4.3.1-hdcbee5b_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt-22.1.8-h57928b3_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt22-22.1.8-h49e36cd_1.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/krb5-1.22.2-h0ea6238_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libclang13-22.1.8-default_hf735972_3.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libcompiler-rt-22.1.8-h49e36cd_1.conda @@ -758,7 +748,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt-22.1.8-ha770c72_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt22-22.1.8-hb700be7_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/conda-gcc-specs-15.2.0-h53410ce_20.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-h8ddf172_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gxx-15.2.0-h834e499_7.conda @@ -843,7 +832,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/cmake-4.3.1-hc9d863e_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt-22.1.8-h8af1aa0_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt22-22.1.8-hfefdfc9_1.conda - - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/icu-78.3-h7ac5ae9_2.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/keyutils-1.6.3-h86ecc28_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/krb5-1.22.2-hfd895c2_0.conda @@ -925,7 +913,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/cmake-4.3.1-h2426fb6_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt-22.1.8-h694c41f_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt22-22.1.8-h1637cdf_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/krb5-1.22.2-h207b36a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/ld64-956.6-llvm22_1_hc399b6d_4.conda @@ -999,7 +986,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/cmake-4.2.1-h54ad630_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt-22.1.8-hce30654_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt22-22.1.8-hd34ed20_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-78.3-hc7cc350_2.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/krb5-1.21.3-h237132a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/ld64-956.6-llvm22_1_h5b97f1b_4.conda @@ -1065,7 +1051,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/cmake-4.2.1-hdcbee5b_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt-22.1.8-h57928b3_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt22-22.1.8-h49e36cd_1.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/icu-78.1-h637d24d_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/krb5-1.21.3-hdf4eb48_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libclang13-22.1.8-default_hf735972_3.conda @@ -1615,7 +1600,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt-22.1.8-ha770c72_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/compiler-rt22-22.1.8-hb700be7_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/conda-gcc-specs-15.2.0-h53410ce_20.conda - - conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc_impl_linux-64-15.2.0-h8ddf172_20.conda - conda: https://conda.anaconda.org/conda-forge/linux-64/gxx-15.2.0-h834e499_7.conda @@ -1700,7 +1684,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/cmake-4.3.1-hc9d863e_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt-22.1.8-h8af1aa0_1.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/compiler-rt22-22.1.8-hfefdfc9_1.conda - - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/icu-78.3-h7ac5ae9_2.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/keyutils-1.6.3-h86ecc28_0.conda - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/krb5-1.22.2-hfd895c2_0.conda @@ -1782,7 +1765,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-64/cmake-4.3.1-h2426fb6_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt-22.1.8-h694c41f_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/compiler-rt22-22.1.8-h1637cdf_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/krb5-1.22.2-h207b36a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-64/ld64-956.6-llvm22_1_hc399b6d_4.conda @@ -1856,7 +1838,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/osx-arm64/cmake-4.2.1-h54ad630_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt-22.1.8-hce30654_1.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/compiler-rt22-22.1.8-hd34ed20_1.conda - - conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-78.3-hc7cc350_2.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/krb5-1.21.3-h237132a_0.conda - conda: https://conda.anaconda.org/conda-forge/osx-arm64/ld64-956.6-llvm22_1_h5b97f1b_4.conda @@ -1922,7 +1903,6 @@ environments: - conda: https://conda.anaconda.org/conda-forge/win-64/cmake-4.2.1-hdcbee5b_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt-22.1.8-h57928b3_1.conda - conda: https://conda.anaconda.org/conda-forge/win-64/compiler-rt22-22.1.8-h49e36cd_1.conda - - conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/krb5-1.21.3-hdf4eb48_0.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libclang13-22.1.8-default_hf735972_3.conda - conda: https://conda.anaconda.org/conda-forge/win-64/libcompiler-rt-22.1.8-h49e36cd_1.conda @@ -2707,21 +2687,6 @@ packages: license_family: MIT size: 1209421 timestamp: 1757336717570 -- conda: https://conda.anaconda.org/conda-forge/linux-64/flatbuffers-25.2.10-hb7832b1_0.conda - sha256: 0e58114d0e16bc89b94ef9068558e304d2eccae5dbaa55b955274ea60da81dfd - md5: 279ba9719d1afc81538d8260f31e42a0 - depends: - - __glibc >=2.17,<3.0.a0 - - libgcc >=13 - - libstdcxx >=13 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1539958 - timestamp: 1747130572350 - conda: https://conda.anaconda.org/conda-forge/linux-64/gcc-15.2.0-hc6a0c74_20.conda sha256: 559aebbc0b6754681c7eec0b6ef8924a13eaa4c110510e99b8224ef3020fed69 md5: 07a50d62feaa7d8656ee442bdc2c0e8d @@ -4856,20 +4821,6 @@ packages: license_family: MIT size: 1113801 timestamp: 1773352768018 -- conda: https://conda.anaconda.org/conda-forge/linux-aarch64/flatbuffers-25.2.10-ha90f286_0.conda - sha256: 0d802dd9a8b804521a25ee21423a674d73d5ac6cecc2faae4264b5286f9d2deb - md5: 2093f2029d159ec0dc522f42990c0bd2 - depends: - - libgcc >=13 - - libstdcxx >=13 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1380724 - timestamp: 1747130553663 - conda: https://conda.anaconda.org/conda-forge/linux-aarch64/icu-78.3-h7ac5ae9_2.conda sha256: 424a4d54715d11f7fdf033c2ba2fd1b19d6ebd069a15e007ea467b719af197b3 md5: 9a2f5f714c8962bfa551cd9c887b1029 @@ -6765,20 +6716,6 @@ packages: license_family: MIT size: 1133353 timestamp: 1773353161332 -- conda: https://conda.anaconda.org/conda-forge/osx-64/flatbuffers-25.2.10-h2cf7b43_0.conda - sha256: eb6be3a3db53cb53f9300f08cfd6579549787e6ec45007d589f4629fec1b9a42 - md5: 109d4025e003f228844a06f246503177 - depends: - - __osx >=10.13 - - libcxx >=18 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1337567 - timestamp: 1747130405020 - conda: https://conda.anaconda.org/conda-forge/osx-64/icu-78.3-h25d91c4_0.conda sha256: 1294117122d55246bb83ad5b589e2a031aacdf2d0b1f99fd338aa4394f881735 md5: 627eca44e62e2b665eeec57a984a7f00 @@ -8182,20 +8119,6 @@ packages: license_family: MIT size: 1050638 timestamp: 1757337263602 -- conda: https://conda.anaconda.org/conda-forge/osx-arm64/flatbuffers-25.2.10-h3144c11_0.conda - sha256: d339e7b15c6a927b6ecdb27513d001ab037e3d4bb146fa498e330cbec0cdf9fe - md5: 87c66c4a31165b25b9f56da755197a64 - depends: - - __osx >=11.0 - - libcxx >=18 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1286290 - timestamp: 1747130536643 - conda: https://conda.anaconda.org/conda-forge/osx-arm64/icu-75.1-hfee45f7_0.conda sha256: 9ba12c93406f3df5ab0a43db8a4b4ef67a5871dfd401010fbe29b218b2cbe620 md5: 5eb22c1d7b3fc4abb50d92d621583137 @@ -9850,21 +9773,6 @@ packages: license_family: MIT size: 1196708 timestamp: 1757337405047 -- conda: https://conda.anaconda.org/conda-forge/win-64/flatbuffers-25.2.10-hc130f0a_0.conda - sha256: 8c26cca2271d99e8b723847c3a3a7e7de3f5f1908dbd1d2413e6b0b154b97d47 - md5: 29353e2ac55f6192b1a5bb0244021128 - depends: - - ucrt >=10.0.20348.0 - - vc >=14.2,<15 - - vc14_runtime >=14.29.30139 - license: Apache-2.0 - license_family: APACHE - purls: [] - run_exports: - weak: - - flatbuffers >=25.2.10,<25.2.11.0a0 - size: 1753609 - timestamp: 1747130826577 - conda: https://conda.anaconda.org/conda-forge/win-64/icu-78.1-h637d24d_0.conda sha256: bee083d5a0f05c380fcec1f30a71ef5518b23563aeb0a21f6b60b792645f9689 md5: cb8048bed35ef01431184d6a88e46b3e diff --git a/pixi.toml b/pixi.toml index f1b001b20..284973a8a 100644 --- a/pixi.toml +++ b/pixi.toml @@ -42,8 +42,6 @@ lld = "==22.1.8" llvm-tools = "==22.1.8" clang-tools = "==22.1.8" compiler-rt = "==22.1.8" -# Must match the headers kotatsu fetches — generated code static_asserts on it. -flatbuffers = "==25.2.10" # cmake/archive.cmake pipes tar through xz for the symbol archives. xz = ">=5.8.1,<6" diff --git a/src/index/include_graph.h b/src/index/include_graph.h index 98746888f..f83a75f9e 100644 --- a/src/index/include_graph.h +++ b/src/index/include_graph.h @@ -7,6 +7,8 @@ #include "syntax/token.h" +#include "kota/codec/macro.h" +#include "kota/meta/annotation.h" #include "llvm/ADT/ArrayRef.h" #include "llvm/ADT/DenseMap.h" @@ -51,8 +53,11 @@ struct IncludeGraph { /// Each `FileID` represents a new header context and is introduced /// by a new include directive. So a include directive is a new header - /// context. A map between FileID and its include location. - llvm::DenseMap file_table; + /// context. A map between FileID and its include location. Build-time + /// only: FileIDs mean nothing outside the compilation, so the field is + /// excluded from serialization. + KOTATSU_ANNOTATE(skip = true) + > file_table; /// Build the graph for `unit`. The file table covers the union of /// every file included in this parse (from the replayed directives) diff --git a/src/index/merged_index.cpp b/src/index/merged_index.cpp index f64c07a1e..6fc7eddef 100644 --- a/src/index/merged_index.cpp +++ b/src/index/merged_index.cpp @@ -2,6 +2,7 @@ #include #include +#include #include #include "compile/dep_file.h" @@ -11,6 +12,7 @@ #include "kota/ipc/lsp/position.h" #include "llvm/ADT/DenseSet.h" +#include "llvm/ADT/STLExtras.h" #include "llvm/Support/raw_os_ostream.h" #include "llvm/Support/xxhash.h" @@ -49,7 +51,7 @@ struct DenseMapInfo { inline static R getEmptyKey() { return R{ - .kind = clice::RelationKind(), + .kind = clice::RelationKind::Invalid, .range = clice::LocalSourceRange(-1, 0), .target_symbol = 0, }; @@ -57,7 +59,7 @@ struct DenseMapInfo { inline static R getTombstoneKey() { return R{ - .kind = clice::RelationKind(), + .kind = clice::RelationKind::Invalid, .range = clice::LocalSourceRange(-2, 0), .target_symbol = 0, }; @@ -65,7 +67,7 @@ struct DenseMapInfo { /// Contextual doesn’t take part in hashing and equality. static auto getHashValue(const R& relation) { - return dense_hash(relation.kind.value(), + return dense_hash(static_cast(relation.kind), relation.range.begin, relation.range.end, relation.target_symbol); @@ -82,7 +84,6 @@ struct DenseMapInfo { namespace clice::index { /// (path_id, content_hash) captured for one dependency at index-build time. -/// Layout must mirror `binary::DepHash` so `safe_cast` can alias it. struct DepHash { std::uint32_t path_id; std::uint64_t content_hash; @@ -92,7 +93,6 @@ struct DepHash { /// Stat fast path for one dependency, recorded at merge time only when the /// file provably did not change since before the indexed build started. -/// Layout must mirror `binary::DepStamp` so `safe_cast` can alias it. /// The server's mutable, in-place-repairing form of the same fast path is /// `DepState` (server/state/workspace.h). struct DepStamp { @@ -121,7 +121,7 @@ std::uint64_t hash_file(llvm::StringRef path) { /// cannot masquerade as fresh; Layer 2 re-hashes the disk against the /// consumed-content hash and treats a match as a mere touch, not an edit. bool dep_stale(llvm::StringRef path, - const DepStamp* stamp, + const std::optional& stamp, std::optional stored_hash) { fs::file_status status; if(auto err = fs::status(path, status)) { @@ -187,6 +187,11 @@ struct CompilationContext { }; struct MergedIndex::Impl { + /// On-disk shard schema version (index_format_version), stamped by + /// serialize() and gated by load(); version-less blobs from older builds + /// read back as 0. + std::uint32_t format_version = 0; + /// Shard-local path table: every path id stored in this shard indexes /// into it, so shards are self-contained across sessions (runtime pool /// ids never persist). @@ -215,11 +220,15 @@ struct MergedIndex::Impl { /// The max canonical id we have allocated. std::uint32_t max_canonical_id = 0; - /// The reference count of each canonical id. - std::vector canonical_ref_counts; + /// The reference count of each canonical id. Derived state: rebuilt from + /// the context tables when a blob loads in memory. + KOTATSU_ANNOTATE(skip = true) + > canonical_ref_counts; - /// The canonical id set of removed index. - roaring::Roaring removed; + /// The canonical id set of removed index. Never persisted: compact() + /// erases the masked rows for real before a shard reaches disk. + KOTATSU_ANNOTATE(skip = true) + removed; /// All merged symbol occurrences. llvm::DenseMap occurrences; @@ -231,7 +240,8 @@ struct MergedIndex::Impl { SymbolTable symbols; /// Sorted occurrences cache for fast lookup. - std::vector occurrences_cache; + KOTATSU_ANNOTATE(skip = true) + > occurrences_cache; /// Drop one reference to a canonical index; the last reference masks its /// occurrences and relations via the removed bitmap. A later re-merge of @@ -273,9 +283,74 @@ struct MergedIndex::Impl { self.max_canonical_id += 1; } + /// Erase the rows masked by the removed bitmap for real. Queries are + /// unaffected (masked rows were already invisible), but a later re-merge + /// of identical content mints a fresh canonical instead of resurrecting + /// the id — the same behavior a save/load cycle produces. + void compact(this Impl& self) { + if(self.removed.isEmpty()) { + return; + } + + llvm::SmallVector dead_hashes; + for(const auto& entry: self.canonical_cache) { + if(self.removed.contains(entry.getValue())) { + dead_hashes.push_back(entry.getKey()); + } + } + for(auto hash: dead_hashes) { + self.canonical_cache.erase(hash); + } + + llvm::SmallVector dead_occurrences; + for(auto& [occurrence, bitmap]: self.occurrences) { + bitmap -= self.removed; + if(bitmap.isEmpty()) { + dead_occurrences.push_back(occurrence); + } + } + for(auto& occurrence: dead_occurrences) { + self.occurrences.erase(occurrence); + } + + llvm::SmallVector dead_symbols; + for(auto& [symbol, entries]: self.relations) { + llvm::SmallVector dead_relations; + for(auto& [relation, bitmap]: entries) { + bitmap -= self.removed; + if(bitmap.isEmpty()) { + dead_relations.push_back(relation); + } + } + for(auto& relation: dead_relations) { + entries.erase(relation); + } + if(entries.empty()) { + dead_symbols.push_back(symbol); + } + } + for(auto symbol: dead_symbols) { + self.relations.erase(symbol); + } + + self.removed = roaring::Roaring(); + self.occurrences_cache.clear(); + } + friend bool operator==(const Impl&, const Impl&) = default; }; +namespace { + +using ShardView = kota::codec::fbs::table_view; + +/// The blob was fully verified at load(); per-query views skip that cost. +ShardView root_of(const llvm::MemoryBuffer& buffer) { + return ShardView::from_verified_bytes(blob_bytes(buffer.getBuffer())); +} + +} // namespace + MergedIndex::MergedIndex(std::unique_ptr buffer, std::unique_ptr impl) : buffer(std::move(buffer)), impl(std::move(impl)) {} @@ -311,102 +386,53 @@ void MergedIndex::load_in_memory(this Self& self) { } auto& index = *self.impl; - auto root = fbs::GetRoot(self.buffer->getBufferStart()); - - index.max_canonical_id = root->max_canonical_id(); - - if(root->paths()) { - for(auto path: *root->paths()) { - index.paths.path_id(path->string_view()); - } - } - - for(auto entry: *root->canonical_cache()) { - index.canonical_cache.try_emplace(entry->sha256()->string_view(), entry->canonical_id()); - } - - index.canonical_ref_counts.resize(index.max_canonical_id, 0); - - for(auto entry: *root->header_contexts()) { - HeaderContext context; - auto path = entry->path_id(); - context.version = entry->version(); - for(auto include: *entry->includes()) { - index.canonical_ref_counts[include->canonical_id()] += 1; - context.includes.emplace_back(*safe_cast(include)); + // The buffer's structure was verified at load(), but structural + // verification does not constrain field values: a canonical id at or + // past max_canonical_id would index canonical_ref_counts out of bounds + // below (and in every later release_canonical). A blob carrying one — + // like a blob that fails to decode outright — is dropped, so the shard + // reads as empty and the background indexer rebuilds it. + auto usable = [&] { + if(!deserialize_blob(self.buffer->getBuffer(), index)) { + return false; } - index.header_contexts.try_emplace(path, std::move(context)); - } - - for(auto entry: *root->compilation_contexts()) { - CompilationContext context; - auto path = entry->path_id(); - context.version = entry->version(); - context.canonical_id = entry->canonical_id(); - context.build_at = entry->build_at(); - for(auto include: *entry->include_locations()) { - context.include_locations.emplace_back(*safe_cast(include)); + auto in_range = [&](std::uint32_t canonical_id) { + return canonical_id < index.max_canonical_id; + }; + for(const auto& entry: index.canonical_cache) { + if(!in_range(entry.getValue())) { + return false; + } } - if(entry->dep_hashes()) { - for(auto dep: *entry->dep_hashes()) { - context.dep_hashes.emplace_back(*safe_cast(dep)); + for(auto& context: llvm::make_second_range(index.header_contexts)) { + for(auto& include: context.includes) { + if(!in_range(include.canonical_id)) { + return false; + } } } - // Absent from shards written before the stamp field existed: their - // deps simply have no fast path and validate by hash. - if(entry->dep_stamps()) { - for(auto stamp: *entry->dep_stamps()) { - context.dep_stamps.emplace_back(*safe_cast(stamp)); + for(auto& context: llvm::make_second_range(index.compilation_contexts)) { + if(!in_range(context.canonical_id)) { + return false; } } - index.compilation_contexts.try_emplace(path, std::move(context)); - } - - // Count ref counts from compilation contexts. - for(auto entry: *root->compilation_contexts()) { - index.canonical_ref_counts[entry->canonical_id()] += 1; - } - - // Deserialize removed bitmap. - if(root->removed() && root->removed()->size() > 0) { - index.removed = read_bitmap(root->removed()); + return true; + }; + if(!usable()) { + self.impl = std::make_unique(); + self.buffer.reset(); + return; } - for(auto entry: *root->occurrences()) { - index.occurrences.try_emplace(*safe_cast(entry->occurrence()), - read_bitmap(entry->context())); - } + index.canonical_ref_counts.resize(index.max_canonical_id, 0); - for(auto entry: *root->relations()) { - auto& relations = index.relations[entry->symbol()]; - for(auto relation_entry: *entry->relations()) { - relations.try_emplace(*safe_cast(relation_entry->relation()), - read_bitmap(relation_entry->context())); + for(auto& context: llvm::make_second_range(index.header_contexts)) { + for(auto& include: context.includes) { + index.canonical_ref_counts[include.canonical_id] += 1; } } - - if(root->content()) { - index.content = root->content()->str(); - } - - if(root->line_starts() && root->line_starts()->size() > 0) { - auto* ls = root->line_starts(); - index.line_starts.assign(ls->begin(), ls->end()); - } else if(!index.content.empty()) { - index.line_starts = kota::ipc::lsp::build_line_starts(index.content); - } - - if(root->symbols()) { - for(auto entry: *root->symbols()) { - auto& symbol = index.symbols[entry->symbol_id()]; - if(auto* s = entry->symbol()) { - if(s->name()) - symbol.name = s->name()->str(); - symbol.kind = SymbolKind(static_cast(s->kind())); - symbol.scope = static_cast(s->scope()); - symbol.reference_files = read_bitmap(s->refs()); - } - } + for(auto& context: llvm::make_second_range(index.compilation_contexts)) { + index.canonical_ref_counts[context.canonical_id] += 1; } self.buffer.reset(); @@ -418,26 +444,21 @@ MergedIndex MergedIndex::load(llvm::StringRef path) { return MergedIndex(); } - // A stale cache directory from an older build must never crash the server - // or be misread. Verify the blob is a structurally valid flatbuffer, then - // discard any shard whose format version differs (including version-less - // shards, which report 0). A discarded shard is treated as "not on disk" - // and the background indexer rebuilds it. - auto data = reinterpret_cast((*buffer)->getBufferStart()); - fbs::Verifier verifier(data, (*buffer)->getBufferSize()); - if(!verifier.VerifyBuffer(nullptr)) { - return MergedIndex(); - } - - auto root = fbs::GetRoot((*buffer)->getBufferStart()); - if(root->format_version() != index_format_version) { + // A stale cache directory from an older build must never crash the + // server or be misread. from_bytes deep-verifies every offset, string, + // vector and table the views can reach; shards whose format version + // differs are discarded (version-less shards read back 0). A discarded + // shard is treated as "not on disk" and the background indexer rebuilds + // it. + auto root = ShardView::from_bytes(blob_bytes((*buffer)->getBuffer())); + if(!root.valid() || root[&Impl::format_version] != index_format_version) { return MergedIndex(); } return MergedIndex(std::move(*buffer), nullptr); } -void MergedIndex::serialize(this const Self& self, llvm::raw_ostream& out) { +void MergedIndex::serialize(this Self& self, llvm::raw_ostream& out) { if(self.buffer) { out.write(self.buffer->getBufferStart(), self.buffer->getBufferSize()); return; @@ -447,150 +468,12 @@ void MergedIndex::serialize(this const Self& self, llvm::raw_ostream& out) { return; } - auto& index = self.impl; - - fbs::FlatBufferBuilder builder(1024); - - llvm::SmallVector buffer; - - // Compaction: rows whose every canonical was released are masked at - // runtime by the removed bitmap, but the serialized shard is served - // through buffer-only lookups that never consult it — so masked state - // must not reach disk at all. Dead rows are dropped, live bitmaps are - // written pre-subtracted, dead cache entries go with them (a later - // re-merge of identical content mints a fresh canonical), and the - // persisted removed bitmap is always empty. - auto& removed = index->removed; - auto live = [&](const roaring::Roaring& bitmap) { - return removed.isEmpty() ? bitmap : bitmap - removed; - }; - - llvm::SmallVector> canonical_cache; - for(auto& [hash, canonical_id]: index->canonical_cache) { - if(removed.contains(canonical_id)) { - continue; - } - canonical_cache.push_back( - binary::CreateCacheEntry(builder, CreateString(builder, hash.str()), canonical_id)); - } - - auto header_contexts = transform(index->header_contexts, [&](auto&& value) { - auto& [path_id, context] = value; - return binary::CreateHeaderContextEntry( - builder, - path_id, - context.version, - CreateStructVector(builder, context.includes)); - }); - - auto compilation_contexts = transform(index->compilation_contexts, [&](auto&& value) { - auto& [path_id, context] = value; - return binary::CreateCompilationContextEntry( - builder, - path_id, - context.version, - context.canonical_id, - context.build_at, - CreateStructVector(builder, context.include_locations), - CreateStructVector(builder, context.dep_hashes), - CreateStructVector(builder, context.dep_stamps)); - }); - - llvm::SmallVector occurrence_keys; - llvm::SmallVector> occurrences; - occurrence_keys.reserve(index->occurrences.size()); - occurrences.reserve(index->occurrences.size()); - for(auto& [occurrence, bitmap]: index->occurrences) { - auto masked = live(bitmap); - if(masked.isEmpty()) { - continue; - } - buffer.clear(); - buffer.resize_for_overwrite(masked.getSizeInBytes(false)); - masked.write(buffer.data(), false); - occurrence_keys.emplace_back(&occurrence); - occurrences.push_back( - binary::CreateOccurrenceEntry(builder, - safe_cast(&occurrence), - CreateVector(builder, buffer))); - } - std::ranges::sort(std::views::zip(occurrence_keys, occurrences), [](auto lhs, auto rhs) { - const auto& lo = *std::get<0>(lhs); - const auto& ro = *std::get<0>(rhs); - return std::tuple(lo.range.begin, lo.range.end, lo.target) < - std::tuple(ro.range.begin, ro.range.end, ro.target); - }); - - llvm::SmallVector relation_keys; - llvm::SmallVector> relations; - relation_keys.reserve(index->relations.size()); - relations.reserve(index->relations.size()); - for(auto& [symbol_id, symbol_relations]: index->relations) { - llvm::SmallVector> entries; - for(auto& [relation, bitmap]: symbol_relations) { - auto masked = live(bitmap); - if(masked.isEmpty()) { - continue; - } - buffer.clear(); - buffer.resize_for_overwrite(masked.getSizeInBytes(false)); - masked.write(buffer.data(), false); - entries.push_back(binary::CreateRelationEntry(builder, - safe_cast(&relation), - CreateVector(builder, buffer))); - } - if(entries.empty()) { - continue; - } - relation_keys.emplace_back(symbol_id); - relations.push_back( - binary::CreateSymbolRelationsEntry(builder, symbol_id, CreateVector(builder, entries))); - } - std::ranges::sort(std::views::zip(relation_keys, relations), {}, [](auto e) { - return std::get<0>(e); - }); - - // Post-compaction nothing on disk is masked; the persisted removed - // bitmap is always empty. - buffer.clear(); - auto removed_offset = CreateVector(builder, buffer); - - auto content_offset = CreateString(builder, index->content); - auto line_starts_offset = builder.CreateVector(index->line_starts); - - auto symbols = transform(index->symbols, [&](auto&& value) { - auto& [symbol_id, symbol] = value; - buffer.clear(); - buffer.resize_for_overwrite(symbol.reference_files.getSizeInBytes(false)); - symbol.reference_files.write(buffer.data(), false); - return binary::CreateSymbolEntry(builder, - symbol_id, - binary::CreateSymbol(builder, - CreateString(builder, symbol.name), - symbol.kind.value(), - CreateVector(builder, buffer), - static_cast(symbol.scope))); - }); - - auto paths = transform(index->paths.paths, - [&](llvm::StringRef path) { return CreateString(builder, path); }); - - auto merged_index = binary::CreateMergedIndex(builder, - index->max_canonical_id, - CreateVector(builder, paths), - CreateVector(builder, canonical_cache), - CreateVector(builder, header_contexts), - CreateVector(builder, compilation_contexts), - CreateVector(builder, occurrences), - CreateVector(builder, relations), - removed_offset, - content_offset, - line_starts_offset, - CreateVector(builder, symbols), - index_format_version); - builder.Finish(merged_index); - - out.write(safe_cast(builder.GetBufferPointer()), builder.GetSize()); + // The serialized shard is served through buffer-only lookups that never + // consult the removed bitmap, so masked state must not reach disk at + // all: compact first, then reflect the impl directly onto the wire. + self.impl->compact(); + self.impl->format_version = index_format_version; + serialize_blob(*self.impl, out); } void MergedIndex::lookup(this const Self& self, @@ -638,26 +521,12 @@ void MergedIndex::lookup(this const Self& self, break; } } else if(self.buffer) { - auto index = fbs::GetRoot(self.buffer->getBufferStart()); - auto& occurrences = *index->occurrences(); - - auto it = std::ranges::lower_bound(occurrences, offset, {}, [](auto o) { - return o->occurrence()->range().end(); - }); - - while(it != occurrences.end()) { - auto o = safe_cast(it->occurrence()); - if(o->range.contains(offset)) { - if(!callback(*o)) { - break; - } - - it++; - continue; - } - - break; - } + auto occurrences = root_of(*self.buffer)[&Impl::occurrences]; + scan_occurrences_at( + occurrences.size(), + offset, + [&](std::size_t i) { return occurrences.at(i).get<0>(); }, + callback); } } @@ -673,7 +542,7 @@ void MergedIndex::lookup(this const Self& self, auto& relations = it->second; for(auto& [relation, bitmap]: relations) { - if(relation.kind & kind) { + if(RelationKind(relation.kind) & kind) { // Skip relations whose canonical_ids are all removed. if(!self.impl->removed.isEmpty()) { auto remaining = bitmap - self.impl->removed; @@ -688,18 +557,16 @@ void MergedIndex::lookup(this const Self& self, } } } else if(self.buffer) { - auto index = fbs::GetRoot(self.buffer->getBufferStart()); - auto& entries = *index->relations(); - - auto it = std::ranges::lower_bound(entries, symbol, {}, [](auto e) { return e->symbol(); }); - if(it == entries.end() || it->symbol() != symbol) [[unlikely]] { + auto found = root_of(*self.buffer)[&Impl::relations].find(symbol); + if(!found) [[unlikely]] { return; } - for(auto entry: *it->relations()) { - auto r = safe_cast(entry->relation()); - if(r->kind & kind) { - if(!callback(*r)) { + auto entries = found->get<1>(); + for(std::size_t i = 0; i < entries.size(); ++i) { + Relation relation = entries.at(i).get<0>(); + if(RelationKind(relation.kind) & kind) { + if(!callback(relation)) { break; } } @@ -725,9 +592,9 @@ bool MergedIndex::need_update(this const Self& self) { for(auto& dep: context.dep_hashes) { hashes.try_emplace(dep.path_id, dep.content_hash); } - llvm::DenseMap stamps; + llvm::DenseMap stamps; for(auto& stamp: context.dep_stamps) { - stamps.try_emplace(stamp.path_id, &stamp); + stamps.try_emplace(stamp.path_id, stamp); } llvm::DenseSet deps; @@ -742,7 +609,8 @@ bool MergedIndex::need_update(this const Self& self) { auto stamp_it = stamps.find(location.path_id); auto it = hashes.find(location.path_id); if(dep_stale(paths[location.path_id], - stamp_it != stamps.end() ? stamp_it->second : nullptr, + stamp_it != stamps.end() ? std::optional(stamp_it->second) + : std::nullopt, it != hashes.end() ? std::optional(it->second) : std::nullopt)) { return true; } @@ -751,41 +619,46 @@ bool MergedIndex::need_update(this const Self& self) { return false; } else if(self.buffer) { - auto index = fbs::GetRoot(self.buffer->getBufferStart()); - if(index->compilation_contexts()->empty()) { + auto root = root_of(*self.buffer); + auto contexts = root[&Impl::compilation_contexts]; + if(contexts.empty()) { return true; } - auto* paths = index->paths(); + auto paths = root[&Impl::paths]; + + for(std::size_t c = 0; c < contexts.size(); ++c) { + auto context = contexts.at(c).get<1>(); - for(auto context: *index->compilation_contexts()) { llvm::DenseMap hashes; - if(context->dep_hashes()) { - for(auto dep: *context->dep_hashes()) { - hashes.try_emplace(dep->path_id(), dep->content_hash()); - } + auto dep_hashes = context[&CompilationContext::dep_hashes]; + for(std::size_t i = 0; i < dep_hashes.size(); ++i) { + DepHash dep = dep_hashes[i]; + hashes.try_emplace(dep.path_id, dep.content_hash); } - llvm::DenseMap stamps; - if(context->dep_stamps()) { - for(auto stamp: *context->dep_stamps()) { - stamps.try_emplace(stamp->path_id(), stamp); - } + llvm::DenseMap stamps; + auto dep_stamps = context[&CompilationContext::dep_stamps]; + for(std::size_t i = 0; i < dep_stamps.size(); ++i) { + DepStamp stamp = dep_stamps[i]; + stamps.try_emplace(stamp.path_id, stamp); } llvm::DenseSet deps; - for(auto location: *context->include_locations()) { - if(!deps.insert(location->path_id()).second) { + auto locations = context[&CompilationContext::include_locations]; + for(std::size_t i = 0; i < locations.size(); ++i) { + IncludeLocation location = locations[i]; + if(!deps.insert(location.path_id).second) { continue; } // A dep the table does not cover cannot be validated: rebuild. - if(!paths || location->path_id() >= paths->size()) { + if(location.path_id >= paths.size()) { return true; } - auto stamp_it = stamps.find(location->path_id()); - auto it = hashes.find(location->path_id()); - if(dep_stale(paths->Get(location->path_id())->string_view(), - stamp_it != stamps.end() ? safe_cast(stamp_it->second) - : nullptr, + auto stamp_it = stamps.find(location.path_id); + auto it = hashes.find(location.path_id); + if(dep_stale(to_ref(paths[location.path_id]), + stamp_it != stamps.end() ? std::optional(stamp_it->second) + : std::nullopt, it != hashes.end() ? std::optional(it->second) : std::nullopt)) { return true; } @@ -817,13 +690,11 @@ bool MergedIndex::has_contribution(this const Self& self, llvm::StringRef contex } if(self.buffer) { - auto root = fbs::GetRoot(self.buffer->getBufferStart()); - if(!root->paths()) { - return false; - } + auto root = root_of(*self.buffer); + auto paths = root[&Impl::paths]; std::optional local; - for(std::uint32_t i = 0; i < root->paths()->size(); ++i) { - if(llvm::StringRef(root->paths()->Get(i)->string_view()) == context_path) { + for(std::uint32_t i = 0; i < paths.size(); ++i) { + if(to_ref(paths[i]) == context_path) { local = i; break; } @@ -831,16 +702,8 @@ bool MergedIndex::has_contribution(this const Self& self, llvm::StringRef contex if(!local) { return false; } - for(auto entry: *root->header_contexts()) { - if(entry->path_id() == *local) { - return true; - } - } - for(auto entry: *root->compilation_contexts()) { - if(entry->path_id() == *local) { - return true; - } - } + return root[&Impl::header_contexts].contains(*local) || + root[&Impl::compilation_contexts].contains(*local); } return false; @@ -889,18 +752,12 @@ bool MergedIndex::find_symbol(this const Self& self, return true; } } else if(self.buffer) { - auto root = fbs::GetRoot(self.buffer->getBufferStart()); - if(root->symbols()) { - for(auto entry: *root->symbols()) { - if(entry->symbol_id() == hash) { - if(auto* s = entry->symbol()) { - if(s->name()) - name = s->name()->str(); - kind = SymbolKind(static_cast(s->kind())); - } - return true; - } - } + auto found = root_of(*self.buffer)[&Impl::symbols].find(hash); + if(found) { + auto symbol = found->get<1>(); + name = std::string(symbol[&Symbol::name]); + kind = SymbolKind(symbol[&Symbol::kind]); + return true; } } return false; @@ -1022,10 +879,7 @@ llvm::StringRef MergedIndex::content(this const Self& self) { if(self.impl) { return self.impl->content; } else if(self.buffer) { - auto root = fbs::GetRoot(self.buffer->getBufferStart()); - if(root->content()) { - return root->content()->string_view(); - } + return to_ref(root_of(*self.buffer)[&Impl::content]); } return {}; } @@ -1034,10 +888,8 @@ std::span MergedIndex::line_starts(this const Self& self) { if(self.impl) { return self.impl->line_starts; } else if(self.buffer) { - auto root = fbs::GetRoot(self.buffer->getBufferStart()); - if(root->line_starts() && root->line_starts()->size() > 0) { - return {root->line_starts()->data(), root->line_starts()->size()}; - } + auto starts = to_array_ref(root_of(*self.buffer)[&Impl::line_starts]); + return {starts.data(), starts.size()}; } return {}; } diff --git a/src/index/merged_index.h b/src/index/merged_index.h index 4b69e60a5..69d09d00f 100644 --- a/src/index/merged_index.h +++ b/src/index/merged_index.h @@ -25,9 +25,14 @@ struct DepLocation { }; class MergedIndex { -private: +public: + /// The in-memory shard state, defined in merged_index.cpp. Its reflected + /// layout doubles as the persisted shard schema: serialization reflects + /// an Impl directly and the buffer-backed query paths read the blob + /// through a zero-copy view of the same layout. struct Impl; +private: using Self = MergedIndex; MergedIndex(std::unique_ptr buffer, std::unique_ptr impl); @@ -52,8 +57,9 @@ class MergedIndex { /// Load merged index from disk static MergedIndex load(llvm::StringRef path); - /// Serialize it to binary format. - void serialize(this const Self& self, llvm::raw_ostream& out); + /// Serialize it to binary format. Compacts rows masked by removals in + /// place first, so the serialized blob is a direct reflection of Impl. + void serialize(this Self& self, llvm::raw_ostream& out); /// Lookup the occurrence in corresponding offset. void lookup(this const Self& self, diff --git a/src/index/preamble_state.cpp b/src/index/preamble_state.cpp index 81c885394..0724d3b94 100644 --- a/src/index/preamble_state.cpp +++ b/src/index/preamble_state.cpp @@ -1,6 +1,6 @@ #include "index/preamble_state.h" -#include +#include #include #include #include @@ -9,59 +9,74 @@ #include "index/serialization.h" #include "kota/ipc/lsp/text.h" -#include "llvm/ADT/SmallVector.h" namespace clice::index { namespace { -/// Serialize one FileIndex into a PreambleFileEntry, with relations sorted -/// by symbol hash so the read path can binary-search them. -fbs::Offset serialize_entry(fbs::FlatBufferBuilder& builder, - std::uint32_t path_id, - const FileIndex& index, - llvm::StringRef content, - llvm::ArrayRef line_starts) { - auto occs = CreateStructVector(builder, index.occurrences); - - llvm::SmallVector*>, 0> sorted; - sorted.reserve(index.relations.size()); - for(auto& [symbol_id, relations]: index.relations) { - sorted.emplace_back(symbol_id, &relations); - } - std::ranges::sort(sorted, {}, [](const auto& entry) { return entry.first; }); - - auto rels = transform(sorted, [&](const auto& entry) { - return binary::CreateTUFileRelationsEntry( - builder, - entry.first, - CreateStructVector(builder, *entry.second)); - }); - - return binary::CreatePreambleFileEntry( - builder, - path_id, - occs, - CreateVector(builder, rels), - content.empty() ? 0 : CreateString(builder, content), - line_starts.empty() ? 0 : CreateVector(builder, line_starts)); +/// One file covered by the preamble compilation: its rows plus content and +/// line starts for position mapping. The rows are moved out of the TUIndex +/// and the content borrows the compilation's buffers — entries are only +/// ever encoded, never decoded (queries run on the zero-copy view). +struct PreambleFileEntry { + std::uint32_t path_id = 0; + FileIndex index; + llvm::StringRef content; + std::vector line_starts; +}; + +/// find_symbol serves only name and kind, so the blob stores this reduced +/// entry instead of the full Symbol — reflecting that would drag every +/// symbol's scope and reference bitmap into large SDK preamble blobs for +/// nothing. The name borrows the consumed TUIndex (encode-only, like +/// PreambleFileEntry). +struct PreambleSymbol { + llvm::StringRef name; + SymbolKind kind; +}; + +/// The persisted shape of a `.pch.idx` blob. Queries run on a zero-copy +/// view of this layout; nothing is deserialized up front. +struct PreambleBlob { + std::uint32_t format_version = 0; + std::vector paths; + std::vector files; + PreambleFileEntry preamble; + llvm::DenseMap symbols; + llvm::ArrayRef links; + llvm::ArrayRef inactive_regions; + llvm::ArrayRef open_conditionals; +}; + +using StateView = kota::codec::fbs::table_view; +using FileEntryView = kota::codec::fbs::table_view; + +/// The blob was fully verified at load(); per-query views skip that cost. +StateView root_of(const llvm::MemoryBuffer& buffer) { + return StateView::from_verified_bytes(blob_bytes(buffer.getBuffer())); +} + +PreambleState::File file_of(kota::codec::fbs::array_view paths, FileEntryView entry) { + auto line_starts = to_array_ref(entry[&PreambleFileEntry::line_starts]); + return PreambleState::File{ + .path = to_ref(paths[entry[&PreambleFileEntry::path_id]]), + .content = to_ref(entry[&PreambleFileEntry::content]), + .line_starts = std::span(line_starts.data(), line_starts.size()), + }; } } // namespace void PreambleState::serialize(CompilationUnitRef unit, - const TUIndex& index, + TUIndex index, llvm::ArrayRef links, llvm::ArrayRef inactive_regions, llvm::ArrayRef open_conditionals, llvm::raw_ostream& os) { - fbs::FlatBufferBuilder builder(4096); + PreambleBlob blob; + blob.format_version = preamble_format_version; - auto paths = - transform(index.graph.paths, [&](const std::string& p) { return builder.CreateString(p); }); - - Offsets files; - files.reserve(index.file_indices.size()); + blob.files.reserve(index.file_indices.size()); for(auto& [fid, file_index]: index.file_indices) { // A file with no include edge is a synthetic buffer (predefines, // ): it has no real path to attribute rows to, and @@ -73,10 +88,13 @@ void PreambleState::serialize(CompilationUnitRef unit, continue; } auto content = unit.file_content(fid); - auto line_starts = - kota::ipc::lsp::build_line_starts(std::string_view(content.data(), content.size())); - files.push_back( - serialize_entry(builder, index.graph.path_id(fid), file_index, content, line_starts)); + blob.files.push_back({ + .path_id = index.graph.path_id(fid), + .index = std::move(file_index), + .content = content, + .line_starts = + kota::ipc::lsp::build_line_starts(std::string_view(content.data(), content.size())), + }); } // The source file is the last path in graph.paths (convention from @@ -85,46 +103,24 @@ void PreambleState::serialize(CompilationUnitRef unit, // PCH was built from — stored so consumers can compare it against the // live buffer's prefix before serving these rows. auto preamble_text = unit.interested_content(); - auto preamble_starts = kota::ipc::lsp::build_line_starts( - std::string_view(preamble_text.data(), preamble_text.size())); - auto preamble_entry = serialize_entry(builder, - static_cast(index.graph.paths.size() - 1), - index.main_file_index, - preamble_text, - preamble_starts); - - llvm::SmallVector, 0> sorted_symbols; - sorted_symbols.reserve(index.symbols.size()); - for(auto& [symbol_id, symbol]: index.symbols) { - sorted_symbols.emplace_back(symbol_id, &symbol); + blob.preamble = { + .path_id = static_cast(index.graph.paths.size() - 1), + .index = std::move(index.main_file_index), + .content = preamble_text, + .line_starts = kota::ipc::lsp::build_line_starts( + std::string_view(preamble_text.data(), preamble_text.size())), + }; + + blob.symbols.reserve(index.symbols.size()); + for(const auto& [hash, symbol]: index.symbols) { + blob.symbols.try_emplace(hash, PreambleSymbol{.name = symbol.name, .kind = symbol.kind}); } - std::ranges::sort(sorted_symbols, {}, [](const auto& entry) { return entry.first; }); - - auto syms = transform(sorted_symbols, [&](const auto& entry) { - return binary::CreatePreambleSymbolEntry(builder, - entry.first, - CreateString(builder, entry.second->name), - entry.second->kind.value()); - }); - - auto link_entries = transform(links, [&](const feature::DocumentLink& link) { - binary::Range range(link.range.begin, link.range.end); - return binary::CreatePreambleDocumentLink(builder, - &range, - CreateString(builder, link.target)); - }); - - auto root = binary::CreatePreambleState(builder, - preamble_format_version, - CreateVector(builder, paths), - CreateVector(builder, files), - preamble_entry, - CreateVector(builder, syms), - CreateVector(builder, link_entries), - CreateVector(builder, inactive_regions), - CreateVector(builder, open_conditionals)); - builder.Finish(root); - os.write(safe_cast(builder.GetBufferPointer()), builder.GetSize()); + blob.paths = std::move(index.graph.paths); + blob.links = links; + blob.inactive_regions = inactive_regions; + blob.open_conditionals = open_conditionals; + + serialize_blob(blob, os); } PreambleState::PreambleState(std::unique_ptr buffer) : @@ -136,23 +132,13 @@ std::shared_ptr PreambleState::load(llvm::StringRef path) { return nullptr; } - // A stale or truncated blob must never crash the server. Verify it is - // a structurally valid flatbuffer, then discard any blob whose format - // version differs (version-less blobs read back 0). The table budget - // is far above the default: a large preamble's symbol table alone can - // exceed a million entries, and a verification failure here would - // otherwise send every compile into a rebuild loop. - auto data = reinterpret_cast((*buffer)->getBufferStart()); - fbs::Verifier verifier(data, - (*buffer)->getBufferSize(), - /*max_depth=*/64, - /*max_tables=*/1u << 26); - if(!verifier.VerifyBuffer(nullptr)) { - return nullptr; - } - - auto root = fbs::GetRoot((*buffer)->getBufferStart()); - if(root->format_version() != preamble_format_version) { + // A stale or truncated blob must never crash the server. from_bytes + // deep-verifies every offset, string, vector and table the views can + // reach — queries then run unchecked (root_of). Blobs of a different + // format version load as "missing" (version-less blobs read back 0) + // and the PCH pair is rebuilt. + auto root = StateView::from_bytes(blob_bytes((*buffer)->getBuffer())); + if(!root.valid() || root[&PreambleBlob::format_version] != preamble_format_version) { return nullptr; } @@ -162,40 +148,30 @@ std::shared_ptr PreambleState::load(llvm::StringRef path) { void PreambleState::lookup(SymbolHash symbol, RelationKind kind, llvm::function_ref callback) const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto paths = root->paths(); - - for(auto entry: *root->files()) { - auto rels = entry->relations(); - if(!rels) { - continue; - } - - auto it = std::ranges::lower_bound(*rels, symbol, {}, [](auto e) { return e->symbol(); }); - if(it == rels->end() || it->symbol() != symbol || !it->relations()) { + auto root = root_of(*buffer); + auto paths = root[&PreambleBlob::paths]; + auto files = root[&PreambleBlob::files]; + + for(std::size_t i = 0; i < files.size(); ++i) { + auto entry = files[i]; + auto relations = entry[&PreambleFileEntry::index][&FileIndex::relations]; + auto found = relations.find(symbol); + if(!found) { continue; } // The verifier checks structure, not cross-references: a corrupt - // path_id would index out of bounds on the mapped file. - if(entry->path_id() >= paths->size()) { + // path_id must not attribute rows to an arbitrary path. + if(entry[&PreambleFileEntry::path_id] >= paths.size()) { continue; } - auto path = paths->Get(entry->path_id()); - File file{ - .path = llvm::StringRef(path->c_str(), path->size()), - .content = entry->content() - ? llvm::StringRef(entry->content()->c_str(), entry->content()->size()) - : llvm::StringRef(), - .line_starts = entry->line_starts() ? std::span(entry->line_starts()->data(), - entry->line_starts()->size()) - : std::span(), - }; - - for(auto rel: *it->relations()) { - auto r = safe_cast(rel); - if(r->kind & kind) { - if(!callback(file, *r)) { + auto file = file_of(paths, entry); + + auto rels = found->get<1>(); + for(std::size_t j = 0; j < rels.size(); ++j) { + Relation relation = rels[j]; + if(RelationKind(relation.kind) & kind) { + if(!callback(file, relation)) { return; } } @@ -204,68 +180,49 @@ void PreambleState::lookup(SymbolHash symbol, } llvm::StringRef PreambleState::source_path() const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto paths = root->paths(); - if(paths->size() == 0) { + auto root = root_of(*buffer); + auto paths = root[&PreambleBlob::paths]; + if(paths.empty()) { return {}; } // The source file is the last path, by IncludeGraph convention. - auto path = paths->Get(paths->size() - 1); - return llvm::StringRef(path->c_str(), path->size()); + return to_ref(paths[paths.size() - 1]); } llvm::StringRef PreambleState::preamble_content() const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto preamble = root->preamble(); - if(!preamble || !preamble->content()) { - return {}; - } - return llvm::StringRef(preamble->content()->c_str(), preamble->content()->size()); + auto root = root_of(*buffer); + return to_ref(root[&PreambleBlob::preamble][&PreambleFileEntry::content]); } void PreambleState::lookup_preamble(std::uint32_t offset, llvm::function_ref callback) const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto preamble = root->preamble(); - if(!preamble || !preamble->occurrences()) { - return; - } - - auto& occurrences = *preamble->occurrences(); - auto it = - std::ranges::lower_bound(occurrences, offset, {}, [](auto o) { return o->range().end(); }); - - while(it != occurrences.end()) { - auto o = safe_cast(*it); - if(!o->range.contains(offset)) { - break; - } - if(!callback(*o)) { - break; - } - ++it; - } + auto root = root_of(*buffer); + auto occurrences = + root[&PreambleBlob::preamble][&PreambleFileEntry::index][&FileIndex::occurrences]; + + scan_occurrences_at( + occurrences.size(), + offset, + [&](std::size_t i) { return occurrences[i]; }, + callback); } void PreambleState::lookup_preamble(SymbolHash symbol, RelationKind kind, llvm::function_ref callback) const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto preamble = root->preamble(); - if(!preamble || !preamble->relations()) { + auto root = root_of(*buffer); + auto relations = + root[&PreambleBlob::preamble][&PreambleFileEntry::index][&FileIndex::relations]; + auto found = relations.find(symbol); + if(!found) { return; } - auto& rels = *preamble->relations(); - auto it = std::ranges::lower_bound(rels, symbol, {}, [](auto e) { return e->symbol(); }); - if(it == rels.end() || it->symbol() != symbol || !it->relations()) { - return; - } - - for(auto rel: *it->relations()) { - auto r = safe_cast(rel); - if(r->kind & kind) { - if(!callback(*r)) { + auto rels = found->get<1>(); + for(std::size_t i = 0; i < rels.size(); ++i) { + Relation relation = rels[i]; + if(RelationKind(relation.kind) & kind) { + if(!callback(relation)) { return; } } @@ -273,54 +230,42 @@ void PreambleState::lookup_preamble(SymbolHash symbol, } bool PreambleState::find_symbol(SymbolHash hash, std::string& name, SymbolKind& kind) const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto& syms = *root->symbols(); - - auto it = std::ranges::lower_bound(syms, hash, {}, [](auto e) { return e->symbol_id(); }); - if(it == syms.end() || it->symbol_id() != hash) { + auto root = root_of(*buffer); + auto found = root[&PreambleBlob::symbols].find(hash); + if(!found) { return false; } - name = it->name() ? it->name()->str() : std::string(); - kind = SymbolKind(static_cast(it->kind())); + auto symbol = found->get<1>(); + name = std::string(symbol[&PreambleSymbol::name]); + kind = SymbolKind(symbol[&PreambleSymbol::kind]); return true; } std::vector PreambleState::links() const { + auto root = root_of(*buffer); + auto entries = root[&PreambleBlob::links]; + std::vector links; - auto root = fbs::GetRoot(buffer->getBufferStart()); - if(auto ls = root->links()) { - links.reserve(ls->size()); - for(auto entry: *ls) { - feature::DocumentLink link; - if(auto range = entry->range()) { - link.range = *safe_cast(range); - } - if(auto target = entry->target()) { - link.target = target->str(); - } - links.push_back(std::move(link)); - } + links.reserve(entries.size()); + for(std::size_t i = 0; i < entries.size(); ++i) { + auto entry = entries[i]; + links.push_back(feature::DocumentLink{ + .range = entry[&feature::DocumentLink::range], + .target = std::string(entry[&feature::DocumentLink::target]), + }); } return links; } llvm::ArrayRef PreambleState::inactive_regions() const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto regions = root->inactive_regions(); - if(!regions) { - return {}; - } - return llvm::ArrayRef(regions->data(), regions->size()); + auto root = root_of(*buffer); + return to_array_ref(root[&PreambleBlob::inactive_regions]); } llvm::ArrayRef PreambleState::open_conditionals() const { - auto root = fbs::GetRoot(buffer->getBufferStart()); - auto conditionals = root->open_conditionals(); - if(!conditionals) { - return {}; - } - return llvm::ArrayRef(conditionals->data(), conditionals->size()); + auto root = root_of(*buffer); + return to_array_ref(root[&PreambleBlob::open_conditionals]); } } // namespace clice::index diff --git a/src/index/preamble_state.h b/src/index/preamble_state.h index 7af03c0c4..d33f7a2fa 100644 --- a/src/index/preamble_state.h +++ b/src/index/preamble_state.h @@ -15,11 +15,11 @@ namespace clice::index { /// On-disk PreambleState blob schema version (the PCH's `.pch.idx` pair). -/// Bump whenever `schema.fbs` changes the PreambleState layout; a blob -/// carrying a different value loads as "missing" and the PCH pair is -/// rebuilt. cache.json records it so a version change is caught at load -/// time instead of on the first overlay query. -constexpr inline std::uint32_t preamble_format_version = 3; +/// Bump whenever the persisted PreambleState layout (its reflected repr) +/// changes; a blob carrying a different value loads as "missing" and the +/// PCH pair is rebuilt. cache.json records it so a version change is +/// caught at load time instead of on the first overlay query. +constexpr inline std::uint32_t preamble_format_version = 5; /// All master-visible state derived from one PCH build. /// @@ -57,8 +57,10 @@ class PreambleState { /// over the preamble unit with interested_only=false and its /// main_file_index intact (it holds the preamble region's own /// occurrences — macro definitions and references before the bound). + /// Taken by value and consumed: the blob is assembled by moving the + /// index's rows, never copying them. static void serialize(CompilationUnitRef unit, - const TUIndex& index, + TUIndex index, llvm::ArrayRef links, llvm::ArrayRef inactive_regions, llvm::ArrayRef open_conditionals, diff --git a/src/index/project_index.cpp b/src/index/project_index.cpp index 2857af3b3..567f93e47 100644 --- a/src/index/project_index.cpp +++ b/src/index/project_index.cpp @@ -3,6 +3,7 @@ #include "index/serialization.h" #include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/STLExtras.h" namespace clice::index { @@ -33,132 +34,71 @@ llvm::SmallVector ProjectIndex::merge(this ProjectIndex& self, return file_ids_map; } -void ProjectIndex::serialize(this const ProjectIndex& self, +void ProjectIndex::serialize(this ProjectIndex& self, llvm::raw_ostream& os, const clice::PathPool& pool, llvm::ArrayRef shards) { - fbs::FlatBufferBuilder builder(1024); - - // Compact path table: only ids the blob actually references are written, - // in first-seen order. This is where garbage paths get collected — a - // path the pool accumulated but nothing references never reaches disk. - llvm::DenseMap local_ids; - std::vector table; - auto to_local = [&](std::uint32_t pool_id) -> std::uint32_t { - auto [it, inserted] = local_ids.try_emplace(pool_id, table.size()); - if(inserted) { - table.push_back(pool.resolve(pool_id)); - } - return it->second; - }; - - llvm::SmallVector buffer; - llvm::SmallVector remapped; - - auto symbols = transform(self.symbols, [&](auto&& value) { - auto& [symbol_id, symbol] = value; + self.format_version = index_format_version; + self.shards.assign(shards.begin(), shards.end()); - remapped.clear(); - for(auto ref: symbol.reference_files) { - remapped.push_back(to_local(ref)); - } - roaring::Roaring local_refs(remapped.size(), remapped.data()); - - buffer.clear(); - buffer.resize_for_overwrite(local_refs.getSizeInBytes(false)); - local_refs.write(buffer.data(), false); - - return binary::CreateSymbolEntry(builder, - symbol_id, - binary::CreateSymbol(builder, - CreateString(builder, symbol.name), - symbol.kind.value(), - CreateVector(builder, buffer), - static_cast(symbol.scope))); - }); - - llvm::SmallVector shard_locals; - shard_locals.reserve(shards.size()); - for(auto shard: shards) { - shard_locals.push_back(to_local(shard)); + Bitmap referenced(shards.size(), shards.data()); + for(auto& symbol: llvm::make_second_range(self.symbols)) { + referenced |= symbol.reference_files; } - auto paths = - transform(table, [&](llvm::StringRef path) { return CreateString(builder, path); }); - - auto project_index = - binary::CreateProjectIndex(builder, - CreateVector(builder, paths), - CreateVector(builder, symbols), - builder.CreateVector(shard_locals.data(), shard_locals.size()), - index_format_version); + self.paths.clear(); + self.paths.reserve(referenced.cardinality()); + for(auto id: referenced) { + self.paths.emplace_back(id, pool.resolve(id).str()); + } - builder.Finish(project_index); - os.write(safe_cast(builder.GetBufferPointer()), builder.GetSize()); + serialize_blob(self, os); } -std::optional ProjectIndex::from(const void* data, - std::size_t size, +std::optional ProjectIndex::from(llvm::StringRef data, clice::PathPool& pool, llvm::SmallVectorImpl& shards) { - fbs::Verifier verifier(static_cast(data), size); - if(!verifier.VerifyBuffer()) { + std::optional index{std::in_place}; + if(!deserialize_blob(data, *index) || index->format_version != index_format_version) { return std::nullopt; } - auto root = fbs::GetRoot(data); - if(root->format_version() != index_format_version) { - return std::nullopt; - } - - ProjectIndex loaded; - - // Intern the blob's compact path table into the running pool; every id - // in the blob is an index into it. - llvm::SmallVector pool_ids; - if(root->paths()) { - pool_ids.reserve(root->paths()->size()); - for(auto path: *root->paths()) { - pool_ids.push_back(pool.intern(path->string_view())); + // The blob's ids are the writing session's pool ids: intern its path + // table and remap every decoded id into this session's pool. Ids the + // table does not cover are dropped, not misresolved. + llvm::DenseMap remap; + remap.reserve(index->paths.size()); + for(auto& [id, path]: index->paths) { + // The writer only emits interned paths, which are never empty; an + // empty entry marks a corrupt blob and must not become a real pool + // entry. + if(path.empty()) { + return std::nullopt; } + remap.try_emplace(id, pool.intern(path)); } - auto to_pool = [&](std::uint32_t local) -> std::optional { - if(local >= pool_ids.size()) { - return std::nullopt; - } - return pool_ids[local]; - }; - - if(root->symbols()) { - for(auto entry: *root->symbols()) { - auto* fb_symbol = entry->symbol(); - if(!fb_symbol) { - continue; - } - auto& symbol = loaded.symbols[entry->symbol_id()]; - if(auto* name = fb_symbol->name()) { - symbol.name = name->str(); - } - symbol.kind = SymbolKind(static_cast(fb_symbol->kind())); - symbol.scope = static_cast(fb_symbol->scope()); - for(auto local: read_bitmap(fb_symbol->refs())) { - if(auto id = to_pool(local)) { - symbol.reference_files.add(*id); - } + for(auto& symbol: llvm::make_second_range(index->symbols)) { + Bitmap remapped; + for(auto id: symbol.reference_files) { + if(auto it = remap.find(id); it != remap.end()) { + remapped.add(it->second); } } + symbol.reference_files = std::move(remapped); } - if(root->shards()) { - for(auto local: *root->shards()) { - if(auto id = to_pool(local)) { - shards.push_back(*id); - } + for(auto id: index->shards) { + if(auto it = remap.find(id); it != remap.end()) { + shards.push_back(it->second); } } - return loaded; + // The table and manifest were only the wire form; the runtime state is + // the pool and the caller's shard list. + index->paths.clear(); + index->shards.clear(); + return index; } } // namespace clice::index diff --git a/src/index/project_index.h b/src/index/project_index.h index 50f0034a8..7c0b29a0e 100644 --- a/src/index/project_index.h +++ b/src/index/project_index.h @@ -2,6 +2,9 @@ #include #include +#include +#include +#include #include "index/tu_index.h" #include "support/path_pool.h" @@ -15,24 +18,40 @@ namespace clice::index { /// /// There is a single path-id space at runtime: the server-wide /// clice::PathPool. Symbol reference bitmaps carry those ids directly, so -/// queries never translate between pools. Runtime ids are per-session and -/// never persist — serialization remaps every referenced id into a compact -/// self-contained path table (which is also the garbage collection: paths no -/// longer referenced by any symbol or shard are simply not written), and -/// loading interns the table back into the running pool. +/// queries never translate between pools, and they persist as-is — the blob +/// stays self-contained through a path table mapping every referenced id to +/// its path (which is also the garbage collection: paths no longer +/// referenced by any symbol or shard are simply not written). Loading +/// interns the table into the running pool and remaps every id. +/// +/// Serialization reflects this object directly; `format_version`, `paths` +/// and `shards` are serialize-time state populated by serialize() and +/// consumed by from(). struct ProjectIndex { + /// Persisted-blob schema version (index_format_version), stamped by + /// serialize() and gated by from(). + std::uint32_t format_version = 0; + + /// The blob's self-contained path table: pool id → path for every id + /// the symbol bitmaps and the shard manifest reference. + std::vector> paths; + SymbolTable symbols; + /// Pool ids of the files owning a MergedIndex shard blob, persisted so + /// the loader knows which blobs to fetch. + std::vector shards; + /// Merge a TU's external symbols, interning the TU's paths into `pool`. /// Returns the TU-local id → pool id mapping for the TU's path graph. llvm::SmallVector merge(this ProjectIndex& self, TUIndex& index, clice::PathPool& pool); - /// Serialize with a compact path table covering exactly the ids used by - /// the symbol bitmaps plus `shards`, the pool ids of the files owning a - /// MergedIndex shard blob (persisted so the loader knows what to fetch). - void serialize(this const ProjectIndex& self, + /// Serialize with a path table covering exactly the ids used by the + /// symbol bitmaps plus `shards`, the pool ids of the files owning a + /// MergedIndex shard blob. + void serialize(this ProjectIndex& self, llvm::raw_ostream& os, const clice::PathPool& pool, llvm::ArrayRef shards); @@ -42,8 +61,7 @@ struct ProjectIndex { /// the loader should fetch. Returns nullopt for an unreadable or /// old-format blob — the caller treats that as "no index on disk" and /// rebuilds in the background. - static std::optional from(const void* data, - std::size_t size, + static std::optional from(llvm::StringRef data, clice::PathPool& pool, llvm::SmallVectorImpl& shards); }; diff --git a/src/index/schema.fbs b/src/index/schema.fbs deleted file mode 100644 index b9534da52..000000000 --- a/src/index/schema.fbs +++ /dev/null @@ -1,295 +0,0 @@ -namespace clice.index.binary; - -struct Range { - begin : uint; - end : uint; -} - -struct Occurrence { - range : Range; - target : ulong; -} - -struct Relation { - kind : uint; - padding : uint; - range : Range; - target_symbol : ulong; -} - -table CacheEntry { -sha256: - string; -canonical_id: - uint; -} - -struct IncludeContext { - include_id : uint; - canonical_id : uint; -} - -table HeaderContextEntry { -path_id: - uint; -version: - uint; -includes: - [IncludeContext]; -} - -struct IncludeLocation { - path_id : uint; - line : uint; - include_id : uint; -} - -struct DepHash { - path_id : uint; - content_hash : ulong; -} - -// Stat fast path for one dependency: recorded only when the file provably -// did not change since before the indexed build started, so an equal stat -// proves the disk still holds the content the rows were built from. -struct DepStamp { - path_id : uint; - size : ulong; - mtime_ns : long; -} - -table CompilationContextEntry { -path_id: - uint; -version: - uint; -canonical_id: - uint; -build_at: - ulong; -include_locations: - [IncludeLocation]; -dep_hashes: - [DepHash]; - -// Added fields read back as absent from older shards — validation then has -// no fast path and goes by dep_hashes alone, which stays correct. That is -// why this addition does not bump index_format_version. -dep_stamps: - [DepStamp]; -} - -table OccurrenceEntry { -occurrence: - Occurrence; -context: - [ubyte]; -} - -table RelationEntry { -relation: - Relation; -context: - [ubyte]; -} - -table SymbolRelationsEntry { -symbol: - ulong; -relations: - [RelationEntry]; -} - -table Symbol { -name: - string; -kind: - ubyte; -refs: - [ubyte] (required); -scope: - ubyte; -} - -table SymbolEntry { -symbol_id: - ulong; -symbol: - Symbol; -} - -table MergedIndex { -max_canonical_id: - uint; - -// Shard-local path table: every path id inside this shard (dependency -// locations, dependency hashes, context keys) indexes into it. Shards are -// fully self-contained — runtime pool ids are per-session and never persist. -paths: - [string] (required); - -canonical_cache: - [CacheEntry] (required); - -header_contexts: - [HeaderContextEntry] (required); - -compilation_contexts: - [CompilationContextEntry] (required); - -occurrences: - [OccurrenceEntry] (required); - -relations: - [SymbolRelationsEntry] (required); - -removed: - [ubyte]; - -content: - string; - -line_starts: - [uint]; - -symbols: - [SymbolEntry]; - -format_version: - uint; -} - -table TUFileRelationsEntry { -symbol: - ulong; -relations: - [Relation]; -} - -table PreambleDocumentLink { -range: - Range; -target: - string; -} - -// One file covered by a preamble compilation. Content and line starts are -// stored (like MergedIndex shards) so the master converts byte offsets to -// LSP positions without touching the filesystem. -table PreambleFileEntry { -path_id: - uint; - -// Sorted by (range.begin, range.end, target). -occurrences: - [Occurrence]; - -// Sorted by symbol hash for binary search. -relations: - [TUFileRelationsEntry]; - -content: - string; - -line_starts: - [uint]; -} - -table PreambleSymbolEntry { -symbol_id: - ulong; -name: - string; -kind: - ubyte; -} - -// Everything the master needs from one PCH build, written by the stateless -// worker next to the PCH blob and opened as a memory-mapped FlatBuffer: -// the preamble's symbol index plus the PCH-derived feature state that the -// master splices into main-file results (document links, inactive regions, -// open conditional stack). -table PreambleState { -format_version: - uint; - -// Blob-local path table; every path_id indexes into it. -paths: - [string] (required); - -// Header entries covered by the preamble. -files: - [PreambleFileEntry] (required); - -// The source file's own preamble region, in buffer offsets. Its content -// is the exact preamble text the PCH was built from — consumers compare -// it against the live buffer's prefix before serving these rows. -preamble: - PreambleFileEntry; - -// Sorted by symbol_id for binary search. -symbols: - [PreambleSymbolEntry] (required); - -links: - [PreambleDocumentLink]; - -// Flat begin/end offset pairs within the preamble region. -inactive_regions: - [uint]; - -// Conditional stack still open at the preamble bound (see -// feature::InactiveScan::open_stack encoding). -open_conditionals: - [ubyte]; -} - -table TUFileIndexEntry { -file_id: - uint; -occurrences: - [Occurrence]; -relations: - [TUFileRelationsEntry]; -} - -table TUIndex { -built_at: - ulong; -paths: - [string]; -locations: - [IncludeLocation]; -symbols: - [SymbolEntry]; -file_indices: - [TUFileIndexEntry]; -main_file_index: - TUFileIndexEntry; - -// Parallel to `paths`: hash of the bytes the indexing compile actually -// consumed per file (0 = unavailable). The master validates its own disk -// reads against these before pairing rows with content, and stores them -// as the shards' staleness baselines. -path_hashes: - [ulong]; -} - -// The blob's path table is compact: only paths referenced by the symbol -// bitmaps or the shard list are written (garbage paths are collected at -// save time), and every id in the blob is an index into it. Runtime pool -// ids are per-session and never persist. -table ProjectIndex { -paths: - [string] (required); -symbols: - [SymbolEntry] (required); - -// Path-table indices of the files owning a MergedIndex shard blob; the -// loader fetches exactly these (keyed by path hash) and sweeps the rest. -shards: - [uint] (required); - -format_version: - uint; -} diff --git a/src/index/serialization.h b/src/index/serialization.h index bf3e34f66..b8b6dfb77 100644 --- a/src/index/serialization.h +++ b/src/index/serialization.h @@ -1,85 +1,227 @@ -#include -#include -#include +#pragma once -#include "schema_generated.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "index/path_pool.h" +#include "index/tu_index.h" +#include "semantic/symbol.h" #include "support/bitmap.h" -#include "llvm/ADT/SmallVector.h" +#include "kota/codec/fbs/fbs.h" +#include "llvm/ADT/ArrayRef.h" +#include "llvm/ADT/STLExtras.h" +#include "llvm/ADT/StringMap.h" #include "llvm/ADT/StringRef.h" +#include "llvm/Support/raw_ostream.h" -namespace clice::index { +namespace kota::meta { -namespace fbs = flatbuffers; +/// Roaring bitmaps travel as their own serialized image (the non-portable +/// format, matching every in-process reader). +template <> +struct repr { + using type = std::vector; -/// On-disk MergedIndex shard schema version. Bump this whenever `schema.fbs` -/// changes the shard layout so that `MergedIndex::load` silently discards any -/// shard carrying a different value — including version-less shards written by -/// older builds, which read back the field's default of 0. -constexpr inline std::uint32_t index_format_version = 1; + static type to(const clice::Bitmap& bitmap) { + type buffer(bitmap.getSizeInBytes(false)); + bitmap.write(reinterpret_cast(buffer.data()), false); + return buffer; + } -namespace { + static clice::Bitmap from(const type& buffer) { + return clice::Bitmap::read(reinterpret_cast(buffer.data()), false); + } +}; -template -concept sequence_range = std::ranges::input_range && - !requires { typename Range::key_type; } && requires(const Range& r) { - r.data(); - r.size(); - }; +/// SymbolKind hides its enum behind constructors, which keeps it out of +/// reflection; persist the underlying value. +template <> +struct repr { + using type = std::uint8_t; -template -using Offsets = llvm::SmallVector, 0>; - -template -const U* safe_cast(const V* v) { - static_assert(sizeof(U) == sizeof(V), "size mismatch"); - static_assert(alignof(U) == alignof(V), "alignment mismatch"); - static_assert(std::is_trivially_copyable_v && std::is_trivially_copyable_v, - "requires trivially copyable"); - /// If aliasing issues arise, prefer copying into a temporary SmallVector. - return reinterpret_cast(v); -} + static type to(clice::SymbolKind kind) { + return kind.value(); + } -auto CreateString(fbs::FlatBufferBuilder& builder, llvm::StringRef string) { - return builder.CreateString(string.data(), string.size()); -} + static clice::SymbolKind from(type value) { + return clice::SymbolKind(value); + } +}; + +/// A PathPool persists as its path table; ids are the dense indices, so +/// interning the table back in order reproduces them. Both directions drive +/// the visitor: encoding writes the interned StringRefs straight to the +/// wire, decoding interns one path at a time. +template <> +struct repr { + using type = std::vector; + + template + static bool serialize(auto& vis, const clice::index::PathPool& pool) { + return codec::encode_value(vis, pool.paths); + } -template -auto CreateVector(fbs::FlatBufferBuilder& builder, const Range& range) { - return builder.CreateVector(range.data(), range.size()); -} + template + static bool deserialize(auto& vis, clice::index::PathPool& pool) { + type shape; + return vis.visit_seq(shape, [&](auto& sv) -> bool { + while(sv.has_element()) { + std::string path; + if(!sv.visit_element( + [&](auto& ev) -> bool { return codec::decode_value(ev, path); })) { + return false; + } + // The pool never interns an empty path, so a blob carrying + // one is corrupt; reject it instead of tripping the intern + // precondition. + if(path.empty()) { + return false; + } + pool.path_id(path); + } + return true; + }); + } +}; + +/// A StringMap iterates as StringMapEntry, which no codec understands; +/// persist the entries as key/value pairs, sorted by value for +/// deterministic blobs (values are unique canonical ids). Format-agnostic, +/// unlike the reprs above: the pair-list form is not fbs-specific, and the +/// schema layer classifies fields without a format tag — a format-scoped +/// repr would leave it staring at StringMapEntry, which it rejects. +template <> +struct repr> { + using type = std::vector>; + + template + static bool serialize(auto& vis, const llvm::StringMap& map) { + llvm::SmallVector> entries; + entries.reserve(map.size()); + for(const auto& entry: map) { + entries.emplace_back(entry.getKey(), entry.getValue()); + } + llvm::sort(entries, llvm::less_second{}); + return codec::encode_value(vis, entries); + } + + template + static bool deserialize(auto& vis, llvm::StringMap& map) { + type shape; + return vis.visit_seq(shape, [&](auto& sv) -> bool { + while(sv.has_element()) { + std::pair entry; + if(!sv.visit_element( + [&](auto& ev) -> bool { return codec::decode_value(ev, entry); })) { + return false; + } + map.try_emplace(entry.first, entry.second); + } + return true; + }); + } +}; -auto CreateVector(fbs::FlatBufferBuilder& builder, const llvm::SmallVector& range) { - return builder.CreateVector(reinterpret_cast(range.data()), range.size()); +template <> +struct repr { + using type = std::int64_t; + + static type to(std::chrono::milliseconds ms) { + return ms.count(); + } + + static std::chrono::milliseconds from(type count) { + return std::chrono::milliseconds(count); + } +}; + +} // namespace kota::meta + +namespace clice::index { + +/// On-disk index blob schema version. Every persisted blob carries it as a +/// regular field and every loader discards blobs with a different value — +/// including version-less blobs from older builds, which read back as 0. +/// Bump it whenever a persisted type's reflected layout changes. +constexpr inline std::uint32_t index_format_version = 3; + +/// Serialize a reflected index blob to `os` as a verified-readable +/// flatbuffer. Encoding only fails on structural impossibilities (e.g. more +/// fields than slots), which the persisted index types cannot hit. +template +void serialize_blob(const T& value, llvm::raw_ostream& os) { + auto encoded = kota::codec::fbs::to_bytes(value); + assert(encoded.has_value()); + os.write(reinterpret_cast(encoded->data()), encoded->size()); } -template -auto CreateStructVector(fbs::FlatBufferBuilder& builder, const Range& range) { - using V = std::ranges::range_value_t; - (void)sizeof(V); - return builder.CreateVectorOfStructs(safe_cast(range.data()), range.size()); +/// The bytes of `data` as the span every kota fbs entry point takes. +inline std::span blob_bytes(llvm::StringRef data) { + return {reinterpret_cast(data.data()), data.size()}; } -template -auto transform(const Range& range, const Functor& functor) { - using V = std::ranges::range_value_t; - using R = std::invoke_result_t; +/// Verify and deserialize a blob into `out`. The out-parameter overload of +/// from_bytes is deliberate: index types hold llvm::DenseMap members whose +/// explicit default constructors fail std::default_initializable, which the +/// value-returning overload requires. +template +bool deserialize_blob(llvm::StringRef data, T& out) { + return kota::codec::fbs::from_bytes(blob_bytes(data), out).has_value(); +} - llvm::SmallVector result; - result.resize_for_overwrite(std::ranges::size(range)); +/// Scan the occurrences containing `offset` in a sequence sorted by +/// (range.begin, range.end, target): binary-search the first entry whose +/// range ends at or past the offset, then walk while ranges contain it. +/// Binary-searching on range.end is sound because occurrence ranges are +/// name-token spans, pairwise disjoint or identical — never partially +/// overlapping or nested — so under this order range.end is monotonic too. +/// `get(i)` yields the i-th Occurrence. +template +void scan_occurrences_at(std::size_t size, + std::uint32_t offset, + GetOccurrence&& get, + llvm::function_ref callback) { + std::size_t lo = 0; + std::size_t hi = size; + while(lo < hi) { + auto mid = lo + (hi - lo) / 2; + if(get(mid).range.end < offset) { + lo = mid + 1; + } else { + hi = mid; + } + } - auto i = 0; - for(auto&& v: range) { - result[i] = functor(v); - i += 1; + for(; lo < size; ++lo) { + Occurrence occurrence = get(lo); + if(!occurrence.range.contains(offset)) { + break; + } + if(!callback(occurrence)) { + break; + } } - return result; } -Bitmap read_bitmap(const fbs::Vector* buffer) { - return Bitmap::read(reinterpret_cast(buffer->data()), false); +inline llvm::StringRef to_ref(std::string_view text) { + return {text.data(), text.size()}; } -} // namespace +/// A scalar array view as an ArrayRef borrowing the mapped blob. +template +llvm::ArrayRef to_array_ref(kota::codec::fbs::array_view view) { + if(!view.valid()) { + return {}; + } + return {view.raw()->data(), view.raw()->size()}; +} } // namespace clice::index diff --git a/src/index/tu_index.cpp b/src/index/tu_index.cpp index 4b03c9f3d..d88774891 100644 --- a/src/index/tu_index.cpp +++ b/src/index/tu_index.cpp @@ -530,13 +530,8 @@ class Projector { for(auto& [fid, index]: result.file_indices) { for(auto& [symbol_id, relations]: index.relations) { std::ranges::sort(relations, [](const Relation& lhs, const Relation& rhs) { - return std::tuple(lhs.kind.value(), - lhs.range.begin, - lhs.range.end, - lhs.target_symbol) < std::tuple(rhs.kind.value(), - rhs.range.begin, - rhs.range.end, - rhs.target_symbol); + return std::tuple(lhs.kind, lhs.range.begin, lhs.range.end, lhs.target_symbol) < + std::tuple(rhs.kind, rhs.range.begin, rhs.range.end, rhs.target_symbol); }); auto range = std::ranges::unique(relations, [](const Relation& lhs, const Relation& rhs) { @@ -594,7 +589,7 @@ void FileIndex::lookup(SymbolHash symbol, if(it == relations.end()) return; for(auto& r: it->second) { - if(r.kind & kind) { + if(RelationKind(r.kind) & kind) { if(!callback(r)) return; } @@ -640,127 +635,57 @@ TUIndex TUIndex::build(CompilationUnitRef unit, bool interested_only) { return index; } -void TUIndex::serialize(llvm::raw_ostream& os) const { - fbs::FlatBufferBuilder builder(4096); - - llvm::SmallVector buffer; - - auto paths = - transform(graph.paths, [&](const std::string& p) { return builder.CreateString(p); }); - - auto syms = transform(symbols, [&](auto&& value) { - auto& [symbol_id, symbol] = value; - buffer.clear(); - buffer.resize_for_overwrite(symbol.reference_files.getSizeInBytes(false)); - symbol.reference_files.write(buffer.data(), false); - return binary::CreateSymbolEntry(builder, - symbol_id, - binary::CreateSymbol(builder, - CreateString(builder, symbol.name), - symbol.kind.value(), - CreateVector(builder, buffer), - static_cast(symbol.scope))); - }); - - /// Serialize a single FileIndex into a TUFileIndexEntry. - auto serialize_file_index = [&](std::uint32_t fid, const FileIndex& index) { - auto occs = CreateStructVector(builder, index.occurrences); - auto rels = transform(index.relations, [&](auto&& value) { - auto& [symbol_id, relations] = value; - return binary::CreateTUFileRelationsEntry( - builder, - symbol_id, - CreateStructVector(builder, relations)); - }); - return binary::CreateTUFileIndexEntry(builder, fid, occs, CreateVector(builder, rels)); - }; +void TUIndex::serialize(llvm::raw_ostream& os) { + format_version = index_format_version; - /// Convert FileID-keyed file_indices to path_id-keyed entries. - llvm::SmallVector> file_idx_vec; - for(auto& [fid, index]: file_indices) { - auto pid = graph.path_id(fid); - file_idx_vec.push_back(serialize_file_index(pid, index)); + /// Convert the FileID-keyed working state into the persisted + /// path_id-keyed form; multiple FileIDs can share a path id (repeated + /// header contexts), last-wins. A deserialized index has no FileID-keyed + /// state at all — its path-keyed rows already are the persisted form, so + /// re-serializing must not wipe them. + if(!file_indices.empty()) { + path_file_indices.clear(); + for(auto& [fid, file_index]: file_indices) { + path_file_indices[graph.path_id(fid)] = file_index; + } } - /// Main file is the last path in graph.paths (convention from IncludeGraph). - auto main_idx = - serialize_file_index(static_cast(graph.paths.size() - 1), main_file_index); - - auto tu_index = - binary::CreateTUIndex(builder, - static_cast(built_at.count()), - CreateVector(builder, paths), - CreateStructVector(builder, graph.locations), - CreateVector(builder, syms), - builder.CreateVector(file_idx_vec.data(), file_idx_vec.size()), - main_idx, - CreateVector(builder, graph.path_hashes)); - - builder.Finish(tu_index); - os.write(safe_cast(builder.GetBufferPointer()), builder.GetSize()); + serialize_blob(*this, os); } -TUIndex TUIndex::from(const void* data) { - auto root = fbs::GetRoot(data); - - TUIndex index; - index.built_at = std::chrono::milliseconds(root->built_at()); - - for(auto p: *root->paths()) { - index.graph.paths.emplace_back(p->str()); +std::optional TUIndex::from(llvm::StringRef data) { + std::optional index{std::in_place}; + if(!deserialize_blob(data, *index) || index->format_version != index_format_version) { + return std::nullopt; } - - for(auto loc: *root->locations()) { - index.graph.locations.emplace_back(*safe_cast(loc)); - } - - if(root->path_hashes()) { - index.graph.path_hashes.assign(root->path_hashes()->begin(), root->path_hashes()->end()); - } - index.graph.path_hashes.resize(index.graph.paths.size(), 0); - - for(auto entry: *root->symbols()) { - auto& symbol = index.symbols[entry->symbol_id()]; - symbol.name = entry->symbol()->name()->str(); - symbol.kind = SymbolKind(static_cast(entry->symbol()->kind())); - symbol.scope = static_cast(entry->symbol()->scope()); - symbol.reference_files = read_bitmap(entry->symbol()->refs()); - } - - /// Helper to deserialize a TUFileIndexEntry into a FileIndex. - auto deserialize_file_index = [](const binary::TUFileIndexEntry* entry) -> FileIndex { - FileIndex fi; - if(entry->occurrences()) { - fi.occurrences.reserve(entry->occurrences()->size()); - for(auto o: *entry->occurrences()) { - fi.occurrences.emplace_back(*safe_cast(o)); - } - } - if(entry->relations()) { - for(auto rel_entry: *entry->relations()) { - auto& rels = fi.relations[rel_entry->symbol()]; - if(rel_entry->relations()) { - rels.reserve(rel_entry->relations()->size()); - for(auto r: *rel_entry->relations()) { - rels.emplace_back(*safe_cast(r)); - } - } - } - } - return fi; + // The verifier checks structure, not cross-field consistency: consumers + // index path_hashes by path id, so normalize its length to the path + // table's (absent hashes read as 0 = "unavailable"). + index->graph.path_hashes.resize(index->graph.paths.size(), 0); + + // Nor does it constrain field values, and every decoded path id is + // dereferenced against the path table without further checks — graph + // locations and per-file rows in Indexer::merge, reference_files through + // ProjectIndex::merge's file_ids_map. A blob carrying an out-of-range + // one is rejected as a whole. + auto in_range = [count = index->graph.paths.size()](std::uint32_t path_id) { + return path_id < count; }; - - /// Populate path_file_indices keyed by path_id (no clang::FileID needed). - if(root->file_indices()) { - for(auto entry: *root->file_indices()) { - index.path_file_indices[entry->file_id()] = deserialize_file_index(entry); + for(auto& location: index->graph.locations) { + if(!in_range(location.path_id)) { + return std::nullopt; } } - - if(root->main_file_index()) { - index.main_file_index = deserialize_file_index(root->main_file_index()); + for(auto& [path_id, _]: index->path_file_indices) { + if(!in_range(path_id)) { + return std::nullopt; + } + } + for(auto& [_, symbol]: index->symbols) { + if(!symbol.reference_files.isEmpty() && !in_range(symbol.reference_files.maximum())) { + return std::nullopt; + } } - return index; } diff --git a/src/index/tu_index.h b/src/index/tu_index.h index fc91aac93..c727d54c5 100644 --- a/src/index/tu_index.h +++ b/src/index/tu_index.h @@ -4,6 +4,7 @@ #include #include #include +#include #include #include @@ -11,6 +12,8 @@ #include "semantic/symbol.h" #include "support/bitmap.h" +#include "kota/codec/macro.h" +#include "kota/meta/annotation.h" #include "llvm/ADT/STLFunctionalExtras.h" #include "llvm/Support/raw_ostream.h" @@ -34,7 +37,10 @@ enum class SymbolScope : std::uint8_t { }; struct Relation { - RelationKind kind; + /// The raw enum rather than the RelationKind wrapper: the wrapper's + /// constructors hide it from reflection, and reflection is what lets a + /// relation vector persist as one contiguous struct vector. + RelationKind::Kind kind = RelationKind::Invalid; std::uint32_t padding = 0; @@ -62,7 +68,11 @@ struct Occurrence { }; struct FileIndex { - llvm::DenseMap> relations; + /// The braces matter: fbs decode value-constructs map entries with + /// `FileIndex{}`, and without an initializer this member would be + /// copy-initialized from an empty list, which DenseMap's explicit + /// default constructor rejects. + llvm::DenseMap> relations{}; std::vector occurrences; @@ -91,6 +101,13 @@ struct Symbol { using SymbolTable = llvm::DenseMap; struct TUIndex { + /// Persisted-blob schema version (index_format_version), stamped by + /// serialize() and gated by from(). These blobs never touch disk — they + /// travel worker→server over IPC — but a worker respawned after the + /// binary on disk changed can be one build ahead of the server, and a + /// layout change need not be structurally detectable. + std::uint32_t format_version = 0; + /// The building timestamp of this file. std::chrono::milliseconds built_at; @@ -99,10 +116,14 @@ struct TUIndex { SymbolTable symbols; - llvm::DenseMap file_indices; + /// Build-time working state keyed by FileID — clang::FileID means nothing + /// outside the compilation, so it never persists; serialize() converts it + /// through graph.path_id. + KOTATSU_ANNOTATE(skip = true) + > file_indices; - /// File indices keyed by path_id, populated by from() for deserialized data. - /// When built from AST, this is empty and file_indices (keyed by FileID) is used. + /// File indices keyed by path_id: populated from file_indices by + /// serialize(), and directly by from() for deserialized data. llvm::DenseMap path_file_indices; FileIndex main_file_index; @@ -115,9 +136,14 @@ struct TUIndex { /// under the main path id. static TUIndex build(CompilationUnitRef unit, bool interested_only = false); - void serialize(llvm::raw_ostream& os) const; + /// Serialization reflects this object directly (path_file_indices is + /// populated from file_indices first — hence non-const). + void serialize(llvm::raw_ostream& os); - static TUIndex from(const void* data); + /// Verify and deserialize a buffer; nullopt when structural + /// verification fails, the format version differs, or a decoded path id + /// falls outside the blob's own path table. + static std::optional from(llvm::StringRef data); }; } // namespace clice::index diff --git a/src/server/compiler/compiler.cpp b/src/server/compiler/compiler.cpp index af8ea3a47..b93d4af98 100644 --- a/src/server/compiler/compiler.cpp +++ b/src/server/compiler/compiler.cpp @@ -1189,10 +1189,12 @@ kota::task<> Compiler::run_compile(std::shared_ptr session) { pc->succeeded = true; record_deps(*session, result.value().deps, result.value().build_at); - if(!result.value().tu_index_data.empty()) { - auto tu_index = index::TUIndex::from(result.value().tu_index_data.data()); - session->file_index = std::move(tu_index.main_file_index); - session->symbols = std::move(tu_index.symbols); + auto tu_index = result.value().tu_index_data.empty() + ? std::nullopt + : index::TUIndex::from(result.value().tu_index_data); + if(tu_index) { + session->file_index = std::move(tu_index->main_file_index); + session->symbols = std::move(tu_index->symbols); } else { // The AST and the file index settle together — that pairing is // what lets navigation trust the index after ensure_compiled. A diff --git a/src/server/compiler/indexer.cpp b/src/server/compiler/indexer.cpp index 4d90d8226..d9642b3dd 100644 --- a/src/server/compiler/indexer.cpp +++ b/src/server/compiler/indexer.cpp @@ -27,7 +27,13 @@ namespace clice { void Indexer::merge(const void* tu_index_data, std::size_t size) { - auto tu_index = index::TUIndex::from(tu_index_data); + auto loaded = + index::TUIndex::from(llvm::StringRef(static_cast(tu_index_data), size)); + if(!loaded) { + LOG_WARN("Ignoring TUIndex that failed verification"); + return; + } + auto& tu_index = *loaded; if(tu_index.graph.paths.empty()) { LOG_WARN("Ignoring TUIndex with empty path graph"); return; @@ -345,10 +351,7 @@ void Indexer::load() { // An unreadable or old-format blob loads as "no index on disk": // everything is swept and rebuilt once in the background. llvm::SmallVector manifest; - auto loaded = index::ProjectIndex::from((*buf)->getBufferStart(), - (*buf)->getBufferSize(), - workspace.path_pool, - manifest); + auto loaded = index::ProjectIndex::from((*buf)->getBuffer(), workspace.path_pool, manifest); if(loaded) { workspace.project_index = std::move(*loaded); has_project = true; diff --git a/src/server/service/query.cpp b/src/server/service/query.cpp index a72cea1f4..492a301af 100644 --- a/src/server/service/query.cpp +++ b/src/server/service/query.cpp @@ -605,7 +605,7 @@ void IndexQuery::collect_unique_targets(index::SymbolHash hash, if(rel_it == session.file_index->relations.end()) return true; for(auto& r: rel_it->second) { - if(r.kind & kind) { + if(RelationKind(r.kind) & kind) { if(seen.insert(r.target_symbol).second) { targets.push_back(r.target_symbol); } diff --git a/src/server/state/workspace.h b/src/server/state/workspace.h index 2278189bd..035a32a89 100644 --- a/src/server/state/workspace.h +++ b/src/server/state/workspace.h @@ -35,7 +35,7 @@ class ContextResolver; /// On-disk cache layout version (CacheStore root `cache/v{N}`). /// Bump to discard all cached artifacts after incompatible format changes. -constexpr inline std::uint32_t cache_format_version = 4; +constexpr inline std::uint32_t cache_format_version = 5; /// Sentinel for "no path": path pool ids start at 0, so 0 is a real file. constexpr inline std::uint32_t no_path_id = ~0u; diff --git a/src/server/worker/stateless_worker.cpp b/src/server/worker/stateless_worker.cpp index 032d2fd48..6676cd81a 100644 --- a/src/server/worker/stateless_worker.cpp +++ b/src/server/worker/stateless_worker.cpp @@ -67,7 +67,7 @@ static std::string serialize_preamble_state(CompilationUnit& unit, std::uint32_t std::string blob; llvm::raw_string_ostream os(blob); index::PreambleState::serialize(unit, - tu_index, + std::move(tu_index), links, inactive.regions, inactive.open_stack, diff --git a/tests/unit/index/merged_index_tests.cpp b/tests/unit/index/merged_index_tests.cpp index a97a30581..1052c33f9 100644 --- a/tests/unit/index/merged_index_tests.cpp +++ b/tests/unit/index/merged_index_tests.cpp @@ -1,11 +1,16 @@ +#include #include +#include +#include -#include "schema_generated.h" #include "test/temp_dir.h" #include "test/test.h" #include "test/tester.h" #include "index/merged_index.h" +#include "index/serialization.h" +#include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/SmallVector.h" #include "llvm/Support/raw_ostream.h" #include "llvm/Support/xxhash.h" @@ -321,7 +326,7 @@ TEST_CASE(RemergeReplacesContribution) { index::SymbolHash defined{}; for(auto& [symbol, relations]: header_idx.relations) { for(auto& relation: relations) { - if(relation.kind & RelationKind(RelationKind::Definition)) { + if(RelationKind(relation.kind) & RelationKind(RelationKind::Definition)) { defined = symbol; } } @@ -369,7 +374,7 @@ TEST_CASE(RemergePreservesOtherTus) { index::SymbolHash defined{}; for(auto& [symbol, relations]: header_idx.relations) { for(auto& relation: relations) { - if(relation.kind & RelationKind(RelationKind::Definition)) { + if(RelationKind(relation.kind) & RelationKind(RelationKind::Definition)) { defined = symbol; } } @@ -416,6 +421,46 @@ TEST_CASE(CompactionDropsMasked) { ASSERT_FALSE(found); } +TEST_CASE(SerializeCompactsInPlace) { + // Two contributions with distinct content, so each gets its own + // canonical id; only one is removed. + index::FileIndex live_idx; + live_idx.occurrences.emplace_back(index::Range{0, 3}, 100); + index::FileIndex dead_idx; + dead_idx.occurrences.emplace_back(index::Range{10, 13}, 200); + + index::MergedIndex merged; + merged.merge("tu0", std::uint32_t(0), live_idx, "synthetic"); + merged.merge("tu1", std::uint32_t(0), dead_idx, "synthetic"); + merged.remove("tu1"); + + llvm::SmallString<1024> buf; + llvm::raw_svector_ostream os(buf); + merged.serialize(os); + + // The save flip is conditional: when it does not happen, the in-memory + // impl — now compacted by serialize() — keeps serving queries. Surviving + // rows must still resolve and removed ones stay gone. + auto hits_at = [&](std::uint32_t offset) { + std::size_t hits = 0; + merged.lookup(offset, [&](const index::Occurrence&) { + hits += 1; + return true; + }); + return hits; + }; + ASSERT_EQ(hits_at(1), 1u); + ASSERT_EQ(hits_at(11), 0u); + ASSERT_TRUE(merged.has_contribution("tu0")); + ASSERT_FALSE(merged.has_contribution("tu1")); + + // A second serialize of the compacted impl round-trips identically. + llvm::SmallString<1024> again; + llvm::raw_svector_ostream os2(again); + merged.serialize(os2); + ASSERT_EQ(llvm::StringRef(buf), llvm::StringRef(again)); +} + TEST_CASE(HasContributionTracking) { add_file("header.h", R"( #pragma once @@ -673,43 +718,290 @@ TEST_CASE(OldShardDiscarded) { // A version-less (format_version=0) shard from an older build is silently // discarded — load returns an empty index, as if nothing were on disk. + // Only the version slot is written: every other field reads back absent, + // which is structurally valid — rejection must come from the version + // check. { - namespace binary = clice::index::binary; - flatbuffers::FlatBufferBuilder builder; - auto content = builder.CreateString("stale-shard"); - auto paths = builder.CreateVector>({}); - auto cache = builder.CreateVector>({}); - auto headers = builder.CreateVector>({}); - auto compilations = - builder.CreateVector>({}); - auto occurrences = builder.CreateVector>({}); - auto relations = - builder.CreateVector>({}); - auto root = binary::CreateMergedIndex(builder, - 0, - paths, - cache, - headers, - compilations, - occurrences, - relations, - 0, - content, - 0, - 0, - /*format_version=*/0); - builder.Finish(root); + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto blob = kota::codec::fbs::to_bytes(VersionOnly{}); + ASSERT_TRUE(blob.has_value()); auto path = dir.path("stale.idx"); std::error_code ec; llvm::raw_fd_ostream os(path, ec); - os.write(reinterpret_cast(builder.GetBufferPointer()), builder.GetSize()); + os.write(reinterpret_cast(blob->data()), blob->size()); os.flush(); auto loaded = index::MergedIndex::load(path); ASSERT_TRUE(loaded.content().empty()); ASSERT_TRUE(loaded.need_update()); } + + // Positive control: the same single-slot shape carrying the CURRENT + // version is kept — slot 0 really is the version slot and the rejection + // above comes from its value, not from the blob's shape. + { + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto blob = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); + ASSERT_TRUE(blob.has_value()); + + auto path = dir.path("current.idx"); + std::error_code ec; + llvm::raw_fd_ostream os(path, ec); + os.write(reinterpret_cast(blob->data()), blob->size()); + os.flush(); + + ASSERT_TRUE(index::MergedIndex::load(path).loaded()); + } +} + +TEST_CASE(GarbageLoadRejected) { + TempDir dir; + dir.touch("garbage.idx", "not a flatbuffer"); + + auto loaded = index::MergedIndex::load(dir.path("garbage.idx")); + ASSERT_FALSE(loaded.loaded()); + ASSERT_TRUE(loaded.content().empty()); + ASSERT_TRUE(loaded.need_update()); + + // Queries on the rejected shard answer with silence, not UB. + bool visited = false; + loaded.lookup(0, [&](const index::Occurrence&) { + visited = true; + return true; + }); + ASSERT_FALSE(visited); + + loaded.lookup(index::SymbolHash(1), RelationKind::Definition, [&](const index::Relation&) { + visited = true; + return true; + }); + ASSERT_FALSE(visited); + + std::string name; + SymbolKind kind; + ASSERT_FALSE(loaded.find_symbol(1, name, kind)); +} + +TEST_CASE(CorruptShardRejected) { + TempDir dir; + + index::MergedIndex merged; + index::FileIndex file_idx; + merged.merge("tu0", std::chrono::milliseconds(1), {}, file_idx, "corrupt-me"); + + llvm::SmallString<1024> blob; + llvm::raw_svector_ostream os(blob); + merged.serialize(os); + ASSERT_TRUE(blob.size() > 8); + + auto write = [&](llvm::StringRef name, llvm::StringRef bytes) { + dir.touch(name, bytes); + return dir.path(name); + }; + + // Sanity: the intact bytes load, so the rejections below are earned. + ASSERT_TRUE(index::MergedIndex::load(write("valid.idx", blob)).loaded()); + + llvm::StringRef bytes(blob.data(), blob.size()); + ASSERT_FALSE( + index::MergedIndex::load(write("half.idx", bytes.take_front(bytes.size() / 2))).loaded()); + ASSERT_FALSE(index::MergedIndex::load(write("minus1.idx", bytes.drop_back(1))).loaded()); + + // Bytes 4-7 carry the buffer identifier; a blob from another format + // must be rejected up front. + std::string clobbered = bytes.str(); + for(std::size_t i = 4; i < 8; ++i) { + clobbered[i] = 'X'; + } + ASSERT_FALSE(index::MergedIndex::load(write("clobbered.idx", clobbered)).loaded()); +} + +TEST_CASE(OutOfRangeCanonicalIdRejected) { + TempDir dir; + + // Field order MUST mirror the persisted shapes in merged_index.cpp + // (MergedIndex::Impl prefix — skip-annotated fields occupy no slot — + // HeaderContext, IncludeContext, CompilationContext prefix); the + // trailing fields read back absent, which is structurally valid. + struct IncludeContextMirror { + std::uint32_t include_id = 0; + std::uint32_t canonical_id = 0; + }; + + struct HeaderContextMirror { + std::uint32_t version = 0; + llvm::SmallVector includes; + }; + + // The vector matters beyond field parity: without one the mirror would + // be trivially copyable and encode as an inline struct, while the real + // CompilationContext encodes as a table — the verifier tells them apart. + struct CompilationContextMirror { + std::uint32_t version = 0; + std::uint32_t canonical_id = 0; + std::uint64_t build_at = 0; + std::vector include_locations; + }; + + struct ReprMirror { + std::uint32_t format_version = 0; + std::vector paths; + std::string content; + std::vector line_starts; + llvm::SmallDenseMap header_contexts; + llvm::SmallDenseMap compilation_contexts; + std::vector> canonical_cache; + std::uint32_t max_canonical_id = 0; + }; + + // A consistent base: one path, one canonical id, one header context + // referencing it. + auto base = [] { + ReprMirror mirror; + mirror.format_version = index::index_format_version; + mirror.max_canonical_id = 1; + mirror.paths = {"/proj/tu.cpp"}; + mirror.canonical_cache.emplace_back("hash", 0); + mirror.header_contexts[0].includes.push_back({.include_id = 0, .canonical_id = 0}); + return mirror; + }; + + // Structure and version pass, so load() accepts the blob off disk; the + // first mutation materializes it in memory, where the id values face + // the range check. Nullopt = the blob never reached that check. + auto materialized_contribution = [&](llvm::StringRef name, + const ReprMirror& mirror) -> std::optional { + auto blob = kota::codec::fbs::to_bytes(mirror); + if(!blob) { + return std::nullopt; + } + dir.touch(name, llvm::StringRef(reinterpret_cast(blob->data()), blob->size())); + auto shard = index::MergedIndex::load(dir.path(name)); + if(!shard.loaded()) { + return std::nullopt; + } + shard.remove("/proj/never-indexed.cpp"); + return shard.has_contribution("/proj/tu.cpp"); + }; + + // Positive control first: the base materializes intact, so the + // rejections below come from the hostile ids, not the blob's shape. + auto good = materialized_contribution("good.idx", base()); + ASSERT_TRUE(good.has_value() && *good); + + // Structural verification does not constrain field values: each blob + // carries one canonical id at or past max_canonical_id, which would + // index canonical_ref_counts out of bounds if the in-memory load + // accepted it. The blob is dropped and the shard reads as empty. + auto bad_cache = base(); + bad_cache.canonical_cache.front().second = 5; + auto cache_verdict = materialized_contribution("bad-cache.idx", bad_cache); + ASSERT_TRUE(cache_verdict.has_value() && !*cache_verdict); + + auto bad_include = base(); + bad_include.header_contexts[0].includes.front().canonical_id = 5; + auto include_verdict = materialized_contribution("bad-include.idx", bad_include); + ASSERT_TRUE(include_verdict.has_value() && !*include_verdict); + + auto bad_compilation = base(); + bad_compilation.compilation_contexts[0].canonical_id = 5; + auto compilation_verdict = materialized_contribution("bad-compilation.idx", bad_compilation); + ASSERT_TRUE(compilation_verdict.has_value() && !*compilation_verdict); +} + +TEST_CASE(BufferPathLookupParity) { + build_index(R"( + void §(a)alpha_func() {} + int §(b)beta_var = 1; + )"); + + index::MergedIndex merged; + auto fid = unit->interested_file(); + merged.merge("tu0", + tu_index.graph.include_location_id(fid), + tu_index.main_file_index, + unit->interested_content()); + + auto hash_at = [&](llvm::StringRef pos) { + auto offset = point(pos); + index::SymbolHash hash = 0; + merged.lookup(offset, [&](const index::Occurrence& occ) { + hash = occ.target; + return false; + }); + return hash; + }; + + using Row = std::tuple; + auto definitions = [](index::MergedIndex& index, index::SymbolHash hash) { + std::vector rows; + index.lookup(hash, RelationKind::Definition, [&](const index::Relation& relation) { + rows.emplace_back(relation.range.begin, relation.range.end, relation.target_symbol); + return true; + }); + std::ranges::sort(rows); + return rows; + }; + + index::SymbolHash hashes[2] = {hash_at("a"), hash_at("b")}; + std::vector expected[2]; + for(std::size_t i = 0; i < 2; ++i) { + ASSERT_TRUE(hashes[i] != 0); + expected[i] = definitions(merged, hashes[i]); + ASSERT_FALSE(expected[i].empty()); + } + ASSERT_TRUE(hashes[0] != hashes[1]); + + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + merged.serialize(os); + auto restored = index::MergedIndex(buf); + + // The zero-copy view is the production read path: it must agree BEFORE + // anything materializes the impl (operator== would, so it comes last). + for(std::size_t i = 0; i < 2; ++i) { + ASSERT_TRUE(definitions(restored, hashes[i]) == expected[i]); + } + + ASSERT_FALSE(restored.line_starts().empty()); + ASSERT_TRUE(std::ranges::equal(restored.line_starts(), merged.line_starts())); +} + +TEST_CASE(BufferPathMultiOccurrenceLookup) { + // Synthesized occurrences sorted by (begin, end, target), spaced so each + // probe hits exactly one range (contains() is inclusive at both ends). + index::FileIndex file_idx; + index::SymbolHash target = 100; + for(std::uint32_t begin = 0; begin < 60; begin += 10) { + file_idx.occurrences.emplace_back(index::Range{begin, begin + 3}, target++); + } + + index::MergedIndex merged; + merged.merge("tu0", std::uint32_t(0), file_idx, "synthetic"); + + llvm::SmallString<1024> buf; + llvm::raw_svector_ostream os(buf); + merged.serialize(os); + auto restored = index::MergedIndex(buf); + + // The buffer path binary-searches the serialized rows: every probe must + // land on exactly its own range. + for(auto& occurrence: file_idx.occurrences) { + std::vector hits; + restored.lookup(occurrence.range.begin + 1, [&](const index::Occurrence& hit) { + hits.push_back(hit); + return true; + }); + ASSERT_EQ(hits.size(), 1u); + ASSERT_TRUE(hits.front() == occurrence); + } } std::uint64_t file_hash(llvm::StringRef path) { diff --git a/tests/unit/index/persisted_index_tests.cpp b/tests/unit/index/persisted_index_tests.cpp index 867d6b820..341c1c718 100644 --- a/tests/unit/index/persisted_index_tests.cpp +++ b/tests/unit/index/persisted_index_tests.cpp @@ -1,5 +1,6 @@ #include "test/test.h" #include "index/project_index.h" +#include "index/serialization.h" #include "llvm/Support/raw_ostream.h" @@ -29,7 +30,7 @@ TEST_CASE(SerializeCollectsGarbage) { clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); ASSERT_TRUE(fresh.find("/proj/used.cpp").has_value()); ASSERT_FALSE(fresh.find("/proj/garbage.cpp").has_value()); @@ -48,7 +49,7 @@ TEST_CASE(RemapAcrossSessions) { clice::PathPool fresh; fresh.intern("/proj/opened-first.cpp"); llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); auto id = fresh.find("/proj/used.cpp"); @@ -67,7 +68,7 @@ TEST_CASE(ShardManifestRoundTrip) { clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); ASSERT_EQ(shards.size(), 1u); ASSERT_EQ(fresh.resolve(shards.front()), "/proj/tu.cpp"); @@ -78,7 +79,70 @@ TEST_CASE(OldBlobDiscarded) { llvm::SmallVector shards; // Arbitrary bytes are rejected by verification, not misread. const char junk[] = "not a flatbuffer"; - ASSERT_FALSE(index::ProjectIndex::from(junk, sizeof(junk), pool, shards).has_value()); + ASSERT_FALSE( + index::ProjectIndex::from(llvm::StringRef(junk, sizeof(junk)), pool, shards).has_value()); +} + +llvm::StringRef bytes_of(const std::vector& blob) { + return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); +} + +TEST_CASE(VersionGate) { + // Only the version slot is written: every other field reads back absent, + // which is structurally valid — the verdict must hinge on the value. + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + clice::PathPool pool; + llvm::SmallVector shards; + + auto stale = kota::codec::fbs::to_bytes(VersionOnly{}); + ASSERT_TRUE(stale.has_value()); + ASSERT_FALSE(index::ProjectIndex::from(bytes_of(*stale), pool, shards).has_value()); + + auto current = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); + ASSERT_TRUE(current.has_value()); + auto loaded = index::ProjectIndex::from(bytes_of(*current), pool, shards); + ASSERT_TRUE(loaded.has_value()); + ASSERT_TRUE(loaded->symbols.empty()); + ASSERT_TRUE(shards.empty()); +} + +TEST_CASE(OutOfRangeLocalIdsDropped) { + // Field order MUST mirror ProjectIndex (project_index.h): + // format_version, paths, symbols, shards. + struct ProjectIndexMirror { + std::uint32_t format_version = 0; + std::vector> paths; + index::SymbolTable symbols; + std::vector shards; + }; + + ProjectIndexMirror mirror; + mirror.format_version = index::index_format_version; + mirror.paths = { + {0, "/proj/used.cpp"} + }; + auto& symbol = mirror.symbols[42]; + symbol.name = "sym"; + symbol.reference_files.add(7); // Only pool id 0 is in the table. + mirror.shards = {9}; + + auto blob = kota::codec::fbs::to_bytes(mirror); + ASSERT_TRUE(blob.has_value()); + + // Dangling local ids are dropped, not misresolved into the pool: the + // blob still loads, the symbol survives with an empty bitmap, and no + // shard is fetched. + clice::PathPool pool; + llvm::SmallVector shards; + auto loaded = index::ProjectIndex::from(bytes_of(*blob), pool, shards); + ASSERT_TRUE(loaded.has_value()); + ASSERT_TRUE(loaded->symbols.contains(42)); + ASSERT_EQ(loaded->symbols[42].name, "sym"); + ASSERT_EQ(loaded->symbols[42].reference_files.cardinality(), 0u); + ASSERT_TRUE(shards.empty()); } }; // TEST_SUITE(PersistedIndex) diff --git a/tests/unit/index/preamble_state_tests.cpp b/tests/unit/index/preamble_state_tests.cpp index fa167c0b3..df0004675 100644 --- a/tests/unit/index/preamble_state_tests.cpp +++ b/tests/unit/index/preamble_state_tests.cpp @@ -1,8 +1,8 @@ -#include "schema_generated.h" #include "test/temp_dir.h" #include "test/test.h" #include "test/tester.h" #include "index/preamble_state.h" +#include "index/serialization.h" #include "llvm/Support/raw_ostream.h" @@ -171,6 +171,48 @@ int main() { §(ref)⟦§(ref)foo⟧(); return 0; } EXPECT_TRUE(found_relation); } +TEST_CASE(MoveConsumedIndex) { + // The production path (stateless worker) moves the TUIndex into + // serialize; the blob must be complete even though the index is + // consumed rather than copied. + add_file("foo.h", R"( +inline void §(def)⟦foo⟧() {} +)"); + add_main("main.cpp", R"( +#include "foo.h" +int main() { §(ref)⟦foo⟧(); return 0; } +)"); + ASSERT_TRUE(compile()); + tu_index = index::TUIndex::build(*unit); + auto foo = hash_of("foo"); + + auto blob_path = dir.path("moved.pch.idx"); + std::error_code ec; + llvm::raw_fd_ostream os(blob_path, ec); + ASSERT_FALSE(bool(ec)); + index::PreambleState::serialize(*unit, std::move(tu_index), {}, {}, {}, os); + os.close(); + + state = index::PreambleState::load(blob_path); + ASSERT_TRUE(state != nullptr); + + bool found = false; + state->lookup(foo, + RelationKind::Definition, + [&](const index::PreambleState::File& file, const index::Relation& r) { + EXPECT_TRUE(file.path.ends_with("foo.h")); + EXPECT_EQ(dump(r.range), dump(range("def", "foo.h"))); + found = true; + return false; + }); + EXPECT_TRUE(found); + + std::string name; + SymbolKind kind; + EXPECT_TRUE(state->find_symbol(foo, name, kind)); + EXPECT_EQ(name, "foo"); +} + TEST_CASE(SymbolTableLookup) { add_file("foo.h", R"( inline void §(def)⟦foo⟧() {} @@ -225,22 +267,73 @@ TEST_CASE(RejectBadBlob) { TEST_CASE(RejectVersionMismatch) { // A structurally valid blob written by a different format version (0 is // what a version-less blob reads back) must load as missing, so the - // PCH pair rebuilds instead of serving a stale layout. - flatbuffers::FlatBufferBuilder builder(64); - auto paths = builder.CreateVector(std::vector>{}); - auto files = - builder.CreateVector(std::vector>{}); - auto symbols = builder.CreateVector( - std::vector>{}); - builder.Finish(index::binary::CreatePreambleState(builder, 0, paths, files, 0, symbols)); + // PCH pair rebuilds instead of serving a stale layout. The blob only + // needs the version slot: every other field reads back absent, which is + // structurally valid — rejection must come from the version check. + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto blob = kota::codec::fbs::to_bytes(VersionOnly{}); + ASSERT_TRUE(blob.has_value()); auto blob_path = dir.path("stale.pch.idx"); dir.touch("stale.pch.idx", - llvm::StringRef(reinterpret_cast(builder.GetBufferPointer()), - builder.GetSize())); + llvm::StringRef(reinterpret_cast(blob->data()), blob->size())); EXPECT_TRUE(index::PreambleState::load(blob_path) == nullptr); } +TEST_CASE(AcceptCurrentVersionBlob) { + // Positive control for RejectVersionMismatch: the same single-slot shape + // carrying the CURRENT version loads — slot 0 really is the version slot + // and the rejection comes from its value, not from the blob's shape. + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto blob = kota::codec::fbs::to_bytes(VersionOnly{index::preamble_format_version}); + ASSERT_TRUE(blob.has_value()); + + dir.touch("current.pch.idx", + llvm::StringRef(reinterpret_cast(blob->data()), blob->size())); + EXPECT_TRUE(index::PreambleState::load(dir.path("current.pch.idx")) != nullptr); +} + +TEST_CASE(RejectCorruptBlob) { + add_main("main.cpp", R"( +int main() { return 0; } +)"); + build_state(); + + auto buffer = llvm::MemoryBuffer::getFile(dir.path("state.pch.idx")); + ASSERT_TRUE(bool(buffer)); + auto bytes = (*buffer)->getBuffer(); + ASSERT_TRUE(bytes.size() > 8); + + dir.touch("truncated.pch.idx", bytes.take_front(bytes.size() / 2)); + EXPECT_TRUE(index::PreambleState::load(dir.path("truncated.pch.idx")) == nullptr); + + // Bytes 4-7 carry the buffer identifier; a blob from another format + // must be rejected up front. + std::string clobbered = bytes.str(); + for(std::size_t i = 4; i < 8; ++i) { + clobbered[i] = 'X'; + } + dir.touch("clobbered.pch.idx", clobbered); + EXPECT_TRUE(index::PreambleState::load(dir.path("clobbered.pch.idx")) == nullptr); +} + +TEST_CASE(SourcePathAndContent) { + add_main("main.cpp", R"( +int value = 42; +int other = 1; +)"); + build_state(); + + EXPECT_TRUE(state->source_path().ends_with("main.cpp")); + EXPECT_EQ(state->preamble_content(), unit->interested_content()); +} + }; // TEST_SUITE(PreambleState) } // namespace diff --git a/tests/unit/index/project_index_tests.cpp b/tests/unit/index/project_index_tests.cpp index d12951e56..5e795bddd 100644 --- a/tests/unit/index/project_index_tests.cpp +++ b/tests/unit/index/project_index_tests.cpp @@ -1,6 +1,7 @@ #include "test/test.h" #include "test/tester.h" #include "index/project_index.h" +#include "index/serialization.h" namespace clice::testing { namespace { @@ -136,7 +137,7 @@ TEST_CASE(SerializationRoundTrip) { // Deserialize into a fresh pool, as a new session would. clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); auto& restored = *loaded; @@ -201,7 +202,7 @@ TEST_CASE(NameSurvivesRoundTrip) { project.serialize(os, pool, {}); clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); auto& restored = *loaded; @@ -243,6 +244,23 @@ TEST_CASE(LocalSymbolsExcluded) { ASSERT_FALSE(found_local); } +TEST_CASE(EmptyPathRejected) { + // A codec-valid blob whose path table carries an empty entry is corrupt: + // the writer only emits interned (never empty) paths. + index::ProjectIndex corrupt; + corrupt.format_version = index::index_format_version; + corrupt.paths.emplace_back(0, ""); + + llvm::SmallString<128> buf; + llvm::raw_svector_ostream os(buf); + index::serialize_blob(corrupt, os); + + clice::PathPool pool; + llvm::SmallVector shards; + ASSERT_FALSE(index::ProjectIndex::from(buf.str(), pool, shards).has_value()); + ASSERT_TRUE(pool.paths.empty()); +} + TEST_CASE(ScopeRoundTrip) { index::TUIndex tu; ASSERT_TRUE(build_and_index(R"( @@ -261,7 +279,7 @@ TEST_CASE(ScopeRoundTrip) { project.serialize(os, pool, {}); clice::PathPool fresh; llvm::SmallVector shards; - auto loaded = index::ProjectIndex::from(buf.data(), buf.size(), fresh, shards); + auto loaded = index::ProjectIndex::from(buf.str(), fresh, shards); ASSERT_TRUE(loaded.has_value()); auto& restored = *loaded; diff --git a/tests/unit/index/tu_index_tests.cpp b/tests/unit/index/tu_index_tests.cpp index 947dad1c5..c60554830 100644 --- a/tests/unit/index/tu_index_tests.cpp +++ b/tests/unit/index/tu_index_tests.cpp @@ -5,6 +5,7 @@ #include "test/test.h" #include "test/tester.h" #include "feature/feature.h" +#include "index/serialization.h" #include "index/tu_index.h" #include "semantic/selection.h" @@ -91,7 +92,7 @@ void GO_TO_DEFINITION(llvm::StringRef pos, auto& relations = it->second; auto target = std::ranges::find_if(relations, [](const index::Relation& relation) { - return relation.kind.value() == static_cast(RelationKind::Definition); + return relation.kind == RelationKind::Definition; }); ASSERT_TRUE(target != relations.end()); @@ -273,7 +274,7 @@ TEST_CASE(Reference) { auto& relations = it->second; auto ref = std::ranges::find_if(relations, [](const index::Relation& r) { - return r.kind.value() == static_cast(RelationKind::Reference); + return r.kind == RelationKind::Reference; }); ASSERT_TRUE(ref != relations.end()); } @@ -297,18 +298,19 @@ TEST_CASE(BaseAndDerived) { auto base_hash = base_occs.front().target; auto derived_hash = derived_occs.front().target; - auto has_pair = [&](index::SymbolHash source, RelationKind kind, index::SymbolHash target) { - auto it = index.relations.find(source); - if(it == index.relations.end()) { - return false; - } - for(auto& r: it->second) { - if(r.kind.value() == static_cast(kind) && r.target_symbol == target) { - return true; + auto has_pair = + [&](index::SymbolHash source, RelationKind::Kind kind, index::SymbolHash target) { + auto it = index.relations.find(source); + if(it == index.relations.end()) { + return false; } - } - return false; - }; + for(auto& r: it->second) { + if(r.kind == kind && r.target_symbol == target) { + return true; + } + } + return false; + }; ASSERT_TRUE(has_pair(derived_hash, RelationKind::Base, base_hash)); ASSERT_TRUE(has_pair(base_hash, RelationKind::Derived, derived_hash)); @@ -335,7 +337,7 @@ TEST_CASE(CallerAndCallee) { bool found_callee = false; for(auto& r: caller_it->second) { - if(r.kind.value() == static_cast(RelationKind::Callee)) { + if(r.kind == RelationKind::Callee) { found_callee = true; break; } @@ -352,7 +354,7 @@ TEST_CASE(CallerAndCallee) { bool found_caller = false; for(auto& r: callee_it->second) { - if(r.kind.value() == static_cast(RelationKind::Caller)) { + if(r.kind == RelationKind::Caller) { found_caller = true; break; } @@ -388,8 +390,7 @@ TEST_CASE(MethodCallerCallee) { bool found_callee = false; for(auto& r: method_it->second) { - if(r.kind.value() == static_cast(RelationKind::Callee) && - r.target_symbol == callee_hash) { + if(r.kind == RelationKind::Callee && r.target_symbol == callee_hash) { found_callee = true; break; } @@ -401,8 +402,7 @@ TEST_CASE(MethodCallerCallee) { bool found_caller = false; for(auto& r: callee_it->second) { - if(r.kind.value() == static_cast(RelationKind::Caller) && - r.target_symbol == method_hash) { + if(r.kind == RelationKind::Caller && r.target_symbol == method_hash) { found_caller = true; break; } @@ -430,8 +430,7 @@ TEST_CASE(UsingRelationKey) { bool found_use = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::WeakReference) && - r.range == range("use")) { + if(r.kind == RelationKind::WeakReference && r.range == range("use")) { found_use = true; break; } @@ -515,8 +514,7 @@ TEST_CASE(DependentWeakReference) { bool found_weak = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::WeakReference) && - r.range == range("use")) { + if(r.kind == RelationKind::WeakReference && r.range == range("use")) { found_weak = true; break; } @@ -552,8 +550,7 @@ TEST_CASE(TypeDefinitionRelations) { return false; } for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::TypeDefinition) && - r.target_symbol == target_hash) { + if(r.kind == RelationKind::TypeDefinition && r.target_symbol == target_hash) { return true; } } @@ -587,11 +584,10 @@ TEST_CASE(ConstructorDestructorRelations) { bool found_ctor = false; bool found_dtor = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Constructor) && - r.target_symbol == ctor_hash) { + if(r.kind == RelationKind::Constructor && r.target_symbol == ctor_hash) { found_ctor = true; } - if(r.kind.value() == static_cast(RelationKind::Destructor)) { + if(r.kind == RelationKind::Destructor) { found_dtor = true; } } @@ -604,8 +600,7 @@ TEST_CASE(ConstructorDestructorRelations) { bool found_type = false; for(auto& r: ctor_it->second) { - if(r.kind.value() == static_cast(RelationKind::TypeDefinition) && - r.target_symbol == class_hash) { + if(r.kind == RelationKind::TypeDefinition && r.target_symbol == class_hash) { found_type = true; } } @@ -631,12 +626,10 @@ TEST_CASE(MacroRelations) { bool found_definition = false; bool found_reference = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Definition) && - r.range == range("def")) { + if(r.kind == RelationKind::Definition && r.range == range("def")) { found_definition = true; } - if(r.kind.value() == static_cast(RelationKind::Reference) && - r.range == range("use")) { + if(r.kind == RelationKind::Reference && r.range == range("use")) { found_reference = true; } } @@ -656,7 +649,7 @@ TEST_CASE(ModuleName) { bool found_definition = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Definition)) { + if(r.kind == RelationKind::Definition) { found_definition = true; } } @@ -677,7 +670,7 @@ TEST_CASE(ModulePartitionName) { bool found_definition = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Definition)) { + if(r.kind == RelationKind::Definition) { found_definition = true; } } @@ -707,10 +700,10 @@ module §(m)⟦§(m)foo⟧; bool found_reference = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Definition)) { + if(r.kind == RelationKind::Definition) { ASSERT_TRUE(false); } - if(r.kind.value() == static_cast(RelationKind::Reference)) { + if(r.kind == RelationKind::Reference) { found_reference = true; } } @@ -738,9 +731,9 @@ TEST_CASE(OverrideRelation) { auto check_relations = [&](index::FileIndex& idx) { for(auto& [hash, rels]: idx.relations) { for(auto& r: rels) { - if(r.kind.value() == RelationKind::Interface) + if(r.kind == RelationKind::Interface) found_interface = true; - if(r.kind.value() == RelationKind::Implementation) + if(r.kind == RelationKind::Implementation) found_implementation = true; } } @@ -775,10 +768,10 @@ TEST_CASE(DeclarationAndDefinition) { bool found_decl = false; bool found_def = false; for(auto& r: it->second) { - if(r.kind.value() == static_cast(RelationKind::Declaration)) { + if(r.kind == RelationKind::Declaration) { found_decl = true; } - if(r.kind.value() == static_cast(RelationKind::Definition)) { + if(r.kind == RelationKind::Definition) { found_def = true; } } @@ -1192,6 +1185,214 @@ TEST_CASE(SuperQualifierRef) { GO_TO_DEFINITION("use", "def"); } +TEST_CASE(SerializeRoundTrip) { + add_file("header.h", R"( + #pragma once + inline int §(hdr)helper() { return 1; } + )"); + add_main("main.cpp", R"( + #include "header.h" + int main() { return §(use)helper(); } + )"); + ASSERT_TRUE(compile()); + tu_index = index::TUIndex::build(*unit); + ASSERT_FALSE(tu_index.file_indices.empty()); + + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + tu_index.serialize(os); + + auto loaded = index::TUIndex::from(buf); + ASSERT_TRUE(loaded.has_value()); + + ASSERT_EQ(loaded->built_at.count(), tu_index.built_at.count()); + ASSERT_TRUE(loaded->graph.paths == tu_index.graph.paths); + ASSERT_TRUE(loaded->graph.locations == tu_index.graph.locations); + ASSERT_TRUE(loaded->graph.path_hashes == tu_index.graph.path_hashes); + + // The persisted per-file rows are keyed by path id; recompute the + // expected conversion from the build-time FileID-keyed state. + llvm::DenseMap> expected; + for(auto& [fid, file_index]: tu_index.file_indices) { + expected[tu_index.graph.path_id(fid)] = {file_index.occurrences.size(), + file_index.relations.size()}; + } + ASSERT_FALSE(expected.empty()); + ASSERT_EQ(loaded->path_file_indices.size(), expected.size()); + for(auto& [path_id, counts]: expected) { + auto it = loaded->path_file_indices.find(path_id); + ASSERT_TRUE(it != loaded->path_file_indices.end()); + ASSERT_EQ(it->second.occurrences.size(), counts.first); + ASSERT_EQ(it->second.relations.size(), counts.second); + } + + ASSERT_TRUE(loaded->main_file_index.occurrences == tu_index.main_file_index.occurrences); + ASSERT_EQ(loaded->main_file_index.relations.size(), tu_index.main_file_index.relations.size()); + + ASSERT_EQ(loaded->symbols.size(), tu_index.symbols.size()); + for(auto& [hash, symbol]: tu_index.symbols) { + auto it = loaded->symbols.find(hash); + ASSERT_TRUE(it != loaded->symbols.end()); + ASSERT_EQ(it->second.name, symbol.name); + ASSERT_EQ(it->second.kind.value(), symbol.kind.value()); + ASSERT_EQ(static_cast(it->second.scope), static_cast(symbol.scope)); + ASSERT_TRUE(it->second.reference_files == symbol.reference_files); + } +} + +TEST_CASE(FromRejectsHostileInput) { + ASSERT_FALSE(index::TUIndex::from("not a flatbuffer at all").has_value()); + + build_index(R"( + int foo() { return 42; } + )"); + + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + tu_index.serialize(os); + + // Sanity: the intact blob loads, so the rejections below are earned. + ASSERT_TRUE(index::TUIndex::from(buf).has_value()); + + ASSERT_FALSE(index::TUIndex::from(llvm::StringRef(buf.data(), buf.size() / 2)).has_value()); + + // Bytes 4-7 carry the buffer identifier; a blob from another format + // must be rejected up front. + ASSERT_TRUE(buf.size() > 8); + std::string clobbered(buf.data(), buf.size()); + for(std::size_t i = 4; i < 8; ++i) { + clobbered[i] = 'X'; + } + ASSERT_FALSE(index::TUIndex::from(clobbered).has_value()); +} + +TEST_CASE(FromRejectsStaleFormatVersion) { + // Only the version slot is written: every other field reads back + // absent, which is structurally valid — the verdict must hinge on the + // value. Field order MUST mirror TUIndex (tu_index.h): format_version + // is slot 0. + struct VersionOnly { + std::uint32_t format_version = 0; + }; + + auto bytes_of = [](const std::vector& blob) { + return llvm::StringRef(reinterpret_cast(blob.data()), blob.size()); + }; + + auto stale = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version + 1}); + ASSERT_TRUE(stale.has_value()); + ASSERT_FALSE(index::TUIndex::from(bytes_of(*stale)).has_value()); + + // Positive control: the same shape carrying the current version loads, + // so the rejection above comes from the value, not the blob's shape. + auto current = kota::codec::fbs::to_bytes(VersionOnly{index::index_format_version}); + ASSERT_TRUE(current.has_value()); + ASSERT_TRUE(index::TUIndex::from(bytes_of(*current)).has_value()); +} + +TEST_CASE(FromRejectsOutOfRangePathIds) { + // Structural verification does not constrain field values, and the + // merge pipeline dereferences every decoded path id against the path + // table without further checks (Indexer::merge indexes paths and + // path_hashes, ProjectIndex::merge indexes file_ids_map with + // reference_files values) — a blob pointing outside its own table must + // be rejected as a whole. + auto serialized = [](index::TUIndex& index) { + std::string buf; + llvm::raw_string_ostream os(buf); + index.serialize(os); + return buf; + }; + + // Positive control first: the same shapes with in-range ids load, so + // the rejections below come from the hostile values. + index::TUIndex honest; + honest.built_at = std::chrono::milliseconds(0); + honest.graph.paths = {"/proj/main.cpp"}; + honest.graph.locations.push_back({.path_id = 0, .line = 1, .include = 0}); + honest.path_file_indices.try_emplace(0); + honest.symbols[42].reference_files.add(0); + ASSERT_TRUE(index::TUIndex::from(serialized(honest)).has_value()); + + { + index::TUIndex hostile; + hostile.built_at = std::chrono::milliseconds(0); + hostile.graph.paths = {"/proj/main.cpp"}; + hostile.graph.locations.push_back({.path_id = 7, .line = 1, .include = 0}); + ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); + } + { + index::TUIndex hostile; + hostile.built_at = std::chrono::milliseconds(0); + hostile.graph.paths = {"/proj/main.cpp"}; + hostile.path_file_indices.try_emplace(7); // Only path id 0 exists. + ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); + } + { + index::TUIndex hostile; + hostile.built_at = std::chrono::milliseconds(0); + hostile.graph.paths = {"/proj/main.cpp"}; + hostile.symbols[42].reference_files.add(7); + ASSERT_FALSE(index::TUIndex::from(serialized(hostile)).has_value()); + } +} + +TEST_CASE(FromNormalizesPathHashes) { + build_index(R"( + int foo() { return 42; } + )"); + ASSERT_FALSE(tu_index.graph.paths.empty()); + + // A blob without path hashes (structurally valid: the field reads back + // empty) must come back resized to the path table, all "unavailable". + tu_index.graph.path_hashes.clear(); + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + tu_index.serialize(os); + + auto loaded = index::TUIndex::from(buf); + ASSERT_TRUE(loaded.has_value()); + ASSERT_EQ(loaded->graph.path_hashes.size(), loaded->graph.paths.size()); + for(auto hash: loaded->graph.path_hashes) { + ASSERT_EQ(hash, 0u); + } +} + +TEST_CASE(ReserializeKeepsPathIndices) { + add_file("header.h", R"( + #pragma once + inline int helper() { return 1; } + )"); + add_main("main.cpp", R"( + #include "header.h" + int main() { return helper(); } + )"); + ASSERT_TRUE(compile()); + tu_index = index::TUIndex::build(*unit); + + llvm::SmallString<4096> buf; + llvm::raw_svector_ostream os(buf); + tu_index.serialize(os); + + auto loaded = index::TUIndex::from(buf); + ASSERT_TRUE(loaded.has_value()); + ASSERT_TRUE(loaded->file_indices.empty()); + ASSERT_FALSE(loaded->path_file_indices.empty()); + + // A deserialized index has no FileID-keyed state; re-serializing must + // keep the path-keyed rows instead of wiping them from an empty map. + llvm::SmallString<4096> again; + llvm::raw_svector_ostream os2(again); + loaded->serialize(os2); + + auto reloaded = index::TUIndex::from(again); + ASSERT_TRUE(reloaded.has_value()); + ASSERT_EQ(reloaded->path_file_indices.size(), loaded->path_file_indices.size()); + for(auto& [path_id, file_index]: loaded->path_file_indices) { + ASSERT_TRUE(reloaded->path_file_indices.contains(path_id)); + } +} + }; // TEST_SUITE(tu_index) } // namespace diff --git a/tests/unit/server/indexer_tests.cpp b/tests/unit/server/indexer_tests.cpp index 75d74e3b2..380251a0f 100644 --- a/tests/unit/server/indexer_tests.cpp +++ b/tests/unit/server/indexer_tests.cpp @@ -101,6 +101,19 @@ IndexedTU index_file(TempDir& tmp, llvm::StringRef file) { return result; } +TEST_CASE(MergeRejectsGarbage) { + // A worker shipping corrupted bytes (torn write, stale format) must not + // crash the master or leave partial state behind. + ASSERT_TRUE(workspace.merged_indices.empty()); + ASSERT_TRUE(workspace.project_index.symbols.empty()); + + std::string garbage = "definitely not a flatbuffer, but long enough to try"; + indexer.merge(garbage.data(), garbage.size()); + + ASSERT_TRUE(workspace.merged_indices.empty()); + ASSERT_TRUE(workspace.project_index.symbols.empty()); +} + TEST_CASE(MergeSkipsMovedDisk) { TempDir tmp; tmp.touch("main.cpp", "int value() { return 1; }\n"); diff --git a/tools/client/workspace.ts b/tools/client/workspace.ts index 9f1a2d69f..f6018636b 100644 --- a/tools/client/workspace.ts +++ b/tools/client/workspace.ts @@ -10,7 +10,7 @@ import { generateCDB } from "../compile_commands.ts"; /// Versioned root of the unified cache store; bump together with /// cache_format_version in src/server/state/workspace.h. -const CACHE_ROOT = path.join(".clice", "cache", "v4"); +const CACHE_ROOT = path.join(".clice", "cache", "v5"); /// The harness-wide canonical URI spelling: percent-decoded. vscode-uri /// encodes the drive colon (file:///c%3A/...) while the server emits it