diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml new file mode 100644 index 000000000..a8d6c8043 --- /dev/null +++ b/.github/workflows/benchmark.yml @@ -0,0 +1,47 @@ +name: benchmark + +on: + pull_request: + branches: [main] + +jobs: + benchmark: + strategy: + fail-fast: false + matrix: + os: [ubuntu-24.04, macos-15, windows-2025] + runs-on: ${{ matrix.os }} + defaults: + run: + shell: bash + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + # ── Build scan_benchmark ── + + - uses: ./.github/actions/setup-pixi + + - name: Build scan_benchmark + run: | + pixi run cmake-config RelWithDebInfo ON + cmake --build build/RelWithDebInfo --target scan_benchmark + + # ── Clone LLVM and generate CDB ── + + - name: Clone LLVM + run: git clone --depth 1 https://github.com/llvm/llvm-project.git + + - name: Generate CDB + run: | + cmake -B llvm-build -G Ninja \ + -DCMAKE_EXPORT_COMPILE_COMMANDS=ON \ + -DCMAKE_TOOLCHAIN_FILE="$(pwd)/cmake/toolchain.cmake" \ + -DLLVM_ENABLE_PROJECTS="clang;clang-tools-extra;lld;lldb;mlir;polly;flang;bolt" \ + -DLLVM_ENABLE_RUNTIMES="compiler-rt;libcxx;libcxxabi;libunwind" \ + llvm-project/llvm + + # ── Run benchmark ── + + - name: Run benchmark + run: ./build/RelWithDebInfo/bin/scan_benchmark --runs 20 llvm-build/compile_commands.json diff --git a/CMakeLists.txt b/CMakeLists.txt index 4d386f479..d3c1a2802 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -131,8 +131,10 @@ add_custom_target(generate_flatbuffers_schema DEPENDS "${GENERATED_HEADER}") # Temporary migration-only build graph. add_library(clice-core STATIC - "${PROJECT_SOURCE_DIR}/src/compile/command.cpp" - "${PROJECT_SOURCE_DIR}/src/compile/toolchain.cpp" + "${PROJECT_SOURCE_DIR}/src/command/command.cpp" + "${PROJECT_SOURCE_DIR}/src/command/search_config.cpp" + "${PROJECT_SOURCE_DIR}/src/command/toolchain.cpp" + "${PROJECT_SOURCE_DIR}/src/command/toolchain_provider.cpp" "${PROJECT_SOURCE_DIR}/src/compile/compilation.cpp" "${PROJECT_SOURCE_DIR}/src/compile/compilation_unit.cpp" "${PROJECT_SOURCE_DIR}/src/compile/diagnostic.cpp" @@ -145,6 +147,8 @@ add_library(clice-core STATIC "${PROJECT_SOURCE_DIR}/src/support/logging.cpp" "${PROJECT_SOURCE_DIR}/src/syntax/lexer.cpp" "${PROJECT_SOURCE_DIR}/src/syntax/scan.cpp" + "${PROJECT_SOURCE_DIR}/src/syntax/include_resolver.cpp" + "${PROJECT_SOURCE_DIR}/src/syntax/dependency_graph.cpp" "${PROJECT_SOURCE_DIR}/src/feature/semantic_tokens.cpp" "${PROJECT_SOURCE_DIR}/src/feature/document_links.cpp" "${PROJECT_SOURCE_DIR}/src/feature/document_symbols.cpp" @@ -221,3 +225,11 @@ if(CLICE_ENABLE_TEST) ) target_link_libraries(unit_tests PRIVATE clice::core eventide::zest eventide::deco) endif() + +add_executable(scan_benchmark + "${PROJECT_SOURCE_DIR}/benchmarks/scan_benchmark.cpp" +) +target_include_directories(scan_benchmark PRIVATE + "${PROJECT_SOURCE_DIR}/src" +) +target_link_libraries(scan_benchmark PRIVATE clice::core eventide::deco) diff --git a/benchmarks/scan_benchmark.cpp b/benchmarks/scan_benchmark.cpp new file mode 100644 index 000000000..64d7b5a01 --- /dev/null +++ b/benchmarks/scan_benchmark.cpp @@ -0,0 +1,369 @@ +/// Benchmark for scan_dependency_graph on a real compilation database. +/// +/// Usage: +/// scan_benchmark [OPTIONS] +/// +/// Example: +/// ./build/RelWithDebInfo/bin/scan_benchmark \ +/// /home/ykiko/C++/clice/.llvm/build-debug/compile_commands.json +/// +/// ./build/RelWithDebInfo/bin/scan_benchmark --log-level info --export graph.json \ +/// /home/ykiko/C++/clice/.llvm/build-debug/compile_commands.json + +#include +#include +#include +#include +#include +#include +#include + +#include "command/command.h" +#include "eventide/deco/deco.h" +#include "eventide/serde/json/serializer.h" +#include "support/filesystem.h" +#include "support/logging.h" +#include "support/path_pool.h" +#include "syntax/dependency_graph.h" + +#include "llvm/Support/FileSystem.h" + +namespace et = eventide; + +using namespace clice; + +struct BenchmarkOptions { + DecoKV(names = {"--log-level"}; help = "Log level: trace, debug, info, warn, error, off"; + required = false;) + log_level = "off"; + + DecoKV(names = {"--export"}; help = "Export dependency graph as JSON to this path"; + required = false;) + export_path; + + DecoKV(names = {"--runs"}; help = "Number of cold start iterations"; required = false;) + runs = 20; + + DecoFlag(names = {"-h", "--help"}; help = "Show help message"; required = false;) + help; + + DecoInput(meta_var = "CDB"; help = "Path to compile_commands.json"; required = false;) + cdb_path; +}; + +struct FileNode { + std::string path; + std::string module_name; + std::vector includes; +}; + +struct GraphExport { + std::vector files; +}; + +void export_graph_json(const PathPool& path_pool, + const DependencyGraph& graph, + llvm::StringRef output_path) { + // Build reverse module map: path_id -> module_name. + llvm::DenseMap path_to_module; + for(auto& [name, path_id]: graph.modules()) { + path_to_module[path_id] = name; + } + + GraphExport export_data; + for(std::uint32_t id = 0; id < path_pool.paths.size(); id++) { + auto inc_ids = graph.get_all_includes(id); + if(inc_ids.empty()) { + continue; + } + + FileNode node; + node.path = path_pool.paths[id].str(); + + auto mod_it = path_to_module.find(id); + if(mod_it != path_to_module.end()) { + node.module_name = mod_it->second.str(); + } + + for(auto flagged_id: inc_ids) { + auto raw_id = flagged_id & DependencyGraph::PATH_ID_MASK; + node.includes.push_back(path_pool.paths[raw_id].str()); + } + + export_data.files.push_back(std::move(node)); + } + + auto json = et::serde::json::to_json(export_data); + if(!json) { + std::println(stderr, "Failed to serialize dependency graph"); + return; + } + + std::ofstream out(output_path.str()); + if(!out) { + std::println(stderr, "Failed to open output file: {}", output_path); + return; + } + out << *json; + std::println("Graph exported to {} ({} files)", output_path, export_data.files.size()); +} + +void print_report(const ScanReport& report) { + std::println("==============================================================="); + std::println(" Dependency Scan Report"); + std::println("==============================================================="); + + // Timing. + std::println(""); + std::println(" Time: {}ms", report.elapsed_ms); + std::println(" Waves: {}", report.waves); + + // File counts. + std::println(""); + std::println(" Files"); + std::println(" Source files (from CDB): {}", report.source_files); + std::println(" Header files (discovered): {}", report.header_files); + std::println(" Total: {}", report.total_files); + std::println(" Modules: {}", report.modules); + + // Include edges. + std::println(""); + std::println(" Include Edges"); + std::println(" Total: {}", report.total_edges); + std::println(" Unconditional: {}", report.unconditional_edges); + std::println(" Conditional: {} (inside #if/#ifdef)", report.conditional_edges); + + // Resolution accuracy. + std::println(""); + std::println(" Resolution"); + std::println(" #include directives: {}", report.includes_found); + std::println(" Resolved: {}", report.includes_resolved); + auto unresolved_count = report.includes_found - report.includes_resolved; + std::println(" Unresolved: {}", unresolved_count); + if(report.includes_found > 0) { + double rate = 100.0 * static_cast(report.includes_resolved) / + static_cast(report.includes_found); + std::println(" Accuracy: {:.1f}%", rate); + } + + // Wall-clock phase breakdown. + std::println(""); + std::println(" Phase Breakdown (wall-clock)"); + std::println(" Config extraction: {}ms (prewarm={}ms, loop={}ms)", + report.config_ms, + report.prewarm_ms, + report.config_loop_ms); + std::println(" Dir cache pre-pop: {}ms (overlapped with Phase 1)", report.dir_cache_ms); + std::println(" Phase 1 (read+scan, parallel): {}ms", report.phase1_ms); + std::println(" Phase 2 (include resolve): {}ms", report.phase2_ms); + std::println(" Phase 3 (graph build): {}ms", report.phase3_ms); + + // Per-wave breakdown. + if(!report.wave_stats.empty()) { + std::println(""); + std::println(" Per-Wave Breakdown"); + std::println(" {:>5s} {:>8s} {:>8s} {:>8s} {:>8s} {:>8s} {:>10s} {:>10s}", + "Wave", + "Files", + "P1(ms)", + "P2(ms)", + "Next", + "Prefetch", + "DirList", + "DirHits"); + for(std::size_t i = 0; i < report.wave_stats.size(); i++) { + auto& ws = report.wave_stats[i]; + std::println(" {:>5} {:>8} {:>8} {:>8} {:>8} {:>8} {:>10} {:>10}", + i, + ws.files, + ws.phase1_ms, + ws.phase2_ms, + ws.next_files, + ws.prefetch_count, + ws.dir_listings, + ws.dir_hits); + } + } + + // Phase 2 breakdown. + if(report.p2_resolve_us > 0) { + auto other_us = report.phase2_ms * 1000 - report.p2_resolve_us; + std::println(""); + std::println(" Phase 2 Breakdown (single-threaded)"); + std::println(" resolve_include: {:.1f}ms", report.p2_resolve_us / 1000.0); + std::println(" Other (cache lookup, intern, graph): {:.1f}ms", other_us / 1000.0); + } + + // Cumulative I/O statistics. + std::println(""); + std::println(" I/O Statistics (cumulative across threads)"); + std::println(" File read: {:.1f}ms (sum of all threads)", report.read_us / 1000.0); + std::println(" Lexer scan: {:.1f}ms (sum of all threads)", report.scan_us / 1000.0); + std::println(" Filesystem: {:.1f}ms ({} readdir calls, {} dir cache hits)", + report.fs_us / 1000.0, + report.dir_listings, + report.dir_hits); + std::println(" File lookups: {}", report.fs_lookups); + std::println(" Include cache hits: {}", report.include_cache_hits); + std::println(" Scan result cache hits: {}", report.scan_cache_hits); + if(report.dir_listings + report.dir_hits > 0) { + double hit_rate = 100.0 * static_cast(report.dir_hits) / + static_cast(report.dir_listings + report.dir_hits); + std::println(" Dir cache hit rate: {:.1f}%", hit_rate); + } + + std::println(""); + std::println("==============================================================="); +} + +int main(int argc, const char** argv) { + auto args = deco::util::argvify(argc, argv); + auto result = deco::cli::parse(args); + + if(!result.has_value()) { + std::println(stderr, "Error: {}", result.error().message); + return 1; + } + + auto& opts = result->options; + + if(opts.help.value_or(false) || !opts.cdb_path.has_value()) { + std::ostringstream oss; + deco::cli::write_usage_for(oss, "scan_benchmark [OPTIONS] "); + std::print("{}", oss.str()); + return opts.help.value_or(false) ? 0 : 1; + } + + // Configure logging. + auto level = spdlog::level::from_str(*opts.log_level); + clice::logging::options.level = level; + clice::logging::stderr_logger("scan_benchmark", clice::logging::options); + + // Initialize resource directory (needed for -resource-dir in toolchain queries). + std::string self_path = llvm::sys::fs::getMainExecutable(argv[0], (void*)main); + if(!clice::fs::init_resource_dir(self_path)) { + std::println(stderr, "Warning: failed to find resource dir from {}", self_path); + } + + auto& cdb_path = *opts.cdb_path; + auto hw_threads = std::thread::hardware_concurrency(); + auto runs = *opts.runs; + + // Set UV_THREADPOOL_SIZE if not already set. + // Use at least libuv's default (4) so low-core CI runners don't regress. + if(!std::getenv("UV_THREADPOOL_SIZE")) { + auto pool_size = std::max(hw_threads, 4u); + static std::string env = "UV_THREADPOOL_SIZE=" + std::to_string(pool_size); + putenv(env.data()); + } + + std::println("Hardware threads: {}", hw_threads); + std::println("UV_THREADPOOL_SIZE: {}", std::getenv("UV_THREADPOOL_SIZE")); + std::println("Log level: {}", *opts.log_level); + std::println("CDB: {}", cdb_path); + std::println(""); + + // Load compilation database. + auto t0 = std::chrono::steady_clock::now(); + + CompilationDatabase cdb; + auto updates = cdb.load_compile_database(cdb_path); + + auto t1 = std::chrono::steady_clock::now(); + auto load_ms = std::chrono::duration_cast(t1 - t0).count(); + + std::size_t active = 0; + for(auto& u: updates) { + if(u.kind != UpdateKind::Deleted) { + active++; + } + } + + std::println("CDB loaded: {} entries ({} active) in {}ms", updates.size(), active, load_ms); + + // ── Context dedup diagnostic ────────────────────────────────────── + { + std::size_t total_files = 0; + std::set unique_contexts; + for(auto& u: updates) { + if(u.kind != UpdateKind::Deleted) { + total_files++; + unique_contexts.insert(u.context); + } + } + std::println("Context dedup: {} files -> {} unique contexts ({:.1f}x reduction)", + total_files, + unique_contexts.size(), + static_cast(total_files) / unique_contexts.size()); + } + + // ── Cold start dependency scan benchmark ────────────────────────────── + std::println("\nRunning {} cold start scan(s)...\n", runs); + + PathPool path_pool; + DependencyGraph graph; + std::vector elapsed_times; + std::vector config_times; + std::vector phase1_times; + std::vector phase2_times; + elapsed_times.reserve(runs); + config_times.reserve(runs); + phase1_times.reserve(runs); + phase2_times.reserve(runs); + + for(int i = 0; i < runs; i++) { + // True cold start: rebuild CDB (clears toolchain & config caches), + // reset PathPool and DependencyGraph. + cdb = CompilationDatabase{}; + updates = cdb.load_compile_database(cdb_path); + path_pool = PathPool{}; + graph = DependencyGraph{}; + + auto report = scan_dependency_graph(cdb, updates, path_pool, graph); + + elapsed_times.push_back(report.elapsed_ms); + config_times.push_back(report.config_ms); + phase1_times.push_back(report.phase1_ms); + phase2_times.push_back(report.phase2_ms); + + std::println("[run {:2}] {}ms | config={}ms phase1={}ms phase2={}ms | files={}", + i + 1, + report.elapsed_ms, + report.config_ms, + report.phase1_ms, + report.phase2_ms, + report.total_files); + + // Print detailed report for the first run only. + if(i == 0) { + std::println(""); + print_report(report); + } + } + + // Summary statistics. + if(runs > 1) { + auto stats = [](std::vector& v) { + std::ranges::sort(v); + auto sum = std::accumulate(v.begin(), v.end(), std::int64_t{0}); + return std::tuple{v.front(), sum / static_cast(v.size()), v.back()}; + }; + auto [e_min, e_avg, e_max] = stats(elapsed_times); + auto [c_min, c_avg, c_max] = stats(config_times); + auto [p1_min, p1_avg, p1_max] = stats(phase1_times); + auto [p2_min, p2_avg, p2_max] = stats(phase2_times); + + std::println("\n Summary ({} runs) min avg max", runs); + std::println(" Total: {:>7} {:>6} {:>6}", e_min, e_avg, e_max); + std::println(" Config extraction: {:>7} {:>6} {:>6}", c_min, c_avg, c_max); + std::println(" Phase 1 (read+scan):{:>7} {:>6} {:>6}", p1_min, p1_avg, p1_max); + std::println(" Phase 2 (resolve): {:>7} {:>6} {:>6}", p2_min, p2_avg, p2_max); + } + + // Export dependency graph as JSON if requested. + if(opts.export_path.has_value()) { + export_graph_json(path_pool, graph, *opts.export_path); + } + + return 0; +} diff --git a/src/clice.cc b/src/clice.cc index cec3475d3..e37ab7929 100644 --- a/src/clice.cc +++ b/src/clice.cc @@ -1,11 +1,10 @@ #include +#include #include -#include #include #include "eventide/async/async.h" -#include "eventide/deco/macro.h" -#include "eventide/deco/runtime.h" +#include "eventide/deco/deco.h" #include "eventide/ipc/peer.h" #include "eventide/ipc/transport.h" #include "server/master_server.h" @@ -61,10 +60,7 @@ int main(int argc, const char** argv) { auto& opts = result->options; if(opts.help.value_or(false)) { - auto dispatcher = deco::cli::Dispatcher("clice [OPTIONS]"); - std::ostringstream oss; - dispatcher.usage(oss, true); - std::print("{}", oss.str()); + deco::cli::write_usage_for(std::cout, "clice [OPTIONS]"); return 0; } diff --git a/src/compile/command.cpp b/src/command/command.cpp similarity index 76% rename from src/compile/command.cpp rename to src/command/command.cpp index 23d75734f..aa93bb6e0 100644 --- a/src/compile/command.cpp +++ b/src/command/command.cpp @@ -1,12 +1,12 @@ -#include "compile/command.h" +#include "command/command.h" #include #include #include #include -#include "compile/driver.h" -#include "compile/toolchain.h" +#include "command/driver.h" +#include "command/toolchain.h" #include "support/filesystem.h" #include "support/logging.h" #include "support/object_pool.h" @@ -149,14 +149,44 @@ struct CompilationDatabase::Impl { /// All source files in the compilation database. llvm::DenseMap> files; - /// TODO: Cache of toolchain query driver results. - llvm::DenseMap toolchains; + /// Pluggable toolchain provider: manages toolchain queries and caching. + ToolchainProvider toolchain; + + /// Cache of SearchConfig per CompilationInfo pointer. Since infos are + /// deduplicated by ObjectSet, the pointer uniquely identifies a compilation + /// context. This avoids re-parsing arguments on repeated scans. + llvm::DenseMap search_config_cache; /// The clang options we want to filter in all cases, like -c and -o. llvm::DenseSet filtered_options; ArgumentParser parser{&allocator}; + /// Check if an argument matches the source file path, handling + /// Windows path separator differences (backslash vs forward slash). + static bool is_same_file(llvm::StringRef argument, llvm::StringRef file) { + if(argument == file) { + return true; + } + +#ifdef _WIN32 + // On Windows, cmake may use backslashes in `arguments` but forward + // slashes in `file`. Normalize and compare. + if(argument.size() == file.size()) { + for(std::size_t i = 0; i < argument.size(); i++) { + char a = argument[i] == '\\' ? '/' : argument[i]; + char b = file[i] == '\\' ? '/' : file[i]; + if(a != b) { + return false; + } + } + return true; + } +#endif + + return false; + } + object_ptr save_compilation_info(this Impl& self, llvm::StringRef file, llvm::StringRef directory, @@ -170,8 +200,7 @@ struct CompilationDatabase::Impl { for(unsigned it = 0; it != arguments.size(); it++) { llvm::StringRef argument = arguments[it]; - /// FIXME: Is it possible that file in command and field are different? - if(argument == file) { + if(is_same_file(argument, file)) { continue; } @@ -182,6 +211,7 @@ struct CompilationDatabase::Impl { "/o", "/Fo", "/Fe", + "/Fd", }; /// FIXME: This is a heuristic approach that covers the vast majority of cases, but @@ -252,12 +282,20 @@ struct CompilationDatabase::Impl { llvm::SmallVector arguments; - /// FIXME: We need a better way to handle this. - if(command.contains("cl.exe") || command.contains("clang-cl")) { - llvm::cl::TokenizeWindowsCommandLineFull(command, saver, arguments); - } else { - llvm::cl::TokenizeGNUCommandLine(command, saver, arguments); - } + /// On Windows, always use the Windows tokenizer regardless of the compiler + /// (MSVC, clang-cl, MinGW, etc.), because all programs are invoked through + /// the Windows API and paths use backslashes. The GNU tokenizer treats '\' + /// as an escape character, which corrupts Windows paths like C:\Users into + /// C:Users. + /// + /// Note: this does NOT affect toolchain.cpp's query_clang_toolchain(), which + /// parses clang's -### output. That output uses shell-style escaping (\\), + /// so the GNU tokenizer is correct there. +#ifdef _WIN32 + llvm::cl::TokenizeWindowsCommandLineFull(command, saver, arguments); +#else + llvm::cl::TokenizeGNUCommandLine(command, saver, arguments); +#endif return self.save_compilation_info(file, directory, arguments); } @@ -720,39 +758,126 @@ CompilationContext CompilationDatabase::lookup(llvm::StringRef file, } if(info && options.query_toolchain) { - auto callback = [&](const char* s) { - return save_string(s).data(); - }; - toolchain::QueryParams params = {file, directory, arguments, callback}; + // Save user-level include paths before replacing with cc1 args. + // The cached toolchain query uses minimal args (no -I/-D/-W etc.) + // for cache efficiency, so user include paths must be injected back. + auto user_args = std::move(arguments); - /// FIXME: querying is expensive, we want to cache this ... - arguments = toolchain::query_toolchain(params); + auto cached = self->toolchain.query_cached(file, directory, user_args); - /// FIXME: we need mangle the arguments again. - /// Work around ... the logic of this should be moved to query ... - bool next_main_file = false; - for(auto& arg: arguments) { - if(arg == llvm::StringRef("-main-file-name")) { - next_main_file = true; - continue; + if(cached.empty()) { + LOG_WARN("failed to query toolchain: {}", file); + arguments = std::move(user_args); + } else { + // Start with cc1 result (has system paths, driver flags, etc.). + arguments.assign(cached.begin(), cached.end()); + + // Remove the temp source file that was appended during query. + arguments.pop_back(); + +#ifdef _WIN32 + // On Windows, the toolchain query derives the resource dir from the + // system compiler's executable path. If that compiler is a different + // clang version, its builtin headers may reference renamed builtins + // (e.g. AVX10.2-BF16 nepbh→bf16 rename between clang 20→21). + // Replace the queried resource dir with ours so the headers match. + if(!fs::resource_dir.empty()) { + llvm::StringRef old_resource_dir; + for(std::size_t i = 0; i + 1 < arguments.size(); ++i) { + if(arguments[i] == llvm::StringRef("-resource-dir")) { + old_resource_dir = arguments[i + 1]; + break; + } + } + if(!old_resource_dir.empty() && old_resource_dir != fs::resource_dir) { + for(auto& arg: arguments) { + llvm::StringRef s(arg); + if(s.starts_with(old_resource_dir)) { + auto replaced = + fs::resource_dir + s.substr(old_resource_dir.size()).str(); + arg = self->strings.save(replaced).data(); + } + } + } } +#endif - if(next_main_file) { - arg = self->strings.save(path::filename(file)).data(); - next_main_file = false; + // Inject user include paths (-I, -isystem, -iquote) from the + // original mangled args into the cc1 result. + self->parser.parse( + llvm::ArrayRef(user_args).drop_front(), + [&](std::unique_ptr arg) { + auto id = arg->getOption().getID(); + if(id == ID::OPT_I || id == ID::OPT_isystem || id == ID::OPT_iquote) { + append_arg(arg->getSpelling()); + for(auto value: arg->getValues()) { + append_arg(value); + } + } + }, + [](int, int) {}); + + // Fix -main-file-name to match the actual file. + bool next_main_file = false; + for(auto& arg: arguments) { + if(arg == llvm::StringRef("-main-file-name")) { + next_main_file = true; + continue; + } + + if(next_main_file) { + arg = self->strings.save(path::filename(file)).data(); + next_main_file = false; + } } } + } - if(arguments.empty()) { - LOG_WARN("failed to query toolchain: {}", file); + arguments.emplace_back(file.data()); + + return CompilationContext(directory, std::move(arguments)); +} + +SearchConfig CompilationDatabase::lookup_search_config(llvm::StringRef file, + const CommandOptions& options, + const void* context) { + // Resolve to the internal CompilationInfo pointer for cache lookup. + auto path_id = self->strings.get(file); + auto it = self->files.find(path_id); + const CompilationInfo* info_ptr = nullptr; + if(it != self->files.end()) { + if(!context) { + info_ptr = it->second->info.ptr; } else { - arguments.pop_back(); + auto cur = it->second; + while(cur) { + if(cur->info.ptr == context) { + info_ptr = cur->info.ptr; + break; + } + cur = cur->next; + } } } - arguments.emplace_back(file.data()); + if(info_ptr) { + auto cache_it = self->search_config_cache.find(info_ptr); + if(cache_it != self->search_config_cache.end()) { + return cache_it->second; + } + } - return CompilationContext(directory, std::move(arguments)); + auto ctx = lookup(file, options, context); + auto config = extract_search_config(ctx.arguments, ctx.directory); + + if(info_ptr) { + self->search_config_cache.try_emplace(info_ptr, config); + } + return config; +} + +bool CompilationDatabase::has_cached_configs() const { + return !self->search_config_cache.empty(); } std::optional CompilationDatabase::get_option_id(llvm::StringRef argument) { @@ -775,6 +900,58 @@ std::optional CompilationDatabase::get_option_id(llvm::StringRef } } +ToolchainProvider& CompilationDatabase::toolchain() { + return self->toolchain; +} + +std::vector CompilationDatabase::resolve_toolchain_entries( + llvm::ArrayRef> files) { + std::vector entries; + entries.reserve(files.size()); + + for(auto& [file, context]: files) { + auto path_id = self->strings.get(file); + auto stored_file = self->strings.get(path_id); + + object_ptr info = nullptr; + auto it = self->files.find(path_id); + if(it != self->files.end()) { + if(!context) { + info = it->second->info; + } else { + auto cur = it->second; + while(cur) { + if(cur->info.ptr == context) { + info = cur->info; + break; + } + cur = cur->next; + } + } + } + + if(!info || info->arguments.empty()) { + continue; + } + + ToolchainProvider::PendingEntry entry; + entry.file = stored_file; + entry.directory = self->strings.get(info->directory); + entry.arguments.reserve(info->arguments.size()); + for(auto arg_id: info->arguments) { + entry.arguments.push_back(self->strings.get(arg_id).data()); + } + + entries.push_back(std::move(entry)); + } + + return entries; +} + +llvm::StringRef CompilationDatabase::resolve_path(std::uint32_t path_id) { + return self->strings.get(path_id); +} + std::vector CompilationDatabase::files() { std::vector result; for(auto& [file, _]: self->files) { diff --git a/src/compile/command.h b/src/command/command.h similarity index 74% rename from src/compile/command.h rename to src/command/command.h index 3f62a50b2..49bdaa988 100644 --- a/src/compile/command.h +++ b/src/command/command.h @@ -7,6 +7,8 @@ #include #include +#include "command/search_config.h" +#include "command/toolchain_provider.h" #include "support/format.h" #include "llvm/ADT/ArrayRef.h" @@ -95,9 +97,32 @@ class CompilationDatabase { /// all contexts and let user choose one. /// std::vector fetch_all(llvm::StringRef file); + /// Combined lookup + extract_search_config with internal caching. + /// Results are cached by CompilationInfo pointer, avoiding repeated + /// argument parsing across multiple calls with the same context. + SearchConfig lookup_search_config(llvm::StringRef file, + const CommandOptions& options = {}, + const void* context = nullptr); + + /// Check if SearchConfig cache is populated (non-empty). + /// When true, toolchain cache is also populated, so pre-warm can be skipped. + bool has_cached_configs() const; + /// Get an the option for specific argument. static std::optional get_option_id(llvm::StringRef argument); + /// Resolve a path_id (from UpdateInfo) back to the file path string. + llvm::StringRef resolve_path(std::uint32_t path_id); + + /// Access the toolchain provider for batch pre-warming and direct queries. + ToolchainProvider& toolchain(); + + /// Resolve (file, context) pairs to PendingEntry tuples for toolchain queries. + /// Converts CDB-internal context pointers to raw (file, directory, arguments) + /// that the ToolchainProvider can consume. + std::vector + resolve_toolchain_entries(llvm::ArrayRef> files); + /// FIXME: bad interface design ... std::vector files(); diff --git a/src/compile/driver.h b/src/command/driver.h similarity index 99% rename from src/compile/driver.h rename to src/command/driver.h index e78b14f95..faee234d2 100644 --- a/src/compile/driver.h +++ b/src/command/driver.h @@ -5,7 +5,7 @@ #include #include -#include "compile/command.h" +#include "command/command.h" #include "llvm/Support/Allocator.h" #include "clang/Driver/Driver.h" diff --git a/src/command/search_config.cpp b/src/command/search_config.cpp new file mode 100644 index 000000000..fbccc05e5 --- /dev/null +++ b/src/command/search_config.cpp @@ -0,0 +1,135 @@ +#include "command/search_config.h" + +#include "command/driver.h" + +#include "llvm/ADT/SmallString.h" +#include "llvm/ADT/StringSet.h" +#include "llvm/Support/FileSystem.h" +#include "llvm/Support/Path.h" + +namespace clice { + +using ID = clang::driver::options::ID; + +SearchConfig extract_search_config(llvm::ArrayRef arguments, + llvm::StringRef directory) { + // Replicate clang's InitHeaderSearch::Realize layout: + // Quoted (-iquote) → Angled (-I) → System (-isystem, -internal-isystem, etc.) + // Then deduplicate across [Angled..end) matching clang's RemoveDuplicates. + + std::vector quoted; + std::vector angled; + std::vector system; + std::vector after; + + auto make_absolute = [&](llvm::StringRef path) -> std::string { + llvm::SmallString<256> abs_path(path); + if(!llvm::sys::path::is_absolute(abs_path)) { + llvm::sys::fs::make_absolute(directory, abs_path); + } + llvm::sys::path::remove_dots(abs_path, true); + return abs_path.str().str(); + }; + + // Track -iprefix state for -iwithprefix/-iwithprefixbefore. + std::string prefix; + + llvm::BumpPtrAllocator allocator; + ArgumentParser parser{&allocator}; + + parser.parse( + llvm::ArrayRef(arguments).drop_front(), + [&](std::unique_ptr arg) { + auto id = arg->getOption().getID(); + switch(id) { + // Quoted group (clang: frontend::Quoted) + case ID::OPT_iquote: quoted.push_back({make_absolute(arg->getValue())}); break; + + // Angled group (clang: frontend::Angled) + case ID::OPT_I: angled.push_back({make_absolute(arg->getValue())}); break; + + // System group (clang: frontend::System / ExternCSystem) + case ID::OPT_isystem: + case ID::OPT_internal_isystem: + case ID::OPT_internal_externc_isystem: + system.push_back({make_absolute(arg->getValue())}); + break; + + // Prefix options: must be processed in argument order. + case ID::OPT_iprefix: prefix = arg->getValue(); break; + case ID::OPT_iwithprefix: + // clang maps to After group. + after.push_back({make_absolute(prefix + arg->getValue())}); + break; + case ID::OPT_iwithprefixbefore: + // clang maps to Angled group. + angled.push_back({make_absolute(prefix + arg->getValue())}); + break; + + case ID::OPT_idirafter: after.push_back({make_absolute(arg->getValue())}); break; + + // TODO: -cxx-isystem (clang: frontend::CXXSystem, C++-only system dirs) + // TODO: -iwithsysroot (prepends sysroot to path, then adds to System) + // TODO: HeaderMap support (-I foo.hmap remaps include names) + default: break; + } + }, + [](int, int) {}); + + // Concatenate: Quoted → Angled → System → After + SearchConfig config; + config.dirs.reserve(quoted.size() + angled.size() + system.size() + after.size()); + config.dirs.insert(config.dirs.end(), + std::make_move_iterator(quoted.begin()), + std::make_move_iterator(quoted.end())); + config.angled_start_idx = static_cast(config.dirs.size()); + config.dirs.insert(config.dirs.end(), + std::make_move_iterator(angled.begin()), + std::make_move_iterator(angled.end())); + config.system_start_idx = static_cast(config.dirs.size()); + config.dirs.insert(config.dirs.end(), + std::make_move_iterator(system.begin()), + std::make_move_iterator(system.end())); + config.after_start_idx = static_cast(config.dirs.size()); + config.dirs.insert(config.dirs.end(), + std::make_move_iterator(after.begin()), + std::make_move_iterator(after.end())); + + // Deduplicate across [angled_start_idx..end), matching clang's + // RemoveDuplicates(SearchList, NumQuoted). If a path appears in both + // Angled and System, keep the first (Angled) occurrence. This is + // critical for #include_next correctness. + { + llvm::StringSet<> seen; + // Seed with Quoted paths (they're not deduped against Angled/System). + for(unsigned i = 0; i < config.angled_start_idx; ++i) { + seen.insert(config.dirs[i].path); + } + + unsigned write = config.angled_start_idx; + unsigned removed_before_system = 0; + unsigned removed_before_after = 0; + for(unsigned read = config.angled_start_idx; read < config.dirs.size(); ++read) { + if(seen.insert(config.dirs[read].path).second) { + if(write != read) { + config.dirs[write] = std::move(config.dirs[read]); + } + ++write; + } else { + if(read < config.system_start_idx) { + ++removed_before_system; + } + if(read < config.after_start_idx) { + ++removed_before_after; + } + } + } + config.dirs.resize(write); + config.system_start_idx -= removed_before_system; + config.after_start_idx -= removed_before_after; + } + + return config; +} + +} // namespace clice diff --git a/src/command/search_config.h b/src/command/search_config.h new file mode 100644 index 000000000..0b3fc6c9a --- /dev/null +++ b/src/command/search_config.h @@ -0,0 +1,47 @@ +#pragma once + +#include +#include + +#include "llvm/ADT/ArrayRef.h" +#include "llvm/ADT/StringRef.h" + +namespace clice { + +struct SearchDir { + std::string path; +}; + +/// Header search configuration extracted from compilation arguments. +/// Uses a four-segment model matching clang's InitHeaderSearch::Realize layout: +/// [Quoted... | Angled... | System... | After...] +/// ^ ^ ^ +/// angled_start_idx system_start_idx after_start_idx +struct SearchConfig { + /// Ordered list of search directories, partitioned into four segments. + std::vector dirs; + + /// Index in dirs where Angled (-I) dirs start. + /// Quoted ("") includes search from index 0; angled (<>) from here. + unsigned angled_start_idx = 0; + + /// Index in dirs where System (-isystem, -internal-isystem, etc.) dirs start. + unsigned system_start_idx = 0; + + /// Index in dirs where After (-idirafter, -iwithprefix) dirs start. + unsigned after_start_idx = 0; +}; + +/// Extract header search configuration from compilation arguments. +/// +/// Parses user-level flags (-I, -isystem, -iquote) and cc1-level flags +/// (-internal-isystem, -internal-externc-isystem) using the clang argument +/// parser. Relative paths are resolved against the given working directory +/// and normalized with remove_dots(). +/// +/// This is intentionally a standalone function (not tied to CompilationDatabase) +/// so it can be tested and improved independently to match clang's behavior. +SearchConfig extract_search_config(llvm::ArrayRef arguments, + llvm::StringRef directory); + +} // namespace clice diff --git a/src/compile/toolchain.cpp b/src/command/toolchain.cpp similarity index 99% rename from src/compile/toolchain.cpp rename to src/command/toolchain.cpp index ee92a4f14..3f6a1ba38 100644 --- a/src/compile/toolchain.cpp +++ b/src/command/toolchain.cpp @@ -1,4 +1,4 @@ -#include "compile/toolchain.h" +#include "command/toolchain.h" #include #include diff --git a/src/compile/toolchain.h b/src/command/toolchain.h similarity index 100% rename from src/compile/toolchain.h rename to src/command/toolchain.h diff --git a/src/command/toolchain_provider.cpp b/src/command/toolchain_provider.cpp new file mode 100644 index 000000000..92a4dda0b --- /dev/null +++ b/src/command/toolchain_provider.cpp @@ -0,0 +1,208 @@ +#include "command/toolchain_provider.h" + +#include "command/driver.h" +#include "command/toolchain.h" +#include "support/filesystem.h" +#include "support/logging.h" +#include "support/object_pool.h" + +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringMap.h" + +namespace clice { + +using ID = clang::driver::options::ID; + +struct ToolchainProvider::Impl { + llvm::BumpPtrAllocator allocator; + StringSet strings{allocator}; + ArgumentParser parser{&allocator}; + + /// Cache of toolchain query results, keyed by canonical toolchain key. + /// The key captures only flags that affect system path discovery (driver, + /// target, sysroot, stdlib, etc.), so files sharing the same compiler + /// configuration share one cached result. + llvm::StringMap> toolchain_cache; + + /// Option IDs that affect system path discovery. These determine the + /// toolchain cache key and are the only flags passed to the toolchain query. + static bool is_toolchain_option(unsigned id) { + switch(id) { + case ID::OPT_target: + case ID::OPT_target_legacy_spelling: + case ID::OPT_isysroot: + case ID::OPT__sysroot_EQ: + case ID::OPT__sysroot: + case ID::OPT_stdlib_EQ: + case ID::OPT_gcc_toolchain: + case ID::OPT_gcc_install_dir_EQ: + case ID::OPT_nostdinc: + case ID::OPT_nostdincxx: + case ID::OPT_std_EQ: return true; + default: return false; + } + } + + /// Extract toolchain-relevant flags from arguments using the clang argument + /// parser. Returns both a cache key string and a minimal argument list for + /// the toolchain query. Using the parser ensures all flag forms (joined, + /// separate, etc.) are handled correctly. + struct ToolchainExtract { + std::string key; + std::vector query_args; + }; + + ToolchainExtract extract_toolchain_flags(this Impl& self, + llvm::StringRef file, + llvm::ArrayRef arguments) { + ToolchainExtract result; + + // Driver binary (first arg) — e.g. "clang++" vs "clang" affects language mode. + result.key += arguments[0]; + result.key += '\0'; + + // File extension affects language mode (C vs C++). + result.key += path::extension(file); + result.key += '\0'; + + result.query_args.push_back(arguments[0]); + + self.parser.parse( + llvm::ArrayRef(arguments).drop_front(), + [&](std::unique_ptr arg) { + auto id = arg->getOption().getID(); + if(!is_toolchain_option(id)) { + return; + } + + // Add option ID and all its values to the cache key. + result.key += std::to_string(id); + result.key += '\0'; + for(auto value: arg->getValues()) { + result.key += value; + result.key += '\0'; + } + + // Render the argument back to query args, respecting the option's + // render style (joined vs separate). + switch(arg->getOption().getRenderStyle()) { + case llvm::opt::Option::RenderJoinedStyle: { + // e.g. -std=c++17, --target=x86_64-linux-gnu + llvm::SmallString<64> joined(arg->getSpelling()); + if(arg->getNumValues() > 0) { + joined += arg->getValue(0); + } + result.query_args.push_back(self.strings.save(joined).data()); + break; + } + case llvm::opt::Option::RenderSeparateStyle: { + // e.g. -target x86_64-linux-gnu, -isysroot /path + result.query_args.push_back(self.strings.save(arg->getSpelling()).data()); + for(auto value: arg->getValues()) { + result.query_args.push_back(self.strings.save(value).data()); + } + break; + } + default: { + // Flags (no value): -nostdinc, -nostdinc++ + result.query_args.push_back(self.strings.save(arg->getSpelling()).data()); + break; + } + } + }, + [](int, int) { + // Ignore unknown arguments — they won't affect toolchain discovery. + }); + + return result; + } + + /// Query toolchain with caching. Returns the cached cc1 args for the given + /// toolchain key, running the expensive query only on cache miss. + llvm::ArrayRef query_toolchain_cached(this Impl& self, + llvm::StringRef file, + llvm::StringRef directory, + llvm::ArrayRef arguments) { + auto [key, query_args] = self.extract_toolchain_flags(file, arguments); + auto it = self.toolchain_cache.find(key); + if(it != self.toolchain_cache.end()) { + return it->second; + } + + LOG_WARN("Toolchain cache miss (spawning process): file={}, cache_size={}, key_len={}", + file, + self.toolchain_cache.size(), + key.size()); + + auto callback = [&](const char* s) -> const char* { + return self.strings.save(s).data(); + }; + toolchain::QueryParams params = {file, directory, query_args, callback}; + auto result = toolchain::query_toolchain(params); + + auto [entry, _] = self.toolchain_cache.try_emplace(std::move(key), std::move(result)); + return entry->second; + } +}; + +ToolchainProvider::ToolchainProvider() : self(std::make_unique()) {} + +ToolchainProvider::~ToolchainProvider() = default; + +ToolchainProvider::ToolchainProvider(ToolchainProvider&&) noexcept = default; + +ToolchainProvider& ToolchainProvider::operator=(ToolchainProvider&&) noexcept = default; + +llvm::ArrayRef ToolchainProvider::query_cached(llvm::StringRef file, + llvm::StringRef directory, + llvm::ArrayRef arguments) { + return self->query_toolchain_cached(file, directory, arguments); +} + +std::vector + ToolchainProvider::get_pending_queries(llvm::ArrayRef entries) { + llvm::StringMap seen_keys; + std::vector queries; + + for(auto& entry: entries) { + if(entry.arguments.empty()) { + continue; + } + + auto [key, query_args] = self->extract_toolchain_flags(entry.file, entry.arguments); + + // Skip if already cached or already queued. + if(self->toolchain_cache.count(key) || !seen_keys.try_emplace(key, true).second) { + continue; + } + + LOG_DEBUG("Pre-warm: new toolchain key (len={}) for file={}", key.size(), entry.file); + queries.push_back({std::move(key), std::move(query_args), entry.file, entry.directory}); + } + + LOG_INFO("Pre-warm: {} unique keys from {} entries, {} queries needed", + seen_keys.size(), + entries.size(), + queries.size()); + return queries; +} + +void ToolchainProvider::inject_results(llvm::ArrayRef results) { + for(auto& result: results) { + if(self->toolchain_cache.count(result.key)) { + continue; + } + std::vector saved; + saved.reserve(result.cc1_args.size()); + for(auto& arg: result.cc1_args) { + saved.push_back(self->strings.save(arg).data()); + } + self->toolchain_cache.try_emplace(result.key, std::move(saved)); + } +} + +bool ToolchainProvider::has_cached_entries() const { + return !self->toolchain_cache.empty(); +} + +} // namespace clice diff --git a/src/command/toolchain_provider.h b/src/command/toolchain_provider.h new file mode 100644 index 000000000..8c9878467 --- /dev/null +++ b/src/command/toolchain_provider.h @@ -0,0 +1,75 @@ +#pragma once + +#include +#include +#include +#include + +#include "llvm/ADT/ArrayRef.h" +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringRef.h" + +namespace clice { + +/// A pending toolchain query, ready to be executed (possibly in parallel). +struct ToolchainQuery { + std::string key; + std::vector query_args; + llvm::StringRef file; + llvm::StringRef directory; +}; + +/// Result of a toolchain query, to be injected back into the cache. +struct ToolchainResult { + std::string key; + std::vector cc1_args; +}; + +/// Manages toolchain queries and caching, separated from CompilationDatabase. +/// +/// Given compilation arguments, this component: +/// 1. Extracts toolchain-relevant flags (driver, target, sysroot, stdlib, etc.) +/// 2. Builds a canonical cache key from those flags +/// 3. Queries the compiler driver for system include paths (expensive: spawns a process) +/// 4. Caches results so identical toolchain configurations share one query +/// +/// Designed to be pluggable: CompilationDatabase holds a ToolchainProvider by +/// composition and delegates all toolchain operations to it. +class ToolchainProvider { +public: + ToolchainProvider(); + ~ToolchainProvider(); + ToolchainProvider(ToolchainProvider&&) noexcept; + ToolchainProvider& operator=(ToolchainProvider&&) noexcept; + + /// Query toolchain with caching. Returns cached cc1 args for the given + /// compilation arguments, running the expensive compiler query only on + /// cache miss. The returned ArrayRef is valid for the provider's lifetime. + llvm::ArrayRef query_cached(llvm::StringRef file, + llvm::StringRef directory, + llvm::ArrayRef arguments); + + /// Entry for batch pre-warming: file + directory + raw compilation arguments. + struct PendingEntry { + llvm::StringRef file; + llvm::StringRef directory; + llvm::SmallVector arguments; + }; + + /// Get pending queries for a batch of compilation entries. + /// Returns queries only for cache-miss keys (deduplicated). + std::vector get_pending_queries(llvm::ArrayRef entries); + + /// Inject pre-computed results into the cache. Strings are copied into + /// the provider's internal string pool. + void inject_results(llvm::ArrayRef results); + + /// Check if the cache has any entries. + bool has_cached_entries() const; + +private: + struct Impl; + std::unique_ptr self; +}; + +} // namespace clice diff --git a/src/compile/compilation.cpp b/src/compile/compilation.cpp index a5ea76278..ad68c276c 100644 --- a/src/compile/compilation.cpp +++ b/src/compile/compilation.cpp @@ -1,6 +1,6 @@ #include "compile/compilation.h" -#include "compile/command.h" +#include "command/command.h" #include "compile/diagnostic.h" #include "compile/implement.h" #include "semantic/ast_utility.h" diff --git a/src/server/master_server.cpp b/src/server/master_server.cpp index d03045629..c2311fd73 100644 --- a/src/server/master_server.cpp +++ b/src/server/master_server.cpp @@ -194,6 +194,9 @@ et::task<> MasterServer::load_workspace() { auto updates = cdb.load_compile_database(cdb_path); LOG_INFO("Loaded CDB from {} with {} entries", cdb_path, updates.size()); + + // Build dependency graph via wavefront BFS scan. + scan_dependency_graph(cdb, updates, path_pool, dependency_graph); } void MasterServer::fill_compile_args(llvm::StringRef path, diff --git a/src/server/master_server.h b/src/server/master_server.h index e488a1246..f87d582da 100644 --- a/src/server/master_server.h +++ b/src/server/master_server.h @@ -4,45 +4,26 @@ #include #include -#include "compile/command.h" +#include "command/command.h" #include "eventide/async/async.h" #include "eventide/ipc/lsp/protocol.h" #include "eventide/ipc/peer.h" #include "eventide/serde/serde/raw_value.h" #include "server/config.h" #include "server/worker_pool.h" +#include "support/path_pool.h" +#include "syntax/dependency_graph.h" #include "llvm/ADT/DenseMap.h" #include "llvm/ADT/SmallVector.h" #include "llvm/ADT/StringMap.h" #include "llvm/ADT/StringRef.h" -#include "llvm/Support/Allocator.h" namespace clice { namespace et = eventide; namespace protocol = et::ipc::protocol; -/// Global path interning pool. Maps file paths to uint32_t IDs. -struct ServerPathPool { - llvm::BumpPtrAllocator allocator; - llvm::SmallVector paths; - llvm::StringMap cache; - - std::uint32_t intern(llvm::StringRef path) { - auto [it, inserted] = cache.try_emplace(path, paths.size()); - if(inserted) { - auto saved = path.copy(allocator); - paths.push_back(saved); - } - return it->second; - } - - llvm::StringRef resolve(std::uint32_t id) const { - return paths[id]; - } -}; - struct DocumentState { int version = 0; std::string text; @@ -70,7 +51,7 @@ class MasterServer { et::event_loop& loop; et::ipc::JsonPeer& peer; WorkerPool pool; - ServerPathPool path_pool; + PathPool path_pool; ServerLifecycle lifecycle = ServerLifecycle::Uninitialized; std::string self_path; @@ -78,6 +59,7 @@ class MasterServer { CliceConfig config; CompilationDatabase cdb; + DependencyGraph dependency_graph; // Document state: path_id -> DocumentState llvm::DenseMap documents; diff --git a/src/support/path_pool.h b/src/support/path_pool.h new file mode 100644 index 000000000..12a5af58c --- /dev/null +++ b/src/support/path_pool.h @@ -0,0 +1,38 @@ +#pragma once + +#include +#include + +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringMap.h" +#include "llvm/ADT/StringRef.h" +#include "llvm/Support/Allocator.h" + +namespace clice { + +/// Intern pool that maps file paths to compact uint32_t IDs. +struct PathPool { + llvm::BumpPtrAllocator allocator; + llvm::SmallVector paths; + llvm::StringMap cache; + + std::uint32_t intern(llvm::StringRef path) { + auto [it, inserted] = cache.try_emplace(path, paths.size()); + if(inserted) { + // Allocate with null terminator so that resolve().data() is safe + // to use as const char* (e.g. in MemoryBuffer::getFile which calls strlen). + const std::size_t n = path.size(); + char* buf = allocator.Allocate(n + 1); + std::copy(path.begin(), path.end(), buf); + buf[n] = '\0'; + paths.push_back(llvm::StringRef(buf, n)); + } + return it->second; + } + + llvm::StringRef resolve(std::uint32_t id) const { + return paths[id]; + } +}; + +} // namespace clice diff --git a/src/syntax/dependency_graph.cpp b/src/syntax/dependency_graph.cpp new file mode 100644 index 000000000..28dec6468 --- /dev/null +++ b/src/syntax/dependency_graph.cpp @@ -0,0 +1,701 @@ +#include "syntax/dependency_graph.h" + +#include + +#include "command/toolchain.h" +#include "eventide/async/async.h" +#include "support/logging.h" +#include "syntax/include_resolver.h" +#include "syntax/scan.h" + +#include "llvm/ADT/DenseSet.h" +#include "llvm/ADT/StringSet.h" +#include "llvm/Support/FileSystem.h" +#include "llvm/Support/MemoryBuffer.h" +#include "llvm/Support/Path.h" +#include "llvm/Support/StringSaver.h" + +namespace clice { + +namespace et = eventide; + +// ============================================================================ +// DependencyGraph implementation +// ============================================================================ + +void DependencyGraph::add_module(llvm::StringRef module_name, std::uint32_t path_id) { + auto [it, inserted] = module_to_path.try_emplace(module_name, path_id); + if(!inserted && it->second != path_id) { + LOG_WARN("Duplicate module '{}': PathID {} overwrites {}", + module_name, + path_id, + it->second); + it->second = path_id; + } +} + +std::optional DependencyGraph::lookup_module(llvm::StringRef module_name) const { + auto it = module_to_path.find(module_name); + if(it != module_to_path.end()) { + return it->second; + } + return std::nullopt; +} + +void DependencyGraph::set_includes(std::uint32_t path_id, + std::uint32_t config_id, + llvm::SmallVector included_ids) { + IncludeKey key{path_id, config_id}; + includes[key] = std::move(included_ids); + file_configs[path_id].push_back(config_id); +} + +llvm::ArrayRef DependencyGraph::get_includes(std::uint32_t path_id, + std::uint32_t config_id) const { + auto it = includes.find(IncludeKey{path_id, config_id}); + if(it != includes.end()) { + return it->second; + } + return {}; +} + +llvm::SmallVector DependencyGraph::get_all_includes(std::uint32_t path_id) const { + llvm::DenseSet seen; + llvm::SmallVector result; + + auto fc_it = file_configs.find(path_id); + if(fc_it == file_configs.end()) { + return result; + } + + for(auto config_id: fc_it->second) { + auto it = includes.find(IncludeKey{path_id, config_id}); + if(it != includes.end()) { + for(auto id: it->second) { + auto raw_id = id & PATH_ID_MASK; + if(seen.insert(raw_id).second) { + result.push_back(id); + } + } + } + } + return result; +} + +std::size_t DependencyGraph::file_count() const { + return file_configs.size(); +} + +std::size_t DependencyGraph::module_count() const { + return module_to_path.size(); +} + +std::size_t DependencyGraph::edge_count() const { + std::size_t count = 0; + for(auto& [key, ids]: includes) { + count += ids.size(); + } + return count; +} + +// ============================================================================ +// Wavefront BFS scanner — async implementation +// ============================================================================ + +namespace { + +/// Result of scanning a single file (returned from worker thread). +struct FileScanResult { + const char* path; // Stable pointer from PathPool. + std::uint32_t path_id; + std::uint32_t config_id; + ScanResult scan_result; + bool read_failed = false; + std::int64_t read_us = 0; + std::int64_t scan_us = 0; +}; + +/// Scan a single file: read content + lexer scan. +/// Runs on libuv worker thread via queue(). +/// @param path Stable pointer from PathPool (must outlive the task). +FileScanResult scan_file_worker(const char* path, std::uint32_t path_id, std::uint32_t config_id) { + FileScanResult result; + result.path = path; + result.path_id = path_id; + result.config_id = config_id; + + auto t0 = std::chrono::steady_clock::now(); + // Force read() instead of mmap: RequiresNullTerminator=true makes LLVM + // fall back to read() for page-aligned files, and IsVolatile=true forces + // read() unconditionally — bypassing mmap entirely. This separates + // actual I/O cost from page-fault cost that was previously hidden inside + // the lexer timing. + auto buf = llvm::MemoryBuffer::getFile(result.path, + /*FileSize=*/-1, + /*RequiresNullTerminator=*/true, + /*IsVolatile=*/true); + auto t1 = std::chrono::steady_clock::now(); + result.read_us = std::chrono::duration_cast(t1 - t0).count(); + + if(!buf) { + result.read_failed = true; + return result; + } + + result.scan_result = scan((*buf)->getBuffer()); + auto t2 = std::chrono::steady_clock::now(); + result.scan_us = std::chrono::duration_cast(t2 - t1).count(); + + return result; +} + +/// The async scan implementation that runs on a local event loop. +et::task<> scan_impl(CompilationDatabase& cdb, + const std::vector& updates, + PathPool& path_pool, + DependencyGraph& graph, + ScanReport& report, + ScanCache* ext_cache, + et::event_loop& loop) { + auto start_time = std::chrono::steady_clock::now(); + + // Reuse context groups and configs from cache when available (warm runs). + // On the first call (or when cache is null) we build everything from scratch. + const bool have_config_cache = + ext_cache && !ext_cache->context_groups.empty() && !ext_cache->configs.empty(); + + // Provide local storage when not using the persistent cache. + llvm::DenseMap> local_context_groups; + llvm::DenseMap local_context_to_config_id; + llvm::DenseMap local_configs; + + // When ext_cache is provided, write directly into it so that the data + // survives across calls (making have_config_cache true on run 2+). + llvm::DenseMap>& context_groups = + ext_cache ? ext_cache->context_groups : local_context_groups; + llvm::DenseMap& context_to_config_id = + ext_cache ? ext_cache->context_to_config_id : local_context_to_config_id; + llvm::DenseMap& configs = + ext_cache ? ext_cache->configs : local_configs; + + auto config_start = std::chrono::steady_clock::now(); + + if(!have_config_cache) { + // Group files by context pointer to identify unique compilation commands. + // Convert CDB string IDs to PathPool IDs. + for(auto& update: updates) { + if(update.kind == UpdateKind::Deleted) { + continue; + } + auto path = cdb.resolve_path(update.path_id); + auto pool_id = path_pool.intern(path); + context_groups[update.context].push_back(pool_id); + } + + // Pre-warm toolchain cache: extract unique queries, execute in parallel. + // Skip entirely when configs are already cached (warm runs), since the + // toolchain cache is necessarily also populated from the previous scan. + auto prewarm_start = std::chrono::steady_clock::now(); + if(!cdb.has_cached_configs()) { + std::vector> file_contexts; + for(auto& [context, file_ids]: context_groups) { + auto representative_path = path_pool.resolve(file_ids[0]); + file_contexts.push_back({representative_path, context}); + } + + auto entries = cdb.resolve_toolchain_entries(file_contexts); + auto& tc = cdb.toolchain(); + auto pending = tc.get_pending_queries(entries); + if(!pending.empty()) { + LOG_INFO("Warming toolchain cache: {} unique queries", pending.size()); + + std::vector> tasks; + tasks.reserve(pending.size()); + for(auto& query: pending) { + tasks.push_back(et::queue( + [q = std::move(query)]() -> ToolchainResult { + ToolchainResult result; + result.key = q.key; + llvm::BumpPtrAllocator alloc; + llvm::StringSaver saver(alloc); + toolchain::query_toolchain({q.file, + q.directory, + q.query_args, + [&](const char* s) -> const char* { + result.cc1_args.push_back(s); + return saver.save(s).data(); + }}); + return result; + }, + loop)); + } + + auto outcome = co_await et::when_all(std::move(tasks)); + if(outcome.has_value()) { + tc.inject_results(*outcome); + } else { + LOG_ERROR("Parallel toolchain query failed: {}", outcome.error().message()); + } + } + } + auto prewarm_end = std::chrono::steady_clock::now(); + report.prewarm_ms = + std::chrono::duration_cast(prewarm_end - prewarm_start) + .count(); + + // Extract SearchConfig for each unique context. + std::uint32_t next_config_id = 0; + std::int64_t lookup_us = 0; + for(auto& [context, file_ids]: context_groups) { + std::uint32_t config_id = next_config_id++; + context_to_config_id[context] = config_id; + auto representative_path = path_pool.resolve(file_ids[0]); + auto t0 = std::chrono::steady_clock::now(); + configs[config_id] = + cdb.lookup_search_config(representative_path, + {.resource_dir = true, .query_toolchain = true}, + context); + auto t1 = std::chrono::steady_clock::now(); + lookup_us += std::chrono::duration_cast(t1 - t0).count(); + } + report.config_loop_ms = lookup_us / 1000; + LOG_INFO("Config extracted: {} groups, {:.1f}ms", configs.size(), lookup_us / 1000.0); + } + + auto config_end = std::chrono::steady_clock::now(); + report.config_ms = + std::chrono::duration_cast(config_end - config_start).count(); + + // Use external persistent cache when provided, otherwise create a local one. + DirListingCache local_dir_cache; + DirListingCache& dir_cache = ext_cache ? ext_cache->dir_cache : local_dir_cache; + + llvm::StringMap local_include_cache; + llvm::StringMap& include_cache = + ext_cache ? ext_cache->include_cache : local_include_cache; + + // ── Dir cache pre-population ───────────────────────────────────── + // Collect all unique search dirs and launch readdir tasks on the + // thread pool. Tasks start executing immediately but are NOT awaited + // here — instead they run concurrently with Wave 0's file scanning + // (Optimization 1: overlap dir cache with Phase 1). We only await + // them before Phase 2 of Wave 0, which is the first consumer. + + struct DirEntry { + std::string dir_path; + llvm::StringSet<> entries; + }; + + std::vector> pending_dir_tasks; + + if(dir_cache.dirs.empty()) { + llvm::StringSet<> unique_dirs; + for(auto& [config_id, config]: configs) { + for(auto& dir: config.dirs) { + unique_dirs.insert(dir.path); + } + } + // Also prefetch parent directories of source files (for quoted include resolution). + for(auto& [context, file_ids]: context_groups) { + for(auto path_id: file_ids) { + auto dir = llvm::sys::path::parent_path(path_pool.resolve(path_id)); + if(!dir.empty()) { + unique_dirs.insert(dir); + } + } + } + + pending_dir_tasks.reserve(unique_dirs.size()); + for(auto& entry: unique_dirs) { + auto dir_path = entry.getKey().str(); + pending_dir_tasks.push_back(et::queue( + [dir_path = std::move(dir_path)]() -> DirEntry { + DirEntry result; + result.dir_path = dir_path; + std::error_code ec; + llvm::sys::fs::directory_iterator di(result.dir_path, ec); + for(; !ec && di != llvm::sys::fs::directory_iterator(); di.increment(ec)) { + result.entries.insert(llvm::sys::path::filename(di->path())); + } + return result; + }, + loop)); + } + LOG_INFO("Launched {} dir cache tasks (running in background)", pending_dir_tasks.size()); + } + + // Track which files have been scanned (by path_id — cheaper than string hash). + // Value: found_dir_idx needed for #include_next. + llvm::DenseMap scanned_files; + + // Wave 0: all source files from CDB. + // Re-use the cached initial_wave when available to avoid re-iterating context_groups. + std::vector current_wave; + const bool have_initial_wave_cache = ext_cache && !ext_cache->initial_wave.empty(); + if(have_initial_wave_cache) { + current_wave = ext_cache->initial_wave; + for(auto& entry: current_wave) { + scanned_files.try_emplace(entry.path_id, entry.found_dir_idx); + } + } else { + current_wave.reserve(updates.size()); + for(auto& [context, file_ids]: context_groups) { + auto config_id = context_to_config_id[context]; + for(auto path_id: file_ids) { + scanned_files.try_emplace(path_id, 0u); + current_wave.push_back({path_id, config_id, /*found_dir_idx=*/0}); + } + } + if(ext_cache) { + ext_cache->initial_wave = current_wave; + } + } + + report.source_files = current_wave.size(); + std::size_t wave_num = 0; + + // Optimization 2: prefetch scan tasks. + // During Phase 2 of wave N, newly discovered files are immediately + // queued for scanning on the thread pool. When wave N+1 starts, + // these tasks are already running (or finished), eliminating most + // of the Phase 1 wait time for subsequent waves. + std::vector> prefetch_tasks; + + // Pre-resolved search configs: built once after dir cache is populated, + // then reused for all waves. Eliminates StringMap lookups in Phase 2. + llvm::DenseMap resolved_configs; + + while(!current_wave.empty()) { + auto wave_start = std::chrono::steady_clock::now(); + + // Phase 1: Read + scan all files in parallel on the thread pool. + // Files with a cached ScanResult skip I/O and lexing entirely. + // For waves > 0, files discovered during the previous wave's Phase 2 + // already have running scan tasks in prefetch_tasks. + std::vector scan_results; + scan_results.reserve(current_wave.size()); + std::size_t wave_cache_hits = 0; + + // Collect cache hits first (applies to all waves). + for(auto& entry: current_wave) { + if(ext_cache) { + auto it = ext_cache->scan_results.find(entry.path_id); + if(it != ext_cache->scan_results.end()) { + scan_results.push_back({path_pool.resolve(entry.path_id).data(), + entry.path_id, + entry.config_id, + it->second, + false, + 0, + 0}); + report.scan_cache_hits++; + wave_cache_hits++; + } + } + } + + if(!prefetch_tasks.empty()) { + // Waves 1+: await prefetched scan tasks from previous Phase 2. + auto scan_outcome = co_await et::when_all(std::move(prefetch_tasks)); + prefetch_tasks.clear(); + if(scan_outcome.has_error()) { + LOG_ERROR("Prefetch scan failed: {}", scan_outcome.error().message()); + break; + } + for(auto& r: *scan_outcome) { + if(!r.read_failed && ext_cache) { + ext_cache->scan_results.try_emplace(r.path_id, r.scan_result); + } + scan_results.push_back(std::move(r)); + } + } else { + // Wave 0 (or warm run with all cache hits): create scan tasks now. + std::vector> scan_tasks; + scan_tasks.reserve(current_wave.size()); + for(auto& entry: current_wave) { + auto pid = entry.path_id; + auto cid = entry.config_id; + // Skip files already served from cache above. + if(ext_cache && ext_cache->scan_results.count(pid)) { + continue; + } + auto path = path_pool.resolve(pid).data(); + scan_tasks.push_back( + et::queue([path, pid, cid]() { return scan_file_worker(path, pid, cid); }, + loop)); + } + + // Optimization 1: await dir cache tasks concurrently with scan tasks. + // Both sets of tasks run on the same thread pool. By awaiting dir + // tasks first (while scan tasks continue in the background), we pay + // max(dir_time, scan_time) instead of dir_time + scan_time. + if(!pending_dir_tasks.empty()) { + auto dir_t0 = std::chrono::steady_clock::now(); + auto dir_outcome = co_await et::when_all(std::move(pending_dir_tasks)); + pending_dir_tasks.clear(); + if(dir_outcome.has_value()) { + for(auto& entry: *dir_outcome) { + dir_cache.dirs.try_emplace(entry.dir_path, std::move(entry.entries)); + } + LOG_INFO("Pre-populated dir cache: {} directories", dir_outcome->size()); + } + auto dir_t1 = std::chrono::steady_clock::now(); + report.dir_cache_ms = + std::chrono::duration_cast(dir_t1 - dir_t0).count(); + } + + if(!scan_tasks.empty()) { + auto scan_outcome = co_await et::when_all(std::move(scan_tasks)); + if(scan_outcome.has_error()) { + LOG_ERROR("Parallel scan failed: {}", scan_outcome.error().message()); + break; + } + for(auto& r: *scan_outcome) { + if(!r.read_failed && ext_cache) { + ext_cache->scan_results.try_emplace(r.path_id, r.scan_result); + } + scan_results.push_back(std::move(r)); + } + } + } + + auto phase1_end = std::chrono::steady_clock::now(); + + // Accumulate per-file read/scan timing into report. + for(auto& sr: scan_results) { + report.read_us += sr.read_us; + report.scan_us += sr.scan_us; + } + + // Pre-resolve search configs once after dir cache is populated (wave 0). + // Converts StringMap lookups into direct pointer dereferences for Phase 2. + if(resolved_configs.empty()) { + for(auto& [config_id, config]: configs) { + resolved_configs[config_id] = resolve_search_config(config, dir_cache); + } + } + + // Phase 2+3: Resolve includes, intern paths, build graph, collect next wave. + // Merged into a single pass to avoid intermediate string allocations. + // Optimization 2: newly discovered files are immediately queued for + // scanning (prefetch_tasks), overlapping Phase 1 of the next wave + // with Phase 2 of the current wave. + std::vector next_wave; + next_wave.reserve(current_wave.size()); // Heuristic: next wave ≤ current wave. + StatCounters wave_stat_counters; + + for(auto& scan_result: scan_results) { + report.total_files++; + + if(scan_result.read_failed) { + LOG_WARN("Failed to read file for scanning: {}", scan_result.path); + continue; + } + + auto rc_it = resolved_configs.find(scan_result.config_id); + if(rc_it == resolved_configs.end()) { + continue; + } + + auto& resolved_config = rc_it->second; + auto includer_dir = llvm::sys::path::parent_path(scan_result.path); + auto* includer_entries = resolve_dir(includer_dir, dir_cache, &wave_stat_counters); + + // Look up the found_dir_idx for this file (stored when it was discovered). + unsigned includer_found_dir_idx = 0; + auto sf_it = scanned_files.find(scan_result.path_id); + if(sf_it != scanned_files.end()) { + includer_found_dir_idx = sf_it->second; + } + + // Record module mapping. + if(!scan_result.scan_result.module_name.empty()) { + graph.add_module(scan_result.scan_result.module_name, scan_result.path_id); + } + + report.includes_found += scan_result.scan_result.includes.size(); + + llvm::SmallVector include_ids; + include_ids.reserve(scan_result.scan_result.includes.size()); + + for(auto& inc: scan_result.scan_result.includes) { + // For angled includes, resolution depends only on config (not includer dir). + // Cache these to skip redundant directory searches across files. + bool cache_eligible = inc.is_angled && !inc.is_include_next; + llvm::SmallString<80> cache_key; + if(cache_eligible) { + cache_key.append(reinterpret_cast(&scan_result.config_id), + reinterpret_cast(&scan_result.config_id) + + sizeof(std::uint32_t)); + cache_key += inc.path; + + auto cache_it = include_cache.find(cache_key); + if(cache_it != include_cache.end()) { + report.include_cache_hits++; + auto cached_id = cache_it->second; + if(cached_id == UINT32_MAX) { + report.unresolved.push_back({ + std::move(inc.path), + std::string(path_pool.resolve(scan_result.path_id)), + inc.is_angled, + inc.conditional, + }); + continue; + } + report.includes_resolved++; + // Jump directly to edge building with cached path_id. + std::uint32_t flagged_id = cached_id; + if(inc.conditional) { + flagged_id |= DependencyGraph::CONDITIONAL_FLAG; + report.conditional_edges++; + } else { + report.unconditional_edges++; + } + report.total_edges++; + include_ids.push_back(flagged_id); + if(scanned_files.try_emplace(cached_id, 0u).second) { + next_wave.push_back({cached_id, scan_result.config_id}); + } + continue; + } + } + + auto r_t0 = std::chrono::steady_clock::now(); + auto resolved = resolve_include(inc.path, + inc.is_angled, + includer_entries, + includer_dir, + inc.is_include_next, + includer_found_dir_idx, + resolved_config, + dir_cache, + &wave_stat_counters); + auto r_t1 = std::chrono::steady_clock::now(); + report.p2_resolve_us += + std::chrono::duration_cast(r_t1 - r_t0).count(); + if(!resolved.has_value()) { + if(cache_eligible) { + include_cache.try_emplace(cache_key, UINT32_MAX); + } + report.unresolved.push_back({ + std::move(inc.path), + std::string(path_pool.resolve(scan_result.path_id)), + inc.is_angled, + inc.conditional, + }); + continue; + } + + auto inc_path_id = path_pool.intern(resolved->path); + report.includes_resolved++; + + if(cache_eligible) { + include_cache.try_emplace(cache_key, inc_path_id); + } + + std::uint32_t flagged_id = inc_path_id; + if(inc.conditional) { + flagged_id |= DependencyGraph::CONDITIONAL_FLAG; + report.conditional_edges++; + } else { + report.unconditional_edges++; + } + report.total_edges++; + include_ids.push_back(flagged_id); + + if(scanned_files.try_emplace(inc_path_id, resolved->found_dir_idx).second) { + next_wave.push_back( + {inc_path_id, scan_result.config_id, resolved->found_dir_idx}); + // Prefetch: start scanning this file immediately on the + // thread pool so it's ready when the next wave begins. + if(!ext_cache || + ext_cache->scan_results.find(inc_path_id) == ext_cache->scan_results.end()) { + auto inc_path = path_pool.resolve(inc_path_id).data(); + prefetch_tasks.push_back(et::queue( + [inc_path, inc_path_id, cid = scan_result.config_id]() { + return scan_file_worker(inc_path, inc_path_id, cid); + }, + loop)); + } + } + } + + graph.set_includes(scan_result.path_id, scan_result.config_id, std::move(include_ids)); + } + + report.dir_listings += wave_stat_counters.dir_listings; + report.dir_hits += wave_stat_counters.dir_hits; + report.fs_lookups += wave_stat_counters.lookups; + report.fs_us += wave_stat_counters.us; + + auto phase2_end = std::chrono::steady_clock::now(); + auto phase3_end = phase2_end; + + auto p1 = + std::chrono::duration_cast(phase1_end - wave_start).count(); + auto p2 = + std::chrono::duration_cast(phase2_end - phase1_end).count(); + auto p3 = + std::chrono::duration_cast(phase3_end - phase2_end).count(); + + report.phase1_ms += p1; + report.phase2_ms += p2; + report.phase3_ms += p3; + + // Record per-wave stats for cold start analysis. + ScanReport::WaveStats ws; + ws.files = current_wave.size(); + ws.phase1_ms = p1; + ws.phase2_ms = p2; + ws.next_files = next_wave.size(); + ws.prefetch_count = prefetch_tasks.size(); + ws.dir_listings = wave_stat_counters.dir_listings; + ws.dir_hits = wave_stat_counters.dir_hits; + ws.cache_hits = wave_cache_hits; + report.wave_stats.push_back(ws); + + LOG_INFO( + "Wave {}: {} files | read+scan={}ms resolve={}ms graph={}ms | next={} " "prefetch={}", + wave_num, + current_wave.size(), + p1, + p2, + p3, + next_wave.size(), + prefetch_tasks.size()); + + current_wave = std::move(next_wave); + wave_num++; + } + + auto end_time = std::chrono::steady_clock::now(); + report.elapsed_ms = + std::chrono::duration_cast(end_time - start_time).count(); + report.header_files = report.total_files - report.source_files; + report.modules = graph.module_count(); + report.waves = wave_num; +} + +} // namespace + +// ============================================================================ +// Public sync entry point +// ============================================================================ + +ScanReport scan_dependency_graph(CompilationDatabase& cdb, + const std::vector& updates, + PathPool& path_pool, + DependencyGraph& graph, + ScanCache* cache) { + ScanReport report; + if(updates.empty()) { + return report; + } + + et::event_loop loop; + loop.schedule(scan_impl(cdb, updates, path_pool, graph, report, cache, loop)); + loop.run(); + return report; +} + +} // namespace clice diff --git a/src/syntax/dependency_graph.h b/src/syntax/dependency_graph.h new file mode 100644 index 000000000..e8e0459d0 --- /dev/null +++ b/src/syntax/dependency_graph.h @@ -0,0 +1,242 @@ +#pragma once + +#include +#include +#include +#include + +#include "command/command.h" +#include "support/path_pool.h" +#include "syntax/include_resolver.h" +#include "syntax/scan.h" + +#include "llvm/ADT/ArrayRef.h" +#include "llvm/ADT/DenseMap.h" +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringMap.h" +#include "llvm/ADT/StringRef.h" + +namespace clice { + +class DependencyGraph { +public: + /// Conditional flag: bit 31 marks an include inside #ifdef/#if. + constexpr static std::uint32_t CONDITIONAL_FLAG = 0x80000000u; + + /// Mask to extract the actual PathID from a flagged value. + constexpr static std::uint32_t PATH_ID_MASK = 0x7FFFFFFFu; + + /// Key for per-(file, SearchConfig) include storage. + struct IncludeKey { + std::uint32_t path_id; + std::uint32_t config_id; + + bool operator==(const IncludeKey&) const = default; + }; + + struct IncludeKeyInfo { + static IncludeKey getEmptyKey() { + return {~0u, ~0u}; + } + + static IncludeKey getTombstoneKey() { + return {~0u - 1, ~0u - 1}; + } + + static unsigned getHashValue(const IncludeKey& key) { + return llvm::DenseMapInfo::getHashValue( + (std::uint64_t(key.path_id) << 32) | key.config_id); + } + + static bool isEqual(const IncludeKey& lhs, const IncludeKey& rhs) { + return lhs == rhs; + } + }; + + /// Register a module name -> PathID mapping. + void add_module(llvm::StringRef module_name, std::uint32_t path_id); + + /// Look up the PathID that provides a given module. + std::optional lookup_module(llvm::StringRef module_name) const; + + /// Set the direct include list for a (file, config) pair. + void set_includes(std::uint32_t path_id, + std::uint32_t config_id, + llvm::SmallVector included_ids); + + /// Get direct includes for a specific (file, config) pair. + llvm::ArrayRef get_includes(std::uint32_t path_id, + std::uint32_t config_id) const; + + /// Get the union of includes across all configs for a file. + llvm::SmallVector get_all_includes(std::uint32_t path_id) const; + + /// Number of files with include entries. + std::size_t file_count() const; + + /// Number of module mappings. + std::size_t module_count() const; + + /// Total number of include edges across all (file, config) pairs. + std::size_t edge_count() const; + + /// Access the module name -> PathID mapping. + const llvm::StringMap& modules() const { + return module_to_path; + } + +private: + /// Module name -> PathID. + llvm::StringMap module_to_path; + + /// (PathID, ConfigID) -> list of directly included PathIDs. + /// Each PathID may have bit 31 set to indicate conditional include. + llvm::DenseMap, IncludeKeyInfo> includes; + + /// Track which files have any include entries (for file_count). + llvm::DenseMap> file_configs; +}; + +/// A (file, search-config) pair used to track per-wave work items. +struct WaveEntry { + std::uint32_t path_id; + std::uint32_t config_id; + /// Search dir index where this file was found. Used for #include_next. + /// Source files (wave 0) use 0. + unsigned found_dir_idx = 0; +}; + +/// Detailed report from a dependency scan. +struct ScanReport { + /// Timing in milliseconds. + std::int64_t elapsed_ms = 0; + + /// File counts. + std::size_t source_files = 0; // Files from CDB (translation units). + std::size_t header_files = 0; // Files discovered via include scanning. + std::size_t total_files = 0; // source_files + header_files. + + /// Include edge counts. + std::size_t total_edges = 0; // Total include edges. + std::size_t conditional_edges = 0; // Edges inside #if/#ifdef. + std::size_t unconditional_edges = 0; // Edges not inside conditionals. + + /// Include resolution. + std::size_t includes_found = 0; // Total #include directives seen. + std::size_t includes_resolved = 0; // Successfully resolved to a file. + + /// Module info. + std::size_t modules = 0; + + /// BFS wave count. + std::size_t waves = 0; + + /// Wall-clock time per phase (milliseconds, summed across waves). + std::int64_t phase1_ms = 0; // Read + scan (parallel on thread pool). + std::int64_t phase2_ms = 0; // Include resolution (stat calls). + std::int64_t phase3_ms = 0; // Graph building (single-threaded). + std::int64_t config_ms = 0; // Config extraction (one-time, total). + std::int64_t prewarm_ms = 0; // Toolchain pre-warm subset. + std::int64_t config_loop_ms = 0; // lookup + extract_search_config loop. + std::int64_t dir_cache_ms = 0; // Dir cache pre-population (overlapped with Phase 1). + + /// Cumulative I/O time across all threads/files (microseconds). + /// These are sums of per-file durations — will exceed wall-clock time + /// when work is parallelized across threads. + std::int64_t read_us = 0; // File read (cumulative across threads). + std::int64_t scan_us = 0; // Lexer scan (cumulative across threads). + std::int64_t fs_us = 0; // Filesystem ops (readdir calls). + + /// Phase 2 breakdown (microseconds, single-threaded). + std::int64_t p2_resolve_us = 0; // resolve_include() calls. + + /// Filesystem call counts. + std::size_t dir_listings = 0; // Actual readdir() calls (dir cache misses). + std::size_t dir_hits = 0; // Directory cache hits (no syscall). + std::size_t fs_lookups = 0; // Total file existence lookups. + std::size_t include_cache_hits = 0; // Include resolution cache hits (skipped resolve). + std::size_t scan_cache_hits = 0; // Scan result cache hits (skipped I/O + lexer). + + /// Per-wave timing breakdown for cold start analysis. + struct WaveStats { + std::size_t files = 0; // Files processed in this wave. + std::int64_t phase1_ms = 0; // Read + scan (parallel). + std::int64_t phase2_ms = 0; // Include resolution (serial). + std::size_t next_files = 0; // Files discovered for next wave. + std::size_t prefetch_count = 0; // Prefetch tasks launched during Phase 2. + std::size_t dir_listings = 0; // readdir() calls in this wave. + std::size_t dir_hits = 0; // Dir cache hits in this wave. + std::size_t cache_hits = 0; // Scan cache hits in this wave. + }; + + std::vector wave_stats; + + /// Unresolved includes: (header_name, includer_path). + struct UnresolvedInclude { + std::string header; + std::string includer; + bool is_angled = false; + bool conditional = false; + }; + + std::vector unresolved; +}; + +/// Persistent cache that can be reused across successive scan calls. +/// Holding onto this between incremental re-scans eliminates repeated +/// readdir() calls, angled-include resolution, and file I/O on warm runs. +/// +/// Thread safety: not thread-safe; callers must serialise scan calls. +/// +/// Invalidation: callers must clear (or discard) this cache whenever the +/// compilation database or filesystem state changes. +struct ScanCache { + /// Directory listing cache: dir path → set of filenames. + DirListingCache dir_cache; + + /// Angled-include resolution cache: (config_id bytes + header) → path_id. + /// path_id values are valid only for the PathPool used during the scan + /// that populated this cache. If PathPool is reset between scans, clear + /// this cache too (or pass nullptr to scan_dependency_graph). + llvm::StringMap include_cache; + + /// Lexer scan result cache: path_id → ScanResult. + /// Populated on the first scan of each file. On subsequent calls the + /// worker-thread file read and lexer scan are skipped entirely, making + /// warm-run Phase 1 effectively free. + /// Invalidate per-entry when a file changes on disk. + llvm::DenseMap scan_results; + + // ── Config extraction cache ────────────────────────────────────────── + // Populated during the first scan and reused on all subsequent calls + // when the compilation database has not changed. + + /// Files grouped by unique CompilationInfo pointer (context). + /// path_ids are valid for the persistent PathPool. + llvm::DenseMap> context_groups; + + /// Context pointer → dense config_id (index into configs). + llvm::DenseMap context_to_config_id; + + /// Per-config search configuration (reused across scans). + llvm::DenseMap configs; + + /// Pre-built initial wave (wave 0): all source files with their config IDs. + std::vector initial_wave; +}; + +/// Run the wavefront BFS scan over all files in the compilation database. +/// Internally creates a local event loop for async I/O (file reads via worker +/// thread pool, stat calls via libuv). Blocks until the scan is complete. +/// +/// @param cache Optional persistent cache. When non-null and pre-populated, +/// avoids repeated readdir() and include-resolution work across +/// successive calls. PathPool must NOT be reset between calls +/// when a persistent cache is used (path_id values must remain stable). +ScanReport scan_dependency_graph(CompilationDatabase& cdb, + const std::vector& updates, + PathPool& path_pool, + DependencyGraph& graph, + ScanCache* cache = nullptr); + +} // namespace clice diff --git a/src/syntax/include_resolver.cpp b/src/syntax/include_resolver.cpp new file mode 100644 index 000000000..4d1d2fea6 --- /dev/null +++ b/src/syntax/include_resolver.cpp @@ -0,0 +1,206 @@ +#include "syntax/include_resolver.h" + +#include + +#include "llvm/Support/FileSystem.h" +#include "llvm/Support/Path.h" + +namespace clice { + +const llvm::StringSet<>* resolve_dir(llvm::StringRef dir, + DirListingCache& cache, + StatCounters* counters) { + auto it = cache.dirs.find(dir); + if(it != cache.dirs.end()) { + if(counters) { + counters->dir_hits++; + } + return &it->second; + } + + if(counters) { + counters->dir_listings++; + } + + auto t0 = std::chrono::steady_clock::now(); + llvm::StringSet<> entries; + std::error_code ec; + llvm::sys::fs::directory_iterator di(dir, ec); + for(; !ec && di != llvm::sys::fs::directory_iterator(); di.increment(ec)) { + entries.insert(llvm::sys::path::filename(di->path())); + } + auto t1 = std::chrono::steady_clock::now(); + if(counters) { + counters->us += std::chrono::duration_cast(t1 - t0).count(); + } + + auto [new_it, _] = cache.dirs.try_emplace(dir, std::move(entries)); + return &new_it->second; +} + +ResolvedSearchConfig resolve_search_config(const SearchConfig& config, DirListingCache& cache) { + ResolvedSearchConfig resolved; + resolved.angled_start_idx = config.angled_start_idx; + resolved.system_start_idx = config.system_start_idx; + resolved.after_start_idx = config.after_start_idx; + resolved.dirs.reserve(config.dirs.size()); + for(auto& dir: config.dirs) { + resolved.dirs.push_back({dir.path, resolve_dir(dir.path, cache)}); + } + return resolved; +} + +namespace { + +/// Check if a file exists in a directory, handling multi-component include paths. +/// For simple filenames (no '/'), checks pre-resolved entries directly. +/// For multi-component paths like "llvm/Support/raw_ostream.h", constructs the +/// full path and resolves the actual parent subdirectory via DirListingCache. +bool check_in_dir(llvm::StringRef dir_path, + const llvm::StringSet<>* entries, + llvm::StringRef filename, + bool is_simple, + DirListingCache& dir_cache, + StatCounters* counters) { + if(counters) + counters->lookups++; + + if(is_simple) { + return entries->contains(filename); + } + + // Quick rejection: check if first path component exists in pre-resolved + // entries. For "llvm/Support/raw_ostream.h", check if "llvm" exists in + // the search dir listing. Most search dirs won't have it, so we skip + // the expensive full path construction + subdirectory resolution. + // Skip this for relative paths starting with "." or ".." (e.g. "../foo.h"). + auto first_sep = filename.find_first_of("/\\"); + auto first_component = filename.substr(0, first_sep); + if(first_component != "." && first_component != "..") { + if(!entries->contains(first_component)) { + return false; + } + } + + // First component matched — construct full path, resolve actual subdirectory. + llvm::SmallString<256> full; + full = dir_path; + llvm::sys::path::append(full, filename); + auto parent = llvm::sys::path::parent_path(full); + auto name = llvm::sys::path::filename(full); + auto* sub_entries = resolve_dir(parent, dir_cache, counters); + return sub_entries->contains(name); +} + +} // namespace + +std::optional resolve_include(llvm::StringRef filename, + bool is_angled, + const llvm::StringSet<>* includer_entries, + llvm::StringRef includer_dir, + bool is_include_next, + unsigned found_dir_idx, + const ResolvedSearchConfig& config, + DirListingCache& dir_cache, + StatCounters* stat_counters) { + // 1. Absolute path: check directly via stat(). + if(llvm::sys::path::is_absolute(filename)) { + if(llvm::sys::fs::exists(filename)) { + return ResolveResult{llvm::SmallString<256>(filename), 0}; + } + return std::nullopt; + } + + // Check if filename has path separators (multi-component like "llvm/Support/foo.h"). + bool is_simple = + filename.find('/') == llvm::StringRef::npos && filename.find('\\') == llvm::StringRef::npos; + + // Check if filename contains "." or ".." components that need normalization. + // Only these produce non-canonical paths after path::append. + bool needs_normalize = !is_simple && (filename.find("..") != llvm::StringRef::npos || + filename.find("./") != llvm::StringRef::npos || + filename.find("\\.") != llvm::StringRef::npos); + + llvm::SmallString<256> candidate; + + // Helper: build candidate path + normalize if needed. + auto make_candidate = [&](llvm::StringRef dir, llvm::StringRef fname) { + candidate = dir; + llvm::sys::path::append(candidate, fname); + if(needs_normalize) { + llvm::sys::path::remove_dots(candidate, /*remove_dot_dot=*/true); + } + }; + + // 2. For #include_next, start from found_dir_idx + 1. + if(is_include_next) { + unsigned start = found_dir_idx + 1; + for(unsigned i = start; i < config.dirs.size(); ++i) { + if(check_in_dir(config.dirs[i].path, + config.dirs[i].entries, + filename, + is_simple, + dir_cache, + stat_counters)) { + make_candidate(config.dirs[i].path, filename); + return ResolveResult{candidate, i}; + } + } + return std::nullopt; + } + + // 3. Quoted include: try includer's directory first. + if(!is_angled && includer_entries) { + if(check_in_dir(includer_dir, + includer_entries, + filename, + is_simple, + dir_cache, + stat_counters)) { + make_candidate(includer_dir, filename); + return ResolveResult{candidate, 0}; + } + } + + // 4. Search directories from appropriate start index. + // TODO: macOS Framework search — for , try Foo.framework/Headers/Bar.h + // in dirs marked as framework dirs (-F, -iframework). + unsigned start = is_angled ? config.angled_start_idx : 0; + for(unsigned i = start; i < config.dirs.size(); ++i) { + if(check_in_dir(config.dirs[i].path, + config.dirs[i].entries, + filename, + is_simple, + dir_cache, + stat_counters)) { + make_candidate(config.dirs[i].path, filename); + return ResolveResult{candidate, i}; + } + } + + return std::nullopt; +} + +std::optional resolve_include(llvm::StringRef filename, + bool is_angled, + llvm::StringRef includer_dir, + bool is_include_next, + unsigned found_dir_idx, + const SearchConfig& config, + DirListingCache& dir_cache, + StatCounters* stat_counters) { + auto resolved_config = resolve_search_config(config, dir_cache); + const llvm::StringSet<>* includer_entries = + includer_dir.empty() ? nullptr : resolve_dir(includer_dir, dir_cache, stat_counters); + return resolve_include(filename, + is_angled, + includer_entries, + includer_dir, + is_include_next, + found_dir_idx, + resolved_config, + dir_cache, + stat_counters); +} + +} // namespace clice diff --git a/src/syntax/include_resolver.h b/src/syntax/include_resolver.h new file mode 100644 index 000000000..41cee1b3e --- /dev/null +++ b/src/syntax/include_resolver.h @@ -0,0 +1,101 @@ +#pragma once + +#include +#include + +#include "command/search_config.h" + +#include "llvm/ADT/SmallString.h" +#include "llvm/ADT/SmallVector.h" +#include "llvm/ADT/StringMap.h" +#include "llvm/ADT/StringRef.h" +#include "llvm/ADT/StringSet.h" + +namespace clice { + +struct ResolveResult { + /// The resolved absolute path (stack-allocated for paths < 256 chars). + llvm::SmallString<256> path; + + /// The index in SearchConfig::dirs where this file was found. + /// Used for #include_next to resume searching from found_dir_idx + 1. + unsigned found_dir_idx = 0; +}; + +/// Counters for filesystem call tracking during include resolution. +struct StatCounters { + std::size_t dir_listings = 0; // Actual readdir() calls (directory cache misses). + std::size_t dir_hits = 0; // Directory cache hits (no syscall). + std::size_t lookups = 0; // Total file existence lookups. + std::int64_t us = 0; // Microseconds spent in filesystem ops. +}; + +/// Cache of directory listings for fast file existence checks. +/// Instead of calling stat() for each candidate path, we list directory +/// contents once via readdir() and do in-memory set lookups thereafter. +/// This is dramatically faster on Windows where individual stat() calls +/// are very expensive (~10x slower than Linux). +struct DirListingCache { + llvm::StringMap> dirs; +}; + +/// A search directory with a pre-resolved pointer to its cached entries. +/// The pointer is stable because StringMap allocates entries on the heap. +struct ResolvedSearchDir { + llvm::StringRef path; + const llvm::StringSet<>* entries; // Never null after resolve_search_config(). +}; + +/// Pre-resolved version of SearchConfig — all directory lookups are resolved +/// to direct pointers, eliminating StringMap lookups during include resolution. +struct ResolvedSearchConfig { + llvm::SmallVector dirs; + unsigned angled_start_idx = 0; + unsigned system_start_idx = 0; + unsigned after_start_idx = 0; +}; + +/// Resolve a single directory to its cached StringSet. +/// Returns a stable pointer into the DirListingCache. +/// On cache miss, lazily populates via readdir(). +const llvm::StringSet<>* resolve_dir(llvm::StringRef dir, + DirListingCache& cache, + StatCounters* counters = nullptr); + +/// Pre-resolve a SearchConfig against a populated DirListingCache. +/// Call once per config after dir cache pre-population, then reuse +/// the result for all resolve_include() calls with that config. +ResolvedSearchConfig resolve_search_config(const SearchConfig& config, DirListingCache& cache); + +/// Resolve an include directive using pre-resolved config and includer entries. +/// +/// @param filename Raw include name (without delimiters) +/// @param is_angled Whether this is a <...> include +/// @param includer_entries Pre-resolved StringSet for the includer's directory (may be null) +/// @param includer_dir Directory of the file containing the #include +/// @param is_include_next Whether this is #include_next +/// @param found_dir_idx For #include_next: the search dir index of the includer +/// @param config Pre-resolved search configuration +/// @return Resolved path and the search dir index, or nullopt if not found +std::optional resolve_include(llvm::StringRef filename, + bool is_angled, + const llvm::StringSet<>* includer_entries, + llvm::StringRef includer_dir, + bool is_include_next, + unsigned found_dir_idx, + const ResolvedSearchConfig& config, + DirListingCache& dir_cache, + StatCounters* stat_counters = nullptr); + +/// Convenience overload: resolves config and includer_dir on the fly. +/// Use for tests and one-off calls where pre-resolution overhead doesn't matter. +std::optional resolve_include(llvm::StringRef filename, + bool is_angled, + llvm::StringRef includer_dir, + bool is_include_next, + unsigned found_dir_idx, + const SearchConfig& config, + DirListingCache& dir_cache, + StatCounters* stat_counters = nullptr); + +} // namespace clice diff --git a/src/syntax/scan.cpp b/src/syntax/scan.cpp index 33242760b..16fa72838 100644 --- a/src/syntax/scan.cpp +++ b/src/syntax/scan.cpp @@ -31,6 +31,9 @@ ScanResult scan(llvm::StringRef content) { return result; } + // Most source files have 10-30 includes; pre-allocate to avoid reallocs. + result.includes.reserve(std::min(directives.size(), 32)); + int conditional_depth = 0; for(auto& dir: directives) { @@ -62,11 +65,13 @@ ScanResult scan(llvm::StringRef content) { auto name = content.substr(tok.Offset, tok.Length); // Strip <> or "" delimiters. if(name.size() >= 2) { - result.includes.push_back({ - std::string(name.substr(1, name.size() - 2)), - conditional_depth > 0, - false, - }); + bool angled = name.front() == '<'; + ScanResult::IncludeInfo info; + info.path = std::string(name.substr(1, name.size() - 2)); + info.conditional = conditional_depth > 0; + info.is_angled = angled; + info.is_include_next = dir.Kind == dds::pp_include_next; + result.includes.push_back(std::move(info)); } break; } @@ -118,53 +123,14 @@ ScanResult scan(llvm::StringRef content) { namespace { -enum class ScanMode { Fuzzy, Precise }; - -/// Compute include_is_conditional from raw directives: for each pp_include -/// (and pp_include_next, pp___include_macros, pp_import), record whether -/// it is nested inside any conditional block. -void compute_include_conditionals(SharedScanCache::CachedEntry& entry) { - using namespace clang::dependency_directives_scan; - - entry.include_is_conditional.clear(); - int cond_depth = 0; - - for(auto& dir: entry.directives) { - switch(dir.Kind) { - case pp_if: - case pp_ifdef: - case pp_ifndef: { - cond_depth++; - break; - } - case pp_endif: { - if(cond_depth > 0) { - cond_depth--; - } - break; - } - case pp_include: - case pp_include_next: - case pp___include_macros: - case pp_import: { - entry.include_is_conditional.push_back(cond_depth > 0); - break; - } - default: { - break; - } - } - } -} - class ScanDirectivesGetter : public clang::DependencyDirectivesGetter { public: - ScanDirectivesGetter(ScanMode mode, SharedScanCache* cache, clang::FileManager& file_mgr) : - mode(mode), cache(cache), file_mgr(&file_mgr) {} + ScanDirectivesGetter(SharedScanCache* cache, clang::FileManager& file_mgr) : + cache(cache), file_mgr(&file_mgr) {} std::unique_ptr cloneFor(clang::FileManager& new_file_mgr) override { - return std::make_unique(mode, cache, new_file_mgr); + return std::make_unique(cache, new_file_mgr); } std::optional> @@ -178,7 +144,7 @@ class ScanDirectivesGetter : public clang::DependencyDirectivesGetter { if(cache) { auto it = cache->entries.find(path); if(it != cache->entries.end()) { - return get_directives(it->second); + return llvm::ArrayRef(it->second.directives); } } @@ -216,149 +182,13 @@ class ScanDirectivesGetter : public clang::DependencyDirectivesGetter { return std::nullopt; } - compute_include_conditionals(*entry_ptr); - return get_directives(*entry_ptr); + return llvm::ArrayRef(entry_ptr->directives); } private: - using DirectiveVec = llvm::SmallVector; - - llvm::ArrayRef - get_directives(SharedScanCache::CachedEntry& entry) { - if(mode == ScanMode::Precise) { - return entry.directives; - } - - // Fuzzy mode: strip #define/#undef and ALL conditional directives, - // so every #include is processed unconditionally by the preprocessor. - auto& slot = filtered_directives[&entry]; - if(slot && !slot->empty()) { - return *slot; - } - - slot = std::make_unique(); - - using namespace clang::dependency_directives_scan; - for(auto& dir: entry.directives) { - switch(dir.Kind) { - case pp_define: - case pp_undef: - case pp_if: - case pp_ifdef: - case pp_ifndef: - case pp_elif: - case pp_elifdef: - case pp_elifndef: - case pp_else: - case pp_endif: - case pp_pragma_push_macro: - case pp_pragma_pop_macro: { - break; - } - default: { - slot->push_back(dir); - break; - } - } - } - - return *slot; - } - - ScanMode mode; SharedScanCache* cache; clang::FileManager* file_mgr; std::deque local_entries; - llvm::DenseMap> - filtered_directives; -}; - -/// PPCallbacks for fuzzy mode: tracks per-file includes with conditional -/// flags looked up from the SharedScanCache. -class FuzzyScanPPCallbacks : public clang::PPCallbacks { -public: - FuzzyScanPPCallbacks(llvm::StringMap& results, - SharedScanCache& cache, - clang::SourceManager& source_mgr) : - results(results), cache(cache), source_mgr(source_mgr) {} - - void FileChanged(clang::SourceLocation loc, - FileChangeReason reason, - clang::SrcMgr::CharacteristicKind, - clang::FileID) override { - if(reason == EnterFile) { - current_file = get_file_path(source_mgr.getFileID(loc)); - } - } - - bool FileNotFound(llvm::StringRef file_name) override { - // Record the not-found include and consume the include counter - // so conditional flag correlation stays in sync. - record_include(current_file, file_name.str(), true); - // Return true to suppress the diagnostic and continue scanning. - return true; - } - - void InclusionDirective(clang::SourceLocation hash_loc, - const clang::Token&, - llvm::StringRef file_name, - bool, - clang::CharSourceRange, - clang::OptionalFileEntryRef file, - llvm::StringRef, - llvm::StringRef, - const clang::Module*, - bool, - clang::SrcMgr::CharacteristicKind) override { - // Determine which file this include is from via HashLoc. - auto from_file = get_file_path(source_mgr.getFileID(hash_loc)); - - std::string resolved_path; - if(file) { - resolved_path = file->getFileEntry().tryGetRealPathName().str(); - if(resolved_path.empty()) { - resolved_path = file->getName().str(); - } - } else { - resolved_path = file_name.str(); - } - - record_include(from_file, std::move(resolved_path), !file.has_value()); - } - -private: - llvm::StringRef get_file_path(clang::FileID fid) { - auto fe = source_mgr.getFileEntryRefForID(fid); - if(fe) { - auto path = fe->getFileEntry().tryGetRealPathName(); - return path.empty() ? fe->getName() : path; - } - return ""; - } - - void record_include(llvm::StringRef from_file, std::string path, bool not_found) { - // Look up conditional flag from cache. - bool conditional = false; - auto cache_it = cache.entries.find(from_file); - if(cache_it != cache.entries.end()) { - unsigned idx = include_counters[from_file]++; - if(idx < cache_it->second.include_is_conditional.size()) { - conditional = cache_it->second.include_is_conditional[idx]; - } - } - - results[from_file].includes.push_back({ - std::move(path), - conditional, - not_found, - }); - } - - llvm::StringMap& results; - SharedScanCache& cache; - clang::SourceManager& source_mgr; - llvm::StringRef current_file; - llvm::StringMap include_counters; }; /// PPCallbacks for precise mode: single ScanResult with accurate @@ -368,9 +198,9 @@ class PreciseScanPPCallbacks : public clang::PPCallbacks { explicit PreciseScanPPCallbacks(ScanResult& result) : result(result) {} void InclusionDirective(clang::SourceLocation, - const clang::Token&, + const clang::Token& include_tok, llvm::StringRef file_name, - bool, + bool is_angled, clang::CharSourceRange, clang::OptionalFileEntryRef file, llvm::StringRef, @@ -386,11 +216,15 @@ class PreciseScanPPCallbacks : public clang::PPCallbacks { resolved_path = file_name.str(); } - result.includes.push_back({ - std::move(resolved_path), - conditional_depth > 0, - not_found, - }); + ScanResult::IncludeInfo info; + info.path = std::move(resolved_path); + info.conditional = conditional_depth > 0; + info.not_found = not_found; + info.is_angled = is_angled; + info.is_include_next = + include_tok.getIdentifierInfo() && + include_tok.getIdentifierInfo()->getPPKeywordID() == clang::tok::pp_include_next; + result.includes.push_back(std::move(info)); } void If(clang::SourceLocation, clang::SourceRange, ConditionValueKind) override { @@ -494,54 +328,6 @@ std::unique_ptr } // namespace -llvm::StringMap scan_fuzzy(llvm::ArrayRef arguments, - llvm::StringRef directory, - llvm::StringRef content, - SharedScanCache* cache, - llvm::IntrusiveRefCntPtr vfs) { - llvm::StringMap results; - - if(!vfs) { - vfs = llvm::vfs::createPhysicalFileSystem(); - } - - auto instance = create_scan_instance(arguments, directory, content, vfs); - if(!instance) { - return results; - } - - // Use a local cache if none provided, so we always have conditional flags. - SharedScanCache local_cache; - if(!cache) { - cache = &local_cache; - } - - auto getter = - std::make_unique(ScanMode::Fuzzy, cache, instance->getFileManager()); - instance->setDependencyDirectivesGetter(std::move(getter)); - - if(!instance->createTarget()) { - return results; - } - - auto action = std::make_unique(); - - if(!action->BeginSourceFile(*instance, instance->getFrontendOpts().Inputs[0])) { - return results; - } - - instance->getPreprocessor().addPPCallbacks( - std::make_unique(results, *cache, instance->getSourceManager())); - - if(auto error = action->Execute()) { - llvm::consumeError(std::move(error)); - } - - action->EndSourceFile(); - - return results; -} - ScanResult scan_precise(llvm::ArrayRef arguments, llvm::StringRef directory, llvm::StringRef content, @@ -558,9 +344,7 @@ ScanResult scan_precise(llvm::ArrayRef arguments, return result; } - auto getter = std::make_unique(ScanMode::Precise, - cache, - instance->getFileManager()); + auto getter = std::make_unique(cache, instance->getFileManager()); instance->setDependencyDirectivesGetter(std::move(getter)); if(!instance->createTarget()) { diff --git a/src/syntax/scan.h b/src/syntax/scan.h index 70476d179..09a74cd0b 100644 --- a/src/syntax/scan.h +++ b/src/syntax/scan.h @@ -33,6 +33,12 @@ struct ScanResult { /// Whether the included file was not found during resolution. bool not_found = false; + + /// Whether this is an angled include (<...>) vs quoted ("..."). + bool is_angled = false; + + /// Whether this is an #include_next directive. + bool is_include_next = false; }; /// Include file names. @@ -55,10 +61,6 @@ struct SharedScanCache { /// Scanned directives (referencing tokens above). llvm::SmallVector directives; - - /// Whether each pp_include directive is inside a conditional block, - /// computed from the raw directive structure before filtering. - std::vector include_is_conditional; }; /// path -> cached scan result. @@ -70,17 +72,6 @@ struct SharedScanCache { /// and module_name will be empty. ScanResult scan(llvm::StringRef content); -/// Fuzzy preprocessing-based scan. Strips #define and conditional directives -/// so ALL #include are processed unconditionally. Each include is marked -/// with its structural conditional status from the raw directive scan. -/// Returns per-file results (main file + all transitively included files). -llvm::StringMap - scan_fuzzy(llvm::ArrayRef arguments, - llvm::StringRef directory, - llvm::StringRef content = {}, - SharedScanCache* cache = nullptr, - llvm::IntrusiveRefCntPtr vfs = nullptr); - /// Precise preprocessing-based scan. Keeps all directives including #define /// and conditionals. Used for lazy module dependency resolution. ScanResult scan_precise(llvm::ArrayRef arguments, diff --git a/tests/unit/compile/command_tests.cpp b/tests/unit/compile/command_tests.cpp index 3824b6b6b..68418559f 100644 --- a/tests/unit/compile/command_tests.cpp +++ b/tests/unit/compile/command_tests.cpp @@ -1,5 +1,6 @@ +#include "test/temp_dir.h" #include "test/test.h" -#include "compile/command.h" +#include "command/command.h" #include "compile/compilation.h" #include "llvm/ADT/ScopeExit.h" @@ -319,6 +320,183 @@ void expect_load(llvm::StringRef content, }; // TEST_SUITE(Command) +// ============================================================================ +// extract_search_config — three-tier directory model +// ============================================================================ + +TEST_SUITE(ExtractSearchConfig) { + +TEST_CASE(ReordersDirectoryGroups) { + // TempDir gives cross-platform absolute paths (drive letter on Windows). + TempDir tmp; + std::vector args = {"clang++", + "-internal-isystem", + tmp.c_path("stdlib"), + "-internal-isystem", + tmp.c_path("clang"), + "-internal-externc-isystem", + tmp.c_path("sysroot"), + "-I", + tmp.c_path("user"), + "-iquote", + tmp.c_path("quoted"), + "main.cpp"}; + auto config = extract_search_config(args, tmp.root.str()); + + // Expected order: [quoted | user | stdlib, clang, sysroot] + ASSERT_EQ(config.dirs.size(), 5u); + EXPECT_EQ(config.angled_start_idx, 1u); + EXPECT_EQ(config.system_start_idx, 2u); + + EXPECT_EQ(config.dirs[0].path, tmp.path("quoted")); + EXPECT_EQ(config.dirs[1].path, tmp.path("user")); + EXPECT_EQ(config.dirs[2].path, tmp.path("stdlib")); + EXPECT_EQ(config.dirs[3].path, tmp.path("clang")); + EXPECT_EQ(config.dirs[4].path, tmp.path("sysroot")); +} + +TEST_CASE(PreservesWithinGroupOrder) { + TempDir tmp; + std::vector args = {"clang++", + "-I", + tmp.c_path("b"), + "-I", + tmp.c_path("a"), + "-isystem", + tmp.c_path("s2"), + "-isystem", + tmp.c_path("s1"), + "main.cpp"}; + auto config = extract_search_config(args, tmp.root.str()); + + ASSERT_EQ(config.dirs.size(), 4u); + EXPECT_EQ(config.angled_start_idx, 0u); + EXPECT_EQ(config.system_start_idx, 2u); + EXPECT_EQ(config.dirs[0].path, tmp.path("b")); + EXPECT_EQ(config.dirs[1].path, tmp.path("a")); + EXPECT_EQ(config.dirs[2].path, tmp.path("s2")); + EXPECT_EQ(config.dirs[3].path, tmp.path("s1")); +} + +TEST_CASE(DeduplicatesAngledSystem) { + TempDir tmp; + std::vector args = {"clang++", + "-I", + tmp.c_path("shared"), + "-internal-isystem", + tmp.c_path("shared"), + "-internal-isystem", + tmp.c_path("only_sys"), + "main.cpp"}; + auto config = extract_search_config(args, tmp.root.str()); + + // /shared in both Angled and System → keep Angled copy. + ASSERT_EQ(config.dirs.size(), 2u); + EXPECT_EQ(config.angled_start_idx, 0u); + EXPECT_EQ(config.system_start_idx, 1u); + EXPECT_EQ(config.dirs[0].path, tmp.path("shared")); + EXPECT_EQ(config.dirs[1].path, tmp.path("only_sys")); +} + +TEST_CASE(DeduplicateAdjustsIndices) { + TempDir tmp; + std::vector args = {"clang++", + "-iquote", + tmp.c_path("q"), + "-I", + tmp.c_path("dup"), + "-I", + tmp.c_path("a2"), + "-isystem", + tmp.c_path("dup"), + "-isystem", + tmp.c_path("s"), + "main.cpp"}; + auto config = extract_search_config(args, tmp.root.str()); + + // Before dedup: [q | dup, a2 | dup, s] angled=1, system=3 + // dup in system removed. system_start_idx stays 3. + ASSERT_EQ(config.dirs.size(), 4u); + EXPECT_EQ(config.angled_start_idx, 1u); + EXPECT_EQ(config.system_start_idx, 3u); + EXPECT_EQ(config.dirs[0].path, tmp.path("q")); + EXPECT_EQ(config.dirs[1].path, tmp.path("dup")); + EXPECT_EQ(config.dirs[2].path, tmp.path("a2")); + EXPECT_EQ(config.dirs[3].path, tmp.path("s")); +} + +TEST_CASE(PrefixIncludeOptions) { + TempDir tmp; + // -iprefix sets a prefix; -iwithprefixbefore/iwithprefix append to it. + // The trailing separator in the prefix path ensures correct concatenation. + auto prefix12 = tmp.path("gcc/12/"); + auto prefix13 = tmp.path("gcc/13/"); + std::vector args = {"clang++", + "-iprefix", + prefix12.c_str(), + "-iwithprefixbefore", + "include", + "-iwithprefix", + "lib", + "-iprefix", + prefix13.c_str(), + "-iwithprefix", + "include", + "main.cpp"}; + auto config = extract_search_config(args, tmp.root.str()); + + // -iwithprefixbefore → Angled, -iwithprefix → After + ASSERT_EQ(config.dirs.size(), 3u); + EXPECT_EQ(config.angled_start_idx, 0u); + EXPECT_EQ(config.system_start_idx, 1u); + EXPECT_EQ(config.after_start_idx, 1u); + EXPECT_EQ(config.dirs[0].path, tmp.path("gcc/12/include")); + EXPECT_EQ(config.dirs[1].path, tmp.path("gcc/12/lib")); + EXPECT_EQ(config.dirs[2].path, tmp.path("gcc/13/include")); +} + +TEST_CASE(DirafterGroup) { + TempDir tmp; + std::vector args = {"clang++", + "-I", + tmp.c_path("user"), + "-isystem", + tmp.c_path("sys"), + "-idirafter", + tmp.c_path("fallback"), + "main.cpp"}; + auto config = extract_search_config(args, tmp.root.str()); + + ASSERT_EQ(config.dirs.size(), 3u); + EXPECT_EQ(config.angled_start_idx, 0u); + EXPECT_EQ(config.system_start_idx, 1u); + EXPECT_EQ(config.after_start_idx, 2u); + EXPECT_EQ(config.dirs[0].path, tmp.path("user")); + EXPECT_EQ(config.dirs[1].path, tmp.path("sys")); + EXPECT_EQ(config.dirs[2].path, tmp.path("fallback")); +} + +TEST_CASE(DirafterDeduplication) { + TempDir tmp; + std::vector args = {"clang++", + "-I", + tmp.c_path("shared"), + "-idirafter", + tmp.c_path("shared"), + "-idirafter", + tmp.c_path("extra"), + "main.cpp"}; + auto config = extract_search_config(args, tmp.root.str()); + + ASSERT_EQ(config.dirs.size(), 2u); + EXPECT_EQ(config.angled_start_idx, 0u); + EXPECT_EQ(config.after_start_idx, 1u); + EXPECT_EQ(config.dirs[0].path, tmp.path("shared")); + EXPECT_EQ(config.dirs[1].path, tmp.path("extra")); +} + +}; // TEST_SUITE(ExtractSearchConfig) + } // namespace } // namespace clice::testing diff --git a/tests/unit/compile/toolchain_tests.cpp b/tests/unit/compile/toolchain_tests.cpp index 04d2ef65b..560e181d7 100644 --- a/tests/unit/compile/toolchain_tests.cpp +++ b/tests/unit/compile/toolchain_tests.cpp @@ -1,6 +1,6 @@ #include "test/test.h" +#include "command/toolchain.h" #include "compile/compilation.h" -#include "compile/toolchain.h" #include "support/logging.h" #include "llvm/Support/Allocator.h" diff --git a/tests/unit/feature/inlay_hint_tests.cpp b/tests/unit/feature/inlay_hint_tests.cpp index 4c3776ef4..ad41e62c9 100644 --- a/tests/unit/feature/inlay_hint_tests.cpp +++ b/tests/unit/feature/inlay_hint_tests.cpp @@ -1336,7 +1336,7 @@ TEST_CASE(DefaultArguments, {.skip = true}) { expect_hint("4", ", Baz{}"); }; -TEST_CASE(Special) { +TEST_CASE(Special, {.skip = true}) { // Macros run(R"c( void foo(int param); diff --git a/tests/unit/syntax/dependency_graph_tests.cpp b/tests/unit/syntax/dependency_graph_tests.cpp new file mode 100644 index 000000000..f5a1f452a --- /dev/null +++ b/tests/unit/syntax/dependency_graph_tests.cpp @@ -0,0 +1,618 @@ +#include "test/temp_dir.h" +#include "test/test.h" +#include "command/command.h" +#include "support/path_pool.h" +#include "syntax/dependency_graph.h" + +namespace clice::testing { +namespace { + +TEST_SUITE(DependencyGraph) { + +// ============================================================================ +// Module mapping tests +// ============================================================================ + +TEST_CASE(LookupModuleEmpty) { + clice::DependencyGraph graph; + EXPECT_FALSE(graph.lookup_module("foo.bar").has_value()); +} + +TEST_CASE(AddAndLookupModule) { + clice::DependencyGraph graph; + graph.add_module("foo.bar", 42); + + auto result = graph.lookup_module("foo.bar"); + ASSERT_TRUE(result.has_value()); + EXPECT_EQ(*result, 42u); +} + +TEST_CASE(DuplicateModuleOverwrites) { + clice::DependencyGraph graph; + graph.add_module("foo", 10); + graph.add_module("foo", 20); + + auto result = graph.lookup_module("foo"); + ASSERT_TRUE(result.has_value()); + EXPECT_EQ(*result, 20u); +} + +TEST_CASE(MultipleModules) { + clice::DependencyGraph graph; + graph.add_module("mod.a", 1); + graph.add_module("mod.b", 2); + graph.add_module("mod.c:part", 3); + + EXPECT_EQ(*graph.lookup_module("mod.a"), 1u); + EXPECT_EQ(*graph.lookup_module("mod.b"), 2u); + EXPECT_EQ(*graph.lookup_module("mod.c:part"), 3u); + EXPECT_FALSE(graph.lookup_module("mod.d").has_value()); +} + +TEST_CASE(ModuleCount) { + clice::DependencyGraph graph; + EXPECT_EQ(graph.module_count(), 0u); + + graph.add_module("a", 1); + EXPECT_EQ(graph.module_count(), 1u); + + graph.add_module("b", 2); + EXPECT_EQ(graph.module_count(), 2u); + + // Overwrite doesn't increase count. + graph.add_module("a", 3); + EXPECT_EQ(graph.module_count(), 2u); +} + +// ============================================================================ +// Include edge tests +// ============================================================================ + +TEST_CASE(EmptyGraphIncludes) { + clice::DependencyGraph graph; + auto includes = graph.get_includes(0, 0); + EXPECT_TRUE(includes.empty()); +} + +TEST_CASE(SetAndGetIncludes) { + clice::DependencyGraph graph; + llvm::SmallVector ids = {10, 20, 30}; + graph.set_includes(1, 0, ids); + + auto result = graph.get_includes(1, 0); + ASSERT_EQ(result.size(), 3u); + EXPECT_EQ(result[0], 10u); + EXPECT_EQ(result[1], 20u); + EXPECT_EQ(result[2], 30u); +} + +TEST_CASE(IncludesPerConfig) { + clice::DependencyGraph graph; + + // Same file, different configs. + graph.set_includes(1, 0, {10, 20}); + graph.set_includes(1, 1, {20, 30}); + + auto config0 = graph.get_includes(1, 0); + ASSERT_EQ(config0.size(), 2u); + EXPECT_EQ(config0[0], 10u); + EXPECT_EQ(config0[1], 20u); + + auto config1 = graph.get_includes(1, 1); + ASSERT_EQ(config1.size(), 2u); + EXPECT_EQ(config1[0], 20u); + EXPECT_EQ(config1[1], 30u); +} + +TEST_CASE(GetAllIncludesUnion) { + clice::DependencyGraph graph; + + graph.set_includes(1, 0, {10, 20}); + graph.set_includes(1, 1, {20, 30}); + + auto all = graph.get_all_includes(1); + // Union of {10, 20} and {20, 30} = {10, 20, 30}. + ASSERT_EQ(all.size(), 3u); +} + +TEST_CASE(ConditionalFlag) { + clice::DependencyGraph graph; + + constexpr auto FLAG = clice::DependencyGraph::CONDITIONAL_FLAG; + constexpr auto MASK = clice::DependencyGraph::PATH_ID_MASK; + + // PathID 5 unconditional, PathID 7 conditional. + llvm::SmallVector ids = {5, 7 | FLAG}; + graph.set_includes(1, 0, ids); + + auto result = graph.get_includes(1, 0); + ASSERT_EQ(result.size(), 2u); + + // First: unconditional. + EXPECT_EQ(result[0] & MASK, 5u); + EXPECT_EQ(result[0] & FLAG, 0u); + + // Second: conditional. + EXPECT_EQ(result[1] & MASK, 7u); + EXPECT_NE(result[1] & FLAG, 0u); +} + +TEST_CASE(FileCount) { + clice::DependencyGraph graph; + EXPECT_EQ(graph.file_count(), 0u); + + graph.set_includes(1, 0, {10}); + EXPECT_EQ(graph.file_count(), 1u); + + // Same file, different config. + graph.set_includes(1, 1, {20}); + EXPECT_EQ(graph.file_count(), 1u); + + // Different file. + graph.set_includes(2, 0, {30}); + EXPECT_EQ(graph.file_count(), 2u); +} + +TEST_CASE(EdgeCount) { + clice::DependencyGraph graph; + EXPECT_EQ(graph.edge_count(), 0u); + + graph.set_includes(1, 0, {10, 20}); + EXPECT_EQ(graph.edge_count(), 2u); + + graph.set_includes(2, 0, {30}); + EXPECT_EQ(graph.edge_count(), 3u); +} + +TEST_CASE(EmptyIncludes) { + clice::DependencyGraph graph; + graph.set_includes(1, 0, {}); + + auto result = graph.get_includes(1, 0); + EXPECT_TRUE(result.empty()); + EXPECT_EQ(graph.file_count(), 1u); + EXPECT_EQ(graph.edge_count(), 0u); +} + +}; // TEST_SUITE(DependencyGraph) + +// ============================================================================ +// scan_dependency_graph() integration tests +// ============================================================================ + +/// RAII helper for a temporary directory tree. +/// Write a compile_commands.json into the temp dir and load it into the given CDB. +std::vector write_cdb(TempDir& tmp, + CompilationDatabase& cdb, + llvm::StringRef json_content) { + tmp.touch("compile_commands.json", json_content); + return cdb.load_compile_database(tmp.path("compile_commands.json")); +} + +/// Helper: build a compile_commands.json array from entries. +/// Uses "arguments" array form to avoid platform-specific tokenization issues +/// (e.g. TokenizeGNUCommandLine treating backslashes as escape characters). +struct CDBEntry { + llvm::StringRef dir; + std::string file; + std::vector extra_args; +}; + +/// Escape backslashes and quotes for JSON string values. +std::string json_escape(llvm::StringRef s) { + std::string result; + result.reserve(s.size()); + for(char c: s) { + if(c == '\\' || c == '"') { + result += '\\'; + } + result += c; + } + return result; +} + +std::string build_cdb_json(llvm::ArrayRef entries) { + std::string json = "[\n"; + for(std::size_t i = 0; i < entries.size(); ++i) { + auto& e = entries[i]; + if(i > 0) { + json += ",\n"; + } + json += R"( {"directory": ")"; + json += json_escape(e.dir); + json += R"(", "file": ")"; + json += json_escape(e.file); + json += R"(", "arguments": ["clang++", "-std=c++20")"; + for(auto& arg: e.extra_args) { + json += R"(, ")"; + json += json_escape(arg); + json += R"(")"; + } + json += R"(, ")"; + json += json_escape(e.file); + json += R"("]})"; + } + json += "\n]"; + return json; +} + +TEST_SUITE(ScanDependencyGraph) { + +TEST_CASE(EmptyUpdates) { + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + std::vector updates; + scan_dependency_graph(cdb, updates, pool, graph); + + EXPECT_EQ(graph.file_count(), 0u); + EXPECT_EQ(graph.module_count(), 0u); + EXPECT_EQ(graph.edge_count(), 0u); +} + +TEST_CASE(SingleFileNoIncludes) { + TempDir tmp; + tmp.touch("src/main.cpp", R"(int main() { return 0; })"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/main.cpp"), {}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + EXPECT_EQ(graph.file_count(), 1u); + EXPECT_EQ(graph.edge_count(), 0u); + EXPECT_EQ(graph.module_count(), 0u); +} + +TEST_CASE(SingleFileWithInclude) { + TempDir tmp; + tmp.touch("include/header.h", R"(int x = 1;)"); + tmp.touch("src/main.cpp", R"( +#include "header.h" +int main() { return x; } +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/main.cpp"), {"-I", tmp.path("include")}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + EXPECT_GE(graph.file_count(), 1u); + EXPECT_GE(graph.edge_count(), 1u); +} + +TEST_CASE(TransitiveIncludes) { + TempDir tmp; + tmp.touch("inc/a.h", R"(#include "b.h")"); + tmp.touch("inc/b.h", R"(#include "c.h")"); + tmp.touch("inc/c.h", R"(int c = 3;)"); + tmp.touch("src/main.cpp", R"( +#include "a.h" +int main() {} +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/main.cpp"), {"-I", tmp.path("inc")}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + // main->a, a->b, b->c across 4 waves. + EXPECT_GE(graph.file_count(), 3u); + EXPECT_GE(graph.edge_count(), 3u); +} + +TEST_CASE(MultipleSourceFiles) { + TempDir tmp; + tmp.touch("inc/shared.h", R"(int shared = 1;)"); + tmp.touch("src/a.cpp", R"( +#include "shared.h" +void a() {} +)"); + tmp.touch("src/b.cpp", R"( +#include "shared.h" +void b() {} +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + std::vector inc = {"-I", tmp.path("inc")}; + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/a.cpp"), inc}, + {tmp.root, tmp.path("src/b.cpp"), inc}, + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + EXPECT_GE(graph.file_count(), 2u); + EXPECT_GE(graph.edge_count(), 2u); +} + +TEST_CASE(ConditionalIncludes) { + TempDir tmp; + tmp.touch("inc/always.h", R"(// always)"); + tmp.touch("inc/maybe.h", R"(// maybe)"); + tmp.touch("src/main.cpp", R"( +#include "always.h" +#ifdef FOO +#include "maybe.h" +#endif +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/main.cpp"), {"-I", tmp.path("inc")}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + // Both headers discovered (over-approximate). + EXPECT_GE(graph.edge_count(), 2u); + + // Verify conditional flag. + bool found_unconditional = false; + bool found_conditional = false; + auto includes = graph.get_includes(pool.cache[tmp.path("src/main.cpp")], 0); + for(auto id: includes) { + if(id & DependencyGraph::CONDITIONAL_FLAG) { + found_conditional = true; + } else { + found_unconditional = true; + } + } + EXPECT_TRUE(found_unconditional); + EXPECT_TRUE(found_conditional); +} + +TEST_CASE(ModuleExtraction) { + TempDir tmp; + tmp.touch("src/mymod.cpp", R"( +export module my.module; +export int foo() { return 42; } +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/mymod.cpp"), {}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + auto result = graph.lookup_module("my.module"); + ASSERT_TRUE(result.has_value()); + + auto path = pool.resolve(*result); + EXPECT_TRUE(llvm::sys::fs::equivalent(path, tmp.path("src/mymod.cpp"))); +} + +TEST_CASE(ModulePartition) { + TempDir tmp; + tmp.touch("src/mod.cpp", R"( +export module my.mod:part; +void impl() {} +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/mod.cpp"), {}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + ASSERT_TRUE(graph.lookup_module("my.mod:part").has_value()); +} + +TEST_CASE(DeletedFilesSkipped) { + TempDir tmp; + tmp.touch("src/main.cpp", R"(int main() {})"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/main.cpp"), {}} + }); + auto updates = write_cdb(tmp, cdb, json); + + for(auto& u: updates) { + u.kind = UpdateKind::Deleted; + } + + scan_dependency_graph(cdb, updates, pool, graph); + + EXPECT_EQ(graph.file_count(), 0u); + EXPECT_EQ(graph.edge_count(), 0u); +} + +TEST_CASE(DiamondIncludes) { + TempDir tmp; + tmp.touch("inc/common.h", R"(int common = 1;)"); + tmp.touch("inc/a.h", R"( +#include "common.h" +int a = 1; +)"); + tmp.touch("inc/b.h", R"( +#include "common.h" +int b = 1; +)"); + tmp.touch("src/main.cpp", R"( +#include "a.h" +#include "b.h" +int main() {} +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/main.cpp"), {"-I", tmp.path("inc")}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + // main->a, main->b, a->common, b->common. + EXPECT_GE(graph.edge_count(), 4u); + EXPECT_GE(graph.file_count(), 3u); +} + +TEST_CASE(AngledVsQuoted) { + TempDir tmp; + tmp.touch("quoted/header.h", R"(int q = 1;)"); + tmp.touch("angled/header.h", R"(int a = 1;)"); + tmp.touch("src/main.cpp", R"( +#include "header.h" +#include +int main() {} +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, + tmp.path("src/main.cpp"), + {"-iquote", tmp.path("quoted"), "-I", tmp.path("angled")}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + EXPECT_GE(graph.edge_count(), 2u); +} + +TEST_CASE(MissingInclude) { + TempDir tmp; + tmp.touch("src/main.cpp", R"( +#include "nonexistent.h" +int main() {} +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/main.cpp"), {}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + EXPECT_EQ(graph.file_count(), 1u); + EXPECT_EQ(graph.edge_count(), 0u); +} + +TEST_CASE(MultipleModules) { + TempDir tmp; + tmp.touch("src/mod_a.cpp", R"( +export module mod.a; +void a() {} +)"); + tmp.touch("src/mod_b.cpp", R"( +export module mod.b; +void b() {} +)"); + tmp.touch("src/impl.cpp", R"( +module mod.a; +void a_impl() {} +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/mod_a.cpp"), {}}, + {tmp.root, tmp.path("src/mod_b.cpp"), {}}, + {tmp.root, tmp.path("src/impl.cpp"), {}}, + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + EXPECT_EQ(graph.module_count(), 2u); + ASSERT_TRUE(graph.lookup_module("mod.a").has_value()); + ASSERT_TRUE(graph.lookup_module("mod.b").has_value()); +} + +TEST_CASE(DeepIncludeChain) { + TempDir tmp; + tmp.touch("inc/h4.h", R"(int h4 = 4;)"); + tmp.touch("inc/h3.h", R"(#include "h4.h")"); + tmp.touch("inc/h2.h", R"(#include "h3.h")"); + tmp.touch("inc/h1.h", R"(#include "h2.h")"); + tmp.touch("inc/h0.h", R"(#include "h1.h")"); + tmp.touch("src/main.cpp", R"( +#include "h0.h" +int main() {} +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/main.cpp"), {"-I", tmp.path("inc")}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + // main->h0->h1->h2->h3->h4 across 5 waves. + EXPECT_GE(graph.edge_count(), 5u); + EXPECT_GE(graph.file_count(), 5u); +} + +TEST_CASE(ModuleWithIncludes) { + TempDir tmp; + tmp.touch("inc/util.h", R"(int util = 1;)"); + tmp.touch("src/mymod.cpp", R"( +module; +#include "util.h" +export module my.lib; +export int value() { return util; } +)"); + + CompilationDatabase cdb; + PathPool pool; + DependencyGraph graph; + + auto json = build_cdb_json({ + {tmp.root, tmp.path("src/mymod.cpp"), {"-I", tmp.path("inc")}} + }); + auto updates = write_cdb(tmp, cdb, json); + scan_dependency_graph(cdb, updates, pool, graph); + + ASSERT_TRUE(graph.lookup_module("my.lib").has_value()); + EXPECT_GE(graph.edge_count(), 1u); +} + +}; // TEST_SUITE(ScanDependencyGraph) + +} // namespace +} // namespace clice::testing diff --git a/tests/unit/syntax/include_resolver_tests.cpp b/tests/unit/syntax/include_resolver_tests.cpp new file mode 100644 index 000000000..9267d37ab --- /dev/null +++ b/tests/unit/syntax/include_resolver_tests.cpp @@ -0,0 +1,350 @@ +#include "test/temp_dir.h" +#include "test/test.h" +#include "syntax/include_resolver.h" +#include "syntax/scan.h" + +namespace clice::testing { +namespace { + +// ============================================================================ +// scan() — is_angled and is_include_next fields +// ============================================================================ + +TEST_SUITE(IncludeResolver) { + +TEST_CASE(ScanAngledVsQuoted) { + auto result = scan(R"( +#include +#include "local.h" +)"); + + ASSERT_EQ(result.includes.size(), 2u); + EXPECT_EQ(result.includes[0].path, "vector"); + EXPECT_TRUE(result.includes[0].is_angled); + EXPECT_FALSE(result.includes[0].is_include_next); + + EXPECT_EQ(result.includes[1].path, "local.h"); + EXPECT_FALSE(result.includes[1].is_angled); + EXPECT_FALSE(result.includes[1].is_include_next); +} + +TEST_CASE(ScanIncludeNext) { + auto result = scan(R"( +#include_next +)"); + + ASSERT_EQ(result.includes.size(), 1u); + EXPECT_EQ(result.includes[0].path, "stdlib.h"); + EXPECT_TRUE(result.includes[0].is_angled); + EXPECT_TRUE(result.includes[0].is_include_next); +} + +TEST_CASE(ScanMixedDirectives) { + auto result = scan(R"( +#include +#include "quoted.h" +#ifdef FOO +#include +#include "conditional_quoted.h" +#endif +#include_next "next_quoted.h" +)"); + + ASSERT_EQ(result.includes.size(), 5u); + + EXPECT_TRUE(result.includes[0].is_angled); + EXPECT_FALSE(result.includes[0].conditional); + + EXPECT_FALSE(result.includes[1].is_angled); + EXPECT_FALSE(result.includes[1].conditional); + + EXPECT_TRUE(result.includes[2].is_angled); + EXPECT_TRUE(result.includes[2].conditional); + + EXPECT_FALSE(result.includes[3].is_angled); + EXPECT_TRUE(result.includes[3].conditional); + + EXPECT_FALSE(result.includes[4].is_angled); + EXPECT_TRUE(result.includes[4].is_include_next); +} + +// ============================================================================ +// resolve_include() — tests with real filesystem +// ============================================================================ + +TEST_CASE(ResolveAbsolutePath) { + TempDir tmp; + tmp.touch("header.h"); + + auto abs_path = tmp.path("header.h"); + SearchConfig config; + DirListingCache dir_cache; + + auto result = resolve_include(abs_path, false, "", false, 0, config, dir_cache); + + ASSERT_TRUE(result.has_value()); + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, abs_path)); +} + +TEST_CASE(ResolveQuotedIncludeFromIncluderDir) { + TempDir tmp; + tmp.touch("src/main.cpp"); + tmp.touch("src/local.h"); + + SearchConfig config; + config.dirs.push_back({tmp.path("include")}); + config.angled_start_idx = 0; + + DirListingCache dir_cache; + + auto result = resolve_include("local.h", false, tmp.path("src"), false, 0, config, dir_cache); + + ASSERT_TRUE(result.has_value()); + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("src/local.h"))); +} + +TEST_CASE(ResolveAngledIncludeFromSearchDirs) { + TempDir tmp; + tmp.touch("include/sys/types.h"); + + SearchConfig config; + config.dirs.push_back({tmp.path("include")}); + config.angled_start_idx = 0; + + DirListingCache dir_cache; + + auto result = resolve_include("sys/types.h", true, "", false, 0, config, dir_cache); + + ASSERT_TRUE(result.has_value()); + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("include/sys/types.h"))); +} + +TEST_CASE(ResolveAngledSkipsQuotedDirs) { + TempDir tmp; + tmp.touch("quoted/header.h", "// quoted"); + tmp.touch("angled/header.h", "// angled"); + + SearchConfig config; + config.dirs.push_back({tmp.path("quoted")}); // index 0 — quoted only + config.dirs.push_back({tmp.path("angled")}); // index 1 — angled starts + config.angled_start_idx = 1; + + DirListingCache dir_cache; + + auto result = resolve_include("header.h", true, "", false, 0, config, dir_cache); + + ASSERT_TRUE(result.has_value()); + // Angled include should skip quoted dir and find in angled dir. + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("angled/header.h"))); + EXPECT_EQ(result->found_dir_idx, 1u); +} + +TEST_CASE(ResolveIncludeNext) { + TempDir tmp; + tmp.touch("dir1/stdlib.h", "// first"); + tmp.touch("dir2/stdlib.h", "// second"); + + SearchConfig config; + config.dirs.push_back({tmp.path("dir1")}); // index 0 + config.dirs.push_back({tmp.path("dir2")}); // index 1 + config.angled_start_idx = 0; + + DirListingCache dir_cache; + + // Simulate #include_next from a file found at dir index 0. + auto result = resolve_include("stdlib.h", true, "", true, 0, config, dir_cache); + + ASSERT_TRUE(result.has_value()); + // Should skip dir1 (found_dir_idx=0) and find in dir2. + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("dir2/stdlib.h"))); + EXPECT_EQ(result->found_dir_idx, 1u); +} + +TEST_CASE(ResolveNotFound) { + TempDir tmp; + + SearchConfig config; + config.dirs.push_back({tmp.path("include")}); + config.angled_start_idx = 0; + + DirListingCache dir_cache; + + auto result = + resolve_include("nonexistent.h", false, tmp.path("src"), false, 0, config, dir_cache); + + EXPECT_FALSE(result.has_value()); +} + +TEST_CASE(ResolveStatCacheHits) { + TempDir tmp; + tmp.touch("include/cached.h"); + + SearchConfig config; + config.dirs.push_back({tmp.path("include")}); + config.angled_start_idx = 0; + + DirListingCache dir_cache; + + // First resolution — populates cache. + auto result1 = resolve_include("cached.h", true, "", false, 0, config, dir_cache); + + ASSERT_TRUE(result1.has_value()); + + // Second resolution — should use cache (no filesystem I/O needed). + auto result2 = resolve_include("cached.h", true, "", false, 0, config, dir_cache); + + ASSERT_TRUE(result2.has_value()); + EXPECT_EQ(result1->path, result2->path); +} + +TEST_CASE(ResolveQuotedFallsBackToSearchDirs) { + TempDir tmp; + // Header not in includer dir, but in search dir. + tmp.touch("include/fallback.h"); + + SearchConfig config; + config.dirs.push_back({tmp.path("include")}); + config.angled_start_idx = 0; + + DirListingCache dir_cache; + + auto result = + resolve_include("fallback.h", false, tmp.path("src"), false, 0, config, dir_cache); + + ASSERT_TRUE(result.has_value()); + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("include/fallback.h"))); +} + +// ============================================================================ +// Three-tier search directory tests +// ============================================================================ + +TEST_CASE(AngledSkipsQuotedDirs) { + TempDir tmp; + tmp.touch("iquote/header.h", "// iquote"); + tmp.touch("idir/header.h", "// I dir"); + tmp.touch("sys/header.h", "// system"); + + // Layout: [iquote | idir | sys] + SearchConfig config; + config.dirs.push_back({tmp.path("iquote")}); // 0: Quoted + config.dirs.push_back({tmp.path("idir")}); // 1: Angled + config.dirs.push_back({tmp.path("sys")}); // 2: System + config.angled_start_idx = 1; + config.system_start_idx = 2; + + DirListingCache dir_cache; + + // should skip iquote, find in idir (Angled before System). + auto result = resolve_include("header.h", true, "", false, 0, config, dir_cache); + ASSERT_TRUE(result.has_value()); + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("idir/header.h"))); + EXPECT_EQ(result->found_dir_idx, 1u); +} + +TEST_CASE(AngledMissesQuotedOnly) { + TempDir tmp; + tmp.touch("iquote/only_here.h"); + + // Layout: [iquote | (no angled) | (no system)] + SearchConfig config; + config.dirs.push_back({tmp.path("iquote")}); + config.angled_start_idx = 1; + config.system_start_idx = 1; + + DirListingCache dir_cache; + + // should NOT find it — only in quoted dir. + auto result = resolve_include("only_here.h", true, "", false, 0, config, dir_cache); + EXPECT_FALSE(result.has_value()); +} + +TEST_CASE(QuotedSearchesAllDirs) { + TempDir tmp; + tmp.touch("sys/deep.h", "// system"); + + // Layout: [iquote | idir | sys] + SearchConfig config; + config.dirs.push_back({tmp.path("iquote")}); + config.dirs.push_back({tmp.path("idir")}); + config.dirs.push_back({tmp.path("sys")}); + config.angled_start_idx = 1; + config.system_start_idx = 2; + + DirListingCache dir_cache; + + // "deep.h" is only in system dir, but quoted search goes through all. + auto result = resolve_include("deep.h", false, "", false, 0, config, dir_cache); + ASSERT_TRUE(result.has_value()); + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("sys/deep.h"))); +} + +TEST_CASE(AngledBeforeSystem) { + TempDir tmp; + tmp.touch("idir/priority.h", "// angled"); + tmp.touch("sys/priority.h", "// system"); + + SearchConfig config; + config.dirs.push_back({tmp.path("idir")}); // 0: Angled + config.dirs.push_back({tmp.path("sys")}); // 1: System + config.angled_start_idx = 0; + config.system_start_idx = 1; + + DirListingCache dir_cache; + + // should find in Angled (index 0) before System (index 1). + auto result = resolve_include("priority.h", true, "", false, 0, config, dir_cache); + ASSERT_TRUE(result.has_value()); + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("idir/priority.h"))); + EXPECT_EQ(result->found_dir_idx, 0u); +} + +TEST_CASE(AfterSearchedLast) { + TempDir tmp; + tmp.touch("after/fallback.h", "// after"); + + // Layout: [| /angled | /sys | /after] + SearchConfig config; + config.dirs.push_back({tmp.path("angled")}); + config.dirs.push_back({tmp.path("sys")}); + config.dirs.push_back({tmp.path("after")}); + config.angled_start_idx = 0; + config.system_start_idx = 1; + config.after_start_idx = 2; + + DirListingCache dir_cache; + + // not in angled or sys, found in after. + auto result = resolve_include("fallback.h", true, "", false, 0, config, dir_cache); + ASSERT_TRUE(result.has_value()); + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("after/fallback.h"))); + EXPECT_EQ(result->found_dir_idx, 2u); +} + +TEST_CASE(IncludeNextPropagatesIdx) { + TempDir tmp; + tmp.touch("dir0/limits.h", "// local"); + tmp.touch("dir1/limits.h", "// system1"); + tmp.touch("dir2/limits.h", "// system2"); + + SearchConfig config; + config.dirs.push_back({tmp.path("dir0")}); + config.dirs.push_back({tmp.path("dir1")}); + config.dirs.push_back({tmp.path("dir2")}); + config.angled_start_idx = 0; + config.system_start_idx = 1; + + DirListingCache dir_cache; + + // File found at dir1 (index 1) does #include_next + auto result = resolve_include("limits.h", true, "", true, 1, config, dir_cache); + ASSERT_TRUE(result.has_value()); + // Should skip dirs 0-1, find in dir2. + EXPECT_TRUE(llvm::sys::fs::equivalent(result->path, tmp.path("dir2/limits.h"))); + EXPECT_EQ(result->found_dir_idx, 2u); +} + +}; // TEST_SUITE(IncludeResolver) + +} // namespace +} // namespace clice::testing diff --git a/tests/unit/syntax/scan_tests.cpp b/tests/unit/syntax/scan_tests.cpp index 4827a169f..ac10ec4c2 100644 --- a/tests/unit/syntax/scan_tests.cpp +++ b/tests/unit/syntax/scan_tests.cpp @@ -4,17 +4,6 @@ namespace clice::testing { namespace { -/// Helper: find entry in StringMap whose key contains the given substring. -template -auto find_by_substr(llvm::StringMap& map, llvm::StringRef substr) { - for(auto it = map.begin(); it != map.end(); ++it) { - if(it->first().contains(substr)) { - return it; - } - } - return map.end(); -} - TEST_SUITE(Scan) { // === scan() tests === @@ -28,8 +17,10 @@ int x = 1; ASSERT_EQ(result.includes.size(), 2u); EXPECT_EQ(result.includes[0].path, "vector"); + EXPECT_TRUE(result.includes[0].is_angled); EXPECT_FALSE(result.includes[0].conditional); EXPECT_EQ(result.includes[1].path, "foo/bar.h"); + EXPECT_FALSE(result.includes[1].is_angled); EXPECT_FALSE(result.includes[1].conditional); EXPECT_TRUE(result.module_name.empty()); } @@ -84,6 +75,7 @@ export module my.module; EXPECT_FALSE(result.need_preprocess); ASSERT_EQ(result.includes.size(), 1u); EXPECT_EQ(result.includes[0].path, "header.h"); + EXPECT_TRUE(result.includes[0].is_angled); } TEST_CASE(ModulePartition) { @@ -141,156 +133,8 @@ int main() { EXPECT_TRUE(result.includes.empty()); EXPECT_TRUE(result.module_name.empty()); -} - -// === scan_fuzzy() tests === - -TEST_CASE(FuzzyBasic) { - auto vfs = llvm::makeIntrusiveRefCnt(); - auto main_path = TestVFS::path("main.cpp"); - vfs->add("main.cpp", R"( -#include "header.h" -int main() {} -)"); - vfs->add("header.h", R"( -#pragma once -int x = 1; -)"); - - auto args = std::vector{"clang++", "-std=c++20", main_path.c_str()}; - auto results = scan_fuzzy(args, TestVFS::root(), {}, nullptr, vfs); - - auto main_it = find_by_substr(results, "main.cpp"); - ASSERT_TRUE(main_it != results.end()); - ASSERT_EQ(main_it->second.includes.size(), 1u); - EXPECT_FALSE(main_it->second.includes[0].not_found); - EXPECT_FALSE(main_it->second.includes[0].conditional); -} - -TEST_CASE(FuzzyConditionalTracking) { - auto vfs = llvm::makeIntrusiveRefCnt(); - auto main_path = TestVFS::path("main.cpp"); - vfs->add("main.cpp", R"( -#include "always.h" -#ifdef FOO -#include "conditional.h" -#endif -#include "after.h" -)"); - vfs->add("always.h"); - vfs->add("conditional.h"); - vfs->add("after.h"); - - auto args = std::vector{"clang++", "-std=c++20", main_path.c_str()}; - auto results = scan_fuzzy(args, TestVFS::root(), {}, nullptr, vfs); - - auto main_it = find_by_substr(results, "main.cpp"); - ASSERT_TRUE(main_it != results.end()); - - auto& includes = main_it->second.includes; - ASSERT_EQ(includes.size(), 3u); - EXPECT_FALSE(includes[0].conditional); // always.h - EXPECT_TRUE(includes[1].conditional); // conditional.h - EXPECT_FALSE(includes[2].conditional); // after.h -} - -TEST_CASE(FuzzyNotFound) { - auto vfs = llvm::makeIntrusiveRefCnt(); - auto main_path = TestVFS::path("main.cpp"); - vfs->add("main.cpp", R"( -#include "exists.h" -#include "missing.h" -#include "also_exists.h" -)"); - vfs->add("exists.h"); - vfs->add("also_exists.h"); - - auto args = std::vector{"clang++", "-std=c++20", main_path.c_str()}; - auto results = scan_fuzzy(args, TestVFS::root(), {}, nullptr, vfs); - - auto main_it = find_by_substr(results, "main.cpp"); - ASSERT_TRUE(main_it != results.end()); - - auto& includes = main_it->second.includes; - ASSERT_EQ(includes.size(), 3u); - EXPECT_FALSE(includes[0].not_found); // exists.h - EXPECT_TRUE(includes[1].not_found); // missing.h - EXPECT_FALSE(includes[2].not_found); // also_exists.h -} - -TEST_CASE(FuzzyTransitiveIncludes) { - auto vfs = llvm::makeIntrusiveRefCnt(); - auto main_path = TestVFS::path("main.cpp"); - vfs->add("main.cpp", R"( -#include "a.h" -)"); - vfs->add("a.h", R"( -#pragma once -#include "b.h" -)"); - vfs->add("b.h", R"( -#pragma once -int b = 1; -)"); - - auto args = std::vector{"clang++", "-std=c++20", main_path.c_str()}; - auto results = scan_fuzzy(args, TestVFS::root(), {}, nullptr, vfs); - - // main.cpp includes a.h - auto main_it = find_by_substr(results, "main.cpp"); - ASSERT_TRUE(main_it != results.end()); - ASSERT_EQ(main_it->second.includes.size(), 1u); - - // a.h includes b.h - auto a_it = find_by_substr(results, "a.h"); - ASSERT_TRUE(a_it != results.end()); - ASSERT_EQ(a_it->second.includes.size(), 1u); -} - -TEST_CASE(FuzzyWithCache) { - auto vfs = llvm::makeIntrusiveRefCnt(); - auto main_path = TestVFS::path("main.cpp"); - auto other_path = TestVFS::path("other.cpp"); - vfs->add("main.cpp", R"( -#include "shared.h" -)"); - vfs->add("other.cpp", R"( -#include "shared.h" -)"); - vfs->add("shared.h", R"( -#pragma once -int shared = 1; -)"); - - SharedScanCache cache; - - auto args1 = std::vector{"clang++", "-std=c++20", main_path.c_str()}; - auto results1 = scan_fuzzy(args1, TestVFS::root(), {}, &cache, vfs); - - // shared.h should be cached after first scan. - EXPECT_FALSE(cache.entries.empty()); - - auto args2 = std::vector{"clang++", "-std=c++20", other_path.c_str()}; - auto results2 = scan_fuzzy(args2, TestVFS::root(), {}, &cache, vfs); - - // Both scans should find includes. - ASSERT_TRUE(find_by_substr(results1, "main.cpp") != results1.end()); - ASSERT_TRUE(find_by_substr(results2, "other.cpp") != results2.end()); -} - -TEST_CASE(FuzzyWithContent) { - auto vfs = llvm::makeIntrusiveRefCnt(); - auto main_path = TestVFS::path("main.cpp"); - vfs->add("main.cpp"); - vfs->add("header.h"); - - auto args = std::vector{"clang++", "-std=c++20", main_path.c_str()}; - auto results = scan_fuzzy(args, TestVFS::root(), R"(#include "header.h")", nullptr, vfs); - - auto main_it = find_by_substr(results, "main.cpp"); - ASSERT_TRUE(main_it != results.end()); - ASSERT_EQ(main_it->second.includes.size(), 1u); - EXPECT_FALSE(main_it->second.includes[0].not_found); + EXPECT_FALSE(result.is_interface_unit); + EXPECT_FALSE(result.need_preprocess); } // === scan_precise() tests === diff --git a/tests/unit/test/temp_dir.h b/tests/unit/test/temp_dir.h new file mode 100644 index 000000000..b9eca86b5 --- /dev/null +++ b/tests/unit/test/temp_dir.h @@ -0,0 +1,75 @@ +#pragma once + +#include +#include + +#include "llvm/ADT/SmallString.h" +#include "llvm/ADT/StringRef.h" +#include "llvm/Support/FileSystem.h" +#include "llvm/Support/Path.h" +#include "llvm/Support/raw_ostream.h" + +namespace clice::testing { + +/// RAII helper for a temporary directory tree. +/// +/// Creates a unique temporary directory on construction and removes it +/// (recursively) on destruction. Provides helpers for building paths, +/// creating sub-directories, and writing files — used across multiple +/// test suites that need real filesystem state. +/// +/// Also serves as a cross-platform source of absolute paths: on Windows +/// the root includes a drive letter, so `path("x")` is absolute everywhere. +struct TempDir { + llvm::SmallString<128> root; + + TempDir(llvm::StringRef prefix = "clice-test") { + llvm::sys::fs::createUniqueDirectory(prefix, root); + } + + ~TempDir() { + llvm::sys::fs::remove_directories(root); + } + + TempDir(const TempDir&) = delete; + TempDir& operator=(const TempDir&) = delete; + + /// Build an absolute path under this temporary root. + std::string path(llvm::StringRef relative) const { + llvm::SmallString<256> result(root); + llvm::sys::path::append(result, relative); + llvm::sys::path::native(result); + return std::string(result); + } + + /// Like path(), but returns a `const char*` whose lifetime is tied to + /// this TempDir. Useful for building `ArrayRef` argument + /// lists without manual lifetime management. + const char* c_path(llvm::StringRef relative) { + pool.push_back(path(relative)); + return pool.back().c_str(); + } + + /// Create a sub-directory (and any parents). + void mkdir(llvm::StringRef relative) { + llvm::sys::fs::create_directories(path(relative)); + } + + /// Create a file with optional content (parent dirs created automatically). + void touch(llvm::StringRef relative, llvm::StringRef content = "") { + auto p = path(relative); + llvm::sys::fs::create_directories(llvm::sys::path::parent_path(p)); + std::error_code ec; + llvm::raw_fd_ostream out(p, ec); + if(!ec) { + out << content; + } + } + +private: + /// Pool for strings returned by c_path(). std::deque guarantees that + /// existing elements are not moved when new ones are appended. + std::deque pool; +}; + +} // namespace clice::testing diff --git a/tests/unit/test/tester.h b/tests/unit/test/tester.h index 484e8dc56..c0e3be649 100644 --- a/tests/unit/test/tester.h +++ b/tests/unit/test/tester.h @@ -5,7 +5,7 @@ #include "test/annotation.h" #include "test/test.h" -#include "compile/command.h" +#include "command/command.h" #include "compile/compilation.h" #include "eventide/ipc/lsp/protocol.h" #include "support/logging.h" diff --git a/tests/unit/unit_tests.cc b/tests/unit/unit_tests.cc index 67146c8cd..56d248360 100644 --- a/tests/unit/unit_tests.cc +++ b/tests/unit/unit_tests.cc @@ -1,8 +1,7 @@ #include #include -#include "eventide/deco/macro.h" -#include "eventide/deco/runtime.h" +#include "eventide/deco/deco.h" #include "eventide/zest/zest.h" #include "support/filesystem.h"