Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 7 additions & 3 deletions bun.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion package.json
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,7 @@
},
"dependencies": {
"@anthropic-ai/sdk": "^0.78.0",
"@easier-idx/core": "^0.2.1",
"@easier-idx/core": "^0.3.1",
"@easier-idx/embedding": "^0.2.2",
"@modelcontextprotocol/sdk": "^1.27.1",
"ignore": "^7.0.5",
Expand Down
93 changes: 61 additions & 32 deletions src/dedup/global-store-pg.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
*/

import { getPg } from "../db/pg";
import type { PgTx } from "../db/pg";
import { applyGlobalPgMigrations } from "../db/migrate";
import type { CodeindexConfig } from "../search/types";
import type {
Expand Down Expand Up @@ -64,12 +65,13 @@ class PgGlobalStore implements GlobalDedupStore {
const result = new Map<string, BlobRecord>();
if (hashes.length === 0) return result;
const pg = await getPg();
const hashPlaceholders = hashes.map((_, i) => `$${i + 4}`).join(",");
const rows = (await pg.unsafe(
`SELECT content_hash, skeleton, skeleton_entries, embedding::text AS embedding
FROM file_blobs
WHERE provider = $1 AND model = $2 AND dimensions = $3
AND content_hash = ANY($4::text[])`,
[provider, model, dimensions, hashes],
AND content_hash IN (${hashPlaceholders})`,
[provider, model, dimensions, ...hashes],
)) as Array<{
content_hash: string;
skeleton: string | null;
Expand Down Expand Up @@ -250,18 +252,20 @@ class PgGlobalStore implements GlobalDedupStore {
}>;
return rows.map((r) => r.content_hash);
}
const placeholders = excludePackageIds.map((_, i) => `$${i + 1}`).join(",");
const rows = (await pg.unsafe(
`SELECT DISTINCT content_hash FROM package_files
WHERE NOT (package_id = ANY($1::int[]))`,
[excludePackageIds],
WHERE package_id NOT IN (${placeholders})`,
[...excludePackageIds],
)) as Array<{ content_hash: string }>;
return rows.map((r) => r.content_hash);
}

async deletePackages(packageIds: number[]): Promise<void> {
if (packageIds.length === 0) return;
const pg = await getPg();
await pg.unsafe(`DELETE FROM packages WHERE id = ANY($1::int[])`, [packageIds]);
const placeholders = packageIds.map((_, i) => `$${i + 1}`).join(",");
await pg.unsafe(`DELETE FROM packages WHERE id IN (${placeholders})`, [...packageIds]);
}

async countBlobsExcept(liveHashes: Set<string>): Promise<number> {
Expand All @@ -283,19 +287,24 @@ class PgGlobalStore implements GlobalDedupStore {
)) as Array<{ n: number }>;
return rows[0]?.n ?? 0;
}
const rows = (await pg.unsafe(
`SELECT COUNT(*)::int AS n FROM file_blobs fb
WHERE NOT (fb.content_hash = ANY($1::text[]))
AND NOT EXISTS (
SELECT 1 FROM repo_files rf
WHERE rf.content_hash = fb.content_hash
AND rf.provider = fb.provider
AND rf.model = fb.model
AND rf.dimensions = fb.dimensions
)`,
[Array.from(liveHashes)],
)) as Array<{ n: number }>;
return rows[0]?.n ?? 0;
// Spreading every live hash into its own bind would risk PG's 65535
// parameter cap on large dedup sets; load them into a temp table in
// chunks instead so the filter scales independently of |liveHashes|.
return pg.begin(async (tx) => {
await loadLiveHashesIntoTempTable(tx, liveHashes);
const rows = (await tx.unsafe(
`SELECT COUNT(*)::int AS n FROM file_blobs fb
WHERE NOT EXISTS (SELECT 1 FROM _live_hashes WHERE h = fb.content_hash)
AND NOT EXISTS (
SELECT 1 FROM repo_files rf
WHERE rf.content_hash = fb.content_hash
AND rf.provider = fb.provider
AND rf.model = fb.model
AND rf.dimensions = fb.dimensions
)`,
)) as Array<{ n: number }>;
return rows[0]?.n ?? 0;
});
}

async deleteBlobsExcept(liveHashes: Set<string>): Promise<number> {
Expand All @@ -314,20 +323,22 @@ class PgGlobalStore implements GlobalDedupStore {
)) as Array<{ content_hash: string }>;
return rows.length;
}
const rows = (await pg.unsafe(
`DELETE FROM file_blobs fb
WHERE NOT (fb.content_hash = ANY($1::text[]))
AND NOT EXISTS (
SELECT 1 FROM repo_files rf
WHERE rf.content_hash = fb.content_hash
AND rf.provider = fb.provider
AND rf.model = fb.model
AND rf.dimensions = fb.dimensions
)
RETURNING content_hash`,
[Array.from(liveHashes)],
)) as Array<{ content_hash: string }>;
return rows.length;
return pg.begin(async (tx) => {
await loadLiveHashesIntoTempTable(tx, liveHashes);
const rows = (await tx.unsafe(
`DELETE FROM file_blobs fb
WHERE NOT EXISTS (SELECT 1 FROM _live_hashes WHERE h = fb.content_hash)
AND NOT EXISTS (
SELECT 1 FROM repo_files rf
WHERE rf.content_hash = fb.content_hash
AND rf.provider = fb.provider
AND rf.model = fb.model
AND rf.dimensions = fb.dimensions
)
RETURNING content_hash`,
)) as Array<{ content_hash: string }>;
return rows.length;
});
}

async sweepOrphanedBlobs(opts: { dryRun: boolean }): Promise<number | null> {
Expand Down Expand Up @@ -369,6 +380,24 @@ class PgGlobalStore implements GlobalDedupStore {
}
}

// ---------------------------------------------------------------------------
// Temp-table loader for large hash filters
// ---------------------------------------------------------------------------

// 5000 rows per insert leaves comfortable headroom under PG's 65535
// bind-parameter ceiling without causing many round-trips for typical sets.
const HASH_INSERT_CHUNK = 5000;

async function loadLiveHashesIntoTempTable(tx: PgTx, liveHashes: Set<string>): Promise<void> {
await tx.unsafe(`CREATE TEMP TABLE _live_hashes (h text PRIMARY KEY) ON COMMIT DROP`);
const liveArr = Array.from(liveHashes);
for (let i = 0; i < liveArr.length; i += HASH_INSERT_CHUNK) {
const slice = liveArr.slice(i, i + HASH_INSERT_CHUNK);
const placeholders = slice.map((_, j) => `($${j + 1})`).join(",");
await tx.unsafe(`INSERT INTO _live_hashes (h) VALUES ${placeholders}`, slice);
}
}

// ---------------------------------------------------------------------------
// pgvector literal helpers — pgvector serialises as e.g. "[0.1,0.2,0.3]"
// ---------------------------------------------------------------------------
Expand Down
82 changes: 62 additions & 20 deletions src/mcp/tools/files.ts
Original file line number Diff line number Diff line change
Expand Up @@ -146,9 +146,18 @@ export function registerFileTools(ctx: McpToolContext): void {
const scopedRepoIds = session?.repoIds ?? null;

if (config.store === "pg") {
const repoFilter = scopedRepoIds ? `AND nf.repo_id = ANY($4::int[])` : "";
// Bun 1.3.13's bun:sql cannot encode JS arrays for ANY($N) on PG18
// (Bun PR #29552); expand into numbered IN placeholders instead.
// An empty scoped list is a valid "no access" signal — preserve the
// original ANY-on-empty semantics by short-circuiting to no rows.
if (scopedRepoIds && scopedRepoIds.length === 0) return mcpSuccess([]);
const params: unknown[] = [repoRoot, filePath, depth];
if (scopedRepoIds) params.push(scopedRepoIds);
let repoFilter = "";
if (scopedRepoIds) {
const placeholders = scopedRepoIds.map((_, i) => `$${params.length + i + 1}`).join(",");
repoFilter = `AND nf.repo_id IN (${placeholders})`;
params.push(...scopedRepoIds);
}
const query =
dir === "dependencies"
? `WITH RECURSIVE chain AS (
Expand Down Expand Up @@ -265,18 +274,28 @@ export function registerFileTools(ctx: McpToolContext): void {
const scopedRepoIds = session?.repoIds ?? null;

if (config.store === "pg") {
const query = scopedRepoIds
? `SELECT sr.name AS source_repo, tr.name AS target_repo,
// Bun 1.3.13's bun:sql cannot encode JS arrays for ANY($N) on PG18
// (Bun PR #29552); expand into numbered IN placeholders instead.
// An empty scoped list is a valid "no access" signal — preserve the
// original ANY-on-empty semantics by short-circuiting to no rows.
if (scopedRepoIds && scopedRepoIds.length === 0) return mcpSuccess([]);
let query: string;
let params: unknown[];
if (scopedRepoIds) {
const placeholders = scopedRepoIds.map((_, i) => `$${i + 1}`).join(",");
query = `SELECT sr.name AS source_repo, tr.name AS target_repo,
sf.file_path AS source_file, tf.file_path AS target_file,
e.imported_module
FROM cross_repo_edges e
JOIN repos sr ON sr.id = e.source_repo_id
JOIN repos tr ON tr.id = e.target_repo_id
JOIN files sf ON sf.id = e.source_file_id
JOIN files tf ON tf.id = e.target_file_id
WHERE (e.source_repo_id = ANY($1::int[]) OR e.target_repo_id = ANY($1::int[]))
ORDER BY sr.name, tr.name`
: `SELECT sr.name AS source_repo, tr.name AS target_repo,
WHERE (e.source_repo_id IN (${placeholders}) OR e.target_repo_id IN (${placeholders}))
ORDER BY sr.name, tr.name`;
params = [...scopedRepoIds];
Comment on lines +284 to +296

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 $N placeholder reuse is valid but inconsistent with sibling code

The getCrossRepoEdges query uses the same placeholders string ($1,$2,…$N) for both source_repo_id IN (…) and target_repo_id IN (…), passing a single copy of scopedRepoIds in params. PostgreSQL's wire protocol supports $N reuse within a query, so this is functionally correct. However, withCrossRepoEdges in src/search/query.ts handles the same pattern with separate placeholders/placeholders2 index ranges and two copies of the array — a style divergence that may confuse future readers. Consider aligning to one convention.

Note: If this suggestion doesn't match your team's coding style, reply to this and let me know. I'll remember it for next time!

} else {
query = `SELECT sr.name AS source_repo, tr.name AS target_repo,
sf.file_path AS source_file, tf.file_path AS target_file,
e.imported_module
FROM cross_repo_edges e
Expand All @@ -285,7 +304,8 @@ export function registerFileTools(ctx: McpToolContext): void {
JOIN files sf ON sf.id = e.source_file_id
JOIN files tf ON tf.id = e.target_file_id
ORDER BY sr.name, tr.name`;
const params = scopedRepoIds ? [scopedRepoIds] : [];
params = [];
}
const rows = await withMcpScope(session, async (tx) => tx.unsafe(query, params));
return mcpSuccess(rows);
} else {
Expand Down Expand Up @@ -352,23 +372,34 @@ export function registerFileTools(ctx: McpToolContext): void {
const scopedRepoIds = session?.repoIds ?? null;

if (config.store === "pg") {
const query = scopedRepoIds
? `SELECT f.file_path, f.skeleton_entries, r.name AS repo_name
// Bun 1.3.13's bun:sql cannot encode JS arrays for ANY($N) on PG18
// (Bun PR #29552); expand into numbered IN placeholders instead.
// An empty scoped list is a valid "no access" signal — preserve the
// original ANY-on-empty semantics by short-circuiting to no rows.
if (scopedRepoIds && scopedRepoIds.length === 0) return mcpSuccess([]);
let query: string;
let params: unknown[];
if (scopedRepoIds) {
const placeholders = scopedRepoIds.map((_, i) => `$${i + 2}`).join(",");
query = `SELECT f.file_path, f.skeleton_entries, r.name AS repo_name
FROM files f
JOIN repos r ON r.id = f.repo_id
WHERE f.skeleton LIKE $1 ESCAPE '\\'
AND (f.skeleton LIKE '%implements%' OR f.skeleton LIKE '%extends%'
OR f.skeleton LIKE '%: %' OR f.skeleton LIKE '%conform%')
AND r.id = ANY($2::int[])
LIMIT 100`
: `SELECT f.file_path, f.skeleton_entries, r.name AS repo_name
AND r.id IN (${placeholders})
LIMIT 100`;
params = [pattern, ...scopedRepoIds];
} else {
query = `SELECT f.file_path, f.skeleton_entries, r.name AS repo_name
FROM files f
JOIN repos r ON r.id = f.repo_id
WHERE f.skeleton LIKE $1 ESCAPE '\\'
AND (f.skeleton LIKE '%implements%' OR f.skeleton LIKE '%extends%'
OR f.skeleton LIKE '%: %' OR f.skeleton LIKE '%conform%')
LIMIT 100`;
const params = scopedRepoIds ? [pattern, scopedRepoIds] : [pattern];
params = [pattern];
}
const rows = await withMcpScope(session, async (tx) => tx.unsafe(query, params));
return mcpSuccess(rows);
} else {
Expand Down Expand Up @@ -420,23 +451,34 @@ export function registerFileTools(ctx: McpToolContext): void {
const scopedRepoIds = session?.repoIds ?? null;

if (config.store === "pg") {
const query = scopedRepoIds
? `SELECT DISTINCT sf.file_path, r.name AS repo_name, fi.imported_module
// Bun 1.3.13's bun:sql cannot encode JS arrays for ANY($N) on PG18
// (Bun PR #29552); expand into numbered IN placeholders instead.
// An empty scoped list is a valid "no access" signal — preserve the
// original ANY-on-empty semantics by short-circuiting to no rows.
if (scopedRepoIds && scopedRepoIds.length === 0) return mcpSuccess([]);
let query: string;
let params: unknown[];
if (scopedRepoIds) {
const placeholders = scopedRepoIds.map((_, i) => `$${i + 2}`).join(",");
query = `SELECT DISTINCT sf.file_path, r.name AS repo_name, fi.imported_module
FROM file_imports fi
JOIN files sf ON sf.id = fi.source_file_id
JOIN repos r ON r.id = sf.repo_id
WHERE fi.imported_module LIKE $1 ESCAPE '\\'
AND r.id = ANY($2::int[])
AND r.id IN (${placeholders})
ORDER BY r.name, sf.file_path
LIMIT 100`
: `SELECT DISTINCT sf.file_path, r.name AS repo_name, fi.imported_module
LIMIT 100`;
params = [pattern, ...scopedRepoIds];
} else {
query = `SELECT DISTINCT sf.file_path, r.name AS repo_name, fi.imported_module
FROM file_imports fi
JOIN files sf ON sf.id = fi.source_file_id
JOIN repos r ON r.id = sf.repo_id
WHERE fi.imported_module LIKE $1 ESCAPE '\\'
ORDER BY r.name, sf.file_path
LIMIT 100`;
const params = scopedRepoIds ? [pattern, scopedRepoIds] : [pattern];
params = [pattern];
}
const rows = await withMcpScope(session, async (tx) => tx.unsafe(query, params));
return mcpSuccess(rows);
} else {
Expand Down
27 changes: 5 additions & 22 deletions src/repo.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
import path from "path";
import { existsSync } from "fs";
import { loadConfig } from "./config";
import { getPg, pgUnsafe } from "./db/pg";
import { getPg } from "./db/pg";
import { getSqlite } from "./db/sqlite";
import { ensurePgSchema, ensureSqliteSchema } from "./db/schema";
import { getRepoOrigin, getRepoName } from "./index/commits";
Expand Down Expand Up @@ -36,7 +36,7 @@ interface ListRow {
// ---------------------------------------------------------------------------

import type { StoreOps } from "@easier-idx/core";
import { pgToSqlite } from "@easier-idx/core/db/store";
import { createPgStoreOps, createSqliteStoreOps } from "@easier-idx/core/db/store";

export type { StoreOps };

Expand All @@ -49,29 +49,12 @@ export async function getStoreOps(repoRoot: string): Promise<{ store: string; op
const config = await loadConfig(repoRoot);

if (config.store === "pg") {
return {
store: "pg",
ops: {
query: async <T>(sql: string, params?: unknown[]) => (await pgUnsafe(sql, params)) as T[],
run: async (sql: string, params?: unknown[]) => {
await pgUnsafe(sql, params);
},
},
};
const pg = await getPg();
return { store: "pg", ops: createPgStoreOps(pg) };
}

const db = await getSqlite(repoRoot);
type SqlBindings = (string | number | bigint | boolean | null | Uint8Array)[];
return {
store: "sqlite",
ops: {
query: async <T>(sql: string, params?: unknown[]) =>
db.prepare(pgToSqlite(sql)).all(...((params ?? []) as SqlBindings)) as T[],
run: async (sql: string, params?: unknown[]) => {
db.prepare(pgToSqlite(sql)).run(...((params ?? []) as SqlBindings));
},
},
};
return { store: "sqlite", ops: createSqliteStoreOps(db) };
}

// ---------------------------------------------------------------------------
Expand Down
Loading
Loading