Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion apps/web/src/app/api/cron/sync-model-stats/route.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -323,7 +323,7 @@ describe('GET /api/cron/sync-model-stats', () => {
jest.mocked(getEnhancedOpenRouterModels).mockResolvedValue({ data: [...models, ...models] });

expect((await GET(request())).status).toBe(200);
expect(ids).toHaveLength(70);
expect(ids).toHaveLength(89);
expect(syncOpenRouterModels).toHaveBeenCalledWith(
[monitoredModel, ...models],
[monitoredModel.id],
Expand Down
28 changes: 14 additions & 14 deletions apps/web/src/lib/model-stats/enkrypt-identity.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -511,11 +511,11 @@ describe('expanded reviewed Enkrypt identities', () => {
})
);

it('contains exactly all 70 independently enumerated identities and canonical targets', () => {
expect(ENKRYPT_REVIEWED_CASES).toHaveLength(70);
it('contains exactly all 89 independently enumerated identities and canonical targets', () => {
expect(ENKRYPT_REVIEWED_CASES).toHaveLength(89);
expect(ENKRYPT_MODEL_MAPPINGS).toStrictEqual(expectedMappings);
expect(new Set(expectedMappings.map(({ identity }) => JSON.stringify(identity))).size).toBe(70);
expect(new Set(expectedMappings.map(({ modelId }) => modelId)).size).toBe(70);
expect(new Set(expectedMappings.map(({ identity }) => JSON.stringify(identity))).size).toBe(89);
expect(new Set(expectedMappings.map(({ modelId }) => modelId)).size).toBe(89);
expect(records.every(record => record.risk_score === 0 && record.safety_score === null)).toBe(
true
);
Expand Down Expand Up @@ -616,7 +616,7 @@ describe('expanded reviewed Enkrypt identities', () => {
[false, true],
[true, true],
])(
'matches all 70 without collisions with reversed scores %s and catalog %s',
'matches all 89 without collisions with reversed scores %s and catalog %s',
(scores, models) => {
const result = matchEnkryptScores(
scores ? records.toReversed() : records,
Expand Down Expand Up @@ -674,7 +674,7 @@ describe('expanded reviewed Enkrypt identities', () => {
}
);

it('rejects all 70 canonical targets if they share a storage ID', () => {
it('rejects all 89 canonical targets if they share a storage ID', () => {
const result = matchEnkryptScores(
records,
catalog.map(record => ({ ...record, id: 'same-storage-id' }))
Expand All @@ -687,21 +687,21 @@ describe('expanded reviewed Enkrypt identities', () => {
identity,
modelIds: [modelId],
})),
ambiguousCount: 70,
ambiguousCount: 89,
missingRequiredModelIds: ENKRYPT_REQUIRED_MODEL_IDS,
});
});

it('reports all 70 exact identities without exposing synthetic metrics or expanding the required gate', () => {
it('reports all 89 exact identities without exposing synthetic metrics or expanding the required gate', () => {
const report = buildEnkryptCoverageReport(
parseEnkryptScores(envelope(records)),
catalog,
'fullinput'
);
expect(report.counters).toStrictEqual({
fetchedCount: 70,
acceptedCount: 70,
matchedCount: 70,
fetchedCount: 89,
acceptedCount: 89,
matchedCount: 89,
unmatchedCount: 0,
ambiguousCount: 0,
rejectedCount: 0,
Expand All @@ -720,7 +720,7 @@ describe('expanded reviewed Enkrypt identities', () => {
});

it.each(['scores', 'catalog'] as const)(
'does not require the 67 optional mappings when their %s are absent',
'does not require the 86 optional mappings when their %s are absent',
absent => {
const result = matchEnkryptScores(
absent === 'scores' ? records.slice(0, 3) : records,
Expand All @@ -741,15 +741,15 @@ describe('expanded reviewed Enkrypt identities', () => {
);

it.each(reviewedMappings)(
'still fails the required $modelId gate with all 67 optional identities present',
'still fails the required $modelId gate with all 86 optional identities present',
({ modelId }) => {
const result = matchEnkryptScores(
ENKRYPT_REVIEWED_CASES.filter(record => record.modelId !== modelId).map(
({ score }) => score
),
catalog
);
expect(result.matches).toHaveLength(69);
expect(result.matches).toHaveLength(88);
expect(result.missingRequiredModelIds).toEqual([modelId]);
expect(result.ambiguousCount).toBe(0);
}
Expand Down
80 changes: 80 additions & 0 deletions apps/web/src/lib/model-stats/enkrypt-identity.ts
Original file line number Diff line number Diff line change
Expand Up @@ -316,6 +316,86 @@ export const ENKRYPT_MODEL_MAPPINGS: readonly EnkryptModelMapping[] = [
},
modelId: 'mistralai/mistral-small-24b-instruct-2501',
},
{
identity: { model_name: 'claude-opus-4-8', provider: 'anthropic', source: 'anthropic' },
modelId: 'anthropic/claude-opus-4.8',
},
{
identity: { model_name: 'claude-opus-4-7', provider: 'anthropic', source: 'anthropic' },
modelId: 'anthropic/claude-opus-4.7',
},
{
identity: { model_name: 'claude-sonnet-4-6', provider: 'anthropic', source: 'anthropic' },
modelId: 'anthropic/claude-sonnet-4.6',
},
{
identity: { model_name: 'claude-opus-5', provider: 'openai_compatible', source: 'Anthropic' },
modelId: 'anthropic/claude-opus-5',
},
{
identity: { model_name: 'glm-5.2', provider: 'openai_compatible', source: 'z-ai' },
modelId: 'z-ai/glm-5.2',
},
{
identity: { model_name: 'minimax-m3', provider: 'openai_compatible', source: 'minimax' },
modelId: 'minimax/minimax-m3',
},
{
identity: { model_name: 'mimo-v2.5-pro', provider: 'openai_compatible', source: 'xiaomi' },
modelId: 'xiaomi/mimo-v2.5-pro',
},
{
identity: { model_name: 'gemini-3.6-flash', provider: 'openai_compatible', source: 'google' },
modelId: 'google/gemini-3.6-flash',
},
{
identity: { model_name: 'gemini-3.5-flash', provider: 'openai_compatible', source: 'google' },
modelId: 'google/gemini-3.5-flash',
},
{
identity: {
model_name: 'nemotron-3-ultra-550b-a55b',
provider: 'openai_compatible',
source: 'nvidia',
},
modelId: 'nvidia/nemotron-3-ultra-550b-a55b',
},
{
identity: { model_name: 'kimi-k2.6', provider: 'openai_compatible', source: 'moonshotai' },
modelId: 'moonshotai/kimi-k2.6',
},
{
identity: { model_name: 'kimi-k2.7-code', provider: 'openai_compatible', source: 'moonshotai' },
modelId: 'moonshotai/kimi-k2.7-code',
},
{
identity: { model_name: 'qwen3.8-max', provider: 'openai_compatible', source: 'qwen' },
modelId: 'qwen/qwen3.8-max',
},
{
identity: { model_name: 'qwen3.7-max', provider: 'openai_compatible', source: 'qwen' },
modelId: 'qwen/qwen3.7-max',
},
{
identity: { model_name: 'grok-build-0.1', provider: 'openai_compatible', source: 'xAI' },
modelId: 'x-ai/grok-build-0.1',
},
{
identity: { model_name: 'grok-4.5', provider: 'openai_compatible', source: 'xAI' },
modelId: 'x-ai/grok-4.5',
},
{
identity: { model_name: 'grok-4.6', provider: 'openai_compatible', source: 'xAI' },
modelId: 'x-ai/grok-4.6',
},
{
identity: { model_name: 'laguna-s-2.1', provider: 'openai_compatible', source: 'xAI' },

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

WARNING: laguna-s-2.1 is mapped with source: 'xAI' while its canonical target is poolside/laguna-s-2.1.

source is part of the exact identity key used by matchEnkryptScores (identity.source === score.source), and Laguna models are published by Poolside — the sibling entry laguna-xs-2.1 below correctly uses source: 'poolside'. If Enkrypt reports poolside for this record, the mapping never matches and the model's stats are dropped as unreviewed_identity. The fixture at apps/web/src/tests/fixtures/enkrypt-scores.ts:154 mirrors the same value, so the test suite cannot catch it.

Suggested change
identity: { model_name: 'laguna-s-2.1', provider: 'openai_compatible', source: 'xAI' },
identity: { model_name: 'laguna-s-2.1', provider: 'openai_compatible', source: 'poolside' },

Reply with @kilocode-bot fix it to have Kilo Code address this issue.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The feedback is not accurate. Fresh live API call (seconds ago) confirms Enkrypt reports:
laguna-s-2.1 | provider: openai_compatible | source: xAI
laguna-xs-2.1 | provider: openai_compatible | source: poolside
Our mapping's source: 'xAI' for laguna-s-2.1 is exactly what the real leaderboard returns — captured from the live response when I added the mappings, not assumed. Enkrypt's source field reflects where they served the model for evaluation, not the publisher: they ran Laguna S 2.1 through xAI's endpoint and Laguna XS 2.1 through Poolside's, hence the sibling entries differ.
If we "corrected" it to 'poolside' per the feedback, the identity (laguna-s-2.1, openai_compatible, poolside) would never match any real Enkrypt record — creating exactly the silent unreviewed_identity drop the reviewer's warning describes. The fixture mirroring the mapping is by design (it encodes reviewed identities), so the test suite enforcing mapping↔fixture parity is working as intended; live API validation like this is the actual safety check.
The canonical target poolside/laguna-s-2.1 is right — only the identity triple must match Enkrypt's record, and it does. No change made.

modelId: 'poolside/laguna-s-2.1',
},
{
identity: { model_name: 'laguna-xs-2.1', provider: 'openai_compatible', source: 'poolside' },
modelId: 'poolside/laguna-xs-2.1',
},
];

export const ENKRYPT_REQUIRED_MODEL_IDS: readonly string[] = [
Expand Down
24 changes: 24 additions & 0 deletions apps/web/src/tests/fixtures/enkrypt-scores.ts
Original file line number Diff line number Diff line change
Expand Up @@ -129,6 +129,30 @@ const reviewedIdentities: [string, string, string, string][] = [
'Mistral',
'mistralai/mistral-small-24b-instruct-2501',
],
['claude-opus-4-8', 'anthropic', 'anthropic', 'anthropic/claude-opus-4.8'],
['claude-opus-4-7', 'anthropic', 'anthropic', 'anthropic/claude-opus-4.7'],
['claude-sonnet-4-6', 'anthropic', 'anthropic', 'anthropic/claude-sonnet-4.6'],
['claude-opus-5', 'openai_compatible', 'Anthropic', 'anthropic/claude-opus-5'],
['glm-5.2', 'openai_compatible', 'z-ai', 'z-ai/glm-5.2'],
['minimax-m3', 'openai_compatible', 'minimax', 'minimax/minimax-m3'],
['mimo-v2.5-pro', 'openai_compatible', 'xiaomi', 'xiaomi/mimo-v2.5-pro'],
['gemini-3.6-flash', 'openai_compatible', 'google', 'google/gemini-3.6-flash'],
['gemini-3.5-flash', 'openai_compatible', 'google', 'google/gemini-3.5-flash'],
[
'nemotron-3-ultra-550b-a55b',
'openai_compatible',
'nvidia',
'nvidia/nemotron-3-ultra-550b-a55b',
],
['kimi-k2.6', 'openai_compatible', 'moonshotai', 'moonshotai/kimi-k2.6'],
['kimi-k2.7-code', 'openai_compatible', 'moonshotai', 'moonshotai/kimi-k2.7-code'],
['qwen3.8-max', 'openai_compatible', 'qwen', 'qwen/qwen3.8-max'],
['qwen3.7-max', 'openai_compatible', 'qwen', 'qwen/qwen3.7-max'],
['grok-build-0.1', 'openai_compatible', 'xAI', 'x-ai/grok-build-0.1'],
['grok-4.5', 'openai_compatible', 'xAI', 'x-ai/grok-4.5'],
['grok-4.6', 'openai_compatible', 'xAI', 'x-ai/grok-4.6'],
['laguna-s-2.1', 'openai_compatible', 'xAI', 'poolside/laguna-s-2.1'],
['laguna-xs-2.1', 'openai_compatible', 'poolside', 'poolside/laguna-xs-2.1'],
];

export const ENKRYPT_REVIEWED_CASES = reviewedIdentities.map(
Expand Down