Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
30 commits
Select commit Hold shift + click to select a range
09deda4
QVAC-22629 feat: run an opt-in advisory llama.cpp fit check before lo…
simon-iribarren Sep 2, 2026
2d70d39
QVAC-22629 feat: adopt model-fit 0.8.0 and drop the advisory fit flag
simon-iribarren Sep 2, 2026
300ba6b
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 3, 2026
468aef1
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 3, 2026
1704bce
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 3, 2026
7c1d0ee
QVAC-22629 fix: classify completion flash-attn as fit evidence
simon-iribarren Sep 4, 2026
df306ef
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 4, 2026
92986d7
QVAC-22629 mod: align the example with review feedback
simon-iribarren Sep 4, 2026
eb2aa7c
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 4, 2026
c6bccf6
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 4, 2026
adfa94f
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 4, 2026
e0fa806
QVAC-22629 fix: address review β€” opt-out, margin, log level, stderr t…
simon-iribarren Sep 4, 2026
e507376
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 6, 2026
23d57c9
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 7, 2026
bea852b
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 7, 2026
11c187d
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 7, 2026
d782a9e
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 8, 2026
0f047a6
QVAC-22629 fix: classify projection_model_src as unsupported for the …
simon-iribarren Sep 8, 2026
7f7ecc1
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 9, 2026
7c42a61
QVAC-22629 feat[api]: return the advisory fit verdict on getLoadedMod…
simon-iribarren Sep 9, 2026
a654641
QVAC-22629 chore: export the fitProbe contract artifact
simon-iribarren Sep 9, 2026
12c5cd5
QVAC-22629 chore: regenerate the Python client for fitProbe
simon-iribarren Sep 9, 2026
9196c5d
QVAC-22629 mod: keep the fit probe outcome internal
simon-iribarren Sep 9, 2026
319f7a0
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 9, 2026
1905a08
QVAC-22629 chore: regenerate the Python client after dropping fitProbe
simon-iribarren Sep 9, 2026
15c535a
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 9, 2026
1bfc933
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 9, 2026
34e10c9
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 9, 2026
51c6914
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 9, 2026
b4cb5f6
Merge branch 'main' into feat/qvac-22629-advisory-fit-supervisor
simon-iribarren Sep 9, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
70 changes: 53 additions & 17 deletions packages/inference/NOTICE
Original file line number Diff line number Diff line change
Expand Up @@ -63,14 +63,24 @@ Third-Party Model Licenses

abot-world-0-5b-lf-dit-q8_0
https://huggingface.co/acvlab/ABot-World-0-5B-LF
audio8-codec-decoder-f16
https://huggingface.co/Audio8/Audio8-TTS-Preview-0.6b
audio8-codec-encoder-f16
https://huggingface.co/Audio8/Audio8-TTS-Preview-0.6b
audio8-lm-f16
https://huggingface.co/Audio8/Audio8-TTS-Preview-0.6b
bci-embedder
https://github.com/tetherto/qvac/releases/download/bci-test-assets-v0.1.0/bci-embedder.bin
cosyvoice3-campplus-f32
https://huggingface.co/FunAudioLLM/Fun-CosyVoice3-0.5B-2512
cosyvoice3-flow-f32
https://huggingface.co/FunAudioLLM/Fun-CosyVoice3-0.5B-2512
cosyvoice3-hift-f32
https://huggingface.co/FunAudioLLM/Fun-CosyVoice3-0.5B-2512
cosyvoice3-llm-q8_0
https://huggingface.co/FunAudioLLM/Fun-CosyVoice3-0.5B-2512
cosyvoice3-s3tok-f16
https://huggingface.co/FunAudioLLM/Fun-CosyVoice3-0.5B-2512
craft_mlt_25k
https://www.jaided.ai/easyocr/modelhub/
crnn_mobilenet_v3_small
Expand Down Expand Up @@ -179,6 +189,8 @@ Third-Party Model Licenses
https://huggingface.co/unsloth/Qwen3.6-27B-GGUF
Qwen3.6-35B-A3B-GGUF
https://huggingface.co/unsloth/Qwen3.6-35B-A3B-GGUF
Qwen3.8-27B-GGUF
https://huggingface.co/unsloth/Qwen3.8-27B-GGUF
ru-base-ggml-model-f16
https://huggingface.co/CheeLi03/whisper-base-rus-8
ru-tiny-ggml-model-f16
Expand All @@ -199,6 +211,10 @@ Third-Party Model Licenses
https://huggingface.co/Comfy-Org/Wan_2.1_ComfyUI_repackaged
unlimited-ocr-gguf
https://huggingface.co/vimalnakrani/unlimited-ocr-gguf
VisionPsy-Nano-460M-Flash-GGUFs
https://huggingface.co/qvac/VisionPsy-Nano-460M-Flash-GGUFs
VisionPsy-Nano-460M-GGUFs
https://huggingface.co/qvac/VisionPsy-Nano-460M-GGUFs
voice-en
https://huggingface.co/FunAudioLLM/Fun-CosyVoice3-0.5B-2512
voice-zh-2
Expand Down Expand Up @@ -348,6 +364,12 @@ Third-Party Model Licenses
https://huggingface.co/ChristianAzinn/gte-large-gguf
gte-large-gguf
https://huggingface.co/ChristianAzinn/gte-large-gguf
indic-conformer-ctc.f16
https://huggingface.co/ai4bharat/indic-conformer-600m-multilingual
indic-conformer-ctc.q4_0
https://huggingface.co/ai4bharat/indic-conformer-600m-multilingual
indic-conformer-ctc.q8_0
https://huggingface.co/ai4bharat/indic-conformer-600m-multilingual
lavasr-denoiser-f16
https://huggingface.co/spaces/YatharthS
lavasr-enhancer-f16
Expand Down Expand Up @@ -646,7 +668,9 @@ JavaScript Dependencies
@qvac/error@0.1.1
@qvac/logging@0.1.1
https://github.com/tetherto/qvac
@qvac/rag@0.6.4
@qvac/model-fit@0.8.0
https://github.com/tetherto/qvac
@qvac/rag@0.8.0
https://github.com/tetherto/qvac
@qvac/registry-client@0.6.1
https://github.com/tetherto/qvac
Expand All @@ -672,23 +696,25 @@ JavaScript Dependencies
https://github.com/holepunchto/bare-cpu-info
bare-crypto@1.15.3
https://github.com/holepunchto/bare-crypto
bare-dns@2.1.4
bare-dns@2.2.0
https://github.com/holepunchto/bare-dns
bare-env@3.0.1
https://github.com/holepunchto/bare-env
bare-events@2.9.1
bare-events@2.9.2
https://github.com/holepunchto/bare-events
bare-fetch@3.2.0
https://github.com/holepunchto/bare-fetch
bare-form-data@1.2.2
https://github.com/holepunchto/bare-form-data
bare-fs@4.8.0
bare-fs@4.8.1
https://github.com/holepunchto/bare-fs
bare-gpu-info@0.1.1
https://github.com/holepunchto/bare-gpu-info
bare-hrtime@2.1.1
https://github.com/holepunchto/bare-hrtime
bare-http-parser@1.1.5
bare-http-parser@2.1.4
https://github.com/holepunchto/bare-http-parser
bare-http1@4.5.8
bare-http1@4.6.1
https://github.com/holepunchto/bare-http1
bare-https@3.0.0
https://github.com/holepunchto/bare-https
Expand All @@ -714,17 +740,25 @@ JavaScript Dependencies
https://github.com/holepunchto/bare-process
bare-rpc@1.3.8
https://github.com/holepunchto/bare-rpc
bare-runtime@1.31.0
https://github.com/holepunchto/bare-runtime
bare-runtime-darwin-arm64@1.31.0
https://github.com/holepunchto/bare-runtime
bare-semver@1.1.0
https://github.com/holepunchto/bare-semver
bare-signals@5.0.0
https://github.com/holepunchto/bare-signals
bare-stdio@1.0.3
https://github.com/holepunchto/bare-stdio
bare-stream@2.13.3
bare-stream@2.13.4
https://github.com/holepunchto/bare-stream
bare-structured-clone@1.6.0
https://github.com/holepunchto/bare-structured-clone
bare-subprocess@6.1.0
https://github.com/holepunchto/bare-subprocess
bare-tcp@2.5.4
https://github.com/holepunchto/bare-tcp
bare-tls@3.1.8
bare-tls@3.1.9
https://github.com/holepunchto/bare-tls
bare-tty@5.1.2
https://github.com/holepunchto/bare-tty
Expand All @@ -736,7 +770,7 @@ JavaScript Dependencies
https://github.com/holepunchto/bare-zlib
blind-relay@1.6.1
https://github.com/holepunchto/blind-relay
compact-encoding@3.3.0
compact-encoding@3.3.2
https://github.com/holepunchto/compact-encoding
compact-encoding-bitfield@1.1.0
https://github.com/compact-encoding/compact-encoding-bitfield
Expand All @@ -760,7 +794,7 @@ JavaScript Dependencies
https://github.com/holepunchto/hypercore-stats
hypercore-storage@3.2.1
https://github.com/holepunchto/hypercore-storage
hyperdb@6.8.0
hyperdb@6.9.0
https://github.com/holepunchto/hyperdb
hyperdht-address@1.1.1
https://github.com/holepunchto/hyperdht-address
Expand All @@ -780,7 +814,7 @@ JavaScript Dependencies
https://github.com/holepunchto/mirror-drive
noise-handshake@4.2.0
https://github.com/holepunchto/noise-handshake
paparam@1.12.0
paparam@1.13.0
https://github.com/holepunchto/paparam
passive-core-watcher@1.0.1
https://github.com/holepunchto/passive-core-watcher
Expand All @@ -796,6 +830,8 @@ JavaScript Dependencies
https://github.com/holepunchto/refcounter
require-addon@1.2.0
https://github.com/holepunchto/require-addon
require-asset@1.2.2
https://github.com/holepunchto/require-asset
resource-on-exit@1.0.0
https://github.com/holepunchto/bare-teardown
rocksdb-native@3.17.4
Expand All @@ -808,7 +844,7 @@ JavaScript Dependencies
https://github.com/holepunchto/sub-encoder
text-decoder@1.2.7
https://github.com/holepunchto/text-decoder
udx-native@1.21.0
udx-native@1.21.1
https://github.com/holepunchto/udx-native
unslab@1.3.0
https://github.com/holepunchto/unslab
Expand Down Expand Up @@ -840,7 +876,7 @@ JavaScript Dependencies
https://github.com/mafintosh/bogon
codecs@3.1.0
https://github.com/mafintosh/codecs
corestore@7.12.0
corestore@7.12.2
https://github.com/holepunchto/corestore
debounceify@1.1.0
https://github.com/mafintosh/debounceify
Expand All @@ -858,11 +894,11 @@ JavaScript Dependencies
https://github.com/mafintosh/generate-string
hyperbee@2.27.3
https://github.com/holepunchto/hyperbee
hypercore@11.35.1
hypercore@11.35.2
https://github.com/holepunchto/hypercore
hypercore-crypto@3.7.0
https://github.com/mafintosh/hypercore-crypto
hyperdht@6.33.1
hyperdht@6.33.2
https://github.com/holepunchto/hyperdht
hyperswarm@4.17.0
https://github.com/holepunchto/hyperswarm
Expand Down Expand Up @@ -910,9 +946,9 @@ JavaScript Dependencies
https://github.com/holepunchto/sodium-universal
speedometer@1.1.0
https://github.com/mafintosh/speedometer
streamx@2.28.0
streamx@2.28.1
https://github.com/mafintosh/streamx
tar-stream@3.2.0
tar-stream@3.2.1
https://github.com/mafintosh/tar-stream
teex@1.0.1
https://github.com/mafintosh/teex
Expand Down
3 changes: 3 additions & 0 deletions packages/inference/package.json
Original file line number Diff line number Diff line change
Expand Up @@ -163,6 +163,7 @@
"dependencies": {
"@qvac/error": "^0.1.1",
"@qvac/logging": "^0.1.1",
"@qvac/model-fit": "^0.8.0",
"@qvac/rag": "^0.8.0",
"@qvac/registry-client": "^0.6.1",
"bare-abort-controller": "^1.1.2",
Expand All @@ -177,6 +178,7 @@
"bare-os": "^3.9.3",
"bare-path": "^3.1.1",
"bare-rpc": "^1.3.8",
"bare-runtime": "^1.24.2",
Comment thread
simon-iribarren marked this conversation as resolved.
"bare-stream": "^2.13.3",
"bare-url": "^2.4.5",
"bare-zlib": "^1.4.0",
Expand All @@ -187,6 +189,7 @@
"hyperswarm": "^4.17.0",
"semver": "^7.8.5",
"tar-stream": "^3.2.0",
"which-runtime": "^1.2.1",
"zod": "^4.4.3"
},
"peerDependencies": {
Expand Down
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
import EmbedLlamacpp, { IdMapIndex, type GGMLConfig } from '@qvac/embed-llamacpp'
import EmbedLlamacpp, { IdMapIndex } from '@qvac/embed-llamacpp'
import embedAddonLogging from '@qvac/embed-llamacpp/addonLogging'
import type { TurboVecIndexProvider } from '@qvac/rag'
import {
Expand All @@ -19,6 +19,7 @@ import { embed } from '@/plugins/ops/embed'
import { forwardModelExecution } from '@/profiling/model-execution'
import { isMobile } from '@/runtime/state'
import { stripMultiGpuKeys } from '@/utils/multi-gpu-mobile'
import { transformEmbedConfig } from '@/plugins/builtin/llamacpp-embedding/transform'

const turbovecIndexProvider: TurboVecIndexProvider = {
create(options) {
Expand All @@ -29,53 +30,6 @@ const turbovecIndexProvider: TurboVecIndexProvider = {
}
}

function transformEmbedConfig(embedConfig: EmbedConfig): GGMLConfig {
const config: GGMLConfig = {
device: embedConfig.device as 'gpu' | 'cpu',
gpu_layers: `${embedConfig.gpuLayers}` as `${number}`,
batch_size: `${embedConfig.batchSize}` as `${number}`
}

if (embedConfig.flashAttention) {
config.flash_attn = embedConfig.flashAttention
}

if (embedConfig.pooling) {
config.pooling = embedConfig.pooling
}

if (embedConfig.attention) {
config.attention = embedConfig.attention
}

if (typeof embedConfig.embdNormalize === 'number') {
config.embd_normalize = `${embedConfig.embdNormalize}`
}

if (embedConfig.mainGpu !== undefined) {
config['main-gpu'] =
typeof embedConfig.mainGpu === 'number' ? `${embedConfig.mainGpu}` : embedConfig.mainGpu
}

if (embedConfig.splitMode) {
config['split-mode'] = embedConfig.splitMode
}

if (embedConfig.tensorSplit) {
config['tensor-split'] = embedConfig.tensorSplit
}

if (typeof embedConfig.verbosity === 'number') {
config.verbosity = `${embedConfig.verbosity}`
}

if (embedConfig.openclCacheDir) {
config.openclCacheDir = embedConfig.openclCacheDir
}

return config
}

function createEmbeddingsModel(modelId: string, modelPath: string, embedConfig: EmbedConfig) {
const logger = createStreamLogger(modelId, ModelType.llamacppEmbedding)
registerAddonLogger(modelId, ModelType.llamacppEmbedding, logger)
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
import type { GGMLConfig } from '@qvac/embed-llamacpp'
import { type EmbedConfig } from '@/schemas/index'

/**
* Converts an EmbedConfig into the flat string-keyed map the C++ addon expects.
*
* Extracted from the plugin so the advisory fit check can build its request
* from the same transform the real embedding load uses.
*/
export function transformEmbedConfig(embedConfig: EmbedConfig): GGMLConfig {
const config: GGMLConfig = {
device: embedConfig.device as 'gpu' | 'cpu',
gpu_layers: `${embedConfig.gpuLayers}` as `${number}`,
batch_size: `${embedConfig.batchSize}` as `${number}`
}

if (embedConfig.flashAttention) {
config.flash_attn = embedConfig.flashAttention
}

if (embedConfig.pooling) {
config.pooling = embedConfig.pooling
}

if (embedConfig.attention) {
config.attention = embedConfig.attention
}

if (typeof embedConfig.embdNormalize === 'number') {
config.embd_normalize = `${embedConfig.embdNormalize}`
}

if (embedConfig.mainGpu !== undefined) {
config['main-gpu'] =
typeof embedConfig.mainGpu === 'number' ? `${embedConfig.mainGpu}` : embedConfig.mainGpu
}

if (embedConfig.splitMode) {
config['split-mode'] = embedConfig.splitMode
}

if (embedConfig.tensorSplit) {
config['tensor-split'] = embedConfig.tensorSplit
}

if (typeof embedConfig.verbosity === 'number') {
config.verbosity = `${embedConfig.verbosity}`
}

if (embedConfig.openclCacheDir) {
config.openclCacheDir = embedConfig.openclCacheDir
}

return config
}
19 changes: 18 additions & 1 deletion packages/inference/src/plugins/ops/load-model.ts
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,7 @@ import {
ModelFileLocateFailedError
} from '@/errors/index'
import { getPlugin } from '@/plugins/index'
import { runAdvisoryFitCheck } from '@/resources/model-fit/native-probe/advisory-fit'
import { promises as fsPromises } from 'bare-fs'
import path from 'bare-path'
import { getEngineLogger } from '@/logging/index'
Expand Down Expand Up @@ -93,6 +94,21 @@ export async function loadModel(
}
}

// Advisory: every outcome β€” including a projected insufficiency β€” continues
// to the ordinary load below. Runs after config resolution and path
// validation so it sees the same state the real load uses, and before
// `createModel()` so it never competes with the native load for device
// memory. The outcome is stored on the registry entry for internal use;
// it is deliberately not exposed on any public API yet.
const fitProbe = await runAdvisoryFitCheck({
modelId,
modelType: modelType as CanonicalModelType,
modelPath,
modelConfig,
artifacts,
isShardedModel
})

logger.info(`${modelType}: Loading model ${modelId}...`)
startLogBuffering(modelId)

Expand All @@ -118,7 +134,8 @@ export async function loadModel(
path: modelPath,
config: modelConfig,
modelType: modelType as CanonicalModelType,
name: modelName
name: modelName,
fitProbe
})

const loadResult: LoadModelResult =
Expand Down
Loading
Loading