From 4af422ddb472b3e8e90005f1e28ef45f98be7b82 Mon Sep 17 00:00:00 2001 From: bong-water-water-bong Date: Sun, 9 Aug 2026 15:02:22 -0300 Subject: [PATCH 1/2] fix: mlx backend fails to compile; publish containers to correct ghcr org mlx_server.cpp: replace undefined SPEC with mlx::spec(), keep the anonymous namespace and mlx namespace inside lemon::backends (the early } // namespace backends put them at lemon:: scope), and include mlx.h for the descriptor. The mlx backend is in LEMON_BACKENDS unconditionally, so every Windows build has been broken since the backend was added - first surfaced by the Validate New llama.cpp Release run. build-container.yml: push to ghcr.io/${{ github.repository }} instead of the stale ghcr.io/lemonade-sdk org, which denies the repo's GITHUB_TOKEN (permission_denied: The requested installation does not exist). --- .github/workflows/build-container.yml | 2 +- src/cpp/server/backends/mlx/mlx_server.cpp | 7 +++---- 2 files changed, 4 insertions(+), 5 deletions(-) diff --git a/.github/workflows/build-container.yml b/.github/workflows/build-container.yml index 8aee9e6e6e..2f1fa6c4f1 100644 --- a/.github/workflows/build-container.yml +++ b/.github/workflows/build-container.yml @@ -43,7 +43,7 @@ jobs: id: meta uses: docker/metadata-action@v5 with: - images: ghcr.io/lemonade-sdk/lemonade/build-environment + images: ghcr.io/${{ github.repository }}/build-environment tags: | type=raw,value=ubuntu${{ matrix.ubuntu_version }} type=raw,value=latest,enable=${{ matrix.ubuntu_version == '26.04' }} diff --git a/src/cpp/server/backends/mlx/mlx_server.cpp b/src/cpp/server/backends/mlx/mlx_server.cpp index 1456660f8f..0f65430b55 100644 --- a/src/cpp/server/backends/mlx/mlx_server.cpp +++ b/src/cpp/server/backends/mlx/mlx_server.cpp @@ -1,4 +1,5 @@ #include "lemon/backends/mlx/mlx_server.h" +#include "lemon/backends/mlx/mlx.h" #include "lemon/backends/backend_ops.h" #include "lemon/backends/backend_utils.h" #include "lemon/backend_manager.h" @@ -112,7 +113,7 @@ void MlxServer::load(const std::string& model_name, device_type_ = (mlx_backend == "cpu") ? DEVICE_CPU : DEVICE_GPU; // Install mlx-engine binary if needed. - backend_manager_->install_backend(SPEC.recipe, mlx_backend); + backend_manager_->install_backend(mlx::spec()->recipe, mlx_backend); // MLX identifies models by HuggingFace repo-id or a local directory path. // The ModelManager resolves local paths when available; fall back to the @@ -130,7 +131,7 @@ void MlxServer::load(const std::string& model_name, port_ = choose_port(); - std::string executable = BackendUtils::get_backend_binary_path(SPEC, mlx_backend); + std::string executable = BackendUtils::get_backend_binary_path(*mlx::spec(), mlx_backend); std::vector args; // Positional model argument — pre-load mode. @@ -222,8 +223,6 @@ json MlxServer::responses(const json& request) { ); } -} // namespace backends - namespace { class MlxOps : public BackendOps { public: From 8f355c0a711f772c160e7604b6b4dd2349eeb3a4 Mon Sep 17 00:00:00 2001 From: bong-water-water-bong Date: Sun, 9 Aug 2026 18:08:21 -0300 Subject: [PATCH 2/2] chore: regenerate backend boilerplate (mlx-engine defaults) --- README.md | 18 +++++++++++++++++- docs/assets/models.js | 4 +++- docs/dev/backends-reference.md | 19 +++++++++++++++++++ docs/guide/cli.md | 8 ++++++++ docs/guide/configuration/README.md | 3 +++ docs/guide/configuration/custom-models.md | 2 +- src/cpp/resources/defaults.json | 3 +++ 7 files changed, 54 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 69839d1a55..610066e246 100644 --- a/README.md +++ b/README.md @@ -136,7 +136,7 @@ Lemonade supports multiple inference engines for LLM, speech, TTS, and image gen - Text generation + Text generation llamacpp system x86_64/ARM64 CPU, GPU @@ -185,6 +185,22 @@ Lemonade supports multiple inference engines for LLM, speech, TTS, and image gen Strix Halo iGPU (gfx1151) Linux + + mlx-engine (experimental) + metal + Apple Silicon Metal GPU + macOS + + + rocm + AMD ROCm GPUs (RDNA3/RDNA4) + Linux + + + cpu + CPU fallback + Linux, macOS + Speech-to-text whispercpp diff --git a/docs/assets/models.js b/docs/assets/models.js index cffd2a9842..332e3d266b 100644 --- a/docs/assets/models.js +++ b/docs/assets/models.js @@ -8,6 +8,7 @@ const RECIPE_PRIORITY = [ 'flm', 'kokoro', 'llamacpp', + 'mlx-engine', 'moonshine', 'onnxruntime', 'openmoss', @@ -30,7 +31,8 @@ const RECIPE_DISPLAY_NAMES = { acestep: 'ACE-Step', onnxruntime: 'ONNX Runtime', trellis: 'TRELLIS.2', - openmoss: 'OpenMOSS TTS' + openmoss: 'OpenMOSS TTS', + 'mlx-engine': 'MLX Engine (Apple Silicon)' }; /* END GENERATED: models-js-recipes */ diff --git a/docs/dev/backends-reference.md b/docs/dev/backends-reference.md index 99420ed7e0..76c95fe266 100644 --- a/docs/dev/backends-reference.md +++ b/docs/dev/backends-reference.md @@ -13,6 +13,7 @@ the generator instead. Prose outside the markers is preserved. --> | `flm` | FastFlowLM NPU | no | yes | npu | | `kokoro` | Kokoro | no | no | cpu, metal | | `llamacpp` | Llama.cpp GPU | yes | yes | cpu, cuda, metal, rocm, system, vulkan | +| `mlx-engine` | MLX Engine | yes | yes | cpu, metal, rocm | | `moonshine` | Moonshine | no | no | cpu | | `onnxruntime` | ONNX Runtime | no | no | cpu | | `openmoss` | OpenMOSS TTS | yes | no | cuda, rocm, vulkan | @@ -41,6 +42,9 @@ the generator instead. Prose outside the markers is preserved. --> | `llamacpp` | vulkan | linux, windows | amd_gpu; cpu (arm64, x86_64) | | `llamacpp` | rocm | linux, windows | amd_gpu (gfx103X, gfx110X, gfx1150, gfx1151, gfx1152, gfx120X, gfx942, gfx950) | | `llamacpp` | cpu | linux, windows | cpu (arm64, x86_64) | +| `mlx-engine` | metal | macos | metal | +| `mlx-engine` | rocm | linux | amd_gpu (gfx110X, gfx1150, gfx1151, gfx120X) | +| `mlx-engine` | cpu | linux, macos | cpu (arm64, x86_64) | | `moonshine` | cpu | windows | cpu (x86_64) | | `moonshine` | cpu | linux | cpu (arm64, x86_64) | | `moonshine` | cpu | macos | cpu (arm64) | @@ -102,6 +106,14 @@ the generator instead. Prose outside the markers is preserved. --> | `llamacpp_device` | `--llamacpp-device` | DEVICES | "" | Comma-separated list of accelerator devices to use (e.g. Vulkan0) | | `llamacpp_args` | `--llamacpp-args` | ARGS | "" | Custom arguments to pass to llama-server | +#### `mlx-engine` — MLX Engine + +| Option | CLI flag | Type | Default | Description | +|--------|----------|------|---------|-------------| +| `ctx_size` | `--ctx-size` | SIZE | -1 | Context size for the model | +| `mlx_backend` | `--mlx-backend` | BACKEND | "" | MLX backend to use (metal, rocm, cpu) | +| `mlx_args` | `--mlx-args` | ARGS | "" | Extra arguments passed to lemon-mlx-engine | + #### `moonshine` — Moonshine | Option | CLI flag | Type | Default | Description | @@ -268,6 +280,13 @@ the generator instead. Prose outside the markers is preserved. --> | `nomic-embed-text-v1-GGUF` | 0.0781 | embeddings | | `nomic-embed-text-v2-moe-GGUF` | 0.51 | embeddings | +#### `mlx-engine` — MLX Engine (2 models) + +| Model | Size (GB) | Labels | +|-------|-----------|--------| +| `Qwen3-0.6B-MLX` | 0.42 | reasoning | +| `Qwen3-4B-MLX` | 2.3 | reasoning | + #### `moonshine` — Moonshine (3 models) | Model | Size (GB) | Labels | diff --git a/docs/guide/cli.md b/docs/guide/cli.md index 75d0544519..b8f21628fc 100644 --- a/docs/guide/cli.md +++ b/docs/guide/cli.md @@ -408,6 +408,14 @@ The following options are available depending on the recipe being used: | Option | Description | Default | |--------|-------------|---------| | `--openmoss BACKEND` | OpenMOSS TTS backend to use | Auto-detected | + +#### MLX Engine (`mlx-engine` recipe) + +| Option | Description | Default | +|--------|-------------|---------| +| `--ctx-size SIZE` | Context size for the model | auto | +| `--mlx-backend BACKEND` | MLX backend to use (metal, rocm, cpu) | Auto-detected | +| `--mlx-args ARGS` | Extra arguments passed to lemon-mlx-engine | `""` | **Notes:** - Unspecified options will use the backend's default values diff --git a/docs/guide/configuration/README.md b/docs/guide/configuration/README.md index f8c21322ef..acb701a3fb 100644 --- a/docs/guide/configuration/README.md +++ b/docs/guide/configuration/README.md @@ -72,6 +72,9 @@ Values set in the user's `config.json` always take precedence over these seeded }, "log_level": "info", "max_loaded_models": 1, + "mlx-engine": { + "backend": "auto" + }, "models_dir": "auto", "moonshine": { "args": "", diff --git a/docs/guide/configuration/custom-models.md b/docs/guide/configuration/custom-models.md index 80a7704e08..336f0d20a0 100644 --- a/docs/guide/configuration/custom-models.md +++ b/docs/guide/configuration/custom-models.md @@ -87,7 +87,7 @@ Supported registration flags: |------|-------------| | `--source SOURCE` | Remote registry for every checkpoint in this model: `huggingface` (default) or `modelscope`. | | `--checkpoint TYPE CHECKPOINT` | Add a checkpoint entry. Repeat for multi-file models such as `main` + `mmproj` or `main` + `vae`. | -| `--recipe RECIPE` | Recipe to associate with the new `user.*` model. Common values: `llamacpp`, `whispercpp`, `moonshine`, `kokoro`, `sd-cpp`, `flm`, `ryzenai-llm`, `vllm`, `thinksound`, `acestep`, `onnxruntime`, `trellis`, `openmoss`, `collection.omni`. | +| `--recipe RECIPE` | Recipe to associate with the new `user.*` model. Common values: `llamacpp`, `whispercpp`, `moonshine`, `kokoro`, `sd-cpp`, `flm`, `ryzenai-llm`, `vllm`, `thinksound`, `acestep`, `onnxruntime`, `trellis`, `openmoss`, `mlx-engine`, `collection.omni`. | | `--label LABEL` | Add a label to the new model. Repeatable. Valid labels include `coding`, `embeddings`, `hot`, `mtp`, `reasoning`, `reranking`, `tool-calling`, `vision`. | | `--components MODEL [MODEL ...]` | Components for an omni collection (see below). Use with `--recipe collection.omni`. | diff --git a/src/cpp/resources/defaults.json b/src/cpp/resources/defaults.json index ed015570c0..d7c36cca60 100644 --- a/src/cpp/resources/defaults.json +++ b/src/cpp/resources/defaults.json @@ -37,6 +37,9 @@ }, "log_level": "info", "max_loaded_models": 1, + "mlx-engine": { + "backend": "auto" + }, "models_dir": "auto", "moonshine": { "args": "",