From e56e6847f9d45b7b999e386acb9aca74266cf900 Mon Sep 17 00:00:00 2001 From: raullenchai Date: Tue, 28 Jul 2026 07:36:45 -0700 Subject: [PATCH 1/2] docs: add Rapid-MLX as a custom AI endpoint Rapid-MLX is an Apple-Silicon-native inference server (pure MLX) with an OpenAI-compatible API, like the existing Apple MLX / vLLM / Ollama entries. Adds the endpoint page + registers it in meta.json. --- .../librechat_yaml/ai_endpoints/meta.json | 1 + .../librechat_yaml/ai_endpoints/rapid-mlx.mdx | 33 +++++++++++++++++++ 2 files changed, 34 insertions(+) create mode 100644 content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx diff --git a/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json b/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json index cfde5c3eb..feffaeb1c 100644 --- a/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json +++ b/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json @@ -23,6 +23,7 @@ "openrouter", "perplexity", "portkey", + "rapid-mlx", "shuttleai", "togetherai", "truefoundry", diff --git a/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx b/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx new file mode 100644 index 000000000..8a54d0fd5 --- /dev/null +++ b/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx @@ -0,0 +1,33 @@ +--- +title: Rapid-MLX +description: Configure Rapid-MLX as a custom endpoint in LibreChat. +--- + +[Rapid-MLX](https://rapidmlx.com) is an Apple-Silicon-native inference server (pure MLX) that serves models locally through an OpenAI-compatible API, so you can point LibreChat at your own Mac. + +Install it with `brew install rapid-mlx` (or `pip install rapid-mlx`) and start a model with `rapid-mlx serve ` — it listens on `http://localhost:8000` by default. Browse the model catalog at [models.rapidmlx.com](https://models.rapidmlx.com/). + +## Configuration + +The local Rapid-MLX server doesn't authenticate requests, so the API key is just a placeholder. Point `baseURL` at your running server and add the endpoint under `endpoints.custom` in your `librechat.yaml`: + +```yaml filename="librechat.yaml" + - name: "Rapid-MLX" + apiKey: "rapid-mlx" + baseURL: "http://localhost:8000/v1/" + models: + default: [ + "qwen3.5-9b-4bit" + ] + fetch: false # rapid-mlx serves one model at a time; list it explicitly + titleConvo: true + titleModel: "current_model" + summarize: false + summaryModel: "current_model" + modelDisplayLabel: "Rapid-MLX" +``` + +## Notes + +- `rapid-mlx serve ` runs one model at a time on `:8000`. To serve more than one model, run a separate instance on a different port and add another endpoint with its own `baseURL`. +- Set `default` to the alias you launched (e.g. `qwen3.5-9b-4bit`). Any alias from the [Rapid-MLX catalog](https://models.rapidmlx.com/) works. From 1988ccffb40dd2fcb3114c730c8b6106400fa297 Mon Sep 17 00:00:00 2001 From: raullenchai Date: Wed, 7 Oct 2026 13:16:28 -0700 Subject: [PATCH 2/2] docs: refresh Rapid-MLX entry for v0.15.7 --- .../librechat_yaml/ai_endpoints/rapid-mlx.mdx | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx b/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx index 8a54d0fd5..61e04e02a 100644 --- a/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx +++ b/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx @@ -5,11 +5,11 @@ description: Configure Rapid-MLX as a custom endpoint in LibreChat. [Rapid-MLX](https://rapidmlx.com) is an Apple-Silicon-native inference server (pure MLX) that serves models locally through an OpenAI-compatible API, so you can point LibreChat at your own Mac. -Install it with `brew install rapid-mlx` (or `pip install rapid-mlx`) and start a model with `rapid-mlx serve ` — it listens on `http://localhost:8000` by default. Browse the model catalog at [models.rapidmlx.com](https://models.rapidmlx.com/). +Install it with `brew install rapid-mlx` (or `pip install rapid-mlx`) and start a model with `rapid-mlx serve qwen3.5-4b-4bit --port 8000`. This pins the server to `http://localhost:8000`. Browse the model catalog at [models.rapidmlx.com](https://models.rapidmlx.com/). ## Configuration -The local Rapid-MLX server doesn't authenticate requests, so the API key is just a placeholder. Point `baseURL` at your running server and add the endpoint under `endpoints.custom` in your `librechat.yaml`: +By default, the local Rapid-MLX server does not require an API key, so the value below is a placeholder. Point `baseURL` at your running server and add the endpoint under `endpoints.custom` in your `librechat.yaml`: ```yaml filename="librechat.yaml" - name: "Rapid-MLX" @@ -17,9 +17,9 @@ The local Rapid-MLX server doesn't authenticate requests, so the API key is just baseURL: "http://localhost:8000/v1/" models: default: [ - "qwen3.5-9b-4bit" + "qwen3.5-4b-4bit" ] - fetch: false # rapid-mlx serves one model at a time; list it explicitly + fetch: false # list the model served by this instance explicitly titleConvo: true titleModel: "current_model" summarize: false @@ -29,5 +29,6 @@ The local Rapid-MLX server doesn't authenticate requests, so the API key is just ## Notes -- `rapid-mlx serve ` runs one model at a time on `:8000`. To serve more than one model, run a separate instance on a different port and add another endpoint with its own `baseURL`. -- Set `default` to the alias you launched (e.g. `qwen3.5-9b-4bit`). Any alias from the [Rapid-MLX catalog](https://models.rapidmlx.com/) works. +- Each server instance serves one model. For another model, start a second instance on another port and add an endpoint with its own `baseURL`. +- If LibreChat runs in Docker on the same Mac, use `host.docker.internal` in place of `localhost` in `baseURL`. +- Set `default` to the alias you launched (e.g. `qwen3.5-4b-4bit`). Any alias from the [Rapid-MLX catalog](https://models.rapidmlx.com/) works.