diff --git a/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json b/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json index cfde5c3eb..feffaeb1c 100644 --- a/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json +++ b/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json @@ -23,6 +23,7 @@ "openrouter", "perplexity", "portkey", + "rapid-mlx", "shuttleai", "togetherai", "truefoundry", diff --git a/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx b/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx new file mode 100644 index 000000000..61e04e02a --- /dev/null +++ b/content/docs/configuration/librechat_yaml/ai_endpoints/rapid-mlx.mdx @@ -0,0 +1,34 @@ +--- +title: Rapid-MLX +description: Configure Rapid-MLX as a custom endpoint in LibreChat. +--- + +[Rapid-MLX](https://rapidmlx.com) is an Apple-Silicon-native inference server (pure MLX) that serves models locally through an OpenAI-compatible API, so you can point LibreChat at your own Mac. + +Install it with `brew install rapid-mlx` (or `pip install rapid-mlx`) and start a model with `rapid-mlx serve qwen3.5-4b-4bit --port 8000`. This pins the server to `http://localhost:8000`. Browse the model catalog at [models.rapidmlx.com](https://models.rapidmlx.com/). + +## Configuration + +By default, the local Rapid-MLX server does not require an API key, so the value below is a placeholder. Point `baseURL` at your running server and add the endpoint under `endpoints.custom` in your `librechat.yaml`: + +```yaml filename="librechat.yaml" + - name: "Rapid-MLX" + apiKey: "rapid-mlx" + baseURL: "http://localhost:8000/v1/" + models: + default: [ + "qwen3.5-4b-4bit" + ] + fetch: false # list the model served by this instance explicitly + titleConvo: true + titleModel: "current_model" + summarize: false + summaryModel: "current_model" + modelDisplayLabel: "Rapid-MLX" +``` + +## Notes + +- Each server instance serves one model. For another model, start a second instance on another port and add an endpoint with its own `baseURL`. +- If LibreChat runs in Docker on the same Mac, use `host.docker.internal` in place of `localhost` in `baseURL`. +- Set `default` to the alias you launched (e.g. `qwen3.5-4b-4bit`). Any alias from the [Rapid-MLX catalog](https://models.rapidmlx.com/) works.