From 66d0ae7c302400501516b8ee641dd9c50c832d6b Mon Sep 17 00:00:00 2001 From: kuraito315-tech Date: Sat, 3 Oct 2026 20:33:14 +0530 Subject: [PATCH] docs: add llama.cpp custom endpoint example (#783) Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .../librechat_yaml/ai_endpoints/llama-cpp.mdx | 47 +++++++++++++++++++ .../librechat_yaml/ai_endpoints/meta.json | 1 + 2 files changed, 48 insertions(+) create mode 100644 content/docs/configuration/librechat_yaml/ai_endpoints/llama-cpp.mdx diff --git a/content/docs/configuration/librechat_yaml/ai_endpoints/llama-cpp.mdx b/content/docs/configuration/librechat_yaml/ai_endpoints/llama-cpp.mdx new file mode 100644 index 000000000..16d2504ea --- /dev/null +++ b/content/docs/configuration/librechat_yaml/ai_endpoints/llama-cpp.mdx @@ -0,0 +1,47 @@ +--- +title: llama.cpp +icon: BookOpen +description: Configure llama.cpp as a custom endpoint in LibreChat. +--- + +[llama.cpp](https://github.com/ggml-org/llama.cpp) includes `llama-server`, which provides an OpenAI-compatible API for locally hosted models. + +## Configuration + +Set the same API key for LibreChat and `llama-server`. For example, add this to the `.env` file used by LibreChat: + +```bash filename=".env" +LLAMA_CPP_API_KEY=replace-with-the-same-key +``` + +Start `llama-server` with a model alias and that key: + +```bash +llama-server \ + --model /path/to/model.gguf \ + --alias local-model \ + --port 8080 \ + --api-key "replace-with-the-same-key" +``` + +Add this endpoint under `endpoints.custom` in your `librechat.yaml`: + +```yaml filename="librechat.yaml" + - name: "llama.cpp" + apiKey: "${LLAMA_CPP_API_KEY}" + baseURL: "http://localhost:8080/v1" + models: + default: ["local-model"] + fetch: true + titleConvo: true + titleModel: "current_model" + summarize: false + summaryModel: "current_model" + modelDisplayLabel: "llama.cpp" +``` + +## Notes + +- `--alias local-model` sets the model ID returned by the server, which must match the entry in `models.default`. +- `fetch: true` lets LibreChat discover models from the server's `/v1/models` endpoint. +- The server listens on `127.0.0.1:8080` by default. If LibreChat runs in Docker, use an address reachable from the API container instead of `localhost`. If you change the server's bind address, keep it on a trusted network and retain API-key protection. diff --git a/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json b/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json index c46fe3dfd..48d42e689 100644 --- a/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json +++ b/content/docs/configuration/librechat_yaml/ai_endpoints/meta.json @@ -14,6 +14,7 @@ "helicone", "huggingface", "lemonade", + "llama-cpp", "litellm", "mistral", "mlx",