Commit b42b7e6d3 for llama.cpp
commit b42b7e6d3059503dd4bed6ca5dcf342bb6379f67
Author: Xuan-Son Nguyen <son@huggingface.co>
Date: Fri Oct 9 10:40:03 2026 +0200
server: use port 9931 by default (#30159)
* server: use port 9931 by default
* revert unrelated changes
diff --git a/.devops/cann.Dockerfile b/.devops/cann.Dockerfile
index 36cee7bdb..e12362483 100644
--- a/.devops/cann.Dockerfile
+++ b/.devops/cann.Dockerfile
@@ -155,6 +155,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
COPY --from=build /app/full/llama /app/full/llama-server /app
diff --git a/.devops/cpu.Dockerfile b/.devops/cpu.Dockerfile
index cb92343d6..b2a5ab964 100644
--- a/.devops/cpu.Dockerfile
+++ b/.devops/cpu.Dockerfile
@@ -114,6 +114,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
COPY --from=build /app/full/llama /app/full/llama-server /app
diff --git a/.devops/cuda.Dockerfile b/.devops/cuda.Dockerfile
index c9a498d53..e41b94a47 100644
--- a/.devops/cuda.Dockerfile
+++ b/.devops/cuda.Dockerfile
@@ -123,6 +123,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
COPY --from=build /app/full/llama /app/full/llama-server /app
diff --git a/.devops/intel.Dockerfile b/.devops/intel.Dockerfile
index db46fd868..0f13e59e6 100644
--- a/.devops/intel.Dockerfile
+++ b/.devops/intel.Dockerfile
@@ -151,6 +151,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
COPY --from=build /app/lib/ /app
COPY --from=build /app/full/llama /app/full/llama-server /app
diff --git a/.devops/musa.Dockerfile b/.devops/musa.Dockerfile
index 0e3f63359..e902bf829 100644
--- a/.devops/musa.Dockerfile
+++ b/.devops/musa.Dockerfile
@@ -130,6 +130,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
COPY --from=build /app/full/llama /app/full/llama-server /app
diff --git a/.devops/openvino.Dockerfile b/.devops/openvino.Dockerfile
index 4b5ac734f..680ad7db6 100644
--- a/.devops/openvino.Dockerfile
+++ b/.devops/openvino.Dockerfile
@@ -227,6 +227,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
COPY --from=build /app/full/llama /app/full/llama-server /app/
diff --git a/.devops/rocm.Dockerfile b/.devops/rocm.Dockerfile
index 20f6ad636..f5b07c290 100644
--- a/.devops/rocm.Dockerfile
+++ b/.devops/rocm.Dockerfile
@@ -136,6 +136,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
COPY --from=build /app/full/llama /app/full/llama-server /app
diff --git a/.devops/s390x.Dockerfile b/.devops/s390x.Dockerfile
index 94a715ff2..fa8be7616 100644
--- a/.devops/s390x.Dockerfile
+++ b/.devops/s390x.Dockerfile
@@ -133,6 +133,7 @@ ENTRYPOINT [ "/llama.cpp/bin/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
WORKDIR /llama.cpp/bin
diff --git a/.devops/vulkan.Dockerfile b/.devops/vulkan.Dockerfile
index d3599ffb8..a00094bfb 100644
--- a/.devops/vulkan.Dockerfile
+++ b/.devops/vulkan.Dockerfile
@@ -117,6 +117,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
COPY --from=build /app/full/llama /app/full/llama-server /app
diff --git a/.devops/zendnn.Dockerfile b/.devops/zendnn.Dockerfile
index 8a50b3ef6..b24a04389 100644
--- a/.devops/zendnn.Dockerfile
+++ b/.devops/zendnn.Dockerfile
@@ -107,6 +107,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
FROM base AS server
ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080
COPY --from=build /app/full/llama /app/full/llama-server /app
diff --git a/common/arg.cpp b/common/arg.cpp
index 9efcf79ac..bd5afbdfe 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -1479,7 +1479,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
));
add_opt(common_arg(
{"--server-base"}, "URL",
- string_format("connect to this server instead of starting a new one, example: 'http://localhost:8080' (default: none)"),
+ string_format("connect to this server instead of starting a new one, example: 'http://localhost:9931' (default: none)"),
[](common_params & params, const std::string & value) {
params.server_base = value;
}
diff --git a/common/common.h b/common/common.h
index ef062bb84..557518e36 100644
--- a/common/common.h
+++ b/common/common.h
@@ -625,7 +625,7 @@ struct common_params {
std::string cls_sep = "\t"; // separator of classification sequences
// server params
- int32_t port = 8080; // server listens on this network port
+ int32_t port = 9931; // server listens on this network port
bool reuse_port = false; // allow multiple sockets to bind to the same port
int32_t timeout_read = 3600; // http read timeout in seconds
int32_t timeout_write = timeout_read; // http write timeout in seconds
diff --git a/docs/backend/ZenDNN.md b/docs/backend/ZenDNN.md
index b2f970d8c..28f6fd826 100644
--- a/docs/backend/ZenDNN.md
+++ b/docs/backend/ZenDNN.md
@@ -164,11 +164,11 @@ export ZENDNNL_MATMUL_ALGO=1 # Blocked AOCL DLP algo for best performance
./build/bin/llama-server \
-m models/Llama-3.1-8B-Instruct.BF16.gguf \
--host 0.0.0.0 \
- --port 8080 \
+ --port 9931 \
-t 64
```
-Access the server at `http://localhost:8080`.
+Access the server at `http://localhost:9931`.
**Performance tips**:
- Use `ZENDNNL_MATMUL_ALGO=1` for optimal performance
diff --git a/docs/function-calling.md b/docs/function-calling.md
index 28eecbe25..787667eea 100644
--- a/docs/function-calling.md
+++ b/docs/function-calling.md
@@ -282,7 +282,7 @@ This table can be generated with:
# Usage - need tool-aware Jinja template
-First, start a server with any model, but make sure it has a tools-enabled template: you can verify this by inspecting the `chat_template` or `chat_template_tool_use` properties in `http://localhost:8080/props`).
+First, start a server with any model, but make sure it has a tools-enabled template: you can verify this by inspecting the `chat_template` or `chat_template_tool_use` properties in `http://localhost:9931/props`).
Here are some models known to work (w/ chat template override when needed):
@@ -336,7 +336,7 @@ To get the official template from original HuggingFace repos, you can use [scrip
Test in CLI (or with any library / software that can use OpenAI-compatible API backends):
```bash
-curl http://localhost:8080/v1/chat/completions -d '{
+curl http://localhost:9931/v1/chat/completions -d '{
"model": "gpt-3.5-turbo",
"tools": [
{
@@ -366,7 +366,7 @@ curl http://localhost:8080/v1/chat/completions -d '{
}'
-curl http://localhost:8080/v1/chat/completions -d '{
+curl http://localhost:9931/v1/chat/completions -d '{
"model": "gpt-3.5-turbo",
"messages": [
{"role": "system", "content": "You are a chatbot that uses tools/functions. Dont overthink things."},
diff --git a/examples/json_schema_pydantic_example.py b/examples/json_schema_pydantic_example.py
index 19c0bdb5b..dae2a9b8a 100644
--- a/examples/json_schema_pydantic_example.py
+++ b/examples/json_schema_pydantic_example.py
@@ -10,7 +10,7 @@ import json, requests
if True:
- def create_completion(*, response_model=None, endpoint="http://localhost:8080/v1/chat/completions", messages, **kwargs):
+ def create_completion(*, response_model=None, endpoint="http://localhost:9931/v1/chat/completions", messages, **kwargs):
'''
Creates a chat completion using an OpenAI-compatible endpoint w/ JSON schema support
(llama.cpp server, llama-cpp-python, Anyscale / Together...)
@@ -45,7 +45,7 @@ else:
#! pip install instructor openai
import instructor, openai
client = instructor.patch(
- openai.OpenAI(api_key="123", base_url="http://localhost:8080"),
+ openai.OpenAI(api_key="123", base_url="http://localhost:9931"),
mode=instructor.Mode.JSON_SCHEMA)
create_completion = client.chat.completions.create
diff --git a/examples/model-conversion/scripts/causal/modelcard.template b/examples/model-conversion/scripts/causal/modelcard.template
index a04595032..c2750b2b0 100644
--- a/examples/model-conversion/scripts/causal/modelcard.template
+++ b/examples/model-conversion/scripts/causal/modelcard.template
@@ -10,4 +10,4 @@ Recommended way to run this model:
llama-server -hf {namespace}/{model_name}-GGUF
```
-Then, access http://localhost:8080
+Then, access http://localhost:9931
diff --git a/examples/model-conversion/scripts/embedding/modelcard.template b/examples/model-conversion/scripts/embedding/modelcard.template
index 9e63042b7..3e075b9b4 100644
--- a/examples/model-conversion/scripts/embedding/modelcard.template
+++ b/examples/model-conversion/scripts/embedding/modelcard.template
@@ -10,11 +10,11 @@ Recommended way to run this model:
llama-server -hf {namespace}/{model_name}-GGUF --embeddings
```
-Then the endpoint can be accessed at http://localhost:8080/embedding, for
+Then the endpoint can be accessed at http://localhost:9931/embedding, for
example using `curl`:
```console
curl --request POST \
- --url http://localhost:8080/embedding \
+ --url http://localhost:9931/embedding \
--header "Content-Type: application/json" \
--data '{{"input": "Hello embeddings"}}' \
--silent
diff --git a/examples/model-conversion/scripts/utils/curl-embedding-server.sh b/examples/model-conversion/scripts/utils/curl-embedding-server.sh
index 7ed69e1ea..12fee59fb 100755
--- a/examples/model-conversion/scripts/utils/curl-embedding-server.sh
+++ b/examples/model-conversion/scripts/utils/curl-embedding-server.sh
@@ -1,6 +1,6 @@
#!/usr/bin/env bash
curl --request POST \
- --url http://localhost:8080/embedding \
+ --url http://localhost:9931/embedding \
--header "Content-Type: application/json" \
--data '{"input": "Hello world today"}' \
--silent
diff --git a/examples/pydantic_models_to_grammar_examples.py b/examples/pydantic_models_to_grammar_examples.py
index 6dadb7f3f..35905fb97 100755
--- a/examples/pydantic_models_to_grammar_examples.py
+++ b/examples/pydantic_models_to_grammar_examples.py
@@ -295,7 +295,7 @@ def example_concurrent(host):
def main():
parser = argparse.ArgumentParser(description=sys.modules[__name__].__doc__)
- parser.add_argument("--host", default="localhost:8080", help="llama.cpp server")
+ parser.add_argument("--host", default="localhost:9931", help="llama.cpp server")
parser.add_argument("-v", "--verbose", action="store_true", help="enables logging")
args = parser.parse_args()
logging.basicConfig(level=logging.INFO if args.verbose else logging.ERROR)
diff --git a/scripts/compare-logprobs.py b/scripts/compare-logprobs.py
index ac10085b7..6e42d9198 100644
--- a/scripts/compare-logprobs.py
+++ b/scripts/compare-logprobs.py
@@ -15,8 +15,8 @@ Unlike compare-logits.py, it allows dumping logits from a hosted API endpoint. U
Example usage:
Step 1: Dump logits from two different servers
- python scripts/compare-logprobs.py dump logits_llama.log http://localhost:8080/v1/completions
- python scripts/compare-logprobs.py dump logits_other.log http://other-engine:8000/v1/completions
+ python scripts/compare-logprobs.py dump logits_llama.log http://localhost:9931/v1/completions
+ python scripts/compare-logprobs.py dump logits_other.log http://other-engine/v1/completions
(optionally, you can add --api-key <key> if the endpoint requires authentication)
diff --git a/scripts/server-bench.py b/scripts/server-bench.py
index 2eabb3bce..ce9737148 100755
--- a/scripts/server-bench.py
+++ b/scripts/server-bench.py
@@ -54,8 +54,8 @@ def get_server(path_server: str, path_log: Optional[str]) -> dict:
logger.info("LLAMA_ARG_HOST not explicitly set, using 127.0.0.1")
os.environ["LLAMA_ARG_HOST"] = "127.0.0.1"
if os.environ.get("LLAMA_ARG_PORT") is None:
- logger.info("LLAMA_ARG_PORT not explicitly set, using 8080")
- os.environ["LLAMA_ARG_PORT"] = "8080"
+ logger.info("LLAMA_ARG_PORT not explicitly set, using 9931")
+ os.environ["LLAMA_ARG_PORT"] = "9931"
hostname: Optional[str] = os.environ.get("LLAMA_ARG_HOST")
port: Optional[str] = os.environ.get("LLAMA_ARG_PORT")
assert hostname is not None
diff --git a/scripts/server-test-function-call.py b/scripts/server-test-function-call.py
index c32f17b5e..c4383c96f 100755
--- a/scripts/server-test-function-call.py
+++ b/scripts/server-test-function-call.py
@@ -1096,7 +1096,7 @@ def main():
description="Test llama-server tool-calling capability."
)
parser.add_argument("--host", default="localhost")
- parser.add_argument("--port", default=8080, type=int)
+ parser.add_argument("--port", default=9931, type=int)
parser.add_argument(
"--no-stream", action="store_true", help="Disable streaming mode tests"
)
diff --git a/scripts/server-test-model.py b/scripts/server-test-model.py
index 9049d8027..230afa4fe 100644
--- a/scripts/server-test-model.py
+++ b/scripts/server-test-model.py
@@ -183,7 +183,7 @@ def test_tool_call(url, stream):
def main():
parser = argparse.ArgumentParser(description="Test llama-server functionality.")
parser.add_argument("--host", default="localhost", help="Server host")
- parser.add_argument("--port", default=8080, type=int, help="Server port")
+ parser.add_argument("--port", default=9931, type=int, help="Server port")
args = parser.parse_args()
base_url = f"http://{args.host}:{args.port}/v1/chat/completions"
diff --git a/scripts/server-test-parallel-tc.py b/scripts/server-test-parallel-tc.py
index a166c6d72..4a09f831b 100755
--- a/scripts/server-test-parallel-tc.py
+++ b/scripts/server-test-parallel-tc.py
@@ -938,7 +938,7 @@ def main():
)
)
parser.add_argument("--host", default="localhost")
- parser.add_argument("--port", default=8080, type=int)
+ parser.add_argument("--port", default=9931, type=int)
parser.add_argument(
"--no-stream", action="store_true", help="Disable streaming mode tests"
)
diff --git a/scripts/server-test-structured.py b/scripts/server-test-structured.py
index da217fc46..8fee1d9d1 100755
--- a/scripts/server-test-structured.py
+++ b/scripts/server-test-structured.py
@@ -991,7 +991,7 @@ def main():
description="Test llama-server structured-output capability."
)
parser.add_argument("--host", default="localhost")
- parser.add_argument("--port", default=8080, type=int)
+ parser.add_argument("--port", default=9931, type=int)
parser.add_argument(
"--no-stream", action="store_true", help="Disable streaming mode tests"
)
diff --git a/tools/cli/README.md b/tools/cli/README.md
index 20cf01c6b..284b8f308 100644
--- a/tools/cli/README.md
+++ b/tools/cli/README.md
@@ -143,7 +143,7 @@
| Argument | Explanation |
| -------- | ----------- |
-| `--server-base URL` | connect to this server instead of starting a new one, example: 'http://localhost:8080' (default: none) |
+| `--server-base URL` | connect to this server instead of starting a new one, example: 'http://localhost:9931' (default: none) |
| `--verbose-prompt` | print a verbose prompt before generation (default: false) |
| `--display-prompt, --no-display-prompt` | whether to print prompt at generation (default: true) |
| `-co, --color [on\|off\|auto]` | Colorize output to distinguish prompt and user input from generations ('on', 'off', or 'auto', default: 'auto')<br/>'auto' enables colors when output is to a terminal |
diff --git a/tools/cli/cli-client.h b/tools/cli/cli-client.h
index 9493b4fe6..4f947aecc 100644
--- a/tools/cli/cli-client.h
+++ b/tools/cli/cli-client.h
@@ -5,7 +5,7 @@
// openai-like client for CLI
struct cli_client {
- std::string server_base; // base url, for example "http://127.0.0.1:8080"
+ std::string server_base; // base url, for example "http://127.0.0.1:9931"
std::string last_error; // set when wait_health() fails
std::string model; // optional, set when the server has multiple models (router mode)
diff --git a/tools/fit-params/README.md b/tools/fit-params/README.md
index 8f0c958a2..0ec510e29 100644
--- a/tools/fit-params/README.md
+++ b/tools/fit-params/README.md
@@ -37,7 +37,7 @@ system info: n_threads = 16, n_threads_batch = 16, total_threads = 32
system_info: n_threads = 16 (n_threads_batch = 16) / 32 | CUDA : ARCHS = 890 | USE_GRAPHS = 1 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX_VNNI = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | BMI2 = 1 | AVX512 = 1 | AVX512_VBMI = 1 | AVX512_VNNI = 1 | AVX512_BF16 = 1 | LLAMAFILE = 1 | OPENMP = 1 | REPACK = 1 |
main: binding port with default address family
-main: HTTP server is listening, hostname: 127.0.0.1, port: 8080, http threads: 31
+main: HTTP server is listening, hostname: 127.0.0.1, port: 9931, http threads: 31
main: loading model
srv load_model: loading model '/opt/models/qwen_3-30b3a-f16.gguf'
llama_params_fit_impl: projected to use 19187 MiB of device memory vs. 24077 MiB of free device memory
@@ -45,7 +45,7 @@ llama_params_fit_impl: will leave 1199 >= 1024 MiB of free device memory, no cha
llama_params_fit: successfully fit params to free device memory
llama_params_fit: fitting params to free memory took 0.28 seconds
[...]
-main: server is listening on http://127.0.0.1:8080 - starting the main loop
+main: server is listening on http://127.0.0.1:9931 - starting the main loop
srv update_slots: all slots are idle
^Csrv operator(): operator(): cleaning up before exit...
diff --git a/tools/server/README-dev.md b/tools/server/README-dev.md
index 0f42b2ee1..ea84b9f65 100644
--- a/tools/server/README-dev.md
+++ b/tools/server/README-dev.md
@@ -403,4 +403,4 @@ npm run build
After `public/index.html` has been generated, rebuild `llama-server` as described in the [build](#build) section to include the updated UI.
-**Note:** The Vite dev server automatically proxies API requests to `http://localhost:8080`. Make sure `llama-server` is running on that port during development.
+**Note:** The Vite dev server automatically proxies API requests to `http://localhost:9931`. Make sure `llama-server` is running on that port during development.
diff --git a/tools/server/README.md b/tools/server/README.md
index 08d6f17a4..41dca6152 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -191,7 +191,7 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--tags STRING` | set model tags, comma-separated (informational, not used for routing)<br/>(env: LLAMA_ARG_TAGS) |
| `--embd-normalize N` | normalisation for embeddings (default: 2) (-1=none, 0=max absolute int16, 1=taxicab, 2=euclidean, >2=p-norm) |
| `--host HOST` | IP addresses to listen on, comma-separated, or UNIX socket paths ending in .sock; with multiple TCP addresses, :: binds IPv6 only; overlapping addresses result in undefined behavior (default: 127.0.0.1)<br/>(env: LLAMA_ARG_HOST) |
-| `--port PORT` | port to listen (default: 8080)<br/>(env: LLAMA_ARG_PORT) |
+| `--port PORT` | port to listen (default: 9931)<br/>(env: LLAMA_ARG_PORT) |
| `--reuse-port` | allow multiple sockets to bind to the same port (default: disabled)<br/>(env: LLAMA_ARG_REUSE_PORT) |
| `--path PATH` | path to serve static files from (default: )<br/>(env: LLAMA_ARG_STATIC_PATH) |
| `--cors-origins ORIGINS` | comma-separated list of allowed origins for CORS (default: *)<br/>if set to special value 'localhost', reflect the Origin header only if it is localhost<br/>(env: LLAMA_ARG_CORS_ORIGINS) |
@@ -444,7 +444,7 @@ To get started right away, run the following command, making sure to use the cor
llama-server.exe -m models\7B\ggml-model.gguf -c 2048
```
-The above command will start a server that by default listens on `127.0.0.1:8080`.
+The above command will start a server that by default listens on `127.0.0.1:9931`.
You can consume the endpoints with Postman or NodeJS with axios library. You can visit the web front end at the same url.
### Docker
@@ -462,7 +462,7 @@ Using [curl](https://curl.se/). On Windows, `curl.exe` should be available in th
```sh
curl --request POST \
- --url http://localhost:8080/completion \
+ --url http://localhost:9931/completion \
--header "Content-Type: application/json" \
--data '{"prompt": "Building a website can be done in 10 simple steps:","n_predict": 128}'
```
@@ -1322,7 +1322,7 @@ Example usage with `openai` python library:
import openai
client = openai.OpenAI(
- base_url="http://localhost:8080/v1", # "http://<Your api-server IP>:port"
+ base_url="http://localhost:9931/v1", # "http://<Your api-server IP>:port"
api_key = "sk-no-key-required"
)
@@ -1380,7 +1380,7 @@ You can use either Python `openai` library with appropriate checkpoints:
import openai
client = openai.OpenAI(
- base_url="http://localhost:8080/v1", # "http://<Your api-server IP>:port"
+ base_url="http://localhost:9931/v1", # "http://<Your api-server IP>:port"
api_key = "sk-no-key-required"
)
@@ -1398,7 +1398,7 @@ print(completion.choices[0].message)
... or raw HTTP requests:
```shell
-curl http://localhost:8080/v1/chat/completions \
+curl http://localhost:9931/v1/chat/completions \
-H "Content-Type: application/json" \
-H "Authorization: Bearer no-key" \
-d '{
@@ -1502,7 +1502,7 @@ You can use either Python `openai` library with appropriate checkpoints:
import openai
client = openai.OpenAI(
- base_url="http://localhost:8080/v1", # "http://<Your api-server IP>:port"
+ base_url="http://localhost:9931/v1", # "http://<Your api-server IP>:port"
api_key = "sk-no-key-required"
)
@@ -1518,7 +1518,7 @@ print(response.output_text)
... or raw HTTP requests:
```shell
-curl http://localhost:8080/v1/responses \
+curl http://localhost:9931/v1/responses \
-H "Content-Type: application/json" \
-H "Authorization: Bearer no-key" \
-d '{
@@ -1551,7 +1551,7 @@ Each object gives one embedding. This input shape is not part of the OpenAI Embe
- input as string
```shell
- curl http://localhost:8080/v1/embeddings \
+ curl http://localhost:9931/v1/embeddings \
-H "Content-Type: application/json" \
-H "Authorization: Bearer no-key" \
-d '{
@@ -1564,7 +1564,7 @@ Each object gives one embedding. This input shape is not part of the OpenAI Embe
- `input` as string array
```shell
- curl http://localhost:8080/v1/embeddings \
+ curl http://localhost:9931/v1/embeddings \
-H "Content-Type: application/json" \
-H "Authorization: Bearer no-key" \
-d '{
@@ -1577,7 +1577,7 @@ Each object gives one embedding. This input shape is not part of the OpenAI Embe
- `input` as multimodal content
```shell
- curl http://localhost:8080/v1/embeddings \
+ curl http://localhost:9931/v1/embeddings \
-H "Content-Type: application/json" \
-H "Authorization: Bearer no-key" \
-d '{
@@ -1657,7 +1657,7 @@ See [Anthropic Messages API documentation](https://docs.anthropic.com/en/api/mes
*Examples:*
```shell
-curl http://localhost:8080/v1/messages \
+curl http://localhost:9931/v1/messages \
-H "Content-Type: application/json" \
-H "x-api-key: your-api-key" \
-d '{
@@ -1679,7 +1679,7 @@ Accepts the same parameters as `/v1/messages`. The `max_tokens` parameter is not
*Example:*
```shell
-curl http://localhost:8080/v1/messages/count_tokens \
+curl http://localhost:9931/v1/messages/count_tokens \
-H "Content-Type: application/json" \
-d '{
"model": "gpt-4",
@@ -1760,7 +1760,7 @@ The probabilities are scaled with the temperatures stored in the model file. The
*Examples:*
```shell
-curl http://127.0.0.1:8080/v1/systemone \
+curl http://127.0.0.1:9931/v1/systemone \
-H "Content-Type: application/json" \
-d '{
"state": "Customer message: I was charged twice for my order last week and nobody has replied.",
@@ -1817,7 +1817,7 @@ Response (values are shortened):
Example with an image:
```shell
-curl http://127.0.0.1:8080/v1/systemone \
+curl http://127.0.0.1:9931/v1/systemone \
-H "Content-Type: application/json" \
-d '{
"state": "The document was received by the accounting team this morning.",
diff --git a/tools/server/bench/README.md b/tools/server/bench/README.md
index 9549795ec..18d81a100 100644
--- a/tools/server/bench/README.md
+++ b/tools/server/bench/README.md
@@ -29,11 +29,11 @@ Example for PHI-2
```
#### Start the server
-The server must answer OAI Chat completion requests on `http://localhost:8080/v1` or according to the environment variable `SERVER_BENCH_URL`.
+The server must answer OAI Chat completion requests on `http://localhost:9931/v1` or according to the environment variable `SERVER_BENCH_URL`.
Example:
```shell
-llama-server --host localhost --port 8080 \
+llama-server --host localhost --port 9931 \
--model ggml-model-q4_0.gguf \
--cont-batching \
--metrics \
@@ -51,7 +51,7 @@ For 500 chat completions request with 8 concurrent users during maximum 10 minut
```
The benchmark values can be overridden with:
-- `SERVER_BENCH_URL` server url prefix for chat completions, default `http://localhost:8080/v1`
+- `SERVER_BENCH_URL` server url prefix for chat completions, default `http://localhost:9931/v1`
- `SERVER_BENCH_N_PROMPTS` total prompts to randomly select in the benchmark, default `480`
- `SERVER_BENCH_MODEL_ALIAS` model alias to pass in the completion request, default `my-model`
- `SERVER_BENCH_MAX_TOKENS` max tokens to predict, default: `512`
@@ -85,7 +85,7 @@ The script will fail if too many completions are truncated, see `llamacpp_comple
K6 metrics might be compared against [server metrics](../README.md), with:
```shell
-curl http://localhost:8080/metrics
+curl http://localhost:9931/metrics
```
### Using the CI python script
diff --git a/tools/server/bench/bench.py b/tools/server/bench/bench.py
index 2c56ab5eb..82cb483b3 100644
--- a/tools/server/bench/bench.py
+++ b/tools/server/bench/bench.py
@@ -28,7 +28,7 @@ def main(args_in: list[str] | None = None) -> None:
parser.add_argument("--branch", type=str, help="Branch name", default="detached")
parser.add_argument("--commit", type=str, help="Commit name", default="dirty")
parser.add_argument("--host", type=str, help="Server listen host", default="0.0.0.0")
- parser.add_argument("--port", type=int, help="Server listen host", default="8080")
+ parser.add_argument("--port", type=int, help="Server listen host", default="9931")
parser.add_argument("--model-path-prefix", type=str, help="Prefix where to store the model files", default="models")
parser.add_argument("--n-prompts", type=int,
help="SERVER_BENCH_N_PROMPTS: total prompts to randomly select in the benchmark", required=True)
diff --git a/tools/server/bench/prometheus.yml b/tools/server/bench/prometheus.yml
index b15ee5244..92ad20726 100644
--- a/tools/server/bench/prometheus.yml
+++ b/tools/server/bench/prometheus.yml
@@ -6,4 +6,4 @@ global:
scrape_configs:
- job_name: 'llama.cpp server'
static_configs:
- - targets: ['localhost:8080']
+ - targets: ['localhost:9931']
diff --git a/tools/server/bench/script.js b/tools/server/bench/script.js
index 2772bee5e..a5eb73a4d 100644
--- a/tools/server/bench/script.js
+++ b/tools/server/bench/script.js
@@ -5,7 +5,7 @@ import {Counter, Rate, Trend} from 'k6/metrics'
import exec from 'k6/execution';
// Server chat completions prefix
-const server_url = __ENV.SERVER_BENCH_URL ? __ENV.SERVER_BENCH_URL : 'http://localhost:8080/v1'
+const server_url = __ENV.SERVER_BENCH_URL ? __ENV.SERVER_BENCH_URL : 'http://localhost:9931/v1'
// Number of total prompts in the dataset - default 10m / 10 seconds/request * number of users
const n_prompt = __ENV.SERVER_BENCH_N_PROMPTS ? parseInt(__ENV.SERVER_BENCH_N_PROMPTS) : 600 / 10 * 8
diff --git a/tools/server/bench/speed-bench/README.md b/tools/server/bench/speed-bench/README.md
index 8d3fcd804..a8455e2d5 100644
--- a/tools/server/bench/speed-bench/README.md
+++ b/tools/server/bench/speed-bench/README.md
@@ -18,7 +18,7 @@ The client does not launch the server, so start `llama-server` yourself first. I
llama-server \
-m target.gguf \
-c 8192 \
- --port 8080 \
+ --port 9931 \
-ngl 99 -fa on \
--np 1 \
--jinja
@@ -30,7 +30,7 @@ For speculative decoding, start the server with the appropriate flags for your s
```bash
python tools/server/bench/speed-bench/speed_bench.py \
- --url localhost:8080 \
+ --url localhost:9931 \
--bench qualitative \
--category coding \
--osl 1024 \
@@ -41,7 +41,7 @@ python tools/server/bench/speed-bench/speed_bench.py \
| Option | Default | Description |
| --- | --- | --- |
-| `--url` | `localhost:8080` | Server URL. The scheme and `/v1` are optional and a trailing slash is fine, so `localhost:8080` and `http://localhost:8080/v1/` both work. |
+| `--url` | `localhost:9931` | Server URL. The scheme and `/v1` are optional and a trailing slash is fine, so `localhost:9931` and `http://localhost:9931/v1/` both work. |
| `--model` | none | Optional `model` field sent in each request. |
| `--bench` | `qualitative` | SPEED-Bench config, e.g. `qualitative`, `throughput_1k`. See [available dataset variants](https://github.com/ai-dynamo/aiperf/blob/main/docs/tutorials/speed-bench.md#available-dataset-variants). |
| `--category` | `all` | Category filter within the bench; comma-separated list or `all`. For `qualitative` the categories are `coding`, `humanities`, `math`, `multilingual`, `qa`, `rag`, `reasoning`, `roleplay`, `stem`, `summarization`, `writing`. For the `throughput_{ISL}` splits they are `high_entropy`, `low_entropy`, `mixed`. |
@@ -81,7 +81,7 @@ First, start a plain `llama-server` (no speculative decoding) and save a baselin
```bash
python tools/server/bench/speed-bench/speed_bench.py \
- --url localhost:8080 \
+ --url localhost:9931 \
--bench qualitative \
--category all \
--osl 1024 \
@@ -93,7 +93,7 @@ Then restart `llama-server` with speculative decoding enabled and save another r
```bash
python tools/server/bench/speed-bench/speed_bench.py \
- --url localhost:8080 \
+ --url localhost:9931 \
--bench qualitative \
--category all \
--osl 1024 \
diff --git a/tools/server/bench/speed-bench/speed_bench.py b/tools/server/bench/speed-bench/speed_bench.py
index adb378a6b..83d87227a 100644
--- a/tools/server/bench/speed-bench/speed_bench.py
+++ b/tools/server/bench/speed-bench/speed_bench.py
@@ -369,7 +369,7 @@ def save_output(path: str, args: argparse.Namespace, samples: list[Sample], resu
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description="Run SPEED-Bench against an OpenAI-compatible llama-server.")
- parser.add_argument("--url", default="localhost:8080", help="Server URL, for example localhost:8080 or http://localhost:8080/v1")
+ parser.add_argument("--url", default="localhost:9931", help="Server URL, for example localhost:9931 or http://localhost:9931/v1")
parser.add_argument("--model", default=None, help="Optional model name to send in OpenAI requests")
parser.add_argument("--bench", default="qualitative", help="SPEED-Bench config to run, for example qualitative or throughput_1k")
parser.add_argument("--category", default="all", help="Category to run within the selected bench; use all for no category filter")
diff --git a/tools/server/server-http.h b/tools/server/server-http.h
index 4554b20f4..63fd2c186 100644
--- a/tools/server/server-http.h
+++ b/tools/server/server-http.h
@@ -75,7 +75,7 @@ struct server_http_context {
mutable std::unordered_map<std::string, handler_t> handlers;
std::string path_prefix;
- int port = 8080;
+ int port = 9931;
bool is_ssl = false;
server_http_context();
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index 6e2d8ff9d..871e827be 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -521,16 +521,8 @@ int llama_server(common_params & params, int argc, char ** argv, server_child &
#endif
}
- bool uses_default_port = false;
for (const auto & address : ctx_http.listening_addresses) {
SRV_INF("listening on %s\n", address.c_str());
- uses_default_port |= string_ends_with(address, ":8080");
- }
-
- // TODO: remove this in the future
- // check the string to also handle the .sock case
- if (uses_default_port) {
- SRV_WRN("%s", "notice: server default port will be changed to :9931 in a future release (ref: https://github.com/ggml-org/llama.cpp/pull/26508)\n");
}
if (is_router_server) {
diff --git a/tools/ui/README.md b/tools/ui/README.md
index 04a6ea016..2ecbc8f51 100644
--- a/tools/ui/README.md
+++ b/tools/ui/README.md
@@ -118,7 +118,7 @@ This starts:
- **Vite dev server** at `http://localhost:5173` - The main UI frontend app
- **Storybook** at `http://localhost:6006` - Component documentation
-The Vite dev server proxies API requests to `SERVER_ORIGIN` (with fallback to default llama-server `8080` port):
+The Vite dev server proxies API requests to `SERVER_ORIGIN` (with fallback to default llama-server `9931` port):
```typescript
// vite.config.ts proxy configuration
diff --git a/tools/ui/vite.config.ts b/tools/ui/vite.config.ts
index 0f24a600b..233ba7452 100644
--- a/tools/ui/vite.config.ts
+++ b/tools/ui/vite.config.ts
@@ -26,7 +26,7 @@ const browserBaseConfig: any = {
export default defineConfig(({ mode }) => {
const env = loadEnv(mode, process.cwd(), 'VITE_PUBLIC_');
- const SERVER_ORIGIN = env.VITE_PUBLIC_SERVER_ORIGIN || 'http://localhost:8080';
+ const SERVER_ORIGIN = env.VITE_PUBLIC_SERVER_ORIGIN || 'http://localhost:9931';
return {
build: {