Commit b42b7e6d3 for llama.cpp

commit b42b7e6d3059503dd4bed6ca5dcf342bb6379f67
Author: Xuan-Son Nguyen <son@huggingface.co>
Date:   Fri Oct 9 10:40:03 2026 +0200

    server: use port 9931 by default (#30159)

    * server: use port 9931 by default

    * revert unrelated changes

diff --git a/.devops/cann.Dockerfile b/.devops/cann.Dockerfile
index 36cee7bdb..e12362483 100644
--- a/.devops/cann.Dockerfile
+++ b/.devops/cann.Dockerfile
@@ -155,6 +155,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 COPY --from=build /app/full/llama /app/full/llama-server /app

diff --git a/.devops/cpu.Dockerfile b/.devops/cpu.Dockerfile
index cb92343d6..b2a5ab964 100644
--- a/.devops/cpu.Dockerfile
+++ b/.devops/cpu.Dockerfile
@@ -114,6 +114,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 COPY --from=build /app/full/llama /app/full/llama-server /app

diff --git a/.devops/cuda.Dockerfile b/.devops/cuda.Dockerfile
index c9a498d53..e41b94a47 100644
--- a/.devops/cuda.Dockerfile
+++ b/.devops/cuda.Dockerfile
@@ -123,6 +123,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 COPY --from=build /app/full/llama /app/full/llama-server /app

diff --git a/.devops/intel.Dockerfile b/.devops/intel.Dockerfile
index db46fd868..0f13e59e6 100644
--- a/.devops/intel.Dockerfile
+++ b/.devops/intel.Dockerfile
@@ -151,6 +151,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 COPY --from=build /app/lib/ /app
 COPY --from=build /app/full/llama /app/full/llama-server /app
diff --git a/.devops/musa.Dockerfile b/.devops/musa.Dockerfile
index 0e3f63359..e902bf829 100644
--- a/.devops/musa.Dockerfile
+++ b/.devops/musa.Dockerfile
@@ -130,6 +130,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 COPY --from=build /app/full/llama /app/full/llama-server /app

diff --git a/.devops/openvino.Dockerfile b/.devops/openvino.Dockerfile
index 4b5ac734f..680ad7db6 100644
--- a/.devops/openvino.Dockerfile
+++ b/.devops/openvino.Dockerfile
@@ -227,6 +227,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 COPY --from=build /app/full/llama /app/full/llama-server /app/

diff --git a/.devops/rocm.Dockerfile b/.devops/rocm.Dockerfile
index 20f6ad636..f5b07c290 100644
--- a/.devops/rocm.Dockerfile
+++ b/.devops/rocm.Dockerfile
@@ -136,6 +136,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 COPY --from=build /app/full/llama /app/full/llama-server /app

diff --git a/.devops/s390x.Dockerfile b/.devops/s390x.Dockerfile
index 94a715ff2..fa8be7616 100644
--- a/.devops/s390x.Dockerfile
+++ b/.devops/s390x.Dockerfile
@@ -133,6 +133,7 @@ ENTRYPOINT [ "/llama.cpp/bin/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 WORKDIR /llama.cpp/bin

diff --git a/.devops/vulkan.Dockerfile b/.devops/vulkan.Dockerfile
index d3599ffb8..a00094bfb 100644
--- a/.devops/vulkan.Dockerfile
+++ b/.devops/vulkan.Dockerfile
@@ -117,6 +117,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 COPY --from=build /app/full/llama /app/full/llama-server /app

diff --git a/.devops/zendnn.Dockerfile b/.devops/zendnn.Dockerfile
index 8a50b3ef6..b24a04389 100644
--- a/.devops/zendnn.Dockerfile
+++ b/.devops/zendnn.Dockerfile
@@ -107,6 +107,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
 FROM base AS server

 ENV LLAMA_ARG_HOST=0.0.0.0
+ENV LLAMA_ARG_PORT=8080

 COPY --from=build /app/full/llama /app/full/llama-server /app

diff --git a/common/arg.cpp b/common/arg.cpp
index 9efcf79ac..bd5afbdfe 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -1479,7 +1479,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
     ));
     add_opt(common_arg(
         {"--server-base"}, "URL",
-        string_format("connect to this server instead of starting a new one, example: 'http://localhost:8080' (default: none)"),
+        string_format("connect to this server instead of starting a new one, example: 'http://localhost:9931' (default: none)"),
         [](common_params & params, const std::string & value) {
             params.server_base = value;
         }
diff --git a/common/common.h b/common/common.h
index ef062bb84..557518e36 100644
--- a/common/common.h
+++ b/common/common.h
@@ -625,7 +625,7 @@ struct common_params {
     std::string cls_sep    = "\t";  // separator of classification sequences

     // server params
-    int32_t port                = 8080;          // server listens on this network port
+    int32_t port                = 9931;          // server listens on this network port
     bool    reuse_port          = false;         // allow multiple sockets to bind to the same port
     int32_t timeout_read        = 3600;          // http read timeout in seconds
     int32_t timeout_write       = timeout_read;  // http write timeout in seconds
diff --git a/docs/backend/ZenDNN.md b/docs/backend/ZenDNN.md
index b2f970d8c..28f6fd826 100644
--- a/docs/backend/ZenDNN.md
+++ b/docs/backend/ZenDNN.md
@@ -164,11 +164,11 @@ export ZENDNNL_MATMUL_ALGO=1    # Blocked AOCL DLP algo for best performance
 ./build/bin/llama-server \
     -m models/Llama-3.1-8B-Instruct.BF16.gguf \
     --host 0.0.0.0 \
-    --port 8080 \
+    --port 9931 \
     -t 64
 ```

-Access the server at `http://localhost:8080`.
+Access the server at `http://localhost:9931`.

 **Performance tips**:
 - Use `ZENDNNL_MATMUL_ALGO=1` for optimal performance
diff --git a/docs/function-calling.md b/docs/function-calling.md
index 28eecbe25..787667eea 100644
--- a/docs/function-calling.md
+++ b/docs/function-calling.md
@@ -282,7 +282,7 @@ This table can be generated with:

 # Usage - need tool-aware Jinja template

-First, start a server with any model, but make sure it has a tools-enabled template: you can verify this by inspecting the `chat_template` or `chat_template_tool_use` properties in `http://localhost:8080/props`).
+First, start a server with any model, but make sure it has a tools-enabled template: you can verify this by inspecting the `chat_template` or `chat_template_tool_use` properties in `http://localhost:9931/props`).

 Here are some models known to work (w/ chat template override when needed):

@@ -336,7 +336,7 @@ To get the official template from original HuggingFace repos, you can use [scrip
 Test in CLI (or with any library / software that can use OpenAI-compatible API backends):

 ```bash
-curl http://localhost:8080/v1/chat/completions -d '{
+curl http://localhost:9931/v1/chat/completions -d '{
     "model": "gpt-3.5-turbo",
     "tools": [
         {
@@ -366,7 +366,7 @@ curl http://localhost:8080/v1/chat/completions -d '{
 }'


-curl http://localhost:8080/v1/chat/completions -d '{
+curl http://localhost:9931/v1/chat/completions -d '{
     "model": "gpt-3.5-turbo",
     "messages": [
         {"role": "system", "content": "You are a chatbot that uses tools/functions. Dont overthink things."},
diff --git a/examples/json_schema_pydantic_example.py b/examples/json_schema_pydantic_example.py
index 19c0bdb5b..dae2a9b8a 100644
--- a/examples/json_schema_pydantic_example.py
+++ b/examples/json_schema_pydantic_example.py
@@ -10,7 +10,7 @@ import json, requests

 if True:

-    def create_completion(*, response_model=None, endpoint="http://localhost:8080/v1/chat/completions", messages, **kwargs):
+    def create_completion(*, response_model=None, endpoint="http://localhost:9931/v1/chat/completions", messages, **kwargs):
         '''
         Creates a chat completion using an OpenAI-compatible endpoint w/ JSON schema support
         (llama.cpp server, llama-cpp-python, Anyscale / Together...)
@@ -45,7 +45,7 @@ else:
     #! pip install instructor openai
     import instructor, openai
     client = instructor.patch(
-        openai.OpenAI(api_key="123", base_url="http://localhost:8080"),
+        openai.OpenAI(api_key="123", base_url="http://localhost:9931"),
         mode=instructor.Mode.JSON_SCHEMA)
     create_completion = client.chat.completions.create

diff --git a/examples/model-conversion/scripts/causal/modelcard.template b/examples/model-conversion/scripts/causal/modelcard.template
index a04595032..c2750b2b0 100644
--- a/examples/model-conversion/scripts/causal/modelcard.template
+++ b/examples/model-conversion/scripts/causal/modelcard.template
@@ -10,4 +10,4 @@ Recommended way to run this model:
 llama-server -hf {namespace}/{model_name}-GGUF
 ```

-Then, access http://localhost:8080
+Then, access http://localhost:9931
diff --git a/examples/model-conversion/scripts/embedding/modelcard.template b/examples/model-conversion/scripts/embedding/modelcard.template
index 9e63042b7..3e075b9b4 100644
--- a/examples/model-conversion/scripts/embedding/modelcard.template
+++ b/examples/model-conversion/scripts/embedding/modelcard.template
@@ -10,11 +10,11 @@ Recommended way to run this model:
 llama-server -hf {namespace}/{model_name}-GGUF --embeddings
 ```

-Then the endpoint can be accessed at http://localhost:8080/embedding, for
+Then the endpoint can be accessed at http://localhost:9931/embedding, for
 example using `curl`:
 ```console
 curl --request POST \
-    --url http://localhost:8080/embedding \
+    --url http://localhost:9931/embedding \
     --header "Content-Type: application/json" \
     --data '{{"input": "Hello embeddings"}}' \
     --silent
diff --git a/examples/model-conversion/scripts/utils/curl-embedding-server.sh b/examples/model-conversion/scripts/utils/curl-embedding-server.sh
index 7ed69e1ea..12fee59fb 100755
--- a/examples/model-conversion/scripts/utils/curl-embedding-server.sh
+++ b/examples/model-conversion/scripts/utils/curl-embedding-server.sh
@@ -1,6 +1,6 @@
 #!/usr/bin/env bash
 curl --request POST \
-    --url http://localhost:8080/embedding \
+    --url http://localhost:9931/embedding \
     --header "Content-Type: application/json" \
     --data '{"input": "Hello world today"}' \
     --silent
diff --git a/examples/pydantic_models_to_grammar_examples.py b/examples/pydantic_models_to_grammar_examples.py
index 6dadb7f3f..35905fb97 100755
--- a/examples/pydantic_models_to_grammar_examples.py
+++ b/examples/pydantic_models_to_grammar_examples.py
@@ -295,7 +295,7 @@ def example_concurrent(host):

 def main():
     parser = argparse.ArgumentParser(description=sys.modules[__name__].__doc__)
-    parser.add_argument("--host", default="localhost:8080", help="llama.cpp server")
+    parser.add_argument("--host", default="localhost:9931", help="llama.cpp server")
     parser.add_argument("-v", "--verbose", action="store_true", help="enables logging")
     args = parser.parse_args()
     logging.basicConfig(level=logging.INFO if args.verbose else logging.ERROR)
diff --git a/scripts/compare-logprobs.py b/scripts/compare-logprobs.py
index ac10085b7..6e42d9198 100644
--- a/scripts/compare-logprobs.py
+++ b/scripts/compare-logprobs.py
@@ -15,8 +15,8 @@ Unlike compare-logits.py, it allows dumping logits from a hosted API endpoint. U

 Example usage:
     Step 1: Dump logits from two different servers
-        python scripts/compare-logprobs.py dump logits_llama.log http://localhost:8080/v1/completions
-        python scripts/compare-logprobs.py dump logits_other.log http://other-engine:8000/v1/completions
+        python scripts/compare-logprobs.py dump logits_llama.log http://localhost:9931/v1/completions
+        python scripts/compare-logprobs.py dump logits_other.log http://other-engine/v1/completions

         (optionally, you can add --api-key <key> if the endpoint requires authentication)

diff --git a/scripts/server-bench.py b/scripts/server-bench.py
index 2eabb3bce..ce9737148 100755
--- a/scripts/server-bench.py
+++ b/scripts/server-bench.py
@@ -54,8 +54,8 @@ def get_server(path_server: str, path_log: Optional[str]) -> dict:
         logger.info("LLAMA_ARG_HOST not explicitly set, using 127.0.0.1")
         os.environ["LLAMA_ARG_HOST"] = "127.0.0.1"
     if os.environ.get("LLAMA_ARG_PORT") is None:
-        logger.info("LLAMA_ARG_PORT not explicitly set, using 8080")
-        os.environ["LLAMA_ARG_PORT"] = "8080"
+        logger.info("LLAMA_ARG_PORT not explicitly set, using 9931")
+        os.environ["LLAMA_ARG_PORT"] = "9931"
     hostname: Optional[str] = os.environ.get("LLAMA_ARG_HOST")
     port: Optional[str] = os.environ.get("LLAMA_ARG_PORT")
     assert hostname is not None
diff --git a/scripts/server-test-function-call.py b/scripts/server-test-function-call.py
index c32f17b5e..c4383c96f 100755
--- a/scripts/server-test-function-call.py
+++ b/scripts/server-test-function-call.py
@@ -1096,7 +1096,7 @@ def main():
         description="Test llama-server tool-calling capability."
     )
     parser.add_argument("--host", default="localhost")
-    parser.add_argument("--port", default=8080, type=int)
+    parser.add_argument("--port", default=9931, type=int)
     parser.add_argument(
         "--no-stream", action="store_true", help="Disable streaming mode tests"
     )
diff --git a/scripts/server-test-model.py b/scripts/server-test-model.py
index 9049d8027..230afa4fe 100644
--- a/scripts/server-test-model.py
+++ b/scripts/server-test-model.py
@@ -183,7 +183,7 @@ def test_tool_call(url, stream):
 def main():
     parser = argparse.ArgumentParser(description="Test llama-server functionality.")
     parser.add_argument("--host", default="localhost", help="Server host")
-    parser.add_argument("--port", default=8080, type=int, help="Server port")
+    parser.add_argument("--port", default=9931, type=int, help="Server port")
     args = parser.parse_args()

     base_url = f"http://{args.host}:{args.port}/v1/chat/completions"
diff --git a/scripts/server-test-parallel-tc.py b/scripts/server-test-parallel-tc.py
index a166c6d72..4a09f831b 100755
--- a/scripts/server-test-parallel-tc.py
+++ b/scripts/server-test-parallel-tc.py
@@ -938,7 +938,7 @@ def main():
         )
     )
     parser.add_argument("--host", default="localhost")
-    parser.add_argument("--port", default=8080, type=int)
+    parser.add_argument("--port", default=9931, type=int)
     parser.add_argument(
         "--no-stream", action="store_true", help="Disable streaming mode tests"
     )
diff --git a/scripts/server-test-structured.py b/scripts/server-test-structured.py
index da217fc46..8fee1d9d1 100755
--- a/scripts/server-test-structured.py
+++ b/scripts/server-test-structured.py
@@ -991,7 +991,7 @@ def main():
         description="Test llama-server structured-output capability."
     )
     parser.add_argument("--host", default="localhost")
-    parser.add_argument("--port", default=8080, type=int)
+    parser.add_argument("--port", default=9931, type=int)
     parser.add_argument(
         "--no-stream", action="store_true", help="Disable streaming mode tests"
     )
diff --git a/tools/cli/README.md b/tools/cli/README.md
index 20cf01c6b..284b8f308 100644
--- a/tools/cli/README.md
+++ b/tools/cli/README.md
@@ -143,7 +143,7 @@

 | Argument | Explanation |
 | -------- | ----------- |
-| `--server-base URL` | connect to this server instead of starting a new one, example: 'http://localhost:8080' (default: none) |
+| `--server-base URL` | connect to this server instead of starting a new one, example: 'http://localhost:9931' (default: none) |
 | `--verbose-prompt` | print a verbose prompt before generation (default: false) |
 | `--display-prompt, --no-display-prompt` | whether to print prompt at generation (default: true) |
 | `-co, --color [on\|off\|auto]` | Colorize output to distinguish prompt and user input from generations ('on', 'off', or 'auto', default: 'auto')<br/>'auto' enables colors when output is to a terminal |
diff --git a/tools/cli/cli-client.h b/tools/cli/cli-client.h
index 9493b4fe6..4f947aecc 100644
--- a/tools/cli/cli-client.h
+++ b/tools/cli/cli-client.h
@@ -5,7 +5,7 @@

 // openai-like client for CLI
 struct cli_client {
-    std::string server_base; // base url, for example "http://127.0.0.1:8080"
+    std::string server_base; // base url, for example "http://127.0.0.1:9931"
     std::string last_error;  // set when wait_health() fails

     std::string model; // optional, set when the server has multiple models (router mode)
diff --git a/tools/fit-params/README.md b/tools/fit-params/README.md
index 8f0c958a2..0ec510e29 100644
--- a/tools/fit-params/README.md
+++ b/tools/fit-params/README.md
@@ -37,7 +37,7 @@ system info: n_threads = 16, n_threads_batch = 16, total_threads = 32
 system_info: n_threads = 16 (n_threads_batch = 16) / 32 | CUDA : ARCHS = 890 | USE_GRAPHS = 1 | PEER_MAX_BATCH_SIZE = 128 | CPU : SSE3 = 1 | SSSE3 = 1 | AVX = 1 | AVX_VNNI = 1 | AVX2 = 1 | F16C = 1 | FMA = 1 | BMI2 = 1 | AVX512 = 1 | AVX512_VBMI = 1 | AVX512_VNNI = 1 | AVX512_BF16 = 1 | LLAMAFILE = 1 | OPENMP = 1 | REPACK = 1 |

 main: binding port with default address family
-main: HTTP server is listening, hostname: 127.0.0.1, port: 8080, http threads: 31
+main: HTTP server is listening, hostname: 127.0.0.1, port: 9931, http threads: 31
 main: loading model
 srv    load_model: loading model '/opt/models/qwen_3-30b3a-f16.gguf'
 llama_params_fit_impl: projected to use 19187 MiB of device memory vs. 24077 MiB of free device memory
@@ -45,7 +45,7 @@ llama_params_fit_impl: will leave 1199 >= 1024 MiB of free device memory, no cha
 llama_params_fit: successfully fit params to free device memory
 llama_params_fit: fitting params to free memory took 0.28 seconds
 [...]
-main: server is listening on http://127.0.0.1:8080 - starting the main loop
+main: server is listening on http://127.0.0.1:9931 - starting the main loop
 srv  update_slots: all slots are idle
 ^Csrv    operator(): operator(): cleaning up before exit...

diff --git a/tools/server/README-dev.md b/tools/server/README-dev.md
index 0f42b2ee1..ea84b9f65 100644
--- a/tools/server/README-dev.md
+++ b/tools/server/README-dev.md
@@ -403,4 +403,4 @@ npm run build

 After `public/index.html` has been generated, rebuild `llama-server` as described in the [build](#build) section to include the updated UI.

-**Note:** The Vite dev server automatically proxies API requests to `http://localhost:8080`. Make sure `llama-server` is running on that port during development.
+**Note:** The Vite dev server automatically proxies API requests to `http://localhost:9931`. Make sure `llama-server` is running on that port during development.
diff --git a/tools/server/README.md b/tools/server/README.md
index 08d6f17a4..41dca6152 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -191,7 +191,7 @@ For the full list of features, please refer to [server's changelog](https://gith
 | `--tags STRING` | set model tags, comma-separated (informational, not used for routing)<br/>(env: LLAMA_ARG_TAGS) |
 | `--embd-normalize N` | normalisation for embeddings (default: 2) (-1=none, 0=max absolute int16, 1=taxicab, 2=euclidean, >2=p-norm) |
 | `--host HOST` | IP addresses to listen on, comma-separated, or UNIX socket paths ending in .sock; with multiple TCP addresses, :: binds IPv6 only; overlapping addresses result in undefined behavior (default: 127.0.0.1)<br/>(env: LLAMA_ARG_HOST) |
-| `--port PORT` | port to listen (default: 8080)<br/>(env: LLAMA_ARG_PORT) |
+| `--port PORT` | port to listen (default: 9931)<br/>(env: LLAMA_ARG_PORT) |
 | `--reuse-port` | allow multiple sockets to bind to the same port (default: disabled)<br/>(env: LLAMA_ARG_REUSE_PORT) |
 | `--path PATH` | path to serve static files from (default: )<br/>(env: LLAMA_ARG_STATIC_PATH) |
 | `--cors-origins ORIGINS` | comma-separated list of allowed origins for CORS (default: *)<br/>if set to special value 'localhost', reflect the Origin header only if it is localhost<br/>(env: LLAMA_ARG_CORS_ORIGINS) |
@@ -444,7 +444,7 @@ To get started right away, run the following command, making sure to use the cor
 llama-server.exe -m models\7B\ggml-model.gguf -c 2048
 ```

-The above command will start a server that by default listens on `127.0.0.1:8080`.
+The above command will start a server that by default listens on `127.0.0.1:9931`.
 You can consume the endpoints with Postman or NodeJS with axios library. You can visit the web front end at the same url.

 ### Docker
@@ -462,7 +462,7 @@ Using [curl](https://curl.se/). On Windows, `curl.exe` should be available in th

 ```sh
 curl --request POST \
-    --url http://localhost:8080/completion \
+    --url http://localhost:9931/completion \
     --header "Content-Type: application/json" \
     --data '{"prompt": "Building a website can be done in 10 simple steps:","n_predict": 128}'
 ```
@@ -1322,7 +1322,7 @@ Example usage with `openai` python library:
 import openai

 client = openai.OpenAI(
-    base_url="http://localhost:8080/v1", # "http://<Your api-server IP>:port"
+    base_url="http://localhost:9931/v1", # "http://<Your api-server IP>:port"
     api_key = "sk-no-key-required"
 )

@@ -1380,7 +1380,7 @@ You can use either Python `openai` library with appropriate checkpoints:
 import openai

 client = openai.OpenAI(
-    base_url="http://localhost:8080/v1", # "http://<Your api-server IP>:port"
+    base_url="http://localhost:9931/v1", # "http://<Your api-server IP>:port"
     api_key = "sk-no-key-required"
 )

@@ -1398,7 +1398,7 @@ print(completion.choices[0].message)
 ... or raw HTTP requests:

 ```shell
-curl http://localhost:8080/v1/chat/completions \
+curl http://localhost:9931/v1/chat/completions \
 -H "Content-Type: application/json" \
 -H "Authorization: Bearer no-key" \
 -d '{
@@ -1502,7 +1502,7 @@ You can use either Python `openai` library with appropriate checkpoints:
 import openai

 client = openai.OpenAI(
-    base_url="http://localhost:8080/v1", # "http://<Your api-server IP>:port"
+    base_url="http://localhost:9931/v1", # "http://<Your api-server IP>:port"
     api_key = "sk-no-key-required"
 )

@@ -1518,7 +1518,7 @@ print(response.output_text)
 ... or raw HTTP requests:

 ```shell
-curl http://localhost:8080/v1/responses \
+curl http://localhost:9931/v1/responses \
 -H "Content-Type: application/json" \
 -H "Authorization: Bearer no-key" \
 -d '{
@@ -1551,7 +1551,7 @@ Each object gives one embedding. This input shape is not part of the OpenAI Embe
 - input as string

   ```shell
-  curl http://localhost:8080/v1/embeddings \
+  curl http://localhost:9931/v1/embeddings \
   -H "Content-Type: application/json" \
   -H "Authorization: Bearer no-key" \
   -d '{
@@ -1564,7 +1564,7 @@ Each object gives one embedding. This input shape is not part of the OpenAI Embe
 - `input` as string array

   ```shell
-  curl http://localhost:8080/v1/embeddings \
+  curl http://localhost:9931/v1/embeddings \
   -H "Content-Type: application/json" \
   -H "Authorization: Bearer no-key" \
   -d '{
@@ -1577,7 +1577,7 @@ Each object gives one embedding. This input shape is not part of the OpenAI Embe
 - `input` as multimodal content

   ```shell
-  curl http://localhost:8080/v1/embeddings \
+  curl http://localhost:9931/v1/embeddings \
   -H "Content-Type: application/json" \
   -H "Authorization: Bearer no-key" \
   -d '{
@@ -1657,7 +1657,7 @@ See [Anthropic Messages API documentation](https://docs.anthropic.com/en/api/mes
 *Examples:*

 ```shell
-curl http://localhost:8080/v1/messages \
+curl http://localhost:9931/v1/messages \
   -H "Content-Type: application/json" \
   -H "x-api-key: your-api-key" \
   -d '{
@@ -1679,7 +1679,7 @@ Accepts the same parameters as `/v1/messages`. The `max_tokens` parameter is not
 *Example:*

 ```shell
-curl http://localhost:8080/v1/messages/count_tokens \
+curl http://localhost:9931/v1/messages/count_tokens \
   -H "Content-Type: application/json" \
   -d '{
     "model": "gpt-4",
@@ -1760,7 +1760,7 @@ The probabilities are scaled with the temperatures stored in the model file. The
 *Examples:*

 ```shell
-curl http://127.0.0.1:8080/v1/systemone \
+curl http://127.0.0.1:9931/v1/systemone \
     -H "Content-Type: application/json" \
     -d '{
         "state": "Customer message: I was charged twice for my order last week and nobody has replied.",
@@ -1817,7 +1817,7 @@ Response (values are shortened):
 Example with an image:

 ```shell
-curl http://127.0.0.1:8080/v1/systemone \
+curl http://127.0.0.1:9931/v1/systemone \
     -H "Content-Type: application/json" \
     -d '{
         "state": "The document was received by the accounting team this morning.",
diff --git a/tools/server/bench/README.md b/tools/server/bench/README.md
index 9549795ec..18d81a100 100644
--- a/tools/server/bench/README.md
+++ b/tools/server/bench/README.md
@@ -29,11 +29,11 @@ Example for PHI-2
 ```

 #### Start the server
-The server must answer OAI Chat completion requests on `http://localhost:8080/v1` or according to the environment variable `SERVER_BENCH_URL`.
+The server must answer OAI Chat completion requests on `http://localhost:9931/v1` or according to the environment variable `SERVER_BENCH_URL`.

 Example:
 ```shell
-llama-server --host localhost --port 8080 \
+llama-server --host localhost --port 9931 \
   --model ggml-model-q4_0.gguf \
   --cont-batching \
   --metrics \
@@ -51,7 +51,7 @@ For 500 chat completions request with 8 concurrent users during maximum 10 minut
 ```

 The benchmark values can be overridden with:
-- `SERVER_BENCH_URL` server url prefix for chat completions, default `http://localhost:8080/v1`
+- `SERVER_BENCH_URL` server url prefix for chat completions, default `http://localhost:9931/v1`
 - `SERVER_BENCH_N_PROMPTS` total prompts to randomly select in the benchmark, default `480`
 - `SERVER_BENCH_MODEL_ALIAS` model alias to pass in the completion request, default `my-model`
 - `SERVER_BENCH_MAX_TOKENS` max tokens to predict, default: `512`
@@ -85,7 +85,7 @@ The script will fail if too many completions are truncated, see `llamacpp_comple
 K6 metrics might be compared against [server metrics](../README.md), with:

 ```shell
-curl http://localhost:8080/metrics
+curl http://localhost:9931/metrics
 ```

 ### Using the CI python script
diff --git a/tools/server/bench/bench.py b/tools/server/bench/bench.py
index 2c56ab5eb..82cb483b3 100644
--- a/tools/server/bench/bench.py
+++ b/tools/server/bench/bench.py
@@ -28,7 +28,7 @@ def main(args_in: list[str] | None = None) -> None:
     parser.add_argument("--branch", type=str, help="Branch name", default="detached")
     parser.add_argument("--commit", type=str, help="Commit name", default="dirty")
     parser.add_argument("--host", type=str, help="Server listen host", default="0.0.0.0")
-    parser.add_argument("--port", type=int, help="Server listen host", default="8080")
+    parser.add_argument("--port", type=int, help="Server listen host", default="9931")
     parser.add_argument("--model-path-prefix", type=str, help="Prefix where to store the model files", default="models")
     parser.add_argument("--n-prompts", type=int,
                         help="SERVER_BENCH_N_PROMPTS: total prompts to randomly select in the benchmark", required=True)
diff --git a/tools/server/bench/prometheus.yml b/tools/server/bench/prometheus.yml
index b15ee5244..92ad20726 100644
--- a/tools/server/bench/prometheus.yml
+++ b/tools/server/bench/prometheus.yml
@@ -6,4 +6,4 @@ global:
 scrape_configs:
   - job_name: 'llama.cpp server'
     static_configs:
-      - targets: ['localhost:8080']
+      - targets: ['localhost:9931']
diff --git a/tools/server/bench/script.js b/tools/server/bench/script.js
index 2772bee5e..a5eb73a4d 100644
--- a/tools/server/bench/script.js
+++ b/tools/server/bench/script.js
@@ -5,7 +5,7 @@ import {Counter, Rate, Trend} from 'k6/metrics'
 import exec from 'k6/execution';

 // Server chat completions prefix
-const server_url = __ENV.SERVER_BENCH_URL ? __ENV.SERVER_BENCH_URL : 'http://localhost:8080/v1'
+const server_url = __ENV.SERVER_BENCH_URL ? __ENV.SERVER_BENCH_URL : 'http://localhost:9931/v1'

 // Number of total prompts in the dataset - default 10m / 10 seconds/request * number of users
 const n_prompt = __ENV.SERVER_BENCH_N_PROMPTS ? parseInt(__ENV.SERVER_BENCH_N_PROMPTS) : 600 / 10 * 8
diff --git a/tools/server/bench/speed-bench/README.md b/tools/server/bench/speed-bench/README.md
index 8d3fcd804..a8455e2d5 100644
--- a/tools/server/bench/speed-bench/README.md
+++ b/tools/server/bench/speed-bench/README.md
@@ -18,7 +18,7 @@ The client does not launch the server, so start `llama-server` yourself first. I
 llama-server \
   -m target.gguf \
   -c 8192 \
-  --port 8080 \
+  --port 9931 \
   -ngl 99 -fa on \
   --np 1 \
   --jinja
@@ -30,7 +30,7 @@ For speculative decoding, start the server with the appropriate flags for your s

 ```bash
 python tools/server/bench/speed-bench/speed_bench.py \
-  --url localhost:8080 \
+  --url localhost:9931 \
   --bench qualitative \
   --category coding \
   --osl 1024 \
@@ -41,7 +41,7 @@ python tools/server/bench/speed-bench/speed_bench.py \

 | Option | Default | Description |
 | --- | --- | --- |
-| `--url` | `localhost:8080` | Server URL. The scheme and `/v1` are optional and a trailing slash is fine, so `localhost:8080` and `http://localhost:8080/v1/` both work. |
+| `--url` | `localhost:9931` | Server URL. The scheme and `/v1` are optional and a trailing slash is fine, so `localhost:9931` and `http://localhost:9931/v1/` both work. |
 | `--model` | none | Optional `model` field sent in each request. |
 | `--bench` | `qualitative` | SPEED-Bench config, e.g. `qualitative`, `throughput_1k`. See [available dataset variants](https://github.com/ai-dynamo/aiperf/blob/main/docs/tutorials/speed-bench.md#available-dataset-variants). |
 | `--category` | `all` | Category filter within the bench; comma-separated list or `all`. For `qualitative` the categories are `coding`, `humanities`, `math`, `multilingual`, `qa`, `rag`, `reasoning`, `roleplay`, `stem`, `summarization`, `writing`. For the `throughput_{ISL}` splits they are `high_entropy`, `low_entropy`, `mixed`. |
@@ -81,7 +81,7 @@ First, start a plain `llama-server` (no speculative decoding) and save a baselin

 ```bash
 python tools/server/bench/speed-bench/speed_bench.py \
-  --url localhost:8080 \
+  --url localhost:9931 \
   --bench qualitative \
   --category all \
   --osl 1024 \
@@ -93,7 +93,7 @@ Then restart `llama-server` with speculative decoding enabled and save another r

 ```bash
 python tools/server/bench/speed-bench/speed_bench.py \
-  --url localhost:8080 \
+  --url localhost:9931 \
   --bench qualitative \
   --category all \
   --osl 1024 \
diff --git a/tools/server/bench/speed-bench/speed_bench.py b/tools/server/bench/speed-bench/speed_bench.py
index adb378a6b..83d87227a 100644
--- a/tools/server/bench/speed-bench/speed_bench.py
+++ b/tools/server/bench/speed-bench/speed_bench.py
@@ -369,7 +369,7 @@ def save_output(path: str, args: argparse.Namespace, samples: list[Sample], resu

 def main(argv: list[str] | None = None) -> int:
     parser = argparse.ArgumentParser(description="Run SPEED-Bench against an OpenAI-compatible llama-server.")
-    parser.add_argument("--url", default="localhost:8080", help="Server URL, for example localhost:8080 or http://localhost:8080/v1")
+    parser.add_argument("--url", default="localhost:9931", help="Server URL, for example localhost:9931 or http://localhost:9931/v1")
     parser.add_argument("--model", default=None, help="Optional model name to send in OpenAI requests")
     parser.add_argument("--bench", default="qualitative", help="SPEED-Bench config to run, for example qualitative or throughput_1k")
     parser.add_argument("--category", default="all", help="Category to run within the selected bench; use all for no category filter")
diff --git a/tools/server/server-http.h b/tools/server/server-http.h
index 4554b20f4..63fd2c186 100644
--- a/tools/server/server-http.h
+++ b/tools/server/server-http.h
@@ -75,7 +75,7 @@ struct server_http_context {
     mutable std::unordered_map<std::string, handler_t> handlers;

     std::string path_prefix;
-    int port    = 8080;
+    int port    = 9931;
     bool is_ssl = false;

     server_http_context();
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index 6e2d8ff9d..871e827be 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -521,16 +521,8 @@ int llama_server(common_params & params, int argc, char ** argv, server_child &
 #endif
     }

-    bool uses_default_port = false;
     for (const auto & address : ctx_http.listening_addresses) {
         SRV_INF("listening on %s\n", address.c_str());
-        uses_default_port |= string_ends_with(address, ":8080");
-    }
-
-    // TODO: remove this in the future
-    // check the string to also handle the .sock case
-    if (uses_default_port) {
-        SRV_WRN("%s", "notice: server default port will be changed to :9931 in a future release (ref: https://github.com/ggml-org/llama.cpp/pull/26508)\n");
     }

     if (is_router_server) {
diff --git a/tools/ui/README.md b/tools/ui/README.md
index 04a6ea016..2ecbc8f51 100644
--- a/tools/ui/README.md
+++ b/tools/ui/README.md
@@ -118,7 +118,7 @@ This starts:
 - **Vite dev server** at `http://localhost:5173` - The main UI frontend app
 - **Storybook** at `http://localhost:6006` - Component documentation

-The Vite dev server proxies API requests to `SERVER_ORIGIN` (with fallback to default llama-server `8080` port):
+The Vite dev server proxies API requests to `SERVER_ORIGIN` (with fallback to default llama-server `9931` port):

 ```typescript
 // vite.config.ts proxy configuration
diff --git a/tools/ui/vite.config.ts b/tools/ui/vite.config.ts
index 0f24a600b..233ba7452 100644
--- a/tools/ui/vite.config.ts
+++ b/tools/ui/vite.config.ts
@@ -26,7 +26,7 @@ const browserBaseConfig: any = {

 export default defineConfig(({ mode }) => {
 	const env = loadEnv(mode, process.cwd(), 'VITE_PUBLIC_');
-	const SERVER_ORIGIN = env.VITE_PUBLIC_SERVER_ORIGIN || 'http://localhost:8080';
+	const SERVER_ORIGIN = env.VITE_PUBLIC_SERVER_ORIGIN || 'http://localhost:9931';

 	return {
 		build: {