server: Add support for binding to multiple addresses (#28690)

* Add support for binding llama-server to multiple addresses

Assisted-by: Codex

* remove redundant thread handler

* make it clear about overlapping addr

* reject --port 0 with multiple tcp addr

* improve arg handler

* nits

* fix test

* nits 2

* nits

* nits 2

---------

Co-authored-by: Xuan Son Nguyen <[email protected]>
This commit is contained in:
Emanuil Rusev
2026-09-22 15:16:40 +02:00
committed by GitHub
co-authored by Xuan Son Nguyen
parent 828fdf282e
commit 217f81c266
8 changed files with 218 additions and 85 deletions
+20 -21
View File
@@ -111,7 +111,9 @@ int llama_server(int argc, char ** argv) {
llama_backend_init();
llama_numa_init(params.numa);
return llama_server(params, argc, argv);
const int result = llama_server(params, argc, argv);
common_log_flush(common_log_main());
return result;
}
int llama_server(common_params & params, int argc, char ** argv) {
@@ -183,12 +185,6 @@ int llama_server(common_params & params, int argc, char ** argv) {
// struct that contains llama context and inference
server_context ctx_server;
server_http_context ctx_http;
if (!ctx_http.init(params)) {
SRV_ERR("%s", "failed to initialize HTTP server\n");
return 1;
}
//
// Router
//
@@ -199,6 +195,13 @@ int llama_server(common_params & params, int argc, char ** argv) {
server_tools tools;
std::optional<server_models_routes> models_routes{};
server_http_context ctx_http;
if (!ctx_http.init(params)) {
SRV_ERR("%s", "failed to initialize HTTP server\n");
return 1;
}
if (is_router_server) {
// setup server instances manager
try {
@@ -438,9 +441,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
} catch (const std::exception & e) {
SRV_ERR("failed to load models on startup: %s\n", e.what());
ctx_http.stop();
if (ctx_http.thread.joinable()) {
ctx_http.thread.join();
}
ctx_http.join();
clean_up();
return 1;
}
@@ -473,9 +474,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
if (!ctx_server.load_model(params)) {
clean_up();
if (ctx_http.thread.joinable()) {
ctx_http.thread.join();
}
ctx_http.join();
SRV_ERR("%s", "exiting due to model loading error\n");
return 1;
}
@@ -509,11 +508,15 @@ int llama_server(common_params & params, int argc, char ** argv) {
#endif
}
SRV_INF("listening on %s\n", ctx_http.listening_address.c_str());
bool uses_default_port = false;
for (const auto & address : ctx_http.listening_addresses) {
SRV_INF("listening on %s\n", address.c_str());
uses_default_port |= string_ends_with(address, ":8080");
}
// TODO: remove this in the future
// check the string to also handle the .sock case
if (string_ends_with(ctx_http.listening_address, ":8080")) {
if (uses_default_port) {
SRV_WRN("%s", "notice: server default port will be changed to :9931 in a future release (ref: https://github.com/ggml-org/llama.cpp/pull/26508)\n");
}
@@ -523,9 +526,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
SRV_WRN("%s", " please only use presets that you can trust! Unknown presets may be unsafe\n");
}
if (ctx_http.thread.joinable()) {
ctx_http.thread.join(); // keep the main thread alive
}
ctx_http.join(); // keep the main thread alive
// when the HTTP server stops, clean up and exit
clean_up();
@@ -541,9 +542,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
ctx_server.start_loop();
clean_up();
if (ctx_http.thread.joinable()) {
ctx_http.thread.join();
}
ctx_http.join();
if (monitor_thread.joinable()) {
monitor_thread.join();
}