mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-27 21:46:57 +02:00
server: Add support for binding to multiple addresses (#28690)
* Add support for binding llama-server to multiple addresses Assisted-by: Codex * remove redundant thread handler * make it clear about overlapping addr * reject --port 0 with multiple tcp addr * improve arg handler * nits * fix test * nits 2 * nits * nits 2 --------- Co-authored-by: Xuan Son Nguyen <[email protected]>
This commit is contained in:
co-authored by
Xuan Son Nguyen
parent
828fdf282e
commit
217f81c266
+20
-21
@@ -111,7 +111,9 @@ int llama_server(int argc, char ** argv) {
|
||||
llama_backend_init();
|
||||
llama_numa_init(params.numa);
|
||||
|
||||
return llama_server(params, argc, argv);
|
||||
const int result = llama_server(params, argc, argv);
|
||||
common_log_flush(common_log_main());
|
||||
return result;
|
||||
}
|
||||
|
||||
int llama_server(common_params & params, int argc, char ** argv) {
|
||||
@@ -183,12 +185,6 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
||||
// struct that contains llama context and inference
|
||||
server_context ctx_server;
|
||||
|
||||
server_http_context ctx_http;
|
||||
if (!ctx_http.init(params)) {
|
||||
SRV_ERR("%s", "failed to initialize HTTP server\n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
//
|
||||
// Router
|
||||
//
|
||||
@@ -199,6 +195,13 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
||||
server_tools tools;
|
||||
|
||||
std::optional<server_models_routes> models_routes{};
|
||||
|
||||
server_http_context ctx_http;
|
||||
if (!ctx_http.init(params)) {
|
||||
SRV_ERR("%s", "failed to initialize HTTP server\n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
if (is_router_server) {
|
||||
// setup server instances manager
|
||||
try {
|
||||
@@ -438,9 +441,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
||||
} catch (const std::exception & e) {
|
||||
SRV_ERR("failed to load models on startup: %s\n", e.what());
|
||||
ctx_http.stop();
|
||||
if (ctx_http.thread.joinable()) {
|
||||
ctx_http.thread.join();
|
||||
}
|
||||
ctx_http.join();
|
||||
clean_up();
|
||||
return 1;
|
||||
}
|
||||
@@ -473,9 +474,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
||||
|
||||
if (!ctx_server.load_model(params)) {
|
||||
clean_up();
|
||||
if (ctx_http.thread.joinable()) {
|
||||
ctx_http.thread.join();
|
||||
}
|
||||
ctx_http.join();
|
||||
SRV_ERR("%s", "exiting due to model loading error\n");
|
||||
return 1;
|
||||
}
|
||||
@@ -509,11 +508,15 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
||||
#endif
|
||||
}
|
||||
|
||||
SRV_INF("listening on %s\n", ctx_http.listening_address.c_str());
|
||||
bool uses_default_port = false;
|
||||
for (const auto & address : ctx_http.listening_addresses) {
|
||||
SRV_INF("listening on %s\n", address.c_str());
|
||||
uses_default_port |= string_ends_with(address, ":8080");
|
||||
}
|
||||
|
||||
// TODO: remove this in the future
|
||||
// check the string to also handle the .sock case
|
||||
if (string_ends_with(ctx_http.listening_address, ":8080")) {
|
||||
if (uses_default_port) {
|
||||
SRV_WRN("%s", "notice: server default port will be changed to :9931 in a future release (ref: https://github.com/ggml-org/llama.cpp/pull/26508)\n");
|
||||
}
|
||||
|
||||
@@ -523,9 +526,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
||||
SRV_WRN("%s", " please only use presets that you can trust! Unknown presets may be unsafe\n");
|
||||
}
|
||||
|
||||
if (ctx_http.thread.joinable()) {
|
||||
ctx_http.thread.join(); // keep the main thread alive
|
||||
}
|
||||
ctx_http.join(); // keep the main thread alive
|
||||
|
||||
// when the HTTP server stops, clean up and exit
|
||||
clean_up();
|
||||
@@ -541,9 +542,7 @@ int llama_server(common_params & params, int argc, char ** argv) {
|
||||
ctx_server.start_loop();
|
||||
|
||||
clean_up();
|
||||
if (ctx_http.thread.joinable()) {
|
||||
ctx_http.thread.join();
|
||||
}
|
||||
ctx_http.join();
|
||||
if (monitor_thread.joinable()) {
|
||||
monitor_thread.join();
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user