Compare commits

...
Author SHA1 Message Date
Pascal a2ec6b0fff server: do not advertise a connect code without the tunnel
--connect-code on its own had no effect other than being served in
/props, so the Web UI showed a remote access code that no tunnel was
serving. Clear it when --connect is off, and say so once at startup.

The teardown comment claimed that terminate() closes the child stdout.
It does not: the log thread leaves fgets() only once every writer of
that pipe is gone, so a descendant of llama-connect would block the
join forever. Describe the real contract instead.

Also restore the --reasoning-preserve default in the generated docs.
2026-09-03 10:12:11 +02:00
Xuan Son Nguyen a75348a2ae fix windows build 2026-09-03 00:59:39 +02:00
Xuan Son Nguyen 339444c12a nits 2026-09-03 00:50:00 +02:00
Xuan Son Nguyen e9df25130c add props.connect_code 2026-09-03 00:44:11 +02:00
Xuan Son Nguyen a6cf7b8f18 make it clear that it's experimental 2026-09-03 00:29:12 +02:00
Xuan Son Nguyen a3f62cc133 wire it up 2026-09-03 00:22:15 +02:00
Xuan Son Nguyen dda508de49 wip 2026-09-02 23:28:51 +02:00
Xuan Son Nguyen 8209b1ca06 init 2026-09-02 23:10:48 +02:00
16 changed files with 617 additions and 51 deletions
+2 -1
View File
@@ -30,7 +30,7 @@ on:
env:
GH_TOKEN: ${{ github.token }}
BRANCH_NAME: ${{ github.head_ref || github.ref_name }}
CMAKE_ARGS: "-DLLAMA_BUILD_EXAMPLES=OFF -DLLAMA_BUILD_TESTS=OFF -DLLAMA_BUILD_TOOLS=ON -DLLAMA_BUILD_SERVER=ON -DGGML_RPC=ON"
CMAKE_ARGS: "-DLLAMA_BUILD_EXAMPLES=OFF -DLLAMA_BUILD_TESTS=OFF -DLLAMA_BUILD_TOOLS=ON -DLLAMA_BUILD_SERVER=ON -DGGML_RPC=ON -DLLAMA_CONNECT=ON"
# note: run this workflow one at a time for better cache reuse
concurrency:
@@ -1243,6 +1243,7 @@ jobs:
-DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \
-DLLAMA_OPENSSL=OFF \
-DGGML_NATIVE=OFF \
-DLLAMA_CONNECT=ON \
-DGGML_SYCL_F16=${{ matrix.fp16 }}
time cmake --build build --config Release -j $(nproc)
+1
View File
@@ -136,6 +136,7 @@ option(LLAMA_BUILD_SERVER "llama: build server example"
option(LLAMA_BUILD_APP "llama: build the unified binary" ${LLAMA_STANDALONE})
option(LLAMA_BUILD_UI "llama: build the embedded Web UI for server" OFF)
option(LLAMA_USE_PREBUILT_UI "llama: use prebuilt UI from HF Bucket when available" ON)
option(LLAMA_CONNECT "llama: include the llama-connect tool (downloads a prebuilt binary)" OFF)
option(LLAMA_TOOLS_INSTALL "llama: install tools" ${LLAMA_TOOLS_INSTALL_DEFAULT})
option(LLAMA_TESTS_INSTALL "llama: install tests" ON)
+20
View File
@@ -3491,6 +3491,26 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.ui = value;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_UI"));
add_opt(common_arg(
{"--connect"},
string_format("(experimental) open a peer-to-peer tunnel for server using llama-connect (default: %s)", params.server_connect ? "enabled" : "disabled"),
[](common_params & params) {
params.server_connect = true;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_CONNECT"));
add_opt(common_arg(
{"--connect-code"}, "CODE",
"manually specify a 40-character code for --connect (default: generate a new one for each run)",
[](common_params & params, const std::string & value) {
std::string val = value;
string_replace_all(val, "-", "");
string_replace_all(val, " ", "");
if (val.size() != 40) {
throw std::invalid_argument(string_format("error: invalid connect code '%s', must be 40 characters\n", value.c_str()));
}
params.server_connect_code = val;
}
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_CONNECT_CODE"));
add_opt(common_arg(
{"--embedding", "--embeddings"},
string_format("restrict to only support embedding use case; use only with dedicated embedding models (default: %s)", params.embedding ? "enabled" : "disabled"),
+4
View File
@@ -631,6 +631,10 @@ struct common_params {
int32_t checkpoint_min_step = 8192; // minimum spacing between context checkpoints
int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.
// llama-connect params
bool server_connect = false;
std::string server_connect_code = "";
std::string hostname = "127.0.0.1";
std::string public_path = ""; // NOLINT
std::string api_prefix = ""; // NOLINT
+86
View File
@@ -0,0 +1,86 @@
# download the prebuilt llama-connect binary for the given platform
# the release assets are made by: https://github.com/ggml-org/llama-connect/blob/master/.github/workflows/build.yml
cmake_minimum_required(VERSION 3.18)
set(REPO "" CACHE STRING "GitHub repository to download from (owner/name)")
set(VERSION "" CACHE STRING "Release tag to download, or 'latest'")
set(PLATFORM "" CACHE STRING "Platform suffix of the release asset (ex: linux-x64)")
set(OUT_DIR "" CACHE STRING "Directory to place the binary into")
set(WORK_DIR "" CACHE STRING "Scratch directory used for the download")
if (PLATFORM MATCHES "^win-")
set(ASSET "llama-connect-${PLATFORM}.zip")
set(BINARY "llama-connect.exe")
else()
set(ASSET "llama-connect-${PLATFORM}.tar.gz")
set(BINARY "llama-connect")
endif()
if (VERSION STREQUAL "latest")
set(URL "https://github.com/${REPO}/releases/latest/download/${ASSET}")
else()
set(URL "https://github.com/${REPO}/releases/download/${VERSION}/${ASSET}")
endif()
set(DST "${OUT_DIR}/${BINARY}")
set(ARCHIVE "${WORK_DIR}/${ASSET}")
set(STAMP "${WORK_DIR}/${BINARY}.url")
# skip if already downloaded from the same URL
if (EXISTS "${DST}" AND EXISTS "${STAMP}")
file(READ "${STAMP}" prev_url)
if (prev_url STREQUAL "${URL}")
return()
endif()
endif()
file(REMOVE "${STAMP}")
file(MAKE_DIRECTORY "${WORK_DIR}")
message(STATUS "llama-connect: downloading ${URL}")
# retry a few times, this runs in CI where a transient failure must not break the build
foreach (attempt RANGE 1 3)
file(DOWNLOAD "${URL}" "${ARCHIVE}" STATUS status TLS_VERIFY ON INACTIVITY_TIMEOUT 60)
list(GET status 0 code)
list(GET status 1 msg)
if (code EQUAL 0)
break()
endif()
message(STATUS "llama-connect: download attempt ${attempt} failed (${code}: ${msg})")
file(REMOVE "${ARCHIVE}")
endforeach()
if (NOT code EQUAL 0)
file(REMOVE "${ARCHIVE}")
message(FATAL_ERROR
"llama-connect: failed to download ${URL} (${code}: ${msg})\n"
" use -DLLAMA_CONNECT_VERSION=<tag> to select another release, "
"or -DLLAMA_CONNECT=OFF to skip this tool")
endif()
set(EXTRACT_DIR "${WORK_DIR}/extract")
file(REMOVE_RECURSE "${EXTRACT_DIR}")
file(MAKE_DIRECTORY "${EXTRACT_DIR}")
file(ARCHIVE_EXTRACT INPUT "${ARCHIVE}" DESTINATION "${EXTRACT_DIR}")
if (NOT EXISTS "${EXTRACT_DIR}/${BINARY}")
message(FATAL_ERROR "llama-connect: ${ASSET} does not contain ${BINARY}")
endif()
file(MAKE_DIRECTORY "${OUT_DIR}")
file(COPY "${EXTRACT_DIR}/${BINARY}" DESTINATION "${OUT_DIR}"
FILE_PERMISSIONS
OWNER_READ OWNER_WRITE OWNER_EXECUTE
GROUP_READ GROUP_EXECUTE
WORLD_READ WORLD_EXECUTE
)
file(WRITE "${STAMP}" "${URL}")
message(STATUS "llama-connect: ready at ${DST}")
+3
View File
@@ -32,6 +32,9 @@ else()
if (GGML_RPC)
add_subdirectory(rpc)
endif()
if (LLAMA_CONNECT)
add_subdirectory(connect)
endif()
if (NOT GGML_BACKEND_DL AND GGML_CPU)
# these tools use backends directly (no dynamic loading) and depend on CPU backend symbols
add_subdirectory(cvector-generator)
+85
View File
@@ -0,0 +1,85 @@
# llama-connect is developed in a separate repo, nothing is compiled here
# we only download the prebuilt binary, see scripts/connect-download.cmake
set(TARGET llama-connect)
set(LLAMA_CONNECT_REPO "ggml-org/llama-connect" CACHE STRING "llama-connect: GitHub repository to download from")
set(LLAMA_CONNECT_VERSION "latest" CACHE STRING "llama-connect: release tag to download (ex: v0.0.1), or 'latest'")
# map the target platform to the release asset name
set(LLAMA_CONNECT_PLATFORM "")
if (APPLE)
set(_arch "${CMAKE_SYSTEM_PROCESSOR}")
if (CMAKE_OSX_ARCHITECTURES)
list(LENGTH CMAKE_OSX_ARCHITECTURES _narch)
if (_narch EQUAL 1)
set(_arch "${CMAKE_OSX_ARCHITECTURES}")
else()
set(_arch "universal") # no such asset, so the tool is skipped below
endif()
endif()
# note: there is no macos-x64 asset, it is not built upstream (no Intel macOS runners)
if (_arch MATCHES "^(arm64|aarch64)$")
set(LLAMA_CONNECT_PLATFORM "macos-arm64")
endif()
elseif (WIN32)
# with Visual Studio generators, the target arch comes from -A
set(_arch "${CMAKE_SYSTEM_PROCESSOR}")
if (CMAKE_GENERATOR_PLATFORM)
set(_arch "${CMAKE_GENERATOR_PLATFORM}")
endif()
if (_arch MATCHES "^(x86_64|AMD64|amd64|x64)$")
set(LLAMA_CONNECT_PLATFORM "win-x64")
endif()
elseif (CMAKE_SYSTEM_NAME STREQUAL "Linux")
if (CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64|amd64|x64)$")
set(LLAMA_CONNECT_PLATFORM "linux-x64")
elseif (CMAKE_SYSTEM_PROCESSOR MATCHES "^(arm64|aarch64)$")
set(LLAMA_CONNECT_PLATFORM "linux-arm64")
endif()
endif()
if (LLAMA_CONNECT_PLATFORM STREQUAL "")
message(WARNING "llama-connect: no prebuilt binary for ${CMAKE_SYSTEM_NAME}/${CMAKE_SYSTEM_PROCESSOR}, skipping")
return()
endif()
if (LLAMA_CONNECT_PLATFORM MATCHES "^win-")
set(LLAMA_CONNECT_BINARY "llama-connect.exe")
else()
set(LLAMA_CONNECT_BINARY "llama-connect")
endif()
# multi-config generators put the binaries in a per-config subdir
get_property(LLAMA_CONNECT_MULTI_CONFIG GLOBAL PROPERTY GENERATOR_IS_MULTI_CONFIG)
if (LLAMA_CONNECT_MULTI_CONFIG)
set(LLAMA_CONNECT_OUT_DIR "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/$<CONFIG>")
else()
set(LLAMA_CONNECT_OUT_DIR "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}")
endif()
# the script is a no-op if the binary is already downloaded
add_custom_target(${TARGET} ALL
BYPRODUCTS "${LLAMA_CONNECT_OUT_DIR}/${LLAMA_CONNECT_BINARY}"
COMMAND ${CMAKE_COMMAND}
"-DREPO=${LLAMA_CONNECT_REPO}"
"-DVERSION=${LLAMA_CONNECT_VERSION}"
"-DPLATFORM=${LLAMA_CONNECT_PLATFORM}"
"-DOUT_DIR=${LLAMA_CONNECT_OUT_DIR}"
"-DWORK_DIR=${CMAKE_CURRENT_BINARY_DIR}"
-P "${PROJECT_SOURCE_DIR}/scripts/connect-download.cmake"
COMMENT "Downloading llama-connect (${LLAMA_CONNECT_PLATFORM}, ${LLAMA_CONNECT_VERSION})"
VERBATIM
)
# building llama-server should also pull the binary (for convenience)
if (TARGET llama-server)
add_dependencies(llama-server ${TARGET})
endif()
if (LLAMA_TOOLS_INSTALL)
install(PROGRAMS "${LLAMA_CONNECT_OUT_DIR}/${LLAMA_CONNECT_BINARY}" TYPE BIN)
endif()
+2
View File
@@ -39,6 +39,8 @@ set(TARGET llama-server-impl)
add_library(${TARGET}
server.cpp
server-connect.cpp
server-connect.h
server-http.cpp
server-http.h
server-models.cpp
+2
View File
@@ -209,6 +209,8 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--mcp-servers-json JSON` | experimental: inline JSON with MCP server definitions (Cursor-compatible format) - do not enable in untrusted environments (default: none)<br/>note: for security reasons, this will limit --cors-origins to localhost by default<br/>(env: LLAMA_ARG_MCP_SERVERS_JSON) |
| `-ag, --agent, -no-ag, --no-agent` | whether to enable CORS proxy and all built-in tools - do not enable in untrusted environments (default: disabled)<br/>note: for security reasons, this will limit --cors-origins to localhost by default<br/>(env: LLAMA_ARG_AGENT) |
| `--ui, --webui, --no-ui, --no-webui` | whether to enable the Web UI (default: enabled)<br/>(env: LLAMA_ARG_UI) |
| `--connect` | (experimental) open a peer-to-peer tunnel for server using llama-connect (default: disabled)<br/>(env: LLAMA_ARG_CONNECT) |
| `--connect-code CODE` | manually specify a 40-character code for --connect (default: generate a new one for each run)<br/>(env: LLAMA_ARG_CONNECT_CODE) |
| `--embedding, --embeddings` | restrict to only support embedding use case; use only with dedicated embedding models (default: disabled)<br/>(env: LLAMA_ARG_EMBEDDINGS) |
| `--rerank, --reranking` | enable reranking endpoint on server (default: disabled)<br/>(env: LLAMA_ARG_RERANKING) |
| `--api-key KEY` | API key to use for authentication, multiple keys can be provided as a comma-separated list (default: none)<br/>(env: LLAMA_API_KEY) |
+55
View File
@@ -9,6 +9,7 @@
#include "server-common.h"
#include <filesystem>
#include <random>
#include <sstream>
#include <fstream>
@@ -16,6 +17,60 @@
#include <cstring>
#include <type_traits>
#if defined(_WIN32)
#define WIN32_LEAN_AND_MEAN
#ifndef NOMINMAX
# define NOMINMAX
#endif
#include <windows.h>
#elif defined(__APPLE__) && defined(__MACH__)
#include <mach-o/dyld.h>
#include <limits.h>
#else
#include <unistd.h>
#endif
std::filesystem::path get_server_exec_path() {
#if defined(_WIN32)
wchar_t buf[32768] = { 0 }; // Large buffer to handle long paths
DWORD len = GetModuleFileNameW(nullptr, buf, _countof(buf));
if (len == 0 || len >= _countof(buf)) {
throw std::runtime_error("GetModuleFileNameW failed or path too long");
}
return std::filesystem::path(buf);
#elif defined(__APPLE__) && defined(__MACH__)
char small_path[PATH_MAX];
uint32_t size = sizeof(small_path);
if (_NSGetExecutablePath(small_path, &size) == 0) {
// resolve any symlinks to get absolute path
try {
return std::filesystem::canonical(std::filesystem::path(small_path));
} catch (...) {
return std::filesystem::path(small_path);
}
} else {
// buffer was too small, allocate required size and call again
std::vector<char> buf(size);
if (_NSGetExecutablePath(buf.data(), &size) == 0) {
try {
return std::filesystem::canonical(std::filesystem::path(buf.data()));
} catch (...) {
return std::filesystem::path(buf.data());
}
}
throw std::runtime_error("_NSGetExecutablePath failed after buffer resize");
}
#else
char path[FILENAME_MAX];
ssize_t count = readlink("/proc/self/exe", path, FILENAME_MAX);
if (count <= 0) {
throw std::runtime_error("failed to resolve /proc/self/exe");
}
return std::filesystem::path(std::string(path, count));
#endif
}
json format_error_response(const std::string & message, const enum error_type type) {
std::string type_str;
int code = 500;
+4
View File
@@ -13,6 +13,7 @@
#include <chrono>
#include <condition_variable>
#include <cinttypes>
#include <filesystem>
#include <functional>
#include <mutex>
#include <queue>
@@ -92,6 +93,9 @@ struct server_grammar_trigger {
json format_error_response(const std::string & message, const enum error_type type);
// path of the running llama-server binary, used to spawn siblings. throws on failure
std::filesystem::path get_server_exec_path();
//
// random string / id
//
+272
View File
@@ -0,0 +1,272 @@
#include "server-connect.h"
#include "server-common.h"
#include "subproc.h"
#include <cctype>
#include <chrono>
#include <exception>
#include <filesystem>
#include <random>
#include <string_view>
#include <system_error>
#include <thread>
#include <vector>
// share code = room code + pass code, must match llama-connect and the Web UI
// ref: https://github.com/ggml-org/llama-connect/blob/master/src/protocol.rs
static constexpr size_t CONNECT_ROOM_CODE_LEN = 8;
static constexpr size_t CONNECT_PASS_CODE_LEN = 32;
static const std::string CONNECT_CODE_CHARS = "ABCDEFGHJKMNPQRSTUVWXYZabcdefghjkmnpqrstuvwxyz23456789";
#if defined(_WIN32)
static const std::string CONNECT_EXE_NAME = "llama-connect.exe";
static constexpr char PATH_SEPARATOR = ';';
#else
static const std::string CONNECT_EXE_NAME = "llama-connect";
static constexpr char PATH_SEPARATOR = ':';
#endif
// how long to wait for the child to notice the closed stdin before killing it
static constexpr int CONNECT_STOP_TIMEOUT_MS = 3000;
// the pass code guards the tunnel, so do not use random_string(): its mt19937 is predictable
static std::string gen_share_code() {
std::random_device rd;
std::uniform_int_distribution<size_t> dist(0, CONNECT_CODE_CHARS.size() - 1);
std::string code(CONNECT_ROOM_CODE_LEN + CONNECT_PASS_CODE_LEN, ' ');
for (char & c : code) {
c = CONNECT_CODE_CHARS[dist(rd)];
}
return code;
}
static std::string format_share_code(const std::string & code) {
std::string out;
for (size_t i = 0; i < code.size(); i += 8) {
if (i > 0) {
out += ' ';
}
out += code.substr(i, 8);
}
return out;
}
// whitespace is tolerated, the Web UI shows the code in blocks and users paste it back
// returns an empty string if it is not a valid share code
static std::string normalize_share_code(const std::string & key) {
std::string code;
for (char c : key) {
if (!std::isspace((unsigned char) c)) {
code += c;
}
}
if (code.size() != CONNECT_ROOM_CODE_LEN + CONNECT_PASS_CODE_LEN) {
return "";
}
if (code.find_first_not_of(CONNECT_CODE_CHARS) != std::string::npos) {
return "";
}
return code;
}
// llama-connect logs as "[LEVEL] text", forward at the same level so our verbosity filter applies
static void forward_child_log(const std::string & line) {
static const std::pair<std::string_view, ggml_log_level> tags[] = {
{ "[ERROR] ", GGML_LOG_LEVEL_ERROR },
{ "[WARN] ", GGML_LOG_LEVEL_WARN },
{ "[INFO] ", GGML_LOG_LEVEL_INFO },
{ "[DEBUG] ", GGML_LOG_LEVEL_DEBUG },
{ "[TRACE] ", GGML_LOG_LEVEL_DEBUG },
};
// untagged lines are the startup banner, show them as info
ggml_log_level level = GGML_LOG_LEVEL_INFO;
const char * text = line.c_str();
for (const auto & [tag, tag_level] : tags) {
if (string_starts_with(line, tag)) {
level = tag_level;
text += tag.size();
break;
}
}
switch (level) {
case GGML_LOG_LEVEL_ERROR: LOG_ERR("connect | %s", text); break;
case GGML_LOG_LEVEL_WARN: LOG_WRN("connect | %s", text); break;
case GGML_LOG_LEVEL_DEBUG: LOG_DBG("connect | %s", text); break;
default: LOG_INF("connect | %s", text); break;
}
}
static bool path_is_file(const std::filesystem::path & p) {
std::error_code ec;
return std::filesystem::is_regular_file(p, ec);
}
std::string server_connect::find_binary() {
// prefer the copy shipped next to llama-server over an unrelated one in PATH
try {
auto sibling = get_server_exec_path().parent_path() / CONNECT_EXE_NAME;
if (path_is_file(sibling)) {
return sibling.string();
}
} catch (const std::exception & e) {
SRV_WRN("could not resolve the llama-server path (%s), looking for llama-connect in PATH only\n", e.what());
}
const std::string path_env = common_get_env("PATH");
size_t start = 0;
while (start <= path_env.size()) {
size_t end = path_env.find(PATH_SEPARATOR, start);
if (end == std::string::npos) {
end = path_env.size();
}
const std::string dir = path_env.substr(start, end - start);
if (!dir.empty()) {
auto candidate = std::filesystem::path(dir) / CONNECT_EXE_NAME;
if (path_is_file(candidate)) {
return candidate.string();
}
}
start = end + 1;
}
return "";
}
std::string server_connect::unavailable_reason(const common_params & params) {
if (!common_subproc::is_supported()) {
return "this build has subprocess support disabled, rebuild with -DLLAMA_SUBPROCESS=ON";
}
if (!params.server_connect_code.empty() && normalize_share_code(params.server_connect_code).empty()) {
return "--connect-code must be " + std::to_string(CONNECT_ROOM_CODE_LEN + CONNECT_PASS_CODE_LEN)
+ " characters from '" + CONNECT_CODE_CHARS + "'";
}
if (find_binary().empty()) {
return "could not find '" + CONNECT_EXE_NAME + "' next to llama-server or in PATH.\n"
" it is a separate binary: download it from https://github.com/ggml-org/llama-connect/releases\n"
" or build llama.cpp with -DLLAMA_CONNECT=ON to have it fetched automatically";
}
return "";
}
std::string server_connect::resolve_code(const std::string & code) {
return code.empty() ? gen_share_code() : normalize_share_code(code);
}
bool server_connect::start(const common_params & params) {
const std::string bin = find_binary();
if (bin.empty()) {
SRV_ERR("%s", "llama-connect binary not found\n");
return false;
}
// main() resolves this before anything can read it from /props
const std::string & code = params.server_connect_code;
GGML_ASSERT(!code.empty());
// always loopback, params.hostname may be 0.0.0.0 or a unix socket which the child cannot dial
const std::vector<std::string> args = {
bin,
"--host", "127.0.0.1",
"--port", std::to_string(params.port),
"--code", code,
// the kernel closes our end of this pipe even if we are killed without cleanup
// this is what stops the child from outliving us
"--exit-on-stdin-eof",
};
proc = std::make_unique<common_subproc>();
const int options = subprocess_option_no_window
| subprocess_option_combined_stdout_stderr
| subprocess_option_inherit_environment;
if (!proc->create(args, options)) {
SRV_ERR("failed to spawn '%s'\n", bin.c_str());
proc.reset();
return false;
}
log_thread = std::thread([this]() {
FILE * out = proc->stdout_file();
if (out == nullptr) {
SRV_ERR("%s", "failed to get stdout of the llama-connect process\n");
return;
}
std::vector<char> buf(4096);
while (fgets(buf.data(), (int) buf.size(), out) != nullptr) {
forward_child_log(buf.data());
}
// EOF means the child is gone
if (!stopping.load(std::memory_order_acquire)) {
SRV_ERR("%s", "llama-connect exited on its own, remote access is no longer available\n");
}
});
SRV_INF("%s", "-----------------\n");
SRV_INF("%s", "remote access is enabled via llama-connect\n");
SRV_INF("share code (enter it in the Web UI under Settings -> Remote Access): %s\n",
format_share_code(code).c_str());
SRV_WRN("%s", "anyone with this code can use this server, do not share it publicly\n");
SRV_INF("%s", "-----------------\n");
return true;
}
void server_connect::stop() {
if (!proc) {
return;
}
SRV_INF("%s", "stopping llama-connect...\n");
stopping.store(true, std::memory_order_release);
proc->close_stdin();
for (int elapsed = 0; elapsed < CONNECT_STOP_TIMEOUT_MS && proc->alive(); elapsed += 100) {
std::this_thread::sleep_for(std::chrono::milliseconds(100));
}
// no-op if the child already exited. the log thread leaves fgets() on EOF, which the kernel
// raises once every writer of the stdout pipe is gone, so llama-connect must not spawn
// children of its own: one of them holding that pipe blocks the join below forever
proc->terminate();
if (log_thread.joinable()) {
try {
log_thread.join();
} catch (const std::system_error & e) {
// ~thread() on a still-joinable thread calls std::terminate, detach instead
SRV_ERR("failed to join the llama-connect log thread: %s\n", e.what());
log_thread.detach();
}
}
proc->join(); // reap the zombie
proc.reset();
}
server_connect::server_connect() = default;
server_connect::~server_connect() {
try {
stop();
} catch (const std::exception & e) {
SRV_ERR("failed to stop llama-connect: %s\n", e.what());
} catch (...) {
SRV_ERR("%s", "failed to stop llama-connect\n");
}
}
+38
View File
@@ -0,0 +1,38 @@
#pragma once
#include "common.h"
#include <atomic>
#include <memory>
#include <string>
#include <thread>
struct common_subproc;
// spawns llama-connect, which exposes this server to a remote browser over WebRTC
// core binary is in a separate project: https://github.com/ggml-org/llama-connect
struct server_connect {
server_connect();
~server_connect();
server_connect(const server_connect &) = delete;
server_connect & operator=(const server_connect &) = delete;
// path of the llama-connect binary, empty if not found
static std::string find_binary();
// why --connect cannot work here, empty if it can
static std::string unavailable_reason(const common_params & params);
static std::string resolve_code(const std::string & code);
bool start(const common_params & params);
// idempotent, also called by the destructor
void stop();
private:
std::unique_ptr<common_subproc> proc;
std::thread log_thread;
std::atomic<bool> stopping{false}; // tells the log thread the exit is expected
};
+4
View File
@@ -4618,6 +4618,10 @@ static json get_res_props(const server_context_meta & meta, const common_params
{ "is_sleeping", is_sleeping },
{ "cors_proxy_enabled", params.ui_mcp_proxy },
};
if (!params.server_connect_code.empty()) {
props["connect_code"] = params.server_connect_code;
}
if (params.use_jinja) {
if (!tmpl_tools.empty()) {
props["chat_template_tool_use"] = tmpl_tools;
+8 -50
View File
@@ -24,7 +24,6 @@
#include <atomic>
#include <chrono>
#include <queue>
#include <filesystem>
#include <random>
#include <sstream>
#include <cstring>
@@ -33,12 +32,6 @@
extern char **environ;
#endif
#if defined(__APPLE__) && defined(__MACH__)
// macOS: use _NSGetExecutablePath to get the executable path
#include <mach-o/dyld.h>
#include <limits.h>
#endif
#define DEFAULT_STOP_TIMEOUT 10 // seconds
#define CMD_ROUTER_TO_CHILD_EXIT "cmd_router_to_child:exit"
@@ -256,47 +249,6 @@ struct server_lru_sched {
// delete). distinct from params.timeout_read/write which only applies to the generation proxy
static constexpr int STREAM_LOOKUP_TIMEOUT_MS = 250;
static std::filesystem::path get_server_exec_path() {
#if defined(_WIN32)
wchar_t buf[32768] = { 0 }; // Large buffer to handle long paths
DWORD len = GetModuleFileNameW(nullptr, buf, _countof(buf));
if (len == 0 || len >= _countof(buf)) {
throw std::runtime_error("GetModuleFileNameW failed or path too long");
}
return std::filesystem::path(buf);
#elif defined(__APPLE__) && defined(__MACH__)
char small_path[PATH_MAX];
uint32_t size = sizeof(small_path);
if (_NSGetExecutablePath(small_path, &size) == 0) {
// resolve any symlinks to get absolute path
try {
return std::filesystem::canonical(std::filesystem::path(small_path));
} catch (...) {
return std::filesystem::path(small_path);
}
} else {
// buffer was too small, allocate required size and call again
std::vector<char> buf(size);
if (_NSGetExecutablePath(buf.data(), &size) == 0) {
try {
return std::filesystem::canonical(std::filesystem::path(buf.data()));
} catch (...) {
return std::filesystem::path(buf.data());
}
}
throw std::runtime_error("_NSGetExecutablePath failed after buffer resize");
}
#else
char path[FILENAME_MAX];
ssize_t count = readlink("/proc/self/exe", path, FILENAME_MAX);
if (count <= 0) {
throw std::runtime_error("failed to resolve /proc/self/exe");
}
return std::filesystem::path(std::string(path, count));
#endif
}
static void unset_reserved_args(common_preset & preset, bool unset_model_args) {
preset.unset_option("LLAMA_ARG_SSL_KEY_FILE");
preset.unset_option("LLAMA_ARG_SSL_CERT_FILE");
@@ -305,6 +257,8 @@ static void unset_reserved_args(common_preset & preset, bool unset_model_args) {
preset.unset_option("LLAMA_ARG_MODELS_MAX");
preset.unset_option("LLAMA_ARG_MODELS_PRESET");
preset.unset_option("LLAMA_ARG_MODELS_AUTOLOAD");
preset.unset_option("LLAMA_ARG_CONNECT");
preset.unset_option("LLAMA_ARG_CONNECT_CODE");
if (unset_model_args) {
preset.unset_option("LLAMA_ARG_MODEL");
preset.unset_option("LLAMA_ARG_MMPROJ");
@@ -1875,7 +1829,7 @@ void server_models_routes::init_routes() {
if (name.empty()) {
// main instance
auto res = std::make_unique<server_http_res>();
res_ok(res, {
json props = {
// TODO: add support for this on web UI
{"role", "router"},
{"max_instances", params.models_max},
@@ -1891,7 +1845,11 @@ void server_models_routes::init_routes() {
{"ui_settings", ui_settings},
{"build_info", std::string(llama_build_info())},
{"cors_proxy_enabled", params.ui_mcp_proxy},
});
};
if (!params.server_connect_code.empty()) {
props["connect_code"] = params.server_connect_code;
}
res_ok(res, props);
return res;
}
return proxy_get(req);
+31
View File
@@ -1,3 +1,4 @@
#include "server-connect.h"
#include "server-context.h"
#include "server-http.h"
#include "server-models.h"
@@ -175,6 +176,20 @@ int llama_server(common_params & params, int argc, char ** argv) {
params.model_alias.insert(model_name);
}
// check early, so a missing llama-connect fails-fast
if (params.server_connect) {
const std::string reason = server_connect::unavailable_reason(params);
if (!reason.empty()) {
SRV_ERR("--connect is not available: %s\n", reason.c_str());
return 1;
}
params.server_connect_code = server_connect::resolve_code(params.server_connect_code);
} else if (!params.server_connect_code.empty()) {
// the code is only meaningful while the tunnel runs, and /props keys off it being set
SRV_WRN("%s", "--connect-code is ignored without --connect\n");
params.server_connect_code.clear();
}
// note: this is guaranteed to out-live ctx_http and tools
server_mcp mcp_mgr;
@@ -333,6 +348,10 @@ int llama_server(common_params & params, int argc, char ** argv) {
warn_names.push_back("router mode");
}
if (params.server_connect) {
warn_names.push_back("peer-to-peer tunnel (llama-connect, experimental)");
}
if (params.ui_mcp_proxy) {
ctx_http.get ("/cors-proxy", ex_wrapper(proxy_handler_get));
ctx_http.post("/cors-proxy", ex_wrapper(proxy_handler_post));
@@ -514,6 +533,18 @@ int llama_server(common_params & params, int argc, char ** argv) {
SRV_INF("listening on %s\n", ctx_http.listening_address.c_str());
// spawn only once listening, so the child health check passes. the destructor stops it
server_connect connect_proc;
if (params.server_connect && !connect_proc.start(params)) {
SRV_ERR("%s", "exiting due to llama-connect error\n");
ctx_http.stop();
if (ctx_http.thread.joinable()) {
ctx_http.thread.join();
}
clean_up();
return 1;
}
// TODO: remove this in the future
// check the string to also handle the .sock case
if (string_ends_with(ctx_http.listening_address, ":8080")) {