Commit a7fb71fab for llama.cpp
commit a7fb71fab83b474a0892b9a05aaa3a8ddca2729b
Author: Pascal <admin@serveurperso.com>
Date: Mon Oct 5 01:59:37 2026 +0200
log, server: self contained colors, split child commands from logs in router mode (#29895)
* log, server: make router child lines carry their own colors
The logger writes the color reset after the trailing newline, so the
reset opens the next line. On the shared pipe of a router child it lands
in front of the next state command, which the router then misses, and
the line break that works around it shows up as an empty log line on
every progress update.
The reset now goes before the trailing newlines, so every line is self
contained and the command goes back to its plain framing. The router
passes its effective color setting to its children, whose output ends
up in its terminal, and leaves that option out when comparing presets
on reload.
* log: enable ANSI colors on the Windows console
A Windows console renders ANSI sequences only in virtual terminal mode,
which nothing turns on for the logger, so llama-server prints raw escape
codes on the Windows 10 console while llama-cli, whose console code
enables it, shows colors. The logger now enables virtual terminal mode
on stdout and stderr when it turns colors on, and keeps colors off when
a console cannot render them. Pipes and files take the sequences as is.
* server: separate the router child commands from its logs
The child sent its state commands on the same pipe as its logs, so the
router had to pick them out of the log stream by a line prefix, and any
unterminated write in front of a command made the router miss it. This
resolves the TODO at the spawn that called for splitting stdout and
stderr.
The child now keeps stdout for the commands and points everything else
written to stdout at stderr, before anything is written. The router
reads both pipes, handles the commands from stdout and forwards stderr
as the log, and warns about any other line on the command pipe.
* server: address review from ngxson
The single server_child is now created first in the entry point and its
constructor keeps stdout for the commands, so the stream is a member of
the instance instead of a static, and init() is gone. The instance is
passed down to the server, while the CLI entry point creates its own.
* Update tools/server/server.cpp
---------
Co-authored-by: Xuan-Son Nguyen <thichthat@gmail.com>
diff --git a/common/common.cpp b/common/common.cpp
index 0907765ec..463137e5a 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -1065,6 +1065,20 @@ bool tty_can_use_colors() {
return common_is_tty(stdout) || common_is_tty(stderr);
}
+bool tty_enable_ansi() {
+#if defined(_WIN32)
+ // a Windows console renders ANSI sequences only in virtual terminal mode, pipes and files take them as is
+ for (DWORD id : { STD_OUTPUT_HANDLE, STD_ERROR_HANDLE }) {
+ HANDLE h = GetStdHandle(id);
+ DWORD mode = 0;
+ if (GetConsoleMode(h, &mode) && !SetConsoleMode(h, mode | ENABLE_VIRTUAL_TERMINAL_PROCESSING)) {
+ return false;
+ }
+ }
+#endif
+ return true;
+}
+
//
// Model utils
//
diff --git a/common/common.h b/common/common.h
index fa2ffd9e1..e1ef70a9e 100644
--- a/common/common.h
+++ b/common/common.h
@@ -940,6 +940,7 @@ void fs_write_atomic(const std::filesystem::path & path, const std::string & dat
// Auto-detect if colors can be enabled based on terminal and environment
bool tty_can_use_colors();
+bool tty_enable_ansi(); // false when stdout or stderr is a console that cannot render ANSI sequences
// Check if the given file is attached to a terminal
bool common_is_tty(FILE * file);
diff --git a/common/log.cpp b/common/log.cpp
index b60e8d268..86c98e8e6 100644
--- a/common/log.cpp
+++ b/common/log.cpp
@@ -144,12 +144,16 @@ struct common_log_entry {
}
}
- fprintf(fcur, "%s", msg.data());
+ // the reset goes before the trailing newlines, so that every line carries its own colors
+ const bool reset = level == GGML_LOG_LEVEL_WARN || level == GGML_LOG_LEVEL_ERROR || level == GGML_LOG_LEVEL_DEBUG;
- if (level == GGML_LOG_LEVEL_WARN || level == GGML_LOG_LEVEL_ERROR || level == GGML_LOG_LEVEL_DEBUG) {
- fprintf(fcur, "%s", g_col[COMMON_LOG_COL_DEFAULT]);
+ size_t end = strlen(msg.data());
+ while (end > 0 && msg[end - 1] == '\n') {
+ end--;
}
+ fprintf(fcur, "%.*s%s%s", (int) end, msg.data(), reset ? g_col[COMMON_LOG_COL_DEFAULT] : "", msg.data() + end);
+
fflush(fcur);
}
};
@@ -158,6 +162,7 @@ struct common_log {
// default capacity
common_log(size_t capacity = 512) {
file = nullptr;
+ colors = false;
prefix = false;
timestamps = false;
running = false;
@@ -185,6 +190,7 @@ private:
FILE * file;
+ bool colors;
bool prefix;
bool timestamps;
bool running;
@@ -394,10 +400,16 @@ public:
resume();
}
+ bool get_colors() const {
+ return colors;
+ }
+
void set_colors(bool colors) {
pause();
- if (colors) {
+ this->colors = colors && tty_enable_ansi();
+
+ if (this->colors) {
g_col[COMMON_LOG_COL_DEFAULT] = LOG_COL_DEFAULT;
g_col[COMMON_LOG_COL_BOLD] = LOG_COL_BOLD;
g_col[COMMON_LOG_COL_RED] = LOG_COL_RED;
@@ -500,6 +512,10 @@ void common_log_set_colors(struct common_log * log, log_colors colors) {
log->set_colors(true);
}
+bool common_log_get_colors(struct common_log * log) {
+ return log->get_colors();
+}
+
void common_log_set_prefix(struct common_log * log, bool prefix) {
log->set_prefix(prefix);
}
diff --git a/common/log.h b/common/log.h
index e36b09463..e9e1f1761 100644
--- a/common/log.h
+++ b/common/log.h
@@ -93,6 +93,7 @@ void common_log_add(struct common_log * log, enum ggml_log_level level, const ch
void common_log_set_file (struct common_log * log, const char * file); // not thread-safe
void common_log_set_colors (struct common_log * log, log_colors colors); // not thread-safe
+bool common_log_get_colors (struct common_log * log); // whether colors are enabled
void common_log_set_prefix (struct common_log * log, bool prefix); // whether to output prefix to each log
void common_log_set_timestamps(struct common_log * log, bool timestamps); // whether to output timestamps in the prefix
void common_log_flush (struct common_log * log); // flush all pending log messages
diff --git a/tools/server/server-common.cpp b/tools/server/server-common.cpp
index 2d90fcad2..be87c45f7 100644
--- a/tools/server/server-common.cpp
+++ b/tools/server/server-common.cpp
@@ -1885,38 +1885,71 @@ server_tokens format_prompt_rerank(
// server_subproc
//
+FILE * server_reserve_stdout() {
+ fflush(stdout);
+ // the reserved stream is not inherited, grandchildren get stdout and stderr of their own
+#ifdef _WIN32
+ int fd = _dup(_fileno(stdout));
+ GGML_ASSERT(fd >= 0);
+ SetHandleInformation((HANDLE) _get_osfhandle(fd), HANDLE_FLAG_INHERIT, 0);
+ _dup2(_fileno(stderr), _fileno(stdout));
+ SetStdHandle(STD_OUTPUT_HANDLE, GetStdHandle(STD_ERROR_HANDLE));
+ FILE * f = _fdopen(fd, "w");
+#else
+ int fd = fcntl(fileno(stdout), F_DUPFD_CLOEXEC, 0);
+ GGML_ASSERT(fd >= 0);
+ dup2(fileno(stderr), fileno(stdout));
+ FILE * f = fdopen(fd, "w");
+#endif
+ GGML_ASSERT(f);
+ return f;
+}
+
bool server_subproc::has_output() {
- if (out_handle >= 0) {
- return true;
- }
- FILE * f = sproc.stdout_file(); // combined stdout/stderr
- if (!f) {
- return false;
- }
+ FILE * files[SERVER_SUBPROC_STREAMS] = { sproc.stdout_file(), sproc.stderr_file() };
+ for (int i = 0; i < SERVER_SUBPROC_STREAMS; i++) {
+ if (out_handles[i] >= 0) {
+ continue;
+ }
+ if (!files[i]) {
+ return false;
+ }
#ifdef _WIN32
- HANDLE h = (HANDLE) _get_osfhandle(_fileno(f));
- if (h != INVALID_HANDLE_VALUE) {
- out_handle = (intptr_t) h;
- }
+ HANDLE h = (HANDLE) _get_osfhandle(_fileno(files[i]));
+ if (h == INVALID_HANDLE_VALUE) {
+ return false;
+ }
+ out_handles[i] = (intptr_t) h;
#else
- int fd = fileno(f);
- if (fd >= 0) {
+ int fd = fileno(files[i]);
+ if (fd < 0) {
+ return false;
+ }
fcntl(fd, F_SETFL, fcntl(fd, F_GETFL, 0) | O_NONBLOCK);
- out_handle = fd;
- }
+ out_handles[i] = fd;
#endif
- return out_handle >= 0;
+ }
+ return true;
+}
+
+bool server_subproc::output_closed() const {
+ return out_closed[SERVER_SUBPROC_STDOUT] && out_closed[SERVER_SUBPROC_STDERR];
}
-int server_subproc::read_output(char * buf, size_t len) {
+int server_subproc::read_output(server_subproc_stream stream, char * buf, size_t len) {
+ if (out_closed[stream]) {
+ return -1;
+ }
if (!has_output()) {
+ out_closed[stream] = true;
return -1;
}
#ifdef _WIN32
- HANDLE h = (HANDLE) out_handle;
+ HANDLE h = (HANDLE) out_handles[stream];
DWORD avail = 0;
if (!PeekNamedPipe(h, NULL, 0, NULL, &avail, NULL)) {
- return -1; // pipe broken, child gone
+ out_closed[stream] = true; // pipe broken, child gone
+ return -1;
}
if (avail == 0) {
return 0;
@@ -1924,24 +1957,23 @@ int server_subproc::read_output(char * buf, size_t len) {
DWORD to_read = avail < (DWORD) len ? avail : (DWORD) len;
DWORD got = 0;
if (!ReadFile(h, buf, to_read, &got, NULL) || got == 0) {
+ out_closed[stream] = true;
return -1;
}
return (int) got;
#else
while (true) {
- ssize_t r = read((int) out_handle, buf, len);
+ ssize_t r = read((int) out_handles[stream], buf, len);
if (r > 0) {
return (int) r;
}
- if (r == 0) {
- return -1; // EOF
- }
- if (errno == EINTR) {
+ if (r < 0 && errno == EINTR) {
continue;
}
- if (errno == EAGAIN || errno == EWOULDBLOCK) {
+ if (r < 0 && (errno == EAGAIN || errno == EWOULDBLOCK)) {
return 0;
}
+ out_closed[stream] = true; // EOF or error
return -1;
}
#endif
@@ -1979,11 +2011,15 @@ void server_subproc::waiter::wait(const std::vector<server_subproc *> & procs, s
// no waitable wait exists for anonymous pipes, so poll them in 50 ms steps
bool any = false;
for (size_t i = 0; i < procs.size(); i++) {
- DWORD avail = 0;
- if (!procs[i]->has_output() || !PeekNamedPipe((HANDLE) procs[i]->out_handle, NULL, 0, NULL, &avail, NULL) || avail > 0) {
- ready[i] = true; // data or broken pipe, read_output() tells which
- any = true;
+ server_subproc * p = procs[i];
+ ready[i] = !p->has_output();
+ for (int s = 0; s < SERVER_SUBPROC_STREAMS && !ready[i]; s++) {
+ DWORD avail = 0;
+ if (!p->out_closed[s] && (!PeekNamedPipe((HANDLE) p->out_handles[s], NULL, 0, NULL, &avail, NULL) || avail > 0)) {
+ ready[i] = true; // data or broken pipe, read_output() tells which
+ }
}
+ any = any || ready[i];
}
if (!any) {
int64_t step = timeout_ms < 0 ? 50 : std::min<int64_t>(timeout_ms, 50);
@@ -1991,10 +2027,13 @@ void server_subproc::waiter::wait(const std::vector<server_subproc *> & procs, s
}
#else
std::vector<pollfd> pfds;
- pfds.reserve(procs.size() + 1);
+ pfds.reserve(procs.size() * SERVER_SUBPROC_STREAMS + 1);
pfds.push_back({ (int) wake_fd[0], POLLIN, 0 });
for (auto * p : procs) {
- pfds.push_back({ p->has_output() ? (int) p->out_handle : -1, POLLIN, 0 }); // poll() skips negative fds
+ const bool open = p->has_output();
+ for (int s = 0; s < SERVER_SUBPROC_STREAMS; s++) {
+ pfds.push_back({ open && !p->out_closed[s] ? (int) p->out_handles[s] : -1, POLLIN, 0 }); // poll() skips negative fds
+ }
}
int timeout = timeout_ms < 0 ? -1 : (int) std::min<int64_t>(timeout_ms, std::numeric_limits<int>::max());
int r = poll(pfds.data(), pfds.size(), timeout);
@@ -2006,7 +2045,10 @@ void server_subproc::waiter::wait(const std::vector<server_subproc *> & procs, s
while (read((int) wake_fd[0], buf, sizeof(buf)) > 0) {}
}
for (size_t i = 0; i < procs.size(); i++) {
- ready[i] = pfds[i + 1].fd < 0 || pfds[i + 1].revents != 0;
+ ready[i] = !procs[i]->has_output();
+ for (int s = 0; s < SERVER_SUBPROC_STREAMS; s++) {
+ ready[i] = ready[i] || pfds[1 + i * SERVER_SUBPROC_STREAMS + s].revents != 0;
+ }
}
#endif
}
diff --git a/tools/server/server-common.h b/tools/server/server-common.h
index 7e4ebed2a..20cdeacf2 100644
--- a/tools/server/server-common.h
+++ b/tools/server/server-common.h
@@ -639,6 +639,16 @@ struct server_pipe {
}
};
+// a child server writes its state commands to stdout and its logs to stderr
+enum server_subproc_stream {
+ SERVER_SUBPROC_STDOUT,
+ SERVER_SUBPROC_STDERR,
+ SERVER_SUBPROC_STREAMS,
+};
+
+// gives the current stdout to the caller as a stream of its own, and sends everything else written to stdout to stderr
+FILE * server_reserve_stdout();
+
// wrapper around common_subproc to manage a child server process
// mainly used by router mode
struct server_subproc {
@@ -649,12 +659,15 @@ struct server_subproc {
void terminate() { sproc.terminate(); }
int join() { return sproc.join(); }
- // true if the child's combined stdout/stderr pipe is available (call after create())
+ // true if both output pipes of the child are available (call after create())
bool has_output();
- // non-blocking read
+ // true once both output pipes are closed
+ bool output_closed() const;
+
+ // non-blocking read from one output pipe
// returns the number of bytes read, 0 when nothing is available, -1 when the pipe is closed or broken
- int read_output(char * buf, size_t len);
+ int read_output(server_subproc_stream stream, char * buf, size_t len);
// wait until one of a set of children has output, wake() is called, or a timeout passes
struct waiter {
@@ -664,7 +677,7 @@ struct server_subproc {
// thread-safe; on Windows this is a no-op, wait() returns within 50 ms anyway
void wake();
- // timeout_ms < 0 waits until data or wake(); ready[i] is set for each child with data (or a broken pipe)
+ // timeout_ms < 0 waits until data or wake(); ready[i] is set for each child with data on an open pipe (or a broken pipe)
void wait(const std::vector<server_subproc *> & procs, std::vector<bool> & ready, int64_t timeout_ms);
private:
@@ -674,5 +687,6 @@ struct server_subproc {
};
private:
- intptr_t out_handle = -1; // fd on POSIX, HANDLE on Windows; taken lazily from sproc
+ intptr_t out_handles[SERVER_SUBPROC_STREAMS] = { -1, -1 }; // fd on POSIX, HANDLE on Windows; taken lazily from sproc
+ bool out_closed [SERVER_SUBPROC_STREAMS] = { false, false };
};
diff --git a/tools/server/server-models.cpp b/tools/server/server-models.cpp
index 1a9ccd1ad..d42b523b2 100644
--- a/tools/server/server-models.cpp
+++ b/tools/server/server-models.cpp
@@ -94,7 +94,7 @@ private:
std::shared_ptr<server_subproc> proc;
server_child_mode mode = SERVER_CHILD_MODE_NORMAL;
int port = 0;
- std::string buf; // partial line
+ std::string buf[SERVER_SUBPROC_STREAMS]; // partial line of each pipe
bool eof = false; // output closed, waiting for the process to be reaped
int64_t deadline = 0; // force-kill time in ms, 0 when no stop is pending
};
@@ -147,46 +147,56 @@ private:
return false;
}
- // read what the child wrote, forward complete lines
+ // read what the child wrote, handle its commands and forward its logs, line by line
void read_output(child_t & c) {
+ for (int i = 0; i < SERVER_SUBPROC_STREAMS; i++) {
+ read_stream(c, (server_subproc_stream) i);
+ }
+ c.eof = c.proc->output_closed();
+ }
+
+ void read_stream(child_t & c, server_subproc_stream stream) {
char chunk[4096];
- while (!c.eof) {
- int n = c.proc->read_output(chunk, sizeof(chunk));
+ std::string & buf = c.buf[stream];
+ bool closed = false;
+ while (true) {
+ int n = c.proc->read_output(stream, chunk, sizeof(chunk));
if (n < 0) {
- c.eof = true;
+ closed = true;
break;
}
if (n == 0) {
break;
}
- c.buf.append(chunk, (size_t) n);
+ buf.append(chunk, (size_t) n);
size_t start = 0;
while (true) {
- size_t nl = c.buf.find('\n', start);
+ size_t nl = buf.find('\n', start);
if (nl == std::string::npos) {
break;
}
- std::string line = c.buf.substr(start, nl + 1 - start);
+ on_line(c, stream, buf.substr(start, nl + 1 - start));
start = nl + 1;
- on_line(c, line);
}
- c.buf.erase(0, start);
- if (c.buf.size() > max_line) {
- c.buf.clear(); // a child that never writes a newline must not grow this without bound
+ buf.erase(0, start);
+ if (buf.size() > max_line) {
+ buf.clear(); // a child that never writes a newline must not grow this without bound
}
}
- if (c.eof && !c.buf.empty()) {
- on_line(c, c.buf);
- c.buf.clear();
+ if (closed && !buf.empty()) {
+ on_line(c, stream, buf);
+ buf.clear();
}
}
- void on_line(child_t & c, const std::string & line) {
- if (string_starts_with(line, CMD_CHILD_TO_ROUTER_STATE)) {
+ void on_line(child_t & c, server_subproc_stream stream, const std::string & line) {
+ if (stream == SERVER_SUBPROC_STDERR) {
+ LOG("[%5d] %s", c.port, line.c_str()); // forward log
+ } else if (string_starts_with(line, CMD_CHILD_TO_ROUTER_STATE)) {
LOG_DBG("[%5d] %s", c.port, line.c_str()); // prevent spamming the log
models.handle_child_state(c.name, line);
} else {
- LOG("[%5d] %s", c.port, line.c_str()); // forward log
+ SRV_WRN("[%5d] unexpected output on the command pipe: %s", c.port, line.c_str());
}
}
@@ -513,6 +523,8 @@ void server_model_meta::update_args(common_preset_context & ctx_preset, std::str
preset.set_option(ctx_preset, "LLAMA_ARG_HOST", CHILD_ADDR);
preset.set_option(ctx_preset, "LLAMA_ARG_PORT", std::to_string(port));
preset.set_option(ctx_preset, "LLAMA_ARG_ALIAS", name);
+ // the child output goes through the router to its terminal, so it follows the router colors
+ preset.set_option(ctx_preset, "LLAMA_ARG_LOG_COLORS", common_log_get_colors(common_log_main()) ? "on" : "off");
// TODO: maybe validate preset before rendering ?
// render args
args = preset.to_args(bin_path);
@@ -796,11 +808,12 @@ void server_models::load_models() {
inst.meta.hidden = hidden_models.count(name) > 0;
}
};
- // update_args() injects HOST/PORT/ALIAS, so strip them before comparing presets
+ // update_args() injects HOST/PORT/ALIAS/LOG_COLORS, so strip them before comparing presets
auto preset_options_for_compare = [](common_preset p) {
p.unset_option("LLAMA_ARG_HOST");
p.unset_option("LLAMA_ARG_PORT");
p.unset_option("LLAMA_ARG_ALIAS");
+ p.unset_option("LLAMA_ARG_LOG_COLORS");
return p.options;
};
@@ -1171,9 +1184,8 @@ void server_models::load(const std::string & name, const load_options & opts) {
}
inst.meta.args = child_args; // save for debugging
- // TODO @ngxson : maybe separate stdout and stderr in the future
- // so that we can use stdout for commands and stderr for logging
- int options = subprocess_option_no_window | subprocess_option_combined_stdout_stderr;
+ // the child writes its commands to stdout and its logs to stderr
+ int options = subprocess_option_no_window;
if (!inst.subproc->sproc.create(child_args, options, child_env)) {
throw std::runtime_error("failed to spawn server instance");
}
@@ -1673,6 +1685,18 @@ void server_models::handle_child_state(const std::string & name, const std::stri
// server_child
//
+server_child::server_child() {
+ if (is_child()) {
+ cmd_out = server_reserve_stdout();
+ }
+}
+
+server_child::~server_child() {
+ if (cmd_out) {
+ fclose(cmd_out);
+ }
+}
+
bool server_child::is_child() {
const char * router_port = std::getenv("LLAMA_SERVER_ROUTER_PORT");
return router_port != nullptr;
@@ -1795,14 +1819,8 @@ void server_child::notify_to_router(const std::string & state, const json & payl
{"payload", payload},
};
std::lock_guard<std::mutex> lk(mtx_stdout);
- common_log_pause(common_log_main());
- fflush(stdout);
- // the router matches the command on a line prefix, so the leading newline
- // closes whatever the logger left open on the shared pipe, down to the
- // trailing color reset that carries no newline of its own
- fprintf(stdout, "\n%s%s\n", CMD_CHILD_TO_ROUTER_STATE, safe_json_to_str(data).c_str());
- fflush(stdout);
- common_log_resume(common_log_main());
+ fprintf(cmd_out, "%s%s\n", CMD_CHILD_TO_ROUTER_STATE, safe_json_to_str(data).c_str());
+ fflush(cmd_out);
}
diff --git a/tools/server/server-models.h b/tools/server/server-models.h
index 90161bf34..0a6999ee3 100644
--- a/tools/server/server-models.h
+++ b/tools/server/server-models.h
@@ -320,6 +320,11 @@ struct server_child {
std::mutex mtx_stdout;
std::atomic<bool> is_finished_downloading = false; // set by run_download
+ // in a child, keeps stdout for the commands to the router, so it is created before anything is written;
+ // everything else written to stdout goes to stderr with the logs
+ server_child();
+ ~server_child();
+
// return true if the current process is a child server instance
bool is_child();
server_child_mode get_mode();
@@ -332,6 +337,9 @@ struct server_child {
// notify router server for status changes (e.g. loading, downloading, sleeping, etc.)
// message will be handled by server_models::handle_child_state() on the router side
void notify_to_router(const std::string & state_name, const json & payload);
+
+private:
+ FILE * cmd_out = nullptr; // the stdout the router reads the commands from
};
struct server_models_routes {
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index ad538a6d6..6e2d8ff9d 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -41,6 +41,7 @@ int llama_server(int argc, char ** argv);
// to be used via CLI (argc / argv are used by router mode only)
int llama_server(common_params & params, int argc, char ** argv);
+int llama_server(common_params & params, int argc, char ** argv, server_child & child);
void llama_server_terminate();
void llama_server_terminate() {
if (shutdown_handler) {
@@ -90,6 +91,8 @@ static server_http_context::handler_t ex_wrapper(server_http_context::handler_t
}
int llama_server(int argc, char ** argv) {
+ server_child child;
+
std::setlocale(LC_NUMERIC, "C");
#ifndef _WIN32
@@ -115,12 +118,17 @@ int llama_server(int argc, char ** argv) {
llama_backend_init();
llama_numa_init(params.numa);
- const int result = llama_server(params, argc, argv);
+ const int result = llama_server(params, argc, argv, child);
common_log_flush(common_log_main());
return result;
}
int llama_server(common_params & params, int argc, char ** argv) {
+ server_child child;
+ return llama_server(params, argc, argv, child);
+}
+
+int llama_server(common_params & params, int argc, char ** argv, server_child & child) {
bool is_run_by_cli = (argv == nullptr);
common_models_handler models_handler;
@@ -194,7 +202,6 @@ int llama_server(common_params & params, int argc, char ** argv) {
//
// register API routes
- server_child child; // only used in non-router mode
server_routes routes(params, ctx_server);
server_tools tools;