Commit 8e330954a for llama.cpp

commit 8e330954adb6e86c329c9d7e338f01f93ffe4b88
Author: Xuan-Son Nguyen <son@huggingface.co>
Date:   Sun Sep 13 01:36:34 2026 +0200

    common: add LOG_JSON macro to log structured data (#28586)

    * add LOG_JSON macro

    * fit: add demo LOG_JSON

diff --git a/common/arg.cpp b/common/arg.cpp
index b1c0f2352..c4c4e143c 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -3875,7 +3875,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
         {"--no-log-jsonl"},
         "Log as JSONL (one JSON object per line) to stdout, this also disables colored logging (default: disabled)",
         [](common_params &, bool value) {
-            common_log_set_jsonl(common_log_main(), value);
+            common_log_set_jsonl(value);
         }
     ).set_env("LLAMA_ARG_LOG_JSONL"));
     add_opt(common_arg(
diff --git a/common/fit.cpp b/common/fit.cpp
index c601fe405..7a0300829 100644
--- a/common/fit.cpp
+++ b/common/fit.cpp
@@ -1,5 +1,6 @@
 #include "fit.h"

+#include "json.h"
 #include "log.h"

 #include "../src/llama-ext.h"
@@ -915,6 +916,9 @@ void common_memory_breakdown_print(const struct llama_context * ctx) {

     std::vector<std::array<std::string, 9>> table_data;
     table_data.reserve(devices.size());
+
+    // same data as the table below, for --log-jsonl consumers
+    common_json rows = common_json::array();
     const std::string template_header = "%s: | %s | %s   %s    %s   %s   %s   %s    %s |\n";
     const std::string template_gpu    = "%s: | %s | %s = %s + (%s = %s + %s + %s) + %s |\n";
     const std::string template_other  = "%s: | %s | %s   %s    %s = %s + %s + %s    %s |\n";
@@ -989,6 +993,19 @@ void common_memory_breakdown_print(const struct llama_context * ctx) {
             std::to_string(mb.context / MiB),
             std::to_string(mb.compute / MiB),
             std::to_string(unaccounted / static_cast<int64_t>(MiB))});
+
+        rows.push_back({
+            {"kind",        "device"},
+            {"name",        name},
+            {"description", desc},
+            {"total",       total / MiB},
+            {"free",        free / MiB},
+            {"self",        self / MiB},
+            {"model",       mb.model / MiB},
+            {"context",     mb.context / MiB},
+            {"compute",     mb.compute / MiB},
+            {"unaccounted", unaccounted / static_cast<int64_t>(MiB)},
+        });
     }

     // print memory breakdown for host:
@@ -1004,6 +1021,15 @@ void common_memory_breakdown_print(const struct llama_context * ctx) {
             std::to_string(mb_host.context / MiB),
             std::to_string(mb_host.compute / MiB),
             ""}); // unaccounted
+
+        rows.push_back({
+            {"kind",    "host"},
+            {"name",    "Host"},
+            {"self",    self / MiB},
+            {"model",   mb_host.model / MiB},
+            {"context", mb_host.context / MiB},
+            {"compute", mb_host.compute / MiB},
+        });
     }

     // print memory breakdown for all remaining buffer types:
@@ -1025,6 +1051,16 @@ void common_memory_breakdown_print(const struct llama_context * ctx) {
             std::to_string(mb.context / MiB),
             std::to_string(mb.compute / MiB),
             ""}); // unaccounted
+
+        rows.push_back({
+            {"kind",    "buffer_type"},
+            {"name",    name},
+            {"self",    self / MiB},
+            {"model",   mb.model / MiB},
+            {"context", mb.context / MiB},
+            {"compute", mb.compute / MiB},
+        });
+
         seen_buffer_types.insert(buft);
     }

@@ -1042,6 +1078,11 @@ void common_memory_breakdown_print(const struct llama_context * ctx) {
             __func__, td[1].c_str(), td[2].c_str(), td[3].c_str(), td[4].c_str(), td[5].c_str(),
             td[6].c_str(), td[7].c_str(), td[8].c_str());
     }
+
+    LOG_JSON("fit_memory_breakdown", common_json({
+        {"unit", "MiB"},
+        {"rows", rows},
+    }));
 }

 void common_fit_print(
diff --git a/common/log.cpp b/common/log.cpp
index 42951190c..0a9a4eb9e 100644
--- a/common/log.cpp
+++ b/common/log.cpp
@@ -37,6 +37,16 @@ void common_log_set_verbosity_thold(int verbosity) {
     common_log_verbosity_thold = verbosity;
 }

+static bool common_log_jsonl = false;
+
+bool common_log_get_jsonl(void) {
+    return common_log_jsonl;
+}
+
+void common_log_set_jsonl(bool jsonl) {
+    common_log_jsonl = jsonl;
+}
+
 static int64_t t_us() {
     return std::chrono::duration_cast<std::chrono::microseconds>(std::chrono::system_clock::now().time_since_epoch()).count();
 }
@@ -87,6 +97,7 @@ struct common_log_entry {
     bool is_end       { false }; // signals the worker thread to stop
     bool prefix       { false };
     bool jsonl        { false };
+    bool is_json      { false }; // msg already holds a serialized JSON object

     common_log_entry(size_t size = 256) : msg(size) { }

@@ -107,6 +118,12 @@ struct common_log_entry {
         }

         if (jsonl) {
+            if (is_json) {
+                fprintf(fcur, "%s\n", msg.data());
+                fflush(fcur);
+                return;
+            }
+
             common_json obj = {
                 {"type",  "log"},
                 {"time",  timestamp},
@@ -156,7 +173,6 @@ struct common_log {
         file       = nullptr;
         prefix     = false;
         timestamps = false;
-        jsonl      = false;
         running    = false;
         t_start    = t_us();

@@ -184,7 +200,6 @@ private:

     bool prefix;
     bool timestamps;
-    bool jsonl;
     bool running;

     int64_t t_start;
@@ -273,7 +288,8 @@ public:
         entry.is_end    = false;
         entry.level     = level;
         entry.prefix    = prefix;
-        entry.jsonl     = jsonl;
+        entry.jsonl     = common_log_jsonl;
+        entry.is_json   = false;
         entry.timestamp = 0;
         if (timestamps) {
             entry.timestamp = t_us() - t_start;
@@ -283,6 +299,42 @@ public:
         cv_new.notify_one();
     }

+    void add_json(const char * type, const common_json & obj) {
+        const common_json full = {
+            {"type", type},
+            {"data", obj},
+        };
+
+        const std::string text = full.dump_safe();
+
+        std::unique_lock<std::mutex> lock(mtx);
+
+        // block if the queue is full
+        cv_full.wait(lock, [this]() { return !running || !is_full(); });
+
+        if (!running) {
+            // discard messages while the worker thread is paused
+            return;
+        }
+
+        auto & entry = queue[tail];
+
+        if (entry.msg.size() < text.size() + 1) {
+            entry.msg.resize(text.size() + 1);
+        }
+        memcpy(entry.msg.data(), text.c_str(), text.size() + 1);
+
+        entry.is_end    = false;
+        entry.level     = GGML_LOG_LEVEL_NONE;
+        entry.prefix    = false;
+        entry.jsonl     = true;
+        entry.is_json   = true;
+        entry.timestamp = 0;
+
+        tail = (tail + 1) % queue.size();
+        cv_new.notify_one();
+    }
+
     void resume() {
         std::lock_guard<std::mutex> lock(mtx);

@@ -388,12 +440,6 @@ public:

         this->timestamps = timestamps;
     }
-
-    void set_jsonl(bool jsonl) {
-        std::lock_guard<std::mutex> lock(mtx);
-
-        this->jsonl = jsonl;
-    }
 };

 //
@@ -440,6 +486,14 @@ void common_log_add(struct common_log * log, enum ggml_log_level level, const ch
     va_end(args);
 }

+void common_log_add_json(struct common_log * log, const char * type, const common_json & obj) {
+    if (!common_log_jsonl) {
+        return;
+    }
+
+    log->add_json(type, obj);
+}
+
 void common_log_set_file(struct common_log * log, const char * file) {
     log->set_file(file);
 }
@@ -467,10 +521,6 @@ void common_log_set_timestamps(struct common_log * log, bool timestamps) {
     log->set_timestamps(timestamps);
 }

-void common_log_set_jsonl(struct common_log * log, bool jsonl) {
-    log->set_jsonl(jsonl);
-}
-
 void common_log_flush(struct common_log * log) {
     log->pause();
     log->resume();
diff --git a/common/log.h b/common/log.h
index 37f4de92b..e36b09463 100644
--- a/common/log.h
+++ b/common/log.h
@@ -43,6 +43,10 @@ int  common_log_get_verbosity_thold(void);

 void common_log_set_verbosity_thold(int verbosity); // not thread-safe

+bool common_log_get_jsonl(void);
+
+void common_log_set_jsonl(bool jsonl); // not thread-safe
+
 int common_log_get_verbosity(enum ggml_log_level level);

 void common_log_default_callback(enum ggml_log_level level, const char * text, void * user_data);
@@ -91,7 +95,6 @@ void common_log_set_file      (struct common_log * log, const char * file); // n
 void common_log_set_colors    (struct common_log * log, log_colors colors); // not thread-safe
 void common_log_set_prefix    (struct common_log * log, bool prefix);       // whether to output prefix to each log
 void common_log_set_timestamps(struct common_log * log, bool timestamps);   // whether to output timestamps in the prefix
-void common_log_set_jsonl     (struct common_log * log, bool jsonl);        // print each log as a JSON object on one line, not thread-safe
 void common_log_flush         (struct common_log * log);                    // flush all pending log messages

 // helper macros for logging
@@ -127,3 +130,17 @@ void common_log_flush         (struct common_log * log);                    // f
 #define LOG_WRNV(verbosity, ...) LOG_TMPL(GGML_LOG_LEVEL_WARN,  verbosity, __VA_ARGS__)
 #define LOG_ERRV(verbosity, ...) LOG_TMPL(GGML_LOG_LEVEL_ERROR, verbosity, __VA_ARGS__)
 #define LOG_CNTV(verbosity, ...) LOG_TMPL(GGML_LOG_LEVEL_CONT,  verbosity, __VA_ARGS__)
+
+class common_json; // defined in common/json.h
+
+// helper allows different types of json output
+// no-op if --log-jsonl is not set
+void common_log_add_json(struct common_log * log, const char * type, const common_json & data);
+
+// will only print if --log-jsonl is set
+#define LOG_JSON(type, data) \
+    do { \
+        if (common_log_get_jsonl()) { \
+            common_log_add_json(common_log_main(), type, data); \
+        } \
+    } while (0)