Commit 4cfb6d1c7 for llama.cpp

commit 4cfb6d1c7517f7ce7b1af3bd6987891269d67273
Author: Aleksander Grygier <aleksander.grygier@gmail.com>
Date:   Wed Sep 30 12:42:17 2026 +0200

    ui : model memory-fit estimation (#27957)

    * ui : model memory-fit estimation

    Replace the raw runtime-memory estimate with the app's compatibility check:
    the smallest real Mac memory tier that fits a model file, budgeted as
    RAM x 0.75 minus fixed overhead with headroom on the file size. The constants
    move to lib; the unused runtime-memory estimate is dropped. browser-info's
    OS detection is exported for reuse.

    Assisted-by: pi:GLM-5.3-Flash

    * ui : cover the memory-fit and tool-use heuristics in tests

    Assisted-by: pi:zai-org/GLM-5.3-Flash

diff --git a/tools/ui/src/lib/constants/index.ts b/tools/ui/src/lib/constants/index.ts
index c66a2f771..8d99180ab 100644
--- a/tools/ui/src/lib/constants/index.ts
+++ b/tools/ui/src/lib/constants/index.ts
@@ -46,6 +46,7 @@ export * from './path-display.constants';
 export * from './model-id.constants';
 export * from './model-loading.constants';
 export * from './models-discover.constants';
+export * from './model-compatibility.constants';
 export * from './huggingface.constants';
 export * from './precision.constants';
 export * from './pwa.constants';
diff --git a/tools/ui/src/lib/constants/model-compatibility.constants.ts b/tools/ui/src/lib/constants/model-compatibility.constants.ts
new file mode 100644
index 000000000..5d65f8d62
--- /dev/null
+++ b/tools/ui/src/lib/constants/model-compatibility.constants.ts
@@ -0,0 +1,32 @@
+/**
+ * Model memory-fit constants.
+ *
+ * Mirrors the app's compatibility check (Model+Compatibility.swift):
+ *   budget      = RAM x RAM_BUDGET_RATIO - RAM_OVERHEAD_MB
+ *   weightBytes = fileBytes x QUANT_WEIGHT
+ * a file fits when weightBytes <= budget. Kept here so the estimation util and
+ * any caller share one source.
+ */
+
+/** Bytes in one mebibyte (MiB), used to convert a file size to MB. */
+export const MIB_BYTES = 1_048_576;
+
+/** MiB in one tier unit; the tiers below are binary sizes, i.e. GiB. */
+export const MB_PER_GB = 1024;
+
+/** Overhead multiplier applied to the file size when estimating weight memory. */
+export const QUANT_WEIGHT = 1.05;
+
+/** Share of RAM the app allows the model to occupy. */
+export const RAM_BUDGET_RATIO = 0.75;
+
+/** Fixed RAM overhead (MB) reserved for the system and KV cache. */
+export const RAM_OVERHEAD_MB = 2048;
+
+/**
+ * Memory tiers (GB) covering the RAM sizes common machines ship with, in
+ * small enough steps that the requirement reads honestly. Device-agnostic on
+ * purpose: the server exposes no host RAM, so the UI presents the tier and
+ * lets the user judge.
+ */
+export const MEM_TIERS = [4, 6, 8, 12, 16, 24, 32, 48, 64, 96, 128, 192, 256, 384, 512, 768, 1024];
diff --git a/tools/ui/src/lib/utils/browser-info.ts b/tools/ui/src/lib/utils/browser-info.ts
index c96abb01e..e313d4258 100644
--- a/tools/ui/src/lib/utils/browser-info.ts
+++ b/tools/ui/src/lib/utils/browser-info.ts
@@ -16,7 +16,7 @@ import {
 } from '$lib/constants';
 import type { ToolExecutionResult } from '$lib/types';

-function detectOs(userAgent: string): string {
+export function detectOs(userAgent: string): string {
 	for (const [pattern, os] of BROWSER_INFO_OS_UA_PATTERNS) {
 		if (pattern.test(userAgent)) return os;
 	}
diff --git a/tools/ui/src/lib/utils/chat-template-tool-detector.ts b/tools/ui/src/lib/utils/chat-template-tool-detector.ts
new file mode 100644
index 000000000..2b8f915c5
--- /dev/null
+++ b/tools/ui/src/lib/utils/chat-template-tool-detector.ts
@@ -0,0 +1,29 @@
+/**
+ * Detects whether a model's chat template supports tool calling.
+ *
+ * There is no server flag for tool support, so we infer it from the chat
+ * template. A template that accepts a `tools` array or emits tool-call tokens
+ * is treated as tool-capable.
+ */
+
+/** Tool-call tokens emitted by the template for assistant tool calls, matched case-insensitively. */
+const TOOL_CALL_TOKENS = [
+	'tool_call',
+	'tool_calls',
+	'function_call',
+	'tool_use',
+	'<tool',
+	'<|tool'
+];
+/** Jinja reference to the `tools` array passed in by the caller. */
+const JINJA_TOOLS_VAR = /\{\{[^{}]*\btools\b[^{}]*\}\}|\{%[^{}]*\btools\b[^{}]*%\}/i;
+
+export function detectToolUseSupport(t: string): boolean {
+	if (!t) return false;
+
+	if (JINJA_TOOLS_VAR.test(t)) return true;
+
+	const template = t.toLowerCase();
+
+	return TOOL_CALL_TOKENS.some((token) => template.includes(token));
+}
diff --git a/tools/ui/src/lib/utils/index.ts b/tools/ui/src/lib/utils/index.ts
index 8896ea234..06ab5d7b6 100644
--- a/tools/ui/src/lib/utils/index.ts
+++ b/tools/ui/src/lib/utils/index.ts
@@ -343,7 +343,13 @@ export { buildSandboxToolDefinition, SANDBOX_TOOL_DEFINITION } from './sandbox-t
 export { executeGetDatetimeTool } from './get-datetime';

 // Browser fallback for the server's get_info tool
-export { executeBrowserInfoTool } from './browser-info';
+export { detectOs, executeBrowserInfoTool } from './browser-info';
+
+// Tool-use support detection from a chat template
+export { detectToolUseSupport } from './chat-template-tool-detector';
+
+// Model memory estimation
+export { minMemoryTierGb } from './model-compatibility';

 // Cryptography utilities

diff --git a/tools/ui/src/lib/utils/model-compatibility.ts b/tools/ui/src/lib/utils/model-compatibility.ts
new file mode 100644
index 000000000..4f7e32fe6
--- /dev/null
+++ b/tools/ui/src/lib/utils/model-compatibility.ts
@@ -0,0 +1,37 @@
+/**
+ * Model memory estimation.
+ *
+ * Mirrors the app's compatibility check (Model+Compatibility.swift): the
+ * runtime budget is RAM x 0.75 minus a fixed overhead, and a file fits when
+ * its size with headroom stays under that budget. The result is the smallest
+ * memory tier that can run the model, so the UI presents an honest machine
+ * requirement instead of a raw file size. Context length and
+ * device-specific budgets are deliberately ignored - callers present the
+ * requirement and let the user judge.
+ */
+import {
+	MB_PER_GB,
+	MEM_TIERS,
+	MIB_BYTES,
+	QUANT_WEIGHT,
+	RAM_BUDGET_RATIO,
+	RAM_OVERHEAD_MB
+} from '$lib/constants';
+
+/**
+ * Smallest memory tier (GB) that can run a model of the given file size,
+ * or null if nothing fits even the largest tier.
+ */
+export function minMemoryTierGb(sizeBytes: number): number | null {
+	if (!sizeBytes) return null;
+
+	const weightMb = (sizeBytes / MIB_BYTES) * QUANT_WEIGHT;
+
+	for (const tier of MEM_TIERS) {
+		const budgetMb = tier * MB_PER_GB * RAM_BUDGET_RATIO - RAM_OVERHEAD_MB;
+
+		if (weightMb <= budgetMb) return tier;
+	}
+
+	return null;
+}
diff --git a/tools/ui/tests/unit/model-compatibility.test.ts b/tools/ui/tests/unit/model-compatibility.test.ts
new file mode 100644
index 000000000..0fbd1e39e
--- /dev/null
+++ b/tools/ui/tests/unit/model-compatibility.test.ts
@@ -0,0 +1,44 @@
+import { detectToolUseSupport } from '$lib/utils/chat-template-tool-detector';
+import { minMemoryTierGb } from '$lib/utils/model-compatibility';
+import { describe, expect, it } from 'vitest';
+
+describe('minMemoryTierGb', () => {
+	it('picks the smallest tier whose budget fits the file', () => {
+		// budget(16) = 16 * 1024 * 0.75 - 2048 = 10240 MiB; ~9.2 GiB file fits
+		expect(minMemoryTierGb(9.5 * 1024 * 1024 * 1024)).toBe(16);
+		// budget(12) = 7168 MiB; the same file does not fit
+		expect(minMemoryTierGb(9.5 * 1024 * 1024 * 1024)).not.toBe(12);
+	});
+
+	it('applies the quant headroom to the file size', () => {
+		// exactly the tier-32 budget before the 1.05 headroom; with it the file
+		// spills into the next tier
+		const budget32Mb = 32 * 1024 * 0.75 - 2048;
+		const bytes = (budget32Mb / 1.05) * 1024 * 1024;
+
+		expect(minMemoryTierGb(bytes)).toBe(32);
+		expect(minMemoryTierGb(bytes + 1)).toBe(48);
+	});
+
+	it('returns null for empty sizes and over-budget files', () => {
+		expect(minMemoryTierGb(0)).toBeNull();
+		expect(minMemoryTierGb(4 * 1024 * 1024 * 1024 * 1024)).toBeNull();
+	});
+});
+
+describe('detectToolUseSupport', () => {
+	it('detects the jinja tools variable', () => {
+		expect(detectToolUseSupport('{% for tool in tools %}')).toBe(true);
+		expect(detectToolUseSupport('{{ tools | tojson }}')).toBe(true);
+	});
+
+	it('detects tool-call tokens case-insensitively', () => {
+		expect(detectToolUseSupport('WRITES <tool_call> BLOCKS')).toBe(true);
+		expect(detectToolUseSupport('emits Tool_Call sections')).toBe(true);
+	});
+
+	it('rejects templates without tool references', () => {
+		expect(detectToolUseSupport('')).toBe(false);
+		expect(detectToolUseSupport('{{ prompt }}')).toBe(false);
+	});
+});