Commit 9b4333611 for llama.cpp

commit 9b43336114386fd0dcac8b42226226a1c55692af
Author: Aleksander Grygier <aleksander.grygier@gmail.com>
Date:   Wed Sep 30 12:42:16 2026 +0200

    ui : Hugging Face Hub data layer (#27947)

    * ui : Hugging Face Hub data layer

    Add HuggingFaceService and its constants/enums/types: GGUF repo search, file
    tree and model detail fetching, quant/sidecar filename analysis, shard-set
    collapsing and the llama.app catalog feed, plus an orgOf() helper on the model
    name utils.

    Assisted-by: pi:GLM-5.3-Flash

    * ui : strip provider tilde prefix from hub avatar urls

    Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash

    * ui : trim redundant comments in the HF data layer service

    Per review: drop JSDoc that restates the method name and inline comments
    that restate the code; keep only comments carrying non-obvious context.

    Assisted-by: pi:zai-org/GLM-5.3-Flash

    * ui : harden the HF data layer error typing, cover the helpers in tests

    Carries the HTTP status on retryable fetch errors instead of matching the
    message text. Marks expand-dependent catalog fields optional and documents
    the data/models index pairing. Adds table tests for the pure helpers.

    Assisted-by: pi:zai-org/GLM-5.3-Flash

diff --git a/tools/ui/src/lib/constants/huggingface.constants.ts b/tools/ui/src/lib/constants/huggingface.constants.ts
new file mode 100644
index 000000000..39f6fbe5c
--- /dev/null
+++ b/tools/ui/src/lib/constants/huggingface.constants.ts
@@ -0,0 +1,232 @@
+/**
+ * HuggingFace Hub constants.
+ *
+ * URLs, parsing regexes and formatting units for the HuggingFaceService.
+ * Reference: https://huggingface.co/docs/huggingface_hub/package_reference/hf_api
+ */
+
+// API endpoints
+
+export const HF_BASE_URL = 'https://huggingface.co';
+export const HF_API_MODELS_URL = `${HF_BASE_URL}/api/models`;
+export const HF_AVATARS_URL = `${HF_BASE_URL}/api/avatars`;
+
+// Query params
+
+export const HF_FULL_DETAIL_PARAM = 'full=true';
+export const HF_RECURSIVE_TREE_PARAM = 'recursive=true';
+/** Search filter that restricts results to repos containing GGUF files. */
+export const HF_GGUF_FILTER = 'gguf';
+/** Repeatable `expand` query param selecting fields on the list endpoint. */
+export const HF_EXPAND_PARAM = 'expand';
+/**
+ * Fields the model list endpoint omits by default but the discover list rows
+ * render: `gguf` (chat template, context length, param count) drives the
+ * reasoning / tool-use icons and the context badge, `siblings` the vision and
+ * draft-sidecar badges. Without them those parts of a row stay empty.
+ */
+export const HF_MODEL_LIST_EXPAND: readonly string[] = [
+	'author',
+	'downloads',
+	'gguf',
+	'lastModified',
+	'likes',
+	'pipeline_tag',
+	'siblings',
+	// `base_model:` tags, so search rows can show the base org's avatar as the
+	// main avatar with the quant org as the corner badge, like catalog rows.
+	'tags'
+];
+
+// Repo file conventions
+
+export const HF_MAIN_BRANCH = 'main';
+export const HF_README_FILENAME = 'README.md';
+export const HF_RAW_PATH = 'raw';
+export const HF_TREE_PATH = 'tree';
+
+// Pagination
+
+export const HF_LINK_NEXT_REGEX = /<([^>]+)>;\s*rel="next"/;
+/** `Link` response header carrying the next page URL for cursor pagination. */
+export const HF_LINK_HEADER = 'Link';
+
+// Fetch retry
+
+export const HF_RETRY_ATTEMPTS = 3;
+export const HF_RETRY_DELAY_MS = 1000;
+export const HF_HTTP_NOT_FOUND = 404;
+export const HF_HTTP_SERVER_ERROR_MIN = 500;
+
+// Search limits
+
+export const HF_DEFAULT_LIMIT = 50;
+/** Safety cap on `/tree` pagination: more pages means a misbehaving endpoint. */
+export const HF_TREE_MAX_PAGES = 10;
+export const HF_MAX_LIMIT = 100;
+
+// GGUF shard files
+
+/** Matches a split-shard GGUF file name, e.g. `Model-00001-of-00015.gguf`. */
+export const HF_SHARD_REGEX = /-(\d{5})-of-(\d{5})\.gguf$/i;
+/** Index (1-based) of the first shard in a split-shard set. */
+export const HF_FIRST_SHARD = 1;
+/** Zero-padded width of the shard index in a split-shard file name. */
+export const HF_SHARD_PAD_WIDTH = 5;
+
+// Quantization tokens
+
+/** `UD-` (Unsloth Dynamic) custom quantization prefix, e.g. `UD-Q4_K_XL`. */
+export const HF_UD_QUANT_PREFIX = 'UD';
+export const HF_UD_QUANT_PREFIX_REGEX = /^UD-/i;
+/**
+ * Segment marking an Unsloth `shared-` draft head that borrows the target
+ * model's embedding/output weights, e.g. `...-shared-Q4_K_M.gguf`.
+ */
+export const HF_SHARED_DRAFT_TOKEN = 'shared';
+/**
+ * Extracts the leading precision digits from a quant token, e.g.
+ * `Q4_K_XL` -> 4, `IQ2_XXS` -> 2, `TQ1_0` -> 1, `BF16` -> 16.
+ */
+export const HF_QUANT_PRECISION_REGEX = /^(?:I?Q|TQ|BF|F|MXFP)?(\d+)/i;
+
+// Model card tags
+
+/** Matches the `base_model:` tag (plain or `quantized:`), capturing the repo id. */
+export const HF_BASE_MODEL_TAG_REGEX = /^base_model:(?:quantized:)?(.+)$/;
+export const HF_LICENSE_TAG_PREFIX = 'license:';
+export const HF_GATED_TAG = 'gated';
+export const HF_GGUF_TAG = 'gguf';
+export const HF_SAFETENSORS_TAG = 'safetensors';
+
+// Pipeline tasks (logic use only - matching `pipeline_tag` values against tags)
+
+/**
+ * `pipeline_tag` values grouped by the input/output modality they imply, used
+ * to derive a discover row's modality icons. A tag in more than one group (e.g.
+ * `image-to-video`) lights up each modality it belongs to.
+ */
+export const HF_MODALITY_PIPELINE_TAGS: Readonly<
+	Record<'audio' | 'video' | 'vision', readonly string[]>
+> = {
+	audio: [
+		'audio-classification',
+		'audio-to-audio',
+		'automatic-speech-recognition',
+		'text-to-speech',
+		'voice-activity-detection'
+	],
+	video: ['text-to-video', 'image-to-video', 'video-to-video'],
+	vision: ['image-text-to-text', 'image-to-text', 'text-to-image', 'image-to-video']
+};
+
+/** Filename token marking an mmproj sidecar sibling (unlocks vision / audio). */
+export const HF_MMPROJ_FILENAME_TOKEN = 'mmproj';
+
+export const HF_TASK_TAGS: readonly string[] = [
+	'audio-classification',
+	'audio-to-audio',
+	'automatic-speech-recognition',
+	'conversational',
+	'depth-estimation',
+	'feature-extraction',
+	'fill-mask',
+	'image-classification',
+	'image-feature-extraction',
+	'image-segmentation',
+	'image-text-to-text',
+	'image-to-text',
+	'image-to-video',
+	'object-detection',
+	'question-answering',
+	'reinforcement-learning',
+	'robotics',
+	'sentence-similarity',
+	'summarization',
+	'text2text-generation',
+	'text-classification',
+	'text-generation',
+	'text-to-image',
+	'text-to-speech',
+	'text-to-video',
+	'token-classification',
+	'translation',
+	'video-to-video',
+	'voice-activity-detection',
+	'zero-shot-classification'
+];
+
+// Formatting
+
+export const BYTE = 1;
+export const KILOBYTE = 1_000;
+export const MEGABYTE = 1_000_000;
+export const GIGABYTE = 1_000_000_000;
+export const TERABYTE = 1_000_000_000_000;
+
+/**
+ * Matches a human size string (`177GB`, `1.2 TB`, `500MB`), capturing the
+ * numeric value and its unit suffix. Used by `parseSizeBytes`.
+ */
+export const HF_SIZE_STRING_REGEX = /^\s*([\d.]+)\s*([a-z]+)\s*$/i;
+
+/**
+ * Byte multiplier for a size suffix (`k` kilobyte, `m` megabyte, ...) as used by
+ * the llama.app catalog `size` strings, whose suffix is lowercase.
+ */
+export const HF_SIZE_SUFFIX_BYTES: Readonly<Record<string, number>> = {
+	b: BYTE,
+	g: GIGABYTE,
+	k: KILOBYTE,
+	m: MEGABYTE,
+	t: TERABYTE
+};
+
+export const BYTE_LABEL = 'B';
+export const KILOBYTE_LABEL = 'KB';
+export const MEGABYTE_LABEL = 'MB';
+export const GIGABYTE_LABEL = 'GB';
+
+/** Count suffixes for compact number formatting, e.g. `1.5K`, `2.0M`. */
+export const KILO_LABEL = 'K';
+export const MEGA_LABEL = 'M';
+export const GIGA_LABEL = 'B';
+
+// Relative time
+
+export const MS_PER_DAY = 1000 * 60 * 60 * 24;
+export const DAYS_PER_WEEK = 7;
+/** Rough month length in days, used to bucket relative timestamps. */
+export const DAYS_PER_MONTH = 30;
+export const DAYS_PER_YEAR = 365;
+
+export const TODAY_LABEL = 'Today';
+export const YESTERDAY_LABEL = 'Yesterday';
+export const DAYS_AGO_LABEL = 'days ago';
+export const WEEKS_AGO_LABEL = 'weeks ago';
+export const MONTHS_AGO_LABEL = 'months ago';
+export const YEARS_AGO_LABEL = 'years ago';
+
+// Cache paths
+
+/**
+ * Matches a local HF cache file path
+ * (`.../models--<org>--<name>/snapshots/<sha>/<file>`), capturing the repo
+ * directory name and the repo-relative file path.
+ */
+export const HF_CACHE_PATH_REGEX = /models--(.+?)\/snapshots\/[^/]+\/(.+)$/;
+/** Separator between org and name segments in an HF cache directory name. */
+export const HF_CACHE_DIR_SEPARATOR = '--';
+
+// README
+
+/** Matches a leading YAML frontmatter block (--- ... ---) in a markdown document. */
+export const HF_FRONTMATTER_REGEX = /^---\r?\n[\s\S]*?\r?\n---\r?\n?/;
+
+// Param counts
+
+/**
+ * Best-effort parameter count token in a model id/name, e.g. `27B` from
+ * `Qwen3.8-27B-GGUF` or `300M` from `embeddinggemma-300M-GGUF`.
+ */
+export const HF_PARAM_COUNT_REGEX = /(?:^|[^a-z0-9])(\d+(?:[._]\d+)?)\s*([bm])(?![a-z0-9])/i;
diff --git a/tools/ui/src/lib/constants/index.ts b/tools/ui/src/lib/constants/index.ts
index d93ae6429..c66a2f771 100644
--- a/tools/ui/src/lib/constants/index.ts
+++ b/tools/ui/src/lib/constants/index.ts
@@ -45,6 +45,8 @@ export * from './message-export.constants';
 export * from './path-display.constants';
 export * from './model-id.constants';
 export * from './model-loading.constants';
+export * from './models-discover.constants';
+export * from './huggingface.constants';
 export * from './precision.constants';
 export * from './pwa.constants';
 export * from './routes.constants';
diff --git a/tools/ui/src/lib/constants/models-discover.constants.ts b/tools/ui/src/lib/constants/models-discover.constants.ts
new file mode 100644
index 000000000..f2a6a9463
--- /dev/null
+++ b/tools/ui/src/lib/constants/models-discover.constants.ts
@@ -0,0 +1,8 @@
+/**
+ * Models discover constants.
+ *
+ * Endpoints and settings for the Models Discover dialog.
+ */
+
+/** llama.app model catalog used as the default model list. Online-only source; the discover feature requires an internet connection anyway. */
+export const MODELS_DISCOVER_CATALOG_URL = 'https://llama.app/v1/catalog.json';
diff --git a/tools/ui/src/lib/enums/huggingface.enums.ts b/tools/ui/src/lib/enums/huggingface.enums.ts
new file mode 100644
index 000000000..7a9e0c3b6
--- /dev/null
+++ b/tools/ui/src/lib/enums/huggingface.enums.ts
@@ -0,0 +1,35 @@
+/**
+ * HuggingFace Hub enums.
+ *
+ * Values mirror the strings used by the HF REST API
+ * (https://huggingface.co/docs/huggingface_hub/package_reference/hf_api)
+ * so they can be sent and compared directly.
+ */
+
+/** Sort field for /api/models search queries. */
+export enum HfModelSort {
+	CREATED_AT = 'createdAt',
+	DOWNLOADS = 'downloads',
+	LAST_MODIFIED = 'lastModified',
+	LIKES = 'likes',
+	TRENDING_SCORE = 'trendingScore'
+}
+
+/**
+ * Where the sidecar token (`mtp` / `dflash` / `mmproj` / ...) sits in the
+ * filename.
+ * - `prefix`  sidecar file that lives next to the main weights, e.g. `mtp-Q4_0.gguf`
+ * - `suffix`  embedded draft baked into the main weights, e.g. `Hy3-IQ1_M-mtp.gguf`
+ * - `infix`   standalone sidecar named between head and quant, e.g. `model-mtp-Q8_0.gguf`
+ */
+export enum SidecarForm {
+	INFIX = 'infix',
+	PREFIX = 'prefix',
+	SUFFIX = 'suffix'
+}
+
+/** Entry type in a model repository file tree (`/tree` responses). */
+export enum HfEntryType {
+	DIRECTORY = 'directory',
+	FILE = 'file'
+}
diff --git a/tools/ui/src/lib/enums/index.ts b/tools/ui/src/lib/enums/index.ts
index b20875f80..db4f5b68c 100644
--- a/tools/ui/src/lib/enums/index.ts
+++ b/tools/ui/src/lib/enums/index.ts
@@ -57,6 +57,8 @@ export {
 	SpecialFileType
 } from './files.enums';

+export { HfEntryType, HfModelSort, SidecarForm } from './huggingface.enums';
+
 export {
 	MCPConnectionPhase,
 	MCPLogLevel,
diff --git a/tools/ui/src/lib/services/huggingface.service.ts b/tools/ui/src/lib/services/huggingface.service.ts
new file mode 100644
index 000000000..346da82b5
--- /dev/null
+++ b/tools/ui/src/lib/services/huggingface.service.ts
@@ -0,0 +1,741 @@
+import { PATH_SEPARATOR } from '$lib/constants';
+import {
+	BYTE,
+	BYTE_LABEL,
+	DAYS_AGO_LABEL,
+	DAYS_PER_MONTH,
+	DAYS_PER_WEEK,
+	DAYS_PER_YEAR,
+	GIGA_LABEL,
+	GIGABYTE,
+	GIGABYTE_LABEL,
+	HF_API_MODELS_URL,
+	HF_AVATARS_URL,
+	HF_BASE_MODEL_TAG_REGEX,
+	HF_BASE_URL,
+	HF_CACHE_DIR_SEPARATOR,
+	HF_CACHE_PATH_REGEX,
+	HF_DEFAULT_LIMIT,
+	HF_FIRST_SHARD,
+	HF_FRONTMATTER_REGEX,
+	HF_FULL_DETAIL_PARAM,
+	HF_GATED_TAG,
+	HF_GGUF_FILTER,
+	HF_GGUF_TAG,
+	HF_HTTP_NOT_FOUND,
+	HF_HTTP_SERVER_ERROR_MIN,
+	HF_LICENSE_TAG_PREFIX,
+	HF_LINK_HEADER,
+	HF_LINK_NEXT_REGEX,
+	HF_MAIN_BRANCH,
+	HF_MAX_LIMIT,
+	HF_MODEL_LIST_EXPAND,
+	HF_PARAM_COUNT_REGEX,
+	HF_QUANT_PRECISION_REGEX,
+	HF_RAW_PATH,
+	HF_README_FILENAME,
+	HF_RECURSIVE_TREE_PARAM,
+	HF_RETRY_ATTEMPTS,
+	HF_RETRY_DELAY_MS,
+	HF_SAFETENSORS_TAG,
+	HF_SHARD_PAD_WIDTH,
+	HF_SHARD_REGEX,
+	HF_SHARED_DRAFT_TOKEN,
+	HF_SIZE_STRING_REGEX,
+	HF_SIZE_SUFFIX_BYTES,
+	HF_TASK_TAGS,
+	HF_TREE_MAX_PAGES,
+	HF_TREE_PATH,
+	HF_UD_QUANT_PREFIX,
+	HF_UD_QUANT_PREFIX_REGEX,
+	KILO_LABEL,
+	KILOBYTE,
+	KILOBYTE_LABEL,
+	MEGA_LABEL,
+	MEGABYTE,
+	MEGABYTE_LABEL,
+	MODELS_DISCOVER_CATALOG_URL,
+	MONTHS_AGO_LABEL,
+	MS_PER_DAY,
+	TODAY_LABEL,
+	WEEKS_AGO_LABEL,
+	YEARS_AGO_LABEL,
+	YESTERDAY_LABEL
+} from '$lib/constants';
+import { MODEL_ID, type ModelSidecar } from '$lib/constants';
+import { HfEntryType, HfModelSort, SidecarForm } from '$lib/enums';
+import type {
+	HfCatalogEntry,
+	HfModelDetailInfo,
+	HfModelInfo,
+	HfModelSearchParams,
+	HfModelSibling
+} from '$lib/types/huggingface';
+import { sidecarFromFileToken } from '$lib/utils';
+
+/** Fetch failure carrying the HTTP status, so retry logic tests the code instead of the message. */
+class HfHttpStatusError extends Error {
+	status: number;
+
+	constructor(status: number, statusText: string) {
+		super(`API request failed: ${status} ${statusText}`);
+
+		this.status = status;
+	}
+}
+
+export class HuggingFaceService {
+	private static readonly BASE_URL = HF_API_MODELS_URL;
+
+	// Cached base model lookups keyed by repo id, so repeated selector opens
+	// never re-hit the HF API for the same repo.
+	private static baseModelCache = new Map<string, { org: string; name: string } | null>();
+
+	private static baseModelPending = new Map<
+		string,
+		Promise<{ org: string; name: string } | null>
+	>();
+
+	/**
+	 * Map of quant token to its average bit-depth in bits-per-weight (bpw).
+	 */
+	private static readonly QUANT_BIT_DEPTH: Record<string, number> = {
+		BF16: 16,
+		F16: 16,
+		IQ1_M: 1,
+		IQ1_S: 1,
+		IQ1_XS: 1,
+		IQ1_XXS: 1,
+		IQ2_M: 2,
+		IQ2_S: 2,
+		IQ2_XS: 2,
+		IQ2_XXS: 2,
+		IQ3_M: 3,
+		IQ3_S: 3,
+		IQ3_XS: 3,
+		IQ3_XXS: 3,
+		Q2_K: 2,
+		Q2_K_M: 2,
+		Q2_K_S: 2,
+		Q3_K: 3,
+		Q3_K_L: 3,
+		Q3_K_M: 3,
+		Q3_K_S: 3,
+		Q4_0: 4,
+		Q4_1: 4,
+		Q4_K: 4,
+		Q4_K_M: 4,
+		Q4_K_S: 4,
+		Q5_0: 5,
+		Q5_1: 5,
+		Q5_K: 5,
+		Q5_K_M: 5,
+		Q5_K_S: 5,
+		Q6_K: 6,
+		Q8_0: 8
+	};
+
+	/**
+	 * Collapse split GGUF shard sets (`-00001-of-00015.gguf`, ...) to their first
+	 * shard, summing every shard's size so the kept entry reflects the whole
+	 * quant. Downloads are tag-based (`repo:quant`), so the first shard is
+	 * enough to represent the set.
+	 */
+	static collapseGgufShards(siblings: HfModelSibling[]): HfModelSibling[] {
+		const sizeByPath = new Map(siblings.map((f) => [f.path, f.size ?? 0]));
+		const result: HfModelSibling[] = [];
+
+		for (const file of siblings) {
+			const match = HF_SHARD_REGEX.exec(file.path);
+
+			if (!match) {
+				result.push(file);
+
+				continue;
+			}
+
+			if (Number(match[1]) !== HF_FIRST_SHARD) continue;
+
+			const total = Number(match[2]);
+			const stem = file.path.slice(0, file.path.length - match[0].length);
+
+			let size = 0;
+
+			for (let i = HF_FIRST_SHARD; i <= total; i++) {
+				const shard = HuggingFaceService.shardPath(stem, i, total);
+
+				size += sizeByPath.get(shard) ?? 0;
+			}
+
+			result.push({ ...file, size });
+		}
+
+		return result;
+	}
+
+	// GGUF Model Browsing
+
+	/**
+	 * Extract the GGUF quantization token (e.g. `Q4_K_M`) and any sidecar type
+	 * (`mtp`, `dflash`, `mmproj`, ...) from a `.gguf` filename.
+	 *
+	 * `sidecarForm` records which side of the filename the sidecar token sat on
+	 * so callers can render badges differently. `quant` and `sidecar` are `null`
+	 * when absent; returns `null` for non-GGUF filenames.
+	 */
+	static extractQuantMeta(filename: string): {
+		quant: string | null;
+		/** Draft-head-only variant borrowing embed/output weights from the target model. */
+		shared: boolean;
+		sidecar: ModelSidecar | null;
+		sidecarForm: SidecarForm | null;
+	} | null {
+		if (!MODEL_ID.WEIGHT_EXTENSION_REGEX.test(filename)) return null;
+
+		// HF repos may nest sidecars in a folder (e.g. `MTP/mtp-Model-Q4_0.gguf`);
+		// parse the file name only, the folder adds no quant information.
+		let source = (filename.split(PATH_SEPARATOR).pop() ?? filename).replace(
+			MODEL_ID.WEIGHT_EXTENSION_REGEX,
+			''
+		);
+		let sidecar: ModelSidecar | null = null;
+		let sidecarForm: SidecarForm | null = null;
+
+		// A file named just the sidecar token (`imatrix.gguf`) is the sidecar
+		// itself: no name or quant segments to parse.
+		const bareSidecar = sidecarFromFileToken(source.toLowerCase());
+
+		if (bareSidecar) {
+			return { quant: null, shared: false, sidecar: bareSidecar, sidecarForm: SidecarForm.PREFIX };
+		}
+
+		const prefixMatch = source.match(MODEL_ID.SIDECAR_PREFIX_REGEX);
+
+		if (prefixMatch) {
+			sidecar = sidecarFromFileToken(prefixMatch[1].toLowerCase());
+			sidecarForm = SidecarForm.PREFIX;
+			source = prefixMatch[2];
+		} else {
+			const suffixMatch = source.match(MODEL_ID.SIDECAR_SUFFIX_REGEX);
+
+			if (suffixMatch) {
+				// Take the suffix sidecar even when the head carries no quant:
+				// embedded drafts end in one (`Hy3-IQ1_M-mtp`), standalone sidecar
+				// files do not (`Model-mtp-draft`, `Model-imatrix`).
+				sidecar = sidecarFromFileToken(suffixMatch[2].toLowerCase());
+				sidecarForm = SidecarForm.SUFFIX;
+				source = suffixMatch[1];
+			} else {
+				const infixMatch = source.match(MODEL_ID.SIDECAR_INFIX_REGEX);
+
+				if (infixMatch) {
+					sidecar = sidecarFromFileToken(infixMatch[2].toLowerCase());
+					sidecarForm = SidecarForm.INFIX;
+					source = `${infixMatch[1]}-${infixMatch[3]}`;
+				}
+			}
+		}
+
+		// Scan dash-separated segments left-to-right for the first quant match.
+		// - For sidecars like `mtp-Q4_0-180MB.gguf` the quant is `Q4_0`.
+		// - For embedded MTP like `Hy3-IQ1_M-mtp.gguf` we have `Hy3-IQ1_M` and `IQ1_M` matches.
+		// - For main files like `Llama-3-8B-Q4_K_M.gguf` we land on the trailing quant.
+		const segments = source.split(MODEL_ID.SEGMENT_SEPARATOR);
+		const quantIdx = segments.findIndex((seg) => MODEL_ID.QUANTIZATION_SEGMENT_REGEX.test(seg));
+		// Unsloth ships draft heads in two layouts: `shared-` files borrow the
+		// embedding/output weights from the target model, others are self-contained.
+		const shared = segments.some((seg) => seg.toLowerCase() === HF_SHARED_DRAFT_TOKEN);
+
+		let quant = quantIdx >= 0 ? segments[quantIdx].toUpperCase() : null;
+
+		// Recombine a `UD-` (Unsloth Dynamic) prefix, e.g. `...-UD-Q4_K_XL.gguf`.
+		// The prefix must be the whole previous segment, matching the server's
+		// `UD-<quant>` custom-quant convention (e.g. not `-mtp-Q4_K_M`).
+		const udPrefixIdx = quantIdx - 1;
+
+		if (quant && quantIdx > 0 && segments[udPrefixIdx].toUpperCase() === HF_UD_QUANT_PREFIX) {
+			quant = `${HF_UD_QUANT_PREFIX}-${quant}`;
+		}
+
+		return { quant, shared, sidecar, sidecarForm };
+	}
+
+	static filterByExtension(siblings: HfModelSibling[], ext: string): HfModelSibling[] {
+		return siblings
+			.filter((f) => f.path.toLowerCase().endsWith(ext.toLowerCase()) && (f.size ?? 0) > 0)
+			.sort((a, b) => (b.size ?? 0) - (a.size ?? 0));
+	}
+
+	static formatDownloads(downloads: number): string {
+		if (downloads >= GIGABYTE) {
+			return `${(downloads / GIGABYTE).toFixed(1)}${GIGA_LABEL}`;
+		}
+
+		if (downloads >= MEGABYTE) {
+			return `${(downloads / MEGABYTE).toFixed(1)}${MEGA_LABEL}`;
+		}
+
+		if (downloads >= KILOBYTE) {
+			return `${(downloads / KILOBYTE).toFixed(1)}${KILO_LABEL}`;
+		}
+
+		return downloads.toString();
+	}
+
+	static formatFileSize(bytes: number): string {
+		if (bytes >= GIGABYTE) {
+			return `${(bytes / GIGABYTE).toFixed(1)} ${GIGABYTE_LABEL}`;
+		}
+
+		if (bytes >= MEGABYTE) {
+			return `${(bytes / MEGABYTE).toFixed(1)} ${MEGABYTE_LABEL}`;
+		}
+
+		if (bytes >= KILOBYTE) {
+			return `${(bytes / KILOBYTE).toFixed(1)} ${KILOBYTE_LABEL}`;
+		}
+
+		return `${bytes} ${BYTE_LABEL}`;
+	}
+
+	static formatLikes(likes: number): string {
+		if (likes >= KILOBYTE) {
+			return `${(likes / KILOBYTE).toFixed(1)}${KILO_LABEL}`;
+		}
+
+		return likes.toString();
+	}
+
+	static formatRelativeTime(timestamp: string): string {
+		const date = new Date(timestamp);
+		const now = new Date();
+		const diffMs = now.getTime() - date.getTime();
+		// timestamps can lie in the future (clock skew); clamp so they read as today
+		const diffDays = Math.max(0, Math.floor(diffMs / MS_PER_DAY));
+
+		if (diffDays === 0) return TODAY_LABEL;
+
+		if (diffDays === 1) return YESTERDAY_LABEL;
+
+		if (diffDays < DAYS_PER_WEEK) return `${diffDays} ${DAYS_AGO_LABEL}`;
+
+		if (diffDays < DAYS_PER_MONTH) {
+			return `${Math.floor(diffDays / DAYS_PER_WEEK)} ${WEEKS_AGO_LABEL}`;
+		}
+
+		if (diffDays < DAYS_PER_YEAR) {
+			return `${Math.floor(diffDays / DAYS_PER_MONTH)} ${MONTHS_AGO_LABEL}`;
+		}
+
+		return `${Math.floor(diffDays / DAYS_PER_YEAR)} ${YEARS_AGO_LABEL}`;
+	}
+
+	/**
+	 * Format a min-max size range with one shared unit, e.g. `19.0-28.6 GB`.
+	 */
+	static formatSizeRange(min: number, max: number): string {
+		const unit =
+			max >= GIGABYTE
+				? GIGABYTE_LABEL
+				: max >= MEGABYTE
+					? MEGABYTE_LABEL
+					: max >= KILOBYTE
+						? KILOBYTE_LABEL
+						: BYTE_LABEL;
+		const div =
+			unit === GIGABYTE_LABEL
+				? GIGABYTE
+				: unit === MEGABYTE_LABEL
+					? MEGABYTE
+					: unit === KILOBYTE_LABEL
+						? KILOBYTE
+						: BYTE;
+		const fmt = (n: number) => (div === BYTE ? `${n}` : `${(n / div).toFixed(1)}`);
+
+		return `${fmt(min)}-${fmt(max)} ${unit}`;
+	}
+
+	// Model Details & Files
+
+	/**
+	 * Avatar URL for an author (org or user). 404s when the author does not
+	 * exist, so callers should provide a fallback.
+	 */
+	static getAvatarUrl(author: string): string {
+		// OpenRouter-style model ids prefix the provider with a tilde
+		// (`~openai/gpt-...`); the avatars endpoint only resolves bare names
+		return `${HF_AVATARS_URL}${PATH_SEPARATOR}${author.replace(/^~/, '')}`;
+	}
+
+	/**
+	 * Resolve the original (non-GGUF) base model `{ org, name }` for a GGUF repo
+	 * from its HF card (`cardData.base_model`). Returns null when the card has no
+	 * base model. Results are cached per repo.
+	 */
+	static getBaseModel(repoId: string): Promise<{ org: string; name: string } | null> {
+		const cached = this.baseModelCache.get(repoId);
+
+		if (cached !== undefined) return Promise.resolve(cached);
+
+		const pending = this.baseModelPending.get(repoId);
+
+		if (pending) return pending;
+
+		const promise = (async () => {
+			const details = await this.getDetails(repoId);
+			const base = this.getBaseModels(details)[0];
+
+			if (!base) return null;
+
+			const [org, ...rest] = base.split(PATH_SEPARATOR);
+
+			return { name: rest.join(PATH_SEPARATOR), org };
+		})();
+
+		this.baseModelPending.set(repoId, promise);
+
+		promise
+			.then((result) => this.baseModelCache.set(repoId, result))
+			.finally(() => this.baseModelPending.delete(repoId));
+
+		return promise;
+	}
+
+	/**
+	 * Extract the original (non-GGUF) base model ids for a repo, from
+	 * `cardData.base_model` (string or list) and the `base_model:` tags.
+	 */
+	static getBaseModels(model: HfModelDetailInfo | null): string[] {
+		if (!model) return [];
+
+		const cardBase = model.cardData?.base_model;
+		const fromCard: string[] = Array.isArray(cardBase) ? cardBase : cardBase ? [cardBase] : [];
+		const fromTags = (model.tags ?? [])
+			.map((t) => HF_BASE_MODEL_TAG_REGEX.exec(t)?.[1])
+			.filter((v): v is string => Boolean(v));
+
+		return Array.from(new Set([...fromCard, ...fromTags]));
+	}
+
+	/**
+	 * Look up the average bit-depth for a known GGUF quantization.
+	 * Returns `null` for unrecognized tokens.
+	 */
+	static getBitDepth(quant: string): number | null {
+		const base = quant.replace(HF_UD_QUANT_PREFIX_REGEX, '');
+		const direct = HuggingFaceService.QUANT_BIT_DEPTH[base];
+
+		if (direct !== undefined) return direct;
+
+		// Fall back to the leading precision digits for variants missing from the
+		// map, e.g. `Q4_K_XL` -> 4, `IQ2_XXS` -> 2, `TQ1_0` -> 1, `BF16` -> 16.
+		const match = HF_QUANT_PRECISION_REGEX.exec(base);
+
+		return match ? parseInt(match[1], 10) : null;
+	}
+
+	static async getByTask(
+		pipelineTag: string,
+		params: Omit<HfModelSearchParams, 'pipeline_tag'> = {}
+	): Promise<HfModelInfo[]> {
+		return this.search({
+			...params,
+			pipeline_tag: pipelineTag
+		});
+	}
+
+	static async getCatalog(): Promise<HfCatalogEntry[]> {
+		const response = await fetch(MODELS_DISCOVER_CATALOG_URL);
+
+		if (!response.ok) throw new Error(`Failed to fetch catalog: ${response.status}`);
+
+		return (await response.json()) as HfCatalogEntry[];
+	}
+
+	static async getDetails(modelId: string): Promise<HfModelDetailInfo | null> {
+		// Do not encode the modelId, it contains slashes for author/name.
+		// `full=true` includes cardData (description, base_model) and safetensors.
+		const url = `${HF_API_MODELS_URL}${PATH_SEPARATOR}${modelId}?${HF_FULL_DETAIL_PARAM}`;
+
+		try {
+			const response = await fetch(url);
+
+			if (response.status === HF_HTTP_NOT_FOUND) return null;
+
+			if (!response.ok) throw new Error(`Failed to fetch model details: ${response.status}`);
+
+			const data = (await response.json()) as HfModelDetailInfo;
+
+			return data;
+		} catch (error) {
+			console.error(`Error fetching details for ${modelId}:`, error);
+
+			return null;
+		}
+	}
+
+	static getModelUrl(modelId: string): string {
+		return `${HF_BASE_URL}${PATH_SEPARATOR}${modelId}`;
+	}
+
+	// Utility Methods
+
+	static async getMostLiked(limit: number = HF_DEFAULT_LIMIT): Promise<HfModelInfo[]> {
+		return this.search({ limit, sort: HfModelSort.LIKES });
+	}
+
+	static async getNew(limit: number = HF_DEFAULT_LIMIT): Promise<HfModelInfo[]> {
+		return this.search({ limit, sort: HfModelSort.CREATED_AT });
+	}
+
+	static async getPopular(limit: number = HF_DEFAULT_LIMIT): Promise<HfModelInfo[]> {
+		return this.search({ limit, sort: HfModelSort.DOWNLOADS });
+	}
+
+	/**
+	 * Fetch the raw README.md for a repo, with the YAML frontmatter stripped.
+	 */
+	static async getReadme(modelId: string): Promise<string | null> {
+		// Do not encode the modelId, it contains slashes for author/name
+		const url = `${HF_BASE_URL}${PATH_SEPARATOR}${modelId}${PATH_SEPARATOR}${HF_RAW_PATH}${PATH_SEPARATOR}${HF_MAIN_BRANCH}${PATH_SEPARATOR}${HF_README_FILENAME}`;
+
+		try {
+			const response = await fetch(url);
+
+			if (response.status === HF_HTTP_NOT_FOUND) return null;
+
+			if (!response.ok) throw new Error(`Failed to fetch README: ${response.status}`);
+
+			return HuggingFaceService.stripFrontmatter(await response.text());
+		} catch (error) {
+			console.error(`Error fetching README for ${modelId}:`, error);
+
+			return null;
+		}
+	}
+
+	/**
+	 * Get repository file tree to list available GGUF variants. Recursive so
+	 * repos that keep quants in per-quant subdirectories (e.g. `UD-Q4_K_XL/`)
+	 * are included; follows cursor pagination for repos over one page.
+	 */
+	static async getTree(modelId: string): Promise<HfModelSibling[]> {
+		const files: HfModelSibling[] = [];
+		const firstUrl =
+			`${HF_API_MODELS_URL}${PATH_SEPARATOR}${modelId}${PATH_SEPARATOR}${HF_TREE_PATH}` +
+			`${PATH_SEPARATOR}${HF_MAIN_BRANCH}?${HF_RECURSIVE_TREE_PARAM}`;
+
+		let url: string | null = firstUrl;
+
+		try {
+			for (let page = 0; url && page < HF_TREE_MAX_PAGES; page++) {
+				const response: Response = await fetch(url);
+
+				if (!response.ok) return files;
+
+				const data = (await response.json()) as HfModelSibling[];
+
+				files.push(...data.filter((f) => f.type !== HfEntryType.DIRECTORY));
+
+				url = HuggingFaceService.parseNextPageUrl(response.headers.get(HF_LINK_HEADER));
+			}
+		} catch {
+			// Return whatever was fetched before the failure.
+		}
+
+		return files;
+	}
+
+	static async getTrending(limit: number = HF_DEFAULT_LIMIT): Promise<HfModelInfo[]> {
+		return this.search({ limit, sort: HfModelSort.TRENDING_SCORE });
+	}
+
+	/**
+	 * Parse a local HF cache file path
+	 * (`.../models--<org>--<name>/snapshots/<sha>/<file>`) into its repo id and
+	 * repo-relative file path. Returns null when the path is not an HF cache path.
+	 */
+	static parseCachePath(path: string): { repo: string; file: string } | null {
+		// the paths come from the server's CLI args, which use native separators
+		const match = HF_CACHE_PATH_REGEX.exec(path.replace(/\\/g, PATH_SEPARATOR));
+
+		if (!match) return null;
+
+		const parts = match[1].split(HF_CACHE_DIR_SEPARATOR);
+
+		if (parts.length < 2) return null;
+
+		return {
+			file: match[2],
+			repo: `${parts[0]}${PATH_SEPARATOR}${parts.slice(1).join(HF_CACHE_DIR_SEPARATOR)}`
+		};
+	}
+
+	/**
+	 * Best-effort parameter count parsed from a model id/name, e.g. `27B` from
+	 * `Qwen3.8-27B-GGUF` or `300M` from `embeddinggemma-300M-GGUF`. Returns null
+	 * when no size token is present.
+	 */
+	static parseParamCount(name: string): string | null {
+		const match = HF_PARAM_COUNT_REGEX.exec(name);
+
+		if (!match) return null;
+
+		return `${match[1]}${match[2].toUpperCase()}`;
+	}
+
+	/**
+	 * Parse a human size string (`177GB`, `1.2 TB`, `500MB`) to bytes. Returns
+	 * null when it carries no number or no known suffix, so callers can fall
+	 * back to another source instead of showing a wrong size.
+	 */
+	static parseSizeBytes(size: string): number | null {
+		const match = HF_SIZE_STRING_REGEX.exec(size);
+
+		if (!match) return null;
+
+		const value = parseFloat(match[1]);
+		const multiplier = HF_SIZE_SUFFIX_BYTES[match[2].toLowerCase()];
+
+		if (!Number.isFinite(value) || multiplier === undefined) return null;
+
+		return value * multiplier;
+	}
+
+	static parseTags(tags: string[]): {
+		license: string | null;
+		isGated: boolean;
+		isGguf: boolean;
+		isSafetensors: boolean;
+		tasks: string[];
+	} {
+		const license =
+			tags
+				.find((tag) => tag.startsWith(HF_LICENSE_TAG_PREFIX))
+				?.replace(HF_LICENSE_TAG_PREFIX, '') || null;
+		const isGated = tags.includes(HF_GATED_TAG);
+		const isGguf = tags.includes(HF_GGUF_TAG);
+		const isSafetensors = tags.includes(HF_SAFETENSORS_TAG);
+		const tasks = tags.filter((tag) => HF_TASK_TAGS.includes(tag));
+
+		return { isGated, isGguf, isSafetensors, license, tasks };
+	}
+
+	/**
+	 * Search GGUF models with various filters and options.
+	 *
+	 * Always expands the fields the discover rows render (chat template, context
+	 * length, siblings, ...) so a search result carries the same badges as a
+	 * catalog entry; caller-provided `expand` entries are merged in.
+	 */
+	static async search(params: HfModelSearchParams = {}): Promise<HfModelInfo[]> {
+		const { expand, limit = HF_DEFAULT_LIMIT, ...restParams } = params;
+		const url = this.buildUrl({
+			...restParams,
+			expand: [...new Set([...HF_MODEL_LIST_EXPAND, ...(expand ?? [])])],
+			filter: HF_GGUF_FILTER,
+			limit: Math.min(limit, HF_MAX_LIMIT)
+		});
+
+		return this.fetchWithRetry(url);
+	}
+
+	static async searchByQuery(
+		query: string,
+		params: Omit<HfModelSearchParams, 'search'> = {}
+	): Promise<HfModelInfo[]> {
+		return this.search({
+			...params,
+			search: query
+		});
+	}
+
+	private static buildUrl(params: HfModelSearchParams): string {
+		const url = new URL(this.BASE_URL);
+
+		Object.entries(params).forEach(([key, value]) => {
+			if (value !== undefined && value !== null && value !== '') {
+				if (Array.isArray(value)) {
+					value.forEach((v) => url.searchParams.append(key, v));
+				} else {
+					url.searchParams.set(key, String(value));
+				}
+			}
+		});
+
+		return url.toString();
+	}
+
+	private static delay(ms: number): Promise<void> {
+		return new Promise((resolve) => setTimeout(resolve, ms));
+	}
+
+	private static async fetchWithRetry(url: string, attempt: number = 1): Promise<HfModelInfo[]> {
+		try {
+			const response = await fetch(url);
+
+			if (!response.ok) {
+				if (response.status === HF_HTTP_NOT_FOUND) {
+					return [];
+				}
+
+				if (response.status >= HF_HTTP_SERVER_ERROR_MIN && attempt < HF_RETRY_ATTEMPTS) {
+					await this.delay(HF_RETRY_DELAY_MS * attempt);
+
+					return this.fetchWithRetry(url, attempt + 1);
+				}
+
+				throw new HfHttpStatusError(response.status, response.statusText);
+			}
+
+			const data = await response.json();
+
+			if (Array.isArray(data)) {
+				return data as HfModelInfo[];
+			}
+
+			if (data && Array.isArray(data.data)) {
+				return data.data as HfModelInfo[];
+			}
+
+			throw new Error('Unexpected API response format');
+		} catch (error) {
+			const transient =
+				error instanceof TypeError ||
+				(error instanceof HfHttpStatusError && error.status >= HF_HTTP_SERVER_ERROR_MIN);
+
+			if (transient && attempt < HF_RETRY_ATTEMPTS) {
+				await this.delay(HF_RETRY_DELAY_MS * attempt);
+
+				return this.fetchWithRetry(url, attempt + 1);
+			}
+
+			throw error;
+		}
+	}
+
+	// Internal Methods
+
+	/** Extract the `rel="next"` URL from an RFC 5988 `Link` header, if present. */
+	private static parseNextPageUrl(linkHeader: string | null): string | null {
+		if (!linkHeader) return null;
+
+		const match = HF_LINK_NEXT_REGEX.exec(linkHeader);
+
+		return match ? match[1] : null;
+	}
+
+	/** Full path of one shard in a split-shard GGUF set. */
+	private static shardPath(stem: string, index: number, total: number): string {
+		const pad = (n: number) => String(n).padStart(HF_SHARD_PAD_WIDTH, '0');
+
+		return `${stem}-${pad(index)}-of-${pad(total)}.gguf`;
+	}
+
+	/** Strip a leading YAML frontmatter block (--- ... ---) from a markdown document. */
+	private static stripFrontmatter(text: string): string {
+		const match = text.match(HF_FRONTMATTER_REGEX);
+
+		return match ? text.slice(match[0].length) : text;
+	}
+}
diff --git a/tools/ui/src/lib/services/index.ts b/tools/ui/src/lib/services/index.ts
index be6261bd6..a8072e55b 100644
--- a/tools/ui/src/lib/services/index.ts
+++ b/tools/ui/src/lib/services/index.ts
@@ -147,6 +147,16 @@ export { ConversationTransferService } from './conversation-transfer.service';
  */
 export { ModelsService } from './models.service';

+/**
+ * **HuggingFaceService** - Hugging Face Hub browsing and searching
+ *
+ * Stateless HTTP client for the HF REST API (`/api/models`, `/tree`, raw
+ * README) and the llama.app model catalog. Provides GGUF file analysis
+ * (quant metadata, shard collapsing, size formatting) used by the models
+ * discover UI.
+ */
+export { HuggingFaceService } from './huggingface.service';
+
 /**
  * **PropsService** - Server properties and capabilities retrieval
  *
diff --git a/tools/ui/src/lib/types/huggingface.d.ts b/tools/ui/src/lib/types/huggingface.d.ts
new file mode 100644
index 000000000..675c86b33
--- /dev/null
+++ b/tools/ui/src/lib/types/huggingface.d.ts
@@ -0,0 +1,227 @@
+/**
+ * HuggingFace Hub Model Browsing Types
+ *
+ * Types for the HuggingFace REST API (/api/models)
+ * Reference: https://huggingface.co/docs/huggingface_hub/package_reference/hf_api
+ */
+
+import type { HfEntryType, HfModelSort } from '$lib/enums';
+
+// Search Options
+
+export interface HfModelSearchParams {
+	/** Full-text search query */
+	search?: string;
+	/** Filter by pipeline task (e.g., "text-generation", "image-generation") */
+	pipeline_tag?: string;
+	/** Filter by library (e.g., "transformers", "diffusers", "gguf") */
+	library_name?: string;
+	/** Filter by tag (e.g., "gguf") */
+	filter?: string;
+	/** Filter by author or organization */
+	author?: string;
+	/** Sort field */
+	sort?: HfModelSort;
+	/** Results per page (1-100) */
+	limit?: number;
+	/** Pagination offset */
+	offset?: number;
+	/** Filter by model config */
+	config?: string;
+	/** Return full model info */
+	full?: boolean;
+	/**
+	 * Fields to include beyond the default set (repeated as `expand=<field>`).
+	 * The list endpoint returns only `_id`, `id`, `modelId` and the sort field
+	 * unless this is given, so callers rendering badges must ask for them.
+	 */
+	expand?: string[];
+	/** Filter by visibility */
+	private?: boolean;
+	/** Filter by gated status */
+	gated?: boolean;
+}
+
+// Model Info (from /api/models)
+
+export interface HfModelInfo {
+	/** Unique document ID */
+	_id: string;
+	/** Model ID (e.g., "meta-llama/Llama-3.1-8B-Instruct") */
+	id: string;
+	/** Number of likes */
+	likes: number;
+	/** Trending score; only present when the query sorts by it or expands the field */
+	trendingScore?: number;
+	/** Whether the model is private */
+	private: boolean;
+	/** Number of downloads */
+	downloads: number;
+	/** Model tags */
+	tags: string[];
+	/** Pipeline task (e.g., "text-generation") */
+	pipeline_tag: string | null;
+	/** Library name (e.g., "transformers", "diffusers") */
+	library_name: string | null;
+	/** Creation timestamp; only present when the query expands the field */
+	createdAt?: string;
+	/** Model ID (alias for id) */
+	modelId: string;
+	/** Author / organization (present when full=true) */
+	author?: string;
+	/** Last modified timestamp (present when full=true) */
+	lastModified?: string;
+	/** Repository file listing (present when full=true) */
+	siblings?: HfModelSiblingRef[];
+	/** GGUF metadata (context length, architecture, etc.) */
+	gguf?: HfModelGguf;
+}
+
+// Model Details (with full=true)
+
+export interface HfModelCardData {
+	/** License identifier */
+	license?: string;
+	/** License URL */
+	license_link?: string;
+	/** Model description */
+	description?: string;
+	/** Model library */
+	language?: string[];
+	/** Tags */
+	tags?: string[];
+	/** Original (non-GGUF) model(s) this repo was converted from, e.g. `Qwen/Qwen3.8-27B`. The API returns a single string or a list. */
+	base_model?: string | string[];
+	/** Org that produced the quant, e.g. `bartowski` */
+	quantized_by?: string;
+	[key: string]: unknown;
+}
+
+/** GGUF metadata returned by /api/models/{id}?full=true for GGUF repos. */
+export interface HfModelGguf {
+	/** Total parameter count */
+	total?: number;
+	/** Architecture, e.g. `gemma3`, `qwen3` */
+	architecture?: string;
+	/** Context length */
+	context_length?: number;
+	/** Chat template (Jinja) */
+	chat_template?: string;
+	bos_token?: string;
+	eos_token?: string;
+	/** Total size of all GGUF files in the repo, in bytes */
+	totalFileSize?: number;
+}
+
+export interface HfModelDetails {
+	/** Model ID */
+	id?: string;
+	/** SHA256 digest */
+	sha?: string;
+	/** Last modified timestamp */
+	lastModified?: string;
+	/** Downloads count */
+	downloads?: number;
+	/** Number of likes */
+	likes?: number;
+	/** Whether the model is gated */
+	gated?: boolean;
+	/** Model card data */
+	cardData?: HfModelCardData;
+	/** Tags */
+	tags?: string[];
+	/** Pipeline tag */
+	pipeline_tag?: string | null;
+	/** Library name */
+	library_name?: string | null;
+	/** Safe tensors info */
+	safetensors?: Record<string, unknown>;
+	/** Model size in bytes */
+	size?: number;
+	[key: string]: unknown;
+}
+
+export interface HfModelDetailInfo extends HfModelInfo {
+	/** Whether the model is gated (true/false/'auto') */
+	gated?: boolean | string;
+	/** Repository file listing mirrors of /api/models/{id}/tree/main */
+	siblings?: HfModelSiblingRef[];
+	/** Author / organization */
+	author?: string;
+	/** Last modified timestamp */
+	lastModified?: string;
+	/** Model card YAML data (only present when full=true) */
+	cardData?: HfModelCardData;
+	/** GGUF metadata (only present when full=true for GGUF repos) */
+	gguf?: HfModelGguf;
+	/** Model config (only present when full=true) */
+	config?: Record<string, unknown>;
+	/** Total repo storage in bytes (only present when full=true) */
+	usedStorage?: number;
+	/** Sample widget prompts */
+	widgetData?: Array<{ text?: string }>;
+	/** Related spaces */
+	spaces?: string[];
+}
+
+/** A single entry in a model repository's file tree (`/tree` responses) */
+export interface HfModelSibling {
+	/** Relative path of the file or directory within the repo */
+	path: string;
+	/** Size in bytes (omitted for directories) */
+	size?: number;
+	/** Whether this entry is a directory */
+	type?: HfEntryType;
+	/** OID/hash for the blob */
+	oid?: string;
+	[key: string]: unknown;
+}
+
+/**
+ * A single file entry in a model's `siblings` list. List (`/api/models`) and
+ * detail (`/api/models/{id}`) responses use `rfilename`, unlike `/tree`.
+ */
+export interface HfModelSiblingRef {
+	/** Relative file name within the repo */
+	rfilename: string;
+	[key: string]: unknown;
+}
+
+// API Response
+
+export interface HfModelApiResponse {
+	/** List of models */
+	data: HfModelInfo[];
+	/** Total count (if available) */
+	total?: number;
+}
+
+// llama.app model catalog (https://llama.app/v1/catalog.json)
+
+/** A single GGUF build/repo within a catalog size. */
+export interface HfCatalogBuild {
+	quant: string;
+	size: string;
+	sizeBytes: number;
+	repo: string;
+}
+
+/** A size variant (e.g. `GPT-OSS 20B`) within a catalog entry. */
+export interface HfCatalogSize {
+	name: string;
+	params: string;
+	builds: HfCatalogBuild[];
+}
+
+/** A single model family in the catalog. `featured` marks the staff picks. */
+export interface HfCatalogEntry {
+	name: string;
+	brand: string;
+	description: string;
+	details: string;
+	released: string;
+	license: string;
+	featured?: boolean;
+	maxMemGb?: number;
+	sizes: HfCatalogSize[];
+}
diff --git a/tools/ui/src/lib/types/index.ts b/tools/ui/src/lib/types/index.ts
index a16b5886e..28b7cf601 100644
--- a/tools/ui/src/lib/types/index.ts
+++ b/tools/ui/src/lib/types/index.ts
@@ -32,6 +32,22 @@ export type {
 	ApiStreamSession
 } from './api';

+// HuggingFace types
+export type {
+	HfCatalogBuild,
+	HfCatalogEntry,
+	HfCatalogSize,
+	HfModelApiResponse,
+	HfModelCardData,
+	HfModelDetails,
+	HfModelDetailInfo,
+	HfModelGguf,
+	HfModelInfo,
+	HfModelSearchParams,
+	HfModelSibling,
+	HfModelSiblingRef
+} from './huggingface';
+
 // Chat types
 export type {
 	AttachmentMenuItem,
diff --git a/tools/ui/src/lib/utils/model-names.ts b/tools/ui/src/lib/utils/model-names.ts
index 209f0f2ff..1918eedc8 100644
--- a/tools/ui/src/lib/utils/model-names.ts
+++ b/tools/ui/src/lib/utils/model-names.ts
@@ -1,4 +1,4 @@
-import { FILE_PATH_SEPARATOR_REGEX } from '$lib/constants';
+import { FILE_PATH_SEPARATOR_REGEX, MODEL_ID } from '$lib/constants';

 /**
  * Normalizes a model name by extracting the filename from a path, but preserves Hugging Face repository format.
@@ -56,3 +56,14 @@ export function normalizeModelName(modelName: string): string {
 export function isValidModelName(modelName: string): boolean {
 	return normalizeModelName(modelName).length > 0;
 }
+
+/**
+ * Org segment of a HuggingFace repo id (`ggml-org/Qwen3-8B` -> `ggml-org`).
+ * Returns the input itself when it carries no org separator, and an empty string
+ * for a missing id, so callers can use `||` against their own fallback org.
+ */
+export function orgOf(repoId: string | null | undefined): string {
+	if (!repoId) return '';
+
+	return repoId.split(MODEL_ID.ORG_SEPARATOR)[0] || repoId;
+}
diff --git a/tools/ui/tests/unit/hf-helpers.test.ts b/tools/ui/tests/unit/hf-helpers.test.ts
new file mode 100644
index 000000000..d73a8de22
--- /dev/null
+++ b/tools/ui/tests/unit/hf-helpers.test.ts
@@ -0,0 +1,128 @@
+import { HuggingFaceService } from '$lib/services/huggingface.service';
+import { describe, expect, it } from 'vitest';
+
+const {
+	collapseGgufShards,
+	formatSizeRange,
+	getBitDepth,
+	parseCachePath,
+	parseParamCount,
+	parseSizeBytes
+} = HuggingFaceService;
+
+describe('collapseGgufShards', () => {
+	it('passes non-sharded files through', () => {
+		const files = [
+			{ path: 'Model-Q4_K_M.gguf', size: 100 },
+			{ path: 'mmproj-F16.gguf', size: 10 }
+		];
+
+		expect(collapseGgufShards(files)).toStrictEqual(files);
+	});
+
+	it('collapses a shard set to its first shard with the summed size', () => {
+		const files = [
+			{ path: 'Model-00001-of-00003.gguf', size: 10 },
+			{ path: 'Model-00002-of-00003.gguf', size: 20 },
+			{ path: 'Model-00003-of-00003.gguf', size: 30 }
+		];
+
+		expect(collapseGgufShards(files)).toStrictEqual([
+			{ path: 'Model-00001-of-00003.gguf', size: 60 }
+		]);
+	});
+
+	it('treats a missing shard as zero bytes', () => {
+		const files = [
+			{ path: 'Model-00001-of-00002.gguf', size: 10 },
+			{ path: 'Model-Q8_0.gguf', size: 5 }
+		];
+
+		expect(collapseGgufShards(files)).toStrictEqual([
+			{ path: 'Model-00001-of-00002.gguf', size: 10 },
+			{ path: 'Model-Q8_0.gguf', size: 5 }
+		]);
+	});
+});
+
+describe('getBitDepth', () => {
+	it('resolves known tokens', () => {
+		expect(getBitDepth('Q4_K_M')).toBe(4);
+		expect(getBitDepth('BF16')).toBe(16);
+		expect(getBitDepth('IQ2_XXS')).toBe(2);
+	});
+
+	it('strips the UD prefix', () => {
+		expect(getBitDepth('UD-Q4_K_XL')).toBe(4);
+	});
+
+	it('falls back to the leading precision digits', () => {
+		expect(getBitDepth('TQ1_0')).toBe(1);
+		expect(getBitDepth('MXFP4_MOE')).toBe(4);
+	});
+
+	it('returns null for unrecognized tokens', () => {
+		expect(getBitDepth('xyz')).toBeNull();
+		expect(getBitDepth('QUANT')).toBeNull();
+	});
+});
+
+describe('formatSizeRange', () => {
+	it('formats a gigabyte range without spaces around the dash', () => {
+		expect(formatSizeRange(19e9, 28.6e9)).toBe('19.0-28.6 GB');
+	});
+
+	it('downgrades the unit to the smaller bound when the max is small', () => {
+		expect(formatSizeRange(1e6, 2.5e6)).toBe('1.0-2.5 MB');
+	});
+
+	it('formats sub-kilobyte sizes in bytes', () => {
+		expect(formatSizeRange(100, 900)).toBe('100-900 B');
+	});
+});
+
+describe('parseCachePath', () => {
+	it('parses a posix cache path into repo and file', () => {
+		expect(
+			parseCachePath(
+				'/home/u/.cache/llama.cpp/models--ggml-org--Qwen3-8B-GGUF/snapshots/abc123/Q4_K_M.gguf'
+			)
+		).toStrictEqual({ file: 'Q4_K_M.gguf', repo: 'ggml-org/Qwen3-8B-GGUF' });
+	});
+
+	it('accepts windows separators', () => {
+		expect(
+			parseCachePath('C:\\cache\\models--org--Model\\snapshots\\sha\\sub\\file.gguf')
+		).toStrictEqual({ file: 'sub/file.gguf', repo: 'org/Model' });
+	});
+
+	it('returns null for non-cache paths', () => {
+		expect(parseCachePath('/models/foo.gguf')).toBeNull();
+	});
+});
+
+describe('parseParamCount', () => {
+	it('extracts billions and millions', () => {
+		expect(parseParamCount('Qwen3.8-27B-GGUF')).toBe('27B');
+		expect(parseParamCount('embeddinggemma-300M-GGUF')).toBe('300M');
+		expect(parseParamCount('Model-0.6B-Q4_K_M')).toBe('0.6B');
+	});
+
+	it('returns null when no size token is present', () => {
+		expect(parseParamCount('ggml-org/Laguna-S-GGUF')).toBeNull();
+	});
+});
+
+describe('parseSizeBytes', () => {
+	it('parses single-letter catalog size suffixes', () => {
+		expect(parseSizeBytes('177g')).toBe(177e9);
+		expect(parseSizeBytes('1.2 t')).toBe(1.2e12);
+		expect(parseSizeBytes('500m')).toBe(500e6);
+	});
+
+	it('returns null for malformed input', () => {
+		expect(parseSizeBytes('unknown')).toBeNull();
+		expect(parseSizeBytes('12 parsecs')).toBeNull();
+		expect(parseSizeBytes('')).toBeNull();
+	});
+});