Commit 9b4333611 for llama.cpp
commit 9b43336114386fd0dcac8b42226226a1c55692af
Author: Aleksander Grygier <aleksander.grygier@gmail.com>
Date: Wed Sep 30 12:42:16 2026 +0200
ui : Hugging Face Hub data layer (#27947)
* ui : Hugging Face Hub data layer
Add HuggingFaceService and its constants/enums/types: GGUF repo search, file
tree and model detail fetching, quant/sidecar filename analysis, shard-set
collapsing and the llama.app catalog feed, plus an orgOf() helper on the model
name utils.
Assisted-by: pi:GLM-5.3-Flash
* ui : strip provider tilde prefix from hub avatar urls
Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
* ui : trim redundant comments in the HF data layer service
Per review: drop JSDoc that restates the method name and inline comments
that restate the code; keep only comments carrying non-obvious context.
Assisted-by: pi:zai-org/GLM-5.3-Flash
* ui : harden the HF data layer error typing, cover the helpers in tests
Carries the HTTP status on retryable fetch errors instead of matching the
message text. Marks expand-dependent catalog fields optional and documents
the data/models index pairing. Adds table tests for the pure helpers.
Assisted-by: pi:zai-org/GLM-5.3-Flash
diff --git a/tools/ui/src/lib/constants/huggingface.constants.ts b/tools/ui/src/lib/constants/huggingface.constants.ts
new file mode 100644
index 000000000..39f6fbe5c
--- /dev/null
+++ b/tools/ui/src/lib/constants/huggingface.constants.ts
@@ -0,0 +1,232 @@
+/**
+ * HuggingFace Hub constants.
+ *
+ * URLs, parsing regexes and formatting units for the HuggingFaceService.
+ * Reference: https://huggingface.co/docs/huggingface_hub/package_reference/hf_api
+ */
+
+// API endpoints
+
+export const HF_BASE_URL = 'https://huggingface.co';
+export const HF_API_MODELS_URL = `${HF_BASE_URL}/api/models`;
+export const HF_AVATARS_URL = `${HF_BASE_URL}/api/avatars`;
+
+// Query params
+
+export const HF_FULL_DETAIL_PARAM = 'full=true';
+export const HF_RECURSIVE_TREE_PARAM = 'recursive=true';
+/** Search filter that restricts results to repos containing GGUF files. */
+export const HF_GGUF_FILTER = 'gguf';
+/** Repeatable `expand` query param selecting fields on the list endpoint. */
+export const HF_EXPAND_PARAM = 'expand';
+/**
+ * Fields the model list endpoint omits by default but the discover list rows
+ * render: `gguf` (chat template, context length, param count) drives the
+ * reasoning / tool-use icons and the context badge, `siblings` the vision and
+ * draft-sidecar badges. Without them those parts of a row stay empty.
+ */
+export const HF_MODEL_LIST_EXPAND: readonly string[] = [
+ 'author',
+ 'downloads',
+ 'gguf',
+ 'lastModified',
+ 'likes',
+ 'pipeline_tag',
+ 'siblings',
+ // `base_model:` tags, so search rows can show the base org's avatar as the
+ // main avatar with the quant org as the corner badge, like catalog rows.
+ 'tags'
+];
+
+// Repo file conventions
+
+export const HF_MAIN_BRANCH = 'main';
+export const HF_README_FILENAME = 'README.md';
+export const HF_RAW_PATH = 'raw';
+export const HF_TREE_PATH = 'tree';
+
+// Pagination
+
+export const HF_LINK_NEXT_REGEX = /<([^>]+)>;\s*rel="next"/;
+/** `Link` response header carrying the next page URL for cursor pagination. */
+export const HF_LINK_HEADER = 'Link';
+
+// Fetch retry
+
+export const HF_RETRY_ATTEMPTS = 3;
+export const HF_RETRY_DELAY_MS = 1000;
+export const HF_HTTP_NOT_FOUND = 404;
+export const HF_HTTP_SERVER_ERROR_MIN = 500;
+
+// Search limits
+
+export const HF_DEFAULT_LIMIT = 50;
+/** Safety cap on `/tree` pagination: more pages means a misbehaving endpoint. */
+export const HF_TREE_MAX_PAGES = 10;
+export const HF_MAX_LIMIT = 100;
+
+// GGUF shard files
+
+/** Matches a split-shard GGUF file name, e.g. `Model-00001-of-00015.gguf`. */
+export const HF_SHARD_REGEX = /-(\d{5})-of-(\d{5})\.gguf$/i;
+/** Index (1-based) of the first shard in a split-shard set. */
+export const HF_FIRST_SHARD = 1;
+/** Zero-padded width of the shard index in a split-shard file name. */
+export const HF_SHARD_PAD_WIDTH = 5;
+
+// Quantization tokens
+
+/** `UD-` (Unsloth Dynamic) custom quantization prefix, e.g. `UD-Q4_K_XL`. */
+export const HF_UD_QUANT_PREFIX = 'UD';
+export const HF_UD_QUANT_PREFIX_REGEX = /^UD-/i;
+/**
+ * Segment marking an Unsloth `shared-` draft head that borrows the target
+ * model's embedding/output weights, e.g. `...-shared-Q4_K_M.gguf`.
+ */
+export const HF_SHARED_DRAFT_TOKEN = 'shared';
+/**
+ * Extracts the leading precision digits from a quant token, e.g.
+ * `Q4_K_XL` -> 4, `IQ2_XXS` -> 2, `TQ1_0` -> 1, `BF16` -> 16.
+ */
+export const HF_QUANT_PRECISION_REGEX = /^(?:I?Q|TQ|BF|F|MXFP)?(\d+)/i;
+
+// Model card tags
+
+/** Matches the `base_model:` tag (plain or `quantized:`), capturing the repo id. */
+export const HF_BASE_MODEL_TAG_REGEX = /^base_model:(?:quantized:)?(.+)$/;
+export const HF_LICENSE_TAG_PREFIX = 'license:';
+export const HF_GATED_TAG = 'gated';
+export const HF_GGUF_TAG = 'gguf';
+export const HF_SAFETENSORS_TAG = 'safetensors';
+
+// Pipeline tasks (logic use only - matching `pipeline_tag` values against tags)
+
+/**
+ * `pipeline_tag` values grouped by the input/output modality they imply, used
+ * to derive a discover row's modality icons. A tag in more than one group (e.g.
+ * `image-to-video`) lights up each modality it belongs to.
+ */
+export const HF_MODALITY_PIPELINE_TAGS: Readonly<
+ Record<'audio' | 'video' | 'vision', readonly string[]>
+> = {
+ audio: [
+ 'audio-classification',
+ 'audio-to-audio',
+ 'automatic-speech-recognition',
+ 'text-to-speech',
+ 'voice-activity-detection'
+ ],
+ video: ['text-to-video', 'image-to-video', 'video-to-video'],
+ vision: ['image-text-to-text', 'image-to-text', 'text-to-image', 'image-to-video']
+};
+
+/** Filename token marking an mmproj sidecar sibling (unlocks vision / audio). */
+export const HF_MMPROJ_FILENAME_TOKEN = 'mmproj';
+
+export const HF_TASK_TAGS: readonly string[] = [
+ 'audio-classification',
+ 'audio-to-audio',
+ 'automatic-speech-recognition',
+ 'conversational',
+ 'depth-estimation',
+ 'feature-extraction',
+ 'fill-mask',
+ 'image-classification',
+ 'image-feature-extraction',
+ 'image-segmentation',
+ 'image-text-to-text',
+ 'image-to-text',
+ 'image-to-video',
+ 'object-detection',
+ 'question-answering',
+ 'reinforcement-learning',
+ 'robotics',
+ 'sentence-similarity',
+ 'summarization',
+ 'text2text-generation',
+ 'text-classification',
+ 'text-generation',
+ 'text-to-image',
+ 'text-to-speech',
+ 'text-to-video',
+ 'token-classification',
+ 'translation',
+ 'video-to-video',
+ 'voice-activity-detection',
+ 'zero-shot-classification'
+];
+
+// Formatting
+
+export const BYTE = 1;
+export const KILOBYTE = 1_000;
+export const MEGABYTE = 1_000_000;
+export const GIGABYTE = 1_000_000_000;
+export const TERABYTE = 1_000_000_000_000;
+
+/**
+ * Matches a human size string (`177GB`, `1.2 TB`, `500MB`), capturing the
+ * numeric value and its unit suffix. Used by `parseSizeBytes`.
+ */
+export const HF_SIZE_STRING_REGEX = /^\s*([\d.]+)\s*([a-z]+)\s*$/i;
+
+/**
+ * Byte multiplier for a size suffix (`k` kilobyte, `m` megabyte, ...) as used by
+ * the llama.app catalog `size` strings, whose suffix is lowercase.
+ */
+export const HF_SIZE_SUFFIX_BYTES: Readonly<Record<string, number>> = {
+ b: BYTE,
+ g: GIGABYTE,
+ k: KILOBYTE,
+ m: MEGABYTE,
+ t: TERABYTE
+};
+
+export const BYTE_LABEL = 'B';
+export const KILOBYTE_LABEL = 'KB';
+export const MEGABYTE_LABEL = 'MB';
+export const GIGABYTE_LABEL = 'GB';
+
+/** Count suffixes for compact number formatting, e.g. `1.5K`, `2.0M`. */
+export const KILO_LABEL = 'K';
+export const MEGA_LABEL = 'M';
+export const GIGA_LABEL = 'B';
+
+// Relative time
+
+export const MS_PER_DAY = 1000 * 60 * 60 * 24;
+export const DAYS_PER_WEEK = 7;
+/** Rough month length in days, used to bucket relative timestamps. */
+export const DAYS_PER_MONTH = 30;
+export const DAYS_PER_YEAR = 365;
+
+export const TODAY_LABEL = 'Today';
+export const YESTERDAY_LABEL = 'Yesterday';
+export const DAYS_AGO_LABEL = 'days ago';
+export const WEEKS_AGO_LABEL = 'weeks ago';
+export const MONTHS_AGO_LABEL = 'months ago';
+export const YEARS_AGO_LABEL = 'years ago';
+
+// Cache paths
+
+/**
+ * Matches a local HF cache file path
+ * (`.../models--<org>--<name>/snapshots/<sha>/<file>`), capturing the repo
+ * directory name and the repo-relative file path.
+ */
+export const HF_CACHE_PATH_REGEX = /models--(.+?)\/snapshots\/[^/]+\/(.+)$/;
+/** Separator between org and name segments in an HF cache directory name. */
+export const HF_CACHE_DIR_SEPARATOR = '--';
+
+// README
+
+/** Matches a leading YAML frontmatter block (--- ... ---) in a markdown document. */
+export const HF_FRONTMATTER_REGEX = /^---\r?\n[\s\S]*?\r?\n---\r?\n?/;
+
+// Param counts
+
+/**
+ * Best-effort parameter count token in a model id/name, e.g. `27B` from
+ * `Qwen3.8-27B-GGUF` or `300M` from `embeddinggemma-300M-GGUF`.
+ */
+export const HF_PARAM_COUNT_REGEX = /(?:^|[^a-z0-9])(\d+(?:[._]\d+)?)\s*([bm])(?![a-z0-9])/i;
diff --git a/tools/ui/src/lib/constants/index.ts b/tools/ui/src/lib/constants/index.ts
index d93ae6429..c66a2f771 100644
--- a/tools/ui/src/lib/constants/index.ts
+++ b/tools/ui/src/lib/constants/index.ts
@@ -45,6 +45,8 @@ export * from './message-export.constants';
export * from './path-display.constants';
export * from './model-id.constants';
export * from './model-loading.constants';
+export * from './models-discover.constants';
+export * from './huggingface.constants';
export * from './precision.constants';
export * from './pwa.constants';
export * from './routes.constants';
diff --git a/tools/ui/src/lib/constants/models-discover.constants.ts b/tools/ui/src/lib/constants/models-discover.constants.ts
new file mode 100644
index 000000000..f2a6a9463
--- /dev/null
+++ b/tools/ui/src/lib/constants/models-discover.constants.ts
@@ -0,0 +1,8 @@
+/**
+ * Models discover constants.
+ *
+ * Endpoints and settings for the Models Discover dialog.
+ */
+
+/** llama.app model catalog used as the default model list. Online-only source; the discover feature requires an internet connection anyway. */
+export const MODELS_DISCOVER_CATALOG_URL = 'https://llama.app/v1/catalog.json';
diff --git a/tools/ui/src/lib/enums/huggingface.enums.ts b/tools/ui/src/lib/enums/huggingface.enums.ts
new file mode 100644
index 000000000..7a9e0c3b6
--- /dev/null
+++ b/tools/ui/src/lib/enums/huggingface.enums.ts
@@ -0,0 +1,35 @@
+/**
+ * HuggingFace Hub enums.
+ *
+ * Values mirror the strings used by the HF REST API
+ * (https://huggingface.co/docs/huggingface_hub/package_reference/hf_api)
+ * so they can be sent and compared directly.
+ */
+
+/** Sort field for /api/models search queries. */
+export enum HfModelSort {
+ CREATED_AT = 'createdAt',
+ DOWNLOADS = 'downloads',
+ LAST_MODIFIED = 'lastModified',
+ LIKES = 'likes',
+ TRENDING_SCORE = 'trendingScore'
+}
+
+/**
+ * Where the sidecar token (`mtp` / `dflash` / `mmproj` / ...) sits in the
+ * filename.
+ * - `prefix` sidecar file that lives next to the main weights, e.g. `mtp-Q4_0.gguf`
+ * - `suffix` embedded draft baked into the main weights, e.g. `Hy3-IQ1_M-mtp.gguf`
+ * - `infix` standalone sidecar named between head and quant, e.g. `model-mtp-Q8_0.gguf`
+ */
+export enum SidecarForm {
+ INFIX = 'infix',
+ PREFIX = 'prefix',
+ SUFFIX = 'suffix'
+}
+
+/** Entry type in a model repository file tree (`/tree` responses). */
+export enum HfEntryType {
+ DIRECTORY = 'directory',
+ FILE = 'file'
+}
diff --git a/tools/ui/src/lib/enums/index.ts b/tools/ui/src/lib/enums/index.ts
index b20875f80..db4f5b68c 100644
--- a/tools/ui/src/lib/enums/index.ts
+++ b/tools/ui/src/lib/enums/index.ts
@@ -57,6 +57,8 @@ export {
SpecialFileType
} from './files.enums';
+export { HfEntryType, HfModelSort, SidecarForm } from './huggingface.enums';
+
export {
MCPConnectionPhase,
MCPLogLevel,
diff --git a/tools/ui/src/lib/services/huggingface.service.ts b/tools/ui/src/lib/services/huggingface.service.ts
new file mode 100644
index 000000000..346da82b5
--- /dev/null
+++ b/tools/ui/src/lib/services/huggingface.service.ts
@@ -0,0 +1,741 @@
+import { PATH_SEPARATOR } from '$lib/constants';
+import {
+ BYTE,
+ BYTE_LABEL,
+ DAYS_AGO_LABEL,
+ DAYS_PER_MONTH,
+ DAYS_PER_WEEK,
+ DAYS_PER_YEAR,
+ GIGA_LABEL,
+ GIGABYTE,
+ GIGABYTE_LABEL,
+ HF_API_MODELS_URL,
+ HF_AVATARS_URL,
+ HF_BASE_MODEL_TAG_REGEX,
+ HF_BASE_URL,
+ HF_CACHE_DIR_SEPARATOR,
+ HF_CACHE_PATH_REGEX,
+ HF_DEFAULT_LIMIT,
+ HF_FIRST_SHARD,
+ HF_FRONTMATTER_REGEX,
+ HF_FULL_DETAIL_PARAM,
+ HF_GATED_TAG,
+ HF_GGUF_FILTER,
+ HF_GGUF_TAG,
+ HF_HTTP_NOT_FOUND,
+ HF_HTTP_SERVER_ERROR_MIN,
+ HF_LICENSE_TAG_PREFIX,
+ HF_LINK_HEADER,
+ HF_LINK_NEXT_REGEX,
+ HF_MAIN_BRANCH,
+ HF_MAX_LIMIT,
+ HF_MODEL_LIST_EXPAND,
+ HF_PARAM_COUNT_REGEX,
+ HF_QUANT_PRECISION_REGEX,
+ HF_RAW_PATH,
+ HF_README_FILENAME,
+ HF_RECURSIVE_TREE_PARAM,
+ HF_RETRY_ATTEMPTS,
+ HF_RETRY_DELAY_MS,
+ HF_SAFETENSORS_TAG,
+ HF_SHARD_PAD_WIDTH,
+ HF_SHARD_REGEX,
+ HF_SHARED_DRAFT_TOKEN,
+ HF_SIZE_STRING_REGEX,
+ HF_SIZE_SUFFIX_BYTES,
+ HF_TASK_TAGS,
+ HF_TREE_MAX_PAGES,
+ HF_TREE_PATH,
+ HF_UD_QUANT_PREFIX,
+ HF_UD_QUANT_PREFIX_REGEX,
+ KILO_LABEL,
+ KILOBYTE,
+ KILOBYTE_LABEL,
+ MEGA_LABEL,
+ MEGABYTE,
+ MEGABYTE_LABEL,
+ MODELS_DISCOVER_CATALOG_URL,
+ MONTHS_AGO_LABEL,
+ MS_PER_DAY,
+ TODAY_LABEL,
+ WEEKS_AGO_LABEL,
+ YEARS_AGO_LABEL,
+ YESTERDAY_LABEL
+} from '$lib/constants';
+import { MODEL_ID, type ModelSidecar } from '$lib/constants';
+import { HfEntryType, HfModelSort, SidecarForm } from '$lib/enums';
+import type {
+ HfCatalogEntry,
+ HfModelDetailInfo,
+ HfModelInfo,
+ HfModelSearchParams,
+ HfModelSibling
+} from '$lib/types/huggingface';
+import { sidecarFromFileToken } from '$lib/utils';
+
+/** Fetch failure carrying the HTTP status, so retry logic tests the code instead of the message. */
+class HfHttpStatusError extends Error {
+ status: number;
+
+ constructor(status: number, statusText: string) {
+ super(`API request failed: ${status} ${statusText}`);
+
+ this.status = status;
+ }
+}
+
+export class HuggingFaceService {
+ private static readonly BASE_URL = HF_API_MODELS_URL;
+
+ // Cached base model lookups keyed by repo id, so repeated selector opens
+ // never re-hit the HF API for the same repo.
+ private static baseModelCache = new Map<string, { org: string; name: string } | null>();
+
+ private static baseModelPending = new Map<
+ string,
+ Promise<{ org: string; name: string } | null>
+ >();
+
+ /**
+ * Map of quant token to its average bit-depth in bits-per-weight (bpw).
+ */
+ private static readonly QUANT_BIT_DEPTH: Record<string, number> = {
+ BF16: 16,
+ F16: 16,
+ IQ1_M: 1,
+ IQ1_S: 1,
+ IQ1_XS: 1,
+ IQ1_XXS: 1,
+ IQ2_M: 2,
+ IQ2_S: 2,
+ IQ2_XS: 2,
+ IQ2_XXS: 2,
+ IQ3_M: 3,
+ IQ3_S: 3,
+ IQ3_XS: 3,
+ IQ3_XXS: 3,
+ Q2_K: 2,
+ Q2_K_M: 2,
+ Q2_K_S: 2,
+ Q3_K: 3,
+ Q3_K_L: 3,
+ Q3_K_M: 3,
+ Q3_K_S: 3,
+ Q4_0: 4,
+ Q4_1: 4,
+ Q4_K: 4,
+ Q4_K_M: 4,
+ Q4_K_S: 4,
+ Q5_0: 5,
+ Q5_1: 5,
+ Q5_K: 5,
+ Q5_K_M: 5,
+ Q5_K_S: 5,
+ Q6_K: 6,
+ Q8_0: 8
+ };
+
+ /**
+ * Collapse split GGUF shard sets (`-00001-of-00015.gguf`, ...) to their first
+ * shard, summing every shard's size so the kept entry reflects the whole
+ * quant. Downloads are tag-based (`repo:quant`), so the first shard is
+ * enough to represent the set.
+ */
+ static collapseGgufShards(siblings: HfModelSibling[]): HfModelSibling[] {
+ const sizeByPath = new Map(siblings.map((f) => [f.path, f.size ?? 0]));
+ const result: HfModelSibling[] = [];
+
+ for (const file of siblings) {
+ const match = HF_SHARD_REGEX.exec(file.path);
+
+ if (!match) {
+ result.push(file);
+
+ continue;
+ }
+
+ if (Number(match[1]) !== HF_FIRST_SHARD) continue;
+
+ const total = Number(match[2]);
+ const stem = file.path.slice(0, file.path.length - match[0].length);
+
+ let size = 0;
+
+ for (let i = HF_FIRST_SHARD; i <= total; i++) {
+ const shard = HuggingFaceService.shardPath(stem, i, total);
+
+ size += sizeByPath.get(shard) ?? 0;
+ }
+
+ result.push({ ...file, size });
+ }
+
+ return result;
+ }
+
+ // GGUF Model Browsing
+
+ /**
+ * Extract the GGUF quantization token (e.g. `Q4_K_M`) and any sidecar type
+ * (`mtp`, `dflash`, `mmproj`, ...) from a `.gguf` filename.
+ *
+ * `sidecarForm` records which side of the filename the sidecar token sat on
+ * so callers can render badges differently. `quant` and `sidecar` are `null`
+ * when absent; returns `null` for non-GGUF filenames.
+ */
+ static extractQuantMeta(filename: string): {
+ quant: string | null;
+ /** Draft-head-only variant borrowing embed/output weights from the target model. */
+ shared: boolean;
+ sidecar: ModelSidecar | null;
+ sidecarForm: SidecarForm | null;
+ } | null {
+ if (!MODEL_ID.WEIGHT_EXTENSION_REGEX.test(filename)) return null;
+
+ // HF repos may nest sidecars in a folder (e.g. `MTP/mtp-Model-Q4_0.gguf`);
+ // parse the file name only, the folder adds no quant information.
+ let source = (filename.split(PATH_SEPARATOR).pop() ?? filename).replace(
+ MODEL_ID.WEIGHT_EXTENSION_REGEX,
+ ''
+ );
+ let sidecar: ModelSidecar | null = null;
+ let sidecarForm: SidecarForm | null = null;
+
+ // A file named just the sidecar token (`imatrix.gguf`) is the sidecar
+ // itself: no name or quant segments to parse.
+ const bareSidecar = sidecarFromFileToken(source.toLowerCase());
+
+ if (bareSidecar) {
+ return { quant: null, shared: false, sidecar: bareSidecar, sidecarForm: SidecarForm.PREFIX };
+ }
+
+ const prefixMatch = source.match(MODEL_ID.SIDECAR_PREFIX_REGEX);
+
+ if (prefixMatch) {
+ sidecar = sidecarFromFileToken(prefixMatch[1].toLowerCase());
+ sidecarForm = SidecarForm.PREFIX;
+ source = prefixMatch[2];
+ } else {
+ const suffixMatch = source.match(MODEL_ID.SIDECAR_SUFFIX_REGEX);
+
+ if (suffixMatch) {
+ // Take the suffix sidecar even when the head carries no quant:
+ // embedded drafts end in one (`Hy3-IQ1_M-mtp`), standalone sidecar
+ // files do not (`Model-mtp-draft`, `Model-imatrix`).
+ sidecar = sidecarFromFileToken(suffixMatch[2].toLowerCase());
+ sidecarForm = SidecarForm.SUFFIX;
+ source = suffixMatch[1];
+ } else {
+ const infixMatch = source.match(MODEL_ID.SIDECAR_INFIX_REGEX);
+
+ if (infixMatch) {
+ sidecar = sidecarFromFileToken(infixMatch[2].toLowerCase());
+ sidecarForm = SidecarForm.INFIX;
+ source = `${infixMatch[1]}-${infixMatch[3]}`;
+ }
+ }
+ }
+
+ // Scan dash-separated segments left-to-right for the first quant match.
+ // - For sidecars like `mtp-Q4_0-180MB.gguf` the quant is `Q4_0`.
+ // - For embedded MTP like `Hy3-IQ1_M-mtp.gguf` we have `Hy3-IQ1_M` and `IQ1_M` matches.
+ // - For main files like `Llama-3-8B-Q4_K_M.gguf` we land on the trailing quant.
+ const segments = source.split(MODEL_ID.SEGMENT_SEPARATOR);
+ const quantIdx = segments.findIndex((seg) => MODEL_ID.QUANTIZATION_SEGMENT_REGEX.test(seg));
+ // Unsloth ships draft heads in two layouts: `shared-` files borrow the
+ // embedding/output weights from the target model, others are self-contained.
+ const shared = segments.some((seg) => seg.toLowerCase() === HF_SHARED_DRAFT_TOKEN);
+
+ let quant = quantIdx >= 0 ? segments[quantIdx].toUpperCase() : null;
+
+ // Recombine a `UD-` (Unsloth Dynamic) prefix, e.g. `...-UD-Q4_K_XL.gguf`.
+ // The prefix must be the whole previous segment, matching the server's
+ // `UD-<quant>` custom-quant convention (e.g. not `-mtp-Q4_K_M`).
+ const udPrefixIdx = quantIdx - 1;
+
+ if (quant && quantIdx > 0 && segments[udPrefixIdx].toUpperCase() === HF_UD_QUANT_PREFIX) {
+ quant = `${HF_UD_QUANT_PREFIX}-${quant}`;
+ }
+
+ return { quant, shared, sidecar, sidecarForm };
+ }
+
+ static filterByExtension(siblings: HfModelSibling[], ext: string): HfModelSibling[] {
+ return siblings
+ .filter((f) => f.path.toLowerCase().endsWith(ext.toLowerCase()) && (f.size ?? 0) > 0)
+ .sort((a, b) => (b.size ?? 0) - (a.size ?? 0));
+ }
+
+ static formatDownloads(downloads: number): string {
+ if (downloads >= GIGABYTE) {
+ return `${(downloads / GIGABYTE).toFixed(1)}${GIGA_LABEL}`;
+ }
+
+ if (downloads >= MEGABYTE) {
+ return `${(downloads / MEGABYTE).toFixed(1)}${MEGA_LABEL}`;
+ }
+
+ if (downloads >= KILOBYTE) {
+ return `${(downloads / KILOBYTE).toFixed(1)}${KILO_LABEL}`;
+ }
+
+ return downloads.toString();
+ }
+
+ static formatFileSize(bytes: number): string {
+ if (bytes >= GIGABYTE) {
+ return `${(bytes / GIGABYTE).toFixed(1)} ${GIGABYTE_LABEL}`;
+ }
+
+ if (bytes >= MEGABYTE) {
+ return `${(bytes / MEGABYTE).toFixed(1)} ${MEGABYTE_LABEL}`;
+ }
+
+ if (bytes >= KILOBYTE) {
+ return `${(bytes / KILOBYTE).toFixed(1)} ${KILOBYTE_LABEL}`;
+ }
+
+ return `${bytes} ${BYTE_LABEL}`;
+ }
+
+ static formatLikes(likes: number): string {
+ if (likes >= KILOBYTE) {
+ return `${(likes / KILOBYTE).toFixed(1)}${KILO_LABEL}`;
+ }
+
+ return likes.toString();
+ }
+
+ static formatRelativeTime(timestamp: string): string {
+ const date = new Date(timestamp);
+ const now = new Date();
+ const diffMs = now.getTime() - date.getTime();
+ // timestamps can lie in the future (clock skew); clamp so they read as today
+ const diffDays = Math.max(0, Math.floor(diffMs / MS_PER_DAY));
+
+ if (diffDays === 0) return TODAY_LABEL;
+
+ if (diffDays === 1) return YESTERDAY_LABEL;
+
+ if (diffDays < DAYS_PER_WEEK) return `${diffDays} ${DAYS_AGO_LABEL}`;
+
+ if (diffDays < DAYS_PER_MONTH) {
+ return `${Math.floor(diffDays / DAYS_PER_WEEK)} ${WEEKS_AGO_LABEL}`;
+ }
+
+ if (diffDays < DAYS_PER_YEAR) {
+ return `${Math.floor(diffDays / DAYS_PER_MONTH)} ${MONTHS_AGO_LABEL}`;
+ }
+
+ return `${Math.floor(diffDays / DAYS_PER_YEAR)} ${YEARS_AGO_LABEL}`;
+ }
+
+ /**
+ * Format a min-max size range with one shared unit, e.g. `19.0-28.6 GB`.
+ */
+ static formatSizeRange(min: number, max: number): string {
+ const unit =
+ max >= GIGABYTE
+ ? GIGABYTE_LABEL
+ : max >= MEGABYTE
+ ? MEGABYTE_LABEL
+ : max >= KILOBYTE
+ ? KILOBYTE_LABEL
+ : BYTE_LABEL;
+ const div =
+ unit === GIGABYTE_LABEL
+ ? GIGABYTE
+ : unit === MEGABYTE_LABEL
+ ? MEGABYTE
+ : unit === KILOBYTE_LABEL
+ ? KILOBYTE
+ : BYTE;
+ const fmt = (n: number) => (div === BYTE ? `${n}` : `${(n / div).toFixed(1)}`);
+
+ return `${fmt(min)}-${fmt(max)} ${unit}`;
+ }
+
+ // Model Details & Files
+
+ /**
+ * Avatar URL for an author (org or user). 404s when the author does not
+ * exist, so callers should provide a fallback.
+ */
+ static getAvatarUrl(author: string): string {
+ // OpenRouter-style model ids prefix the provider with a tilde
+ // (`~openai/gpt-...`); the avatars endpoint only resolves bare names
+ return `${HF_AVATARS_URL}${PATH_SEPARATOR}${author.replace(/^~/, '')}`;
+ }
+
+ /**
+ * Resolve the original (non-GGUF) base model `{ org, name }` for a GGUF repo
+ * from its HF card (`cardData.base_model`). Returns null when the card has no
+ * base model. Results are cached per repo.
+ */
+ static getBaseModel(repoId: string): Promise<{ org: string; name: string } | null> {
+ const cached = this.baseModelCache.get(repoId);
+
+ if (cached !== undefined) return Promise.resolve(cached);
+
+ const pending = this.baseModelPending.get(repoId);
+
+ if (pending) return pending;
+
+ const promise = (async () => {
+ const details = await this.getDetails(repoId);
+ const base = this.getBaseModels(details)[0];
+
+ if (!base) return null;
+
+ const [org, ...rest] = base.split(PATH_SEPARATOR);
+
+ return { name: rest.join(PATH_SEPARATOR), org };
+ })();
+
+ this.baseModelPending.set(repoId, promise);
+
+ promise
+ .then((result) => this.baseModelCache.set(repoId, result))
+ .finally(() => this.baseModelPending.delete(repoId));
+
+ return promise;
+ }
+
+ /**
+ * Extract the original (non-GGUF) base model ids for a repo, from
+ * `cardData.base_model` (string or list) and the `base_model:` tags.
+ */
+ static getBaseModels(model: HfModelDetailInfo | null): string[] {
+ if (!model) return [];
+
+ const cardBase = model.cardData?.base_model;
+ const fromCard: string[] = Array.isArray(cardBase) ? cardBase : cardBase ? [cardBase] : [];
+ const fromTags = (model.tags ?? [])
+ .map((t) => HF_BASE_MODEL_TAG_REGEX.exec(t)?.[1])
+ .filter((v): v is string => Boolean(v));
+
+ return Array.from(new Set([...fromCard, ...fromTags]));
+ }
+
+ /**
+ * Look up the average bit-depth for a known GGUF quantization.
+ * Returns `null` for unrecognized tokens.
+ */
+ static getBitDepth(quant: string): number | null {
+ const base = quant.replace(HF_UD_QUANT_PREFIX_REGEX, '');
+ const direct = HuggingFaceService.QUANT_BIT_DEPTH[base];
+
+ if (direct !== undefined) return direct;
+
+ // Fall back to the leading precision digits for variants missing from the
+ // map, e.g. `Q4_K_XL` -> 4, `IQ2_XXS` -> 2, `TQ1_0` -> 1, `BF16` -> 16.
+ const match = HF_QUANT_PRECISION_REGEX.exec(base);
+
+ return match ? parseInt(match[1], 10) : null;
+ }
+
+ static async getByTask(
+ pipelineTag: string,
+ params: Omit<HfModelSearchParams, 'pipeline_tag'> = {}
+ ): Promise<HfModelInfo[]> {
+ return this.search({
+ ...params,
+ pipeline_tag: pipelineTag
+ });
+ }
+
+ static async getCatalog(): Promise<HfCatalogEntry[]> {
+ const response = await fetch(MODELS_DISCOVER_CATALOG_URL);
+
+ if (!response.ok) throw new Error(`Failed to fetch catalog: ${response.status}`);
+
+ return (await response.json()) as HfCatalogEntry[];
+ }
+
+ static async getDetails(modelId: string): Promise<HfModelDetailInfo | null> {
+ // Do not encode the modelId, it contains slashes for author/name.
+ // `full=true` includes cardData (description, base_model) and safetensors.
+ const url = `${HF_API_MODELS_URL}${PATH_SEPARATOR}${modelId}?${HF_FULL_DETAIL_PARAM}`;
+
+ try {
+ const response = await fetch(url);
+
+ if (response.status === HF_HTTP_NOT_FOUND) return null;
+
+ if (!response.ok) throw new Error(`Failed to fetch model details: ${response.status}`);
+
+ const data = (await response.json()) as HfModelDetailInfo;
+
+ return data;
+ } catch (error) {
+ console.error(`Error fetching details for ${modelId}:`, error);
+
+ return null;
+ }
+ }
+
+ static getModelUrl(modelId: string): string {
+ return `${HF_BASE_URL}${PATH_SEPARATOR}${modelId}`;
+ }
+
+ // Utility Methods
+
+ static async getMostLiked(limit: number = HF_DEFAULT_LIMIT): Promise<HfModelInfo[]> {
+ return this.search({ limit, sort: HfModelSort.LIKES });
+ }
+
+ static async getNew(limit: number = HF_DEFAULT_LIMIT): Promise<HfModelInfo[]> {
+ return this.search({ limit, sort: HfModelSort.CREATED_AT });
+ }
+
+ static async getPopular(limit: number = HF_DEFAULT_LIMIT): Promise<HfModelInfo[]> {
+ return this.search({ limit, sort: HfModelSort.DOWNLOADS });
+ }
+
+ /**
+ * Fetch the raw README.md for a repo, with the YAML frontmatter stripped.
+ */
+ static async getReadme(modelId: string): Promise<string | null> {
+ // Do not encode the modelId, it contains slashes for author/name
+ const url = `${HF_BASE_URL}${PATH_SEPARATOR}${modelId}${PATH_SEPARATOR}${HF_RAW_PATH}${PATH_SEPARATOR}${HF_MAIN_BRANCH}${PATH_SEPARATOR}${HF_README_FILENAME}`;
+
+ try {
+ const response = await fetch(url);
+
+ if (response.status === HF_HTTP_NOT_FOUND) return null;
+
+ if (!response.ok) throw new Error(`Failed to fetch README: ${response.status}`);
+
+ return HuggingFaceService.stripFrontmatter(await response.text());
+ } catch (error) {
+ console.error(`Error fetching README for ${modelId}:`, error);
+
+ return null;
+ }
+ }
+
+ /**
+ * Get repository file tree to list available GGUF variants. Recursive so
+ * repos that keep quants in per-quant subdirectories (e.g. `UD-Q4_K_XL/`)
+ * are included; follows cursor pagination for repos over one page.
+ */
+ static async getTree(modelId: string): Promise<HfModelSibling[]> {
+ const files: HfModelSibling[] = [];
+ const firstUrl =
+ `${HF_API_MODELS_URL}${PATH_SEPARATOR}${modelId}${PATH_SEPARATOR}${HF_TREE_PATH}` +
+ `${PATH_SEPARATOR}${HF_MAIN_BRANCH}?${HF_RECURSIVE_TREE_PARAM}`;
+
+ let url: string | null = firstUrl;
+
+ try {
+ for (let page = 0; url && page < HF_TREE_MAX_PAGES; page++) {
+ const response: Response = await fetch(url);
+
+ if (!response.ok) return files;
+
+ const data = (await response.json()) as HfModelSibling[];
+
+ files.push(...data.filter((f) => f.type !== HfEntryType.DIRECTORY));
+
+ url = HuggingFaceService.parseNextPageUrl(response.headers.get(HF_LINK_HEADER));
+ }
+ } catch {
+ // Return whatever was fetched before the failure.
+ }
+
+ return files;
+ }
+
+ static async getTrending(limit: number = HF_DEFAULT_LIMIT): Promise<HfModelInfo[]> {
+ return this.search({ limit, sort: HfModelSort.TRENDING_SCORE });
+ }
+
+ /**
+ * Parse a local HF cache file path
+ * (`.../models--<org>--<name>/snapshots/<sha>/<file>`) into its repo id and
+ * repo-relative file path. Returns null when the path is not an HF cache path.
+ */
+ static parseCachePath(path: string): { repo: string; file: string } | null {
+ // the paths come from the server's CLI args, which use native separators
+ const match = HF_CACHE_PATH_REGEX.exec(path.replace(/\\/g, PATH_SEPARATOR));
+
+ if (!match) return null;
+
+ const parts = match[1].split(HF_CACHE_DIR_SEPARATOR);
+
+ if (parts.length < 2) return null;
+
+ return {
+ file: match[2],
+ repo: `${parts[0]}${PATH_SEPARATOR}${parts.slice(1).join(HF_CACHE_DIR_SEPARATOR)}`
+ };
+ }
+
+ /**
+ * Best-effort parameter count parsed from a model id/name, e.g. `27B` from
+ * `Qwen3.8-27B-GGUF` or `300M` from `embeddinggemma-300M-GGUF`. Returns null
+ * when no size token is present.
+ */
+ static parseParamCount(name: string): string | null {
+ const match = HF_PARAM_COUNT_REGEX.exec(name);
+
+ if (!match) return null;
+
+ return `${match[1]}${match[2].toUpperCase()}`;
+ }
+
+ /**
+ * Parse a human size string (`177GB`, `1.2 TB`, `500MB`) to bytes. Returns
+ * null when it carries no number or no known suffix, so callers can fall
+ * back to another source instead of showing a wrong size.
+ */
+ static parseSizeBytes(size: string): number | null {
+ const match = HF_SIZE_STRING_REGEX.exec(size);
+
+ if (!match) return null;
+
+ const value = parseFloat(match[1]);
+ const multiplier = HF_SIZE_SUFFIX_BYTES[match[2].toLowerCase()];
+
+ if (!Number.isFinite(value) || multiplier === undefined) return null;
+
+ return value * multiplier;
+ }
+
+ static parseTags(tags: string[]): {
+ license: string | null;
+ isGated: boolean;
+ isGguf: boolean;
+ isSafetensors: boolean;
+ tasks: string[];
+ } {
+ const license =
+ tags
+ .find((tag) => tag.startsWith(HF_LICENSE_TAG_PREFIX))
+ ?.replace(HF_LICENSE_TAG_PREFIX, '') || null;
+ const isGated = tags.includes(HF_GATED_TAG);
+ const isGguf = tags.includes(HF_GGUF_TAG);
+ const isSafetensors = tags.includes(HF_SAFETENSORS_TAG);
+ const tasks = tags.filter((tag) => HF_TASK_TAGS.includes(tag));
+
+ return { isGated, isGguf, isSafetensors, license, tasks };
+ }
+
+ /**
+ * Search GGUF models with various filters and options.
+ *
+ * Always expands the fields the discover rows render (chat template, context
+ * length, siblings, ...) so a search result carries the same badges as a
+ * catalog entry; caller-provided `expand` entries are merged in.
+ */
+ static async search(params: HfModelSearchParams = {}): Promise<HfModelInfo[]> {
+ const { expand, limit = HF_DEFAULT_LIMIT, ...restParams } = params;
+ const url = this.buildUrl({
+ ...restParams,
+ expand: [...new Set([...HF_MODEL_LIST_EXPAND, ...(expand ?? [])])],
+ filter: HF_GGUF_FILTER,
+ limit: Math.min(limit, HF_MAX_LIMIT)
+ });
+
+ return this.fetchWithRetry(url);
+ }
+
+ static async searchByQuery(
+ query: string,
+ params: Omit<HfModelSearchParams, 'search'> = {}
+ ): Promise<HfModelInfo[]> {
+ return this.search({
+ ...params,
+ search: query
+ });
+ }
+
+ private static buildUrl(params: HfModelSearchParams): string {
+ const url = new URL(this.BASE_URL);
+
+ Object.entries(params).forEach(([key, value]) => {
+ if (value !== undefined && value !== null && value !== '') {
+ if (Array.isArray(value)) {
+ value.forEach((v) => url.searchParams.append(key, v));
+ } else {
+ url.searchParams.set(key, String(value));
+ }
+ }
+ });
+
+ return url.toString();
+ }
+
+ private static delay(ms: number): Promise<void> {
+ return new Promise((resolve) => setTimeout(resolve, ms));
+ }
+
+ private static async fetchWithRetry(url: string, attempt: number = 1): Promise<HfModelInfo[]> {
+ try {
+ const response = await fetch(url);
+
+ if (!response.ok) {
+ if (response.status === HF_HTTP_NOT_FOUND) {
+ return [];
+ }
+
+ if (response.status >= HF_HTTP_SERVER_ERROR_MIN && attempt < HF_RETRY_ATTEMPTS) {
+ await this.delay(HF_RETRY_DELAY_MS * attempt);
+
+ return this.fetchWithRetry(url, attempt + 1);
+ }
+
+ throw new HfHttpStatusError(response.status, response.statusText);
+ }
+
+ const data = await response.json();
+
+ if (Array.isArray(data)) {
+ return data as HfModelInfo[];
+ }
+
+ if (data && Array.isArray(data.data)) {
+ return data.data as HfModelInfo[];
+ }
+
+ throw new Error('Unexpected API response format');
+ } catch (error) {
+ const transient =
+ error instanceof TypeError ||
+ (error instanceof HfHttpStatusError && error.status >= HF_HTTP_SERVER_ERROR_MIN);
+
+ if (transient && attempt < HF_RETRY_ATTEMPTS) {
+ await this.delay(HF_RETRY_DELAY_MS * attempt);
+
+ return this.fetchWithRetry(url, attempt + 1);
+ }
+
+ throw error;
+ }
+ }
+
+ // Internal Methods
+
+ /** Extract the `rel="next"` URL from an RFC 5988 `Link` header, if present. */
+ private static parseNextPageUrl(linkHeader: string | null): string | null {
+ if (!linkHeader) return null;
+
+ const match = HF_LINK_NEXT_REGEX.exec(linkHeader);
+
+ return match ? match[1] : null;
+ }
+
+ /** Full path of one shard in a split-shard GGUF set. */
+ private static shardPath(stem: string, index: number, total: number): string {
+ const pad = (n: number) => String(n).padStart(HF_SHARD_PAD_WIDTH, '0');
+
+ return `${stem}-${pad(index)}-of-${pad(total)}.gguf`;
+ }
+
+ /** Strip a leading YAML frontmatter block (--- ... ---) from a markdown document. */
+ private static stripFrontmatter(text: string): string {
+ const match = text.match(HF_FRONTMATTER_REGEX);
+
+ return match ? text.slice(match[0].length) : text;
+ }
+}
diff --git a/tools/ui/src/lib/services/index.ts b/tools/ui/src/lib/services/index.ts
index be6261bd6..a8072e55b 100644
--- a/tools/ui/src/lib/services/index.ts
+++ b/tools/ui/src/lib/services/index.ts
@@ -147,6 +147,16 @@ export { ConversationTransferService } from './conversation-transfer.service';
*/
export { ModelsService } from './models.service';
+/**
+ * **HuggingFaceService** - Hugging Face Hub browsing and searching
+ *
+ * Stateless HTTP client for the HF REST API (`/api/models`, `/tree`, raw
+ * README) and the llama.app model catalog. Provides GGUF file analysis
+ * (quant metadata, shard collapsing, size formatting) used by the models
+ * discover UI.
+ */
+export { HuggingFaceService } from './huggingface.service';
+
/**
* **PropsService** - Server properties and capabilities retrieval
*
diff --git a/tools/ui/src/lib/types/huggingface.d.ts b/tools/ui/src/lib/types/huggingface.d.ts
new file mode 100644
index 000000000..675c86b33
--- /dev/null
+++ b/tools/ui/src/lib/types/huggingface.d.ts
@@ -0,0 +1,227 @@
+/**
+ * HuggingFace Hub Model Browsing Types
+ *
+ * Types for the HuggingFace REST API (/api/models)
+ * Reference: https://huggingface.co/docs/huggingface_hub/package_reference/hf_api
+ */
+
+import type { HfEntryType, HfModelSort } from '$lib/enums';
+
+// Search Options
+
+export interface HfModelSearchParams {
+ /** Full-text search query */
+ search?: string;
+ /** Filter by pipeline task (e.g., "text-generation", "image-generation") */
+ pipeline_tag?: string;
+ /** Filter by library (e.g., "transformers", "diffusers", "gguf") */
+ library_name?: string;
+ /** Filter by tag (e.g., "gguf") */
+ filter?: string;
+ /** Filter by author or organization */
+ author?: string;
+ /** Sort field */
+ sort?: HfModelSort;
+ /** Results per page (1-100) */
+ limit?: number;
+ /** Pagination offset */
+ offset?: number;
+ /** Filter by model config */
+ config?: string;
+ /** Return full model info */
+ full?: boolean;
+ /**
+ * Fields to include beyond the default set (repeated as `expand=<field>`).
+ * The list endpoint returns only `_id`, `id`, `modelId` and the sort field
+ * unless this is given, so callers rendering badges must ask for them.
+ */
+ expand?: string[];
+ /** Filter by visibility */
+ private?: boolean;
+ /** Filter by gated status */
+ gated?: boolean;
+}
+
+// Model Info (from /api/models)
+
+export interface HfModelInfo {
+ /** Unique document ID */
+ _id: string;
+ /** Model ID (e.g., "meta-llama/Llama-3.1-8B-Instruct") */
+ id: string;
+ /** Number of likes */
+ likes: number;
+ /** Trending score; only present when the query sorts by it or expands the field */
+ trendingScore?: number;
+ /** Whether the model is private */
+ private: boolean;
+ /** Number of downloads */
+ downloads: number;
+ /** Model tags */
+ tags: string[];
+ /** Pipeline task (e.g., "text-generation") */
+ pipeline_tag: string | null;
+ /** Library name (e.g., "transformers", "diffusers") */
+ library_name: string | null;
+ /** Creation timestamp; only present when the query expands the field */
+ createdAt?: string;
+ /** Model ID (alias for id) */
+ modelId: string;
+ /** Author / organization (present when full=true) */
+ author?: string;
+ /** Last modified timestamp (present when full=true) */
+ lastModified?: string;
+ /** Repository file listing (present when full=true) */
+ siblings?: HfModelSiblingRef[];
+ /** GGUF metadata (context length, architecture, etc.) */
+ gguf?: HfModelGguf;
+}
+
+// Model Details (with full=true)
+
+export interface HfModelCardData {
+ /** License identifier */
+ license?: string;
+ /** License URL */
+ license_link?: string;
+ /** Model description */
+ description?: string;
+ /** Model library */
+ language?: string[];
+ /** Tags */
+ tags?: string[];
+ /** Original (non-GGUF) model(s) this repo was converted from, e.g. `Qwen/Qwen3.8-27B`. The API returns a single string or a list. */
+ base_model?: string | string[];
+ /** Org that produced the quant, e.g. `bartowski` */
+ quantized_by?: string;
+ [key: string]: unknown;
+}
+
+/** GGUF metadata returned by /api/models/{id}?full=true for GGUF repos. */
+export interface HfModelGguf {
+ /** Total parameter count */
+ total?: number;
+ /** Architecture, e.g. `gemma3`, `qwen3` */
+ architecture?: string;
+ /** Context length */
+ context_length?: number;
+ /** Chat template (Jinja) */
+ chat_template?: string;
+ bos_token?: string;
+ eos_token?: string;
+ /** Total size of all GGUF files in the repo, in bytes */
+ totalFileSize?: number;
+}
+
+export interface HfModelDetails {
+ /** Model ID */
+ id?: string;
+ /** SHA256 digest */
+ sha?: string;
+ /** Last modified timestamp */
+ lastModified?: string;
+ /** Downloads count */
+ downloads?: number;
+ /** Number of likes */
+ likes?: number;
+ /** Whether the model is gated */
+ gated?: boolean;
+ /** Model card data */
+ cardData?: HfModelCardData;
+ /** Tags */
+ tags?: string[];
+ /** Pipeline tag */
+ pipeline_tag?: string | null;
+ /** Library name */
+ library_name?: string | null;
+ /** Safe tensors info */
+ safetensors?: Record<string, unknown>;
+ /** Model size in bytes */
+ size?: number;
+ [key: string]: unknown;
+}
+
+export interface HfModelDetailInfo extends HfModelInfo {
+ /** Whether the model is gated (true/false/'auto') */
+ gated?: boolean | string;
+ /** Repository file listing mirrors of /api/models/{id}/tree/main */
+ siblings?: HfModelSiblingRef[];
+ /** Author / organization */
+ author?: string;
+ /** Last modified timestamp */
+ lastModified?: string;
+ /** Model card YAML data (only present when full=true) */
+ cardData?: HfModelCardData;
+ /** GGUF metadata (only present when full=true for GGUF repos) */
+ gguf?: HfModelGguf;
+ /** Model config (only present when full=true) */
+ config?: Record<string, unknown>;
+ /** Total repo storage in bytes (only present when full=true) */
+ usedStorage?: number;
+ /** Sample widget prompts */
+ widgetData?: Array<{ text?: string }>;
+ /** Related spaces */
+ spaces?: string[];
+}
+
+/** A single entry in a model repository's file tree (`/tree` responses) */
+export interface HfModelSibling {
+ /** Relative path of the file or directory within the repo */
+ path: string;
+ /** Size in bytes (omitted for directories) */
+ size?: number;
+ /** Whether this entry is a directory */
+ type?: HfEntryType;
+ /** OID/hash for the blob */
+ oid?: string;
+ [key: string]: unknown;
+}
+
+/**
+ * A single file entry in a model's `siblings` list. List (`/api/models`) and
+ * detail (`/api/models/{id}`) responses use `rfilename`, unlike `/tree`.
+ */
+export interface HfModelSiblingRef {
+ /** Relative file name within the repo */
+ rfilename: string;
+ [key: string]: unknown;
+}
+
+// API Response
+
+export interface HfModelApiResponse {
+ /** List of models */
+ data: HfModelInfo[];
+ /** Total count (if available) */
+ total?: number;
+}
+
+// llama.app model catalog (https://llama.app/v1/catalog.json)
+
+/** A single GGUF build/repo within a catalog size. */
+export interface HfCatalogBuild {
+ quant: string;
+ size: string;
+ sizeBytes: number;
+ repo: string;
+}
+
+/** A size variant (e.g. `GPT-OSS 20B`) within a catalog entry. */
+export interface HfCatalogSize {
+ name: string;
+ params: string;
+ builds: HfCatalogBuild[];
+}
+
+/** A single model family in the catalog. `featured` marks the staff picks. */
+export interface HfCatalogEntry {
+ name: string;
+ brand: string;
+ description: string;
+ details: string;
+ released: string;
+ license: string;
+ featured?: boolean;
+ maxMemGb?: number;
+ sizes: HfCatalogSize[];
+}
diff --git a/tools/ui/src/lib/types/index.ts b/tools/ui/src/lib/types/index.ts
index a16b5886e..28b7cf601 100644
--- a/tools/ui/src/lib/types/index.ts
+++ b/tools/ui/src/lib/types/index.ts
@@ -32,6 +32,22 @@ export type {
ApiStreamSession
} from './api';
+// HuggingFace types
+export type {
+ HfCatalogBuild,
+ HfCatalogEntry,
+ HfCatalogSize,
+ HfModelApiResponse,
+ HfModelCardData,
+ HfModelDetails,
+ HfModelDetailInfo,
+ HfModelGguf,
+ HfModelInfo,
+ HfModelSearchParams,
+ HfModelSibling,
+ HfModelSiblingRef
+} from './huggingface';
+
// Chat types
export type {
AttachmentMenuItem,
diff --git a/tools/ui/src/lib/utils/model-names.ts b/tools/ui/src/lib/utils/model-names.ts
index 209f0f2ff..1918eedc8 100644
--- a/tools/ui/src/lib/utils/model-names.ts
+++ b/tools/ui/src/lib/utils/model-names.ts
@@ -1,4 +1,4 @@
-import { FILE_PATH_SEPARATOR_REGEX } from '$lib/constants';
+import { FILE_PATH_SEPARATOR_REGEX, MODEL_ID } from '$lib/constants';
/**
* Normalizes a model name by extracting the filename from a path, but preserves Hugging Face repository format.
@@ -56,3 +56,14 @@ export function normalizeModelName(modelName: string): string {
export function isValidModelName(modelName: string): boolean {
return normalizeModelName(modelName).length > 0;
}
+
+/**
+ * Org segment of a HuggingFace repo id (`ggml-org/Qwen3-8B` -> `ggml-org`).
+ * Returns the input itself when it carries no org separator, and an empty string
+ * for a missing id, so callers can use `||` against their own fallback org.
+ */
+export function orgOf(repoId: string | null | undefined): string {
+ if (!repoId) return '';
+
+ return repoId.split(MODEL_ID.ORG_SEPARATOR)[0] || repoId;
+}
diff --git a/tools/ui/tests/unit/hf-helpers.test.ts b/tools/ui/tests/unit/hf-helpers.test.ts
new file mode 100644
index 000000000..d73a8de22
--- /dev/null
+++ b/tools/ui/tests/unit/hf-helpers.test.ts
@@ -0,0 +1,128 @@
+import { HuggingFaceService } from '$lib/services/huggingface.service';
+import { describe, expect, it } from 'vitest';
+
+const {
+ collapseGgufShards,
+ formatSizeRange,
+ getBitDepth,
+ parseCachePath,
+ parseParamCount,
+ parseSizeBytes
+} = HuggingFaceService;
+
+describe('collapseGgufShards', () => {
+ it('passes non-sharded files through', () => {
+ const files = [
+ { path: 'Model-Q4_K_M.gguf', size: 100 },
+ { path: 'mmproj-F16.gguf', size: 10 }
+ ];
+
+ expect(collapseGgufShards(files)).toStrictEqual(files);
+ });
+
+ it('collapses a shard set to its first shard with the summed size', () => {
+ const files = [
+ { path: 'Model-00001-of-00003.gguf', size: 10 },
+ { path: 'Model-00002-of-00003.gguf', size: 20 },
+ { path: 'Model-00003-of-00003.gguf', size: 30 }
+ ];
+
+ expect(collapseGgufShards(files)).toStrictEqual([
+ { path: 'Model-00001-of-00003.gguf', size: 60 }
+ ]);
+ });
+
+ it('treats a missing shard as zero bytes', () => {
+ const files = [
+ { path: 'Model-00001-of-00002.gguf', size: 10 },
+ { path: 'Model-Q8_0.gguf', size: 5 }
+ ];
+
+ expect(collapseGgufShards(files)).toStrictEqual([
+ { path: 'Model-00001-of-00002.gguf', size: 10 },
+ { path: 'Model-Q8_0.gguf', size: 5 }
+ ]);
+ });
+});
+
+describe('getBitDepth', () => {
+ it('resolves known tokens', () => {
+ expect(getBitDepth('Q4_K_M')).toBe(4);
+ expect(getBitDepth('BF16')).toBe(16);
+ expect(getBitDepth('IQ2_XXS')).toBe(2);
+ });
+
+ it('strips the UD prefix', () => {
+ expect(getBitDepth('UD-Q4_K_XL')).toBe(4);
+ });
+
+ it('falls back to the leading precision digits', () => {
+ expect(getBitDepth('TQ1_0')).toBe(1);
+ expect(getBitDepth('MXFP4_MOE')).toBe(4);
+ });
+
+ it('returns null for unrecognized tokens', () => {
+ expect(getBitDepth('xyz')).toBeNull();
+ expect(getBitDepth('QUANT')).toBeNull();
+ });
+});
+
+describe('formatSizeRange', () => {
+ it('formats a gigabyte range without spaces around the dash', () => {
+ expect(formatSizeRange(19e9, 28.6e9)).toBe('19.0-28.6 GB');
+ });
+
+ it('downgrades the unit to the smaller bound when the max is small', () => {
+ expect(formatSizeRange(1e6, 2.5e6)).toBe('1.0-2.5 MB');
+ });
+
+ it('formats sub-kilobyte sizes in bytes', () => {
+ expect(formatSizeRange(100, 900)).toBe('100-900 B');
+ });
+});
+
+describe('parseCachePath', () => {
+ it('parses a posix cache path into repo and file', () => {
+ expect(
+ parseCachePath(
+ '/home/u/.cache/llama.cpp/models--ggml-org--Qwen3-8B-GGUF/snapshots/abc123/Q4_K_M.gguf'
+ )
+ ).toStrictEqual({ file: 'Q4_K_M.gguf', repo: 'ggml-org/Qwen3-8B-GGUF' });
+ });
+
+ it('accepts windows separators', () => {
+ expect(
+ parseCachePath('C:\\cache\\models--org--Model\\snapshots\\sha\\sub\\file.gguf')
+ ).toStrictEqual({ file: 'sub/file.gguf', repo: 'org/Model' });
+ });
+
+ it('returns null for non-cache paths', () => {
+ expect(parseCachePath('/models/foo.gguf')).toBeNull();
+ });
+});
+
+describe('parseParamCount', () => {
+ it('extracts billions and millions', () => {
+ expect(parseParamCount('Qwen3.8-27B-GGUF')).toBe('27B');
+ expect(parseParamCount('embeddinggemma-300M-GGUF')).toBe('300M');
+ expect(parseParamCount('Model-0.6B-Q4_K_M')).toBe('0.6B');
+ });
+
+ it('returns null when no size token is present', () => {
+ expect(parseParamCount('ggml-org/Laguna-S-GGUF')).toBeNull();
+ });
+});
+
+describe('parseSizeBytes', () => {
+ it('parses single-letter catalog size suffixes', () => {
+ expect(parseSizeBytes('177g')).toBe(177e9);
+ expect(parseSizeBytes('1.2 t')).toBe(1.2e12);
+ expect(parseSizeBytes('500m')).toBe(500e6);
+ });
+
+ it('returns null for malformed input', () => {
+ expect(parseSizeBytes('unknown')).toBeNull();
+ expect(parseSizeBytes('12 parsecs')).toBeNull();
+ expect(parseSizeBytes('')).toBeNull();
+ });
+});