mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-17 20:31:47 +02:00
ui : add huggingface hub data layer
Add HuggingFaceService for browsing and searching GGUF models on the HF Hub: catalog/model search, model details, repo file tree, raw README fetch, and the llama.app model catalog. Includes GGUF file analysis helpers - extractQuantMeta (quant token plus sidecar type and its form, prefix or suffix), shard collapsing, quant bit-depth lookup, and download/size/likes formatting. Add the HF API types and the curated model list shown in the Discover Models sidebar. Assisted-by: pi
This commit is contained in:
@@ -44,6 +44,7 @@ export * from './message-export.constants';
|
||||
export * from './path-display.constants';
|
||||
export * from './model-id.constants';
|
||||
export * from './model-loading.constants';
|
||||
export * from './models-discover.constants';
|
||||
export * from './precision.constants';
|
||||
export * from './pwa.constants';
|
||||
export * from './routes.constants';
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
/**
|
||||
* Curated default models shown in the Discover Models sidebar, in display order.
|
||||
*/
|
||||
export const CURATED_MODEL_IDS = [
|
||||
'ggml-org/Qwen3.8-27B-GGUF',
|
||||
'ggml-org/DeepSeek-V4-Flash-0731-GGUF',
|
||||
'ggml-org/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-GGUF',
|
||||
'ggml-org/Qwen3.6-35B-A3B-GGUF',
|
||||
'ggml-org/Laguna-S-2.1-GGUF',
|
||||
'ggml-org/gemma-4-31B-it-GGUF',
|
||||
'ggml-org/gemma-4-26B-A4B-it-GGUF',
|
||||
'ggml-org/gemma-4-12B-it-GGUF',
|
||||
'ggml-org/Qwen3.5-0.8B-GGUF',
|
||||
'ggml-org/gemma-4-E2B-it-GGUF',
|
||||
'ggml-org/gemma-4-E4B-it-GGUF',
|
||||
'ggml-org/gpt-oss-120b-GGUF',
|
||||
'ggml-org/gpt-oss-20b-GGUF'
|
||||
];
|
||||
@@ -0,0 +1,800 @@
|
||||
import { MODEL_ID, type ModelSidecar, sidecarFromFileToken } from '$lib/constants';
|
||||
import type {
|
||||
HfCatalogEntry,
|
||||
HfModelDetailInfo,
|
||||
HfModelInfo,
|
||||
HfModelSearchParams,
|
||||
HfModelSibling,
|
||||
HfModelSort
|
||||
} from '$lib/types/huggingface';
|
||||
|
||||
/**
|
||||
* Where the sidecar token (`mtp` / `dflash` / `mmproj` / ...) sits in the filename.
|
||||
* - `prefix` sidecar file that lives next to the main weights, e.g. `mtp-Q4_0.gguf`
|
||||
* - `suffix` embedded draft baked into the main weights, e.g. `Hy3-IQ1_M-mtp.gguf`
|
||||
*/
|
||||
export type SidecarForm = 'prefix' | 'suffix';
|
||||
|
||||
// Constants
|
||||
|
||||
export const HF_TASKS: Record<string, string> = {
|
||||
'audio-classification': 'Audio Classification',
|
||||
'audio-to-audio': 'Audio-to-Audio',
|
||||
'automatic-speech-recognition': 'Speech Recognition',
|
||||
conversational: 'Conversational',
|
||||
'depth-estimation': 'Depth Estimation',
|
||||
'feature-extraction': 'Feature Extraction',
|
||||
'fill-mask': 'Fill Mask',
|
||||
'image-classification': 'Image Classification',
|
||||
'image-feature-extraction': 'Image Feature Extraction',
|
||||
'image-segmentation': 'Image Segmentation',
|
||||
'image-text-to-text': 'Image-Text-to-Text',
|
||||
'image-to-text': 'Image-to-Text',
|
||||
'image-to-video': 'Image-to-Video',
|
||||
'object-detection': 'Object Detection',
|
||||
'question-answering': 'Question Answering',
|
||||
'reinforcement-learning': 'Reinforcement Learning',
|
||||
robotics: 'Robotics',
|
||||
'sentence-similarity': 'Sentence Similarity',
|
||||
summarization: 'Summarization',
|
||||
'text2text-generation': 'Text2Text Generation',
|
||||
'text-classification': 'Text Classification',
|
||||
'text-generation': 'Text Generation',
|
||||
'text-to-image': 'Text-to-Image',
|
||||
'text-to-speech': 'Text to Speech',
|
||||
'text-to-video': 'Text-to-Video',
|
||||
'token-classification': 'Token Classification',
|
||||
translation: 'Translation',
|
||||
'video-to-video': 'Video-to-Video',
|
||||
'voice-activity-detection': 'Voice Activity Detection',
|
||||
'zero-shot-classification': 'Zero-Shot Classification'
|
||||
};
|
||||
|
||||
/**
|
||||
* Best-effort readable label for an HF pipeline tag. Falls back to a
|
||||
* title-cased version of the kebab-case `pipeline_tag` (e.g. `image-text-to-text`
|
||||
* becomes `Image-Text-to-Text`) when we don't have an explicit entry above.
|
||||
*/
|
||||
function pipelineTagLabel(tag: string): string {
|
||||
if (HF_TASKS[tag]) return HF_TASKS[tag];
|
||||
|
||||
return tag
|
||||
.split('-')
|
||||
.map((part) => (part ? part[0].toUpperCase() + part.slice(1) : part))
|
||||
.join('-');
|
||||
}
|
||||
|
||||
/**
|
||||
* Lucide icon name (string identifier, used to lazy-import the Svelte component)
|
||||
* matching the HF pipeline_tag. Used for the filter chips on the model browser.
|
||||
* Returns `null` for unknown tags so the consumer can render a generic icon.
|
||||
*/
|
||||
const HF_PIPELINE_ICONS: Record<string, string> = {
|
||||
'audio-classification': 'mic',
|
||||
'audio-to-audio': 'audio-lines',
|
||||
'automatic-speech-recognition': 'mic',
|
||||
conversational: 'message-circle',
|
||||
'depth-estimation': 'layers',
|
||||
'feature-extraction': 'hash',
|
||||
'fill-mask': 'replace',
|
||||
'image-classification': 'image',
|
||||
'image-feature-extraction': 'image',
|
||||
'image-segmentation': 'image',
|
||||
'image-text-to-text': 'image-plus',
|
||||
'image-to-text': 'image',
|
||||
'image-to-video': 'video',
|
||||
'object-detection': 'scan',
|
||||
'question-answering': 'help-circle',
|
||||
'sentence-similarity': 'equal',
|
||||
summarization: 'list-collapse',
|
||||
'text2text-generation': 'message-square-more',
|
||||
'text-generation': 'message-square',
|
||||
'text-to-image': 'image',
|
||||
'text-to-speech': 'volume-2',
|
||||
'text-to-video': 'video',
|
||||
translation: 'languages',
|
||||
'video-to-video': 'video',
|
||||
'voice-activity-detection': 'mic'
|
||||
};
|
||||
|
||||
function pipelineTagIcon(tag: string): string | null {
|
||||
return HF_PIPELINE_ICONS[tag] ?? null;
|
||||
}
|
||||
|
||||
export const HF_LIBRARIES: Record<string, string> = {
|
||||
gguf: 'GGUF',
|
||||
mlx: 'MLX',
|
||||
onnx: 'ONNX',
|
||||
safetensors: 'Safetensors',
|
||||
transformers: 'Transformers',
|
||||
vllm: 'vLLM'
|
||||
};
|
||||
|
||||
/**
|
||||
* HuggingFaceService - Service for browsing and searching GGUF models on Hugging Face Hub
|
||||
*/
|
||||
export class HuggingFaceService {
|
||||
// Configuration
|
||||
|
||||
/** Available library names with display labels */
|
||||
static readonly LIBRARIES: Record<string, string> = HF_LIBRARIES;
|
||||
/** Sort option display labels */
|
||||
static readonly SORT_LABELS: Record<HfModelSort, string> = {
|
||||
createdAt: 'Newest',
|
||||
downloads: 'Most Downloads',
|
||||
lastModified: 'Recently Updated',
|
||||
likes: 'Most Likes',
|
||||
trendingScore: 'Trending'
|
||||
};
|
||||
/** Available sort options */
|
||||
static readonly SORT_OPTIONS: HfModelSort[] = [
|
||||
'downloads',
|
||||
'likes',
|
||||
'trendingScore',
|
||||
'createdAt'
|
||||
];
|
||||
|
||||
// Available options for filtering
|
||||
|
||||
/** Available pipeline tasks with display labels */
|
||||
static readonly TASKS: Record<string, string> = HF_TASKS;
|
||||
|
||||
private static readonly BASE_URL = 'https://huggingface.co/api/models';
|
||||
|
||||
// Cached base model lookups keyed by repo id, so repeated selector opens
|
||||
// never re-hit the HF API for the same repo.
|
||||
private static baseModelCache = new Map<string, { org: string; name: string } | null>();
|
||||
|
||||
private static baseModelPending = new Map<
|
||||
string,
|
||||
Promise<{ org: string; name: string } | null>
|
||||
>();
|
||||
|
||||
private static readonly DEFAULT_LIMIT = 50;
|
||||
|
||||
private static readonly MAX_LIMIT = 100;
|
||||
|
||||
// GGUF Model Searching
|
||||
|
||||
/**
|
||||
* Map of quant token to its average bit-depth in bits-per-weight (bpw).
|
||||
*/
|
||||
private static readonly QUANT_BIT_DEPTH: Record<string, number> = {
|
||||
BF16: 16,
|
||||
F16: 16,
|
||||
IQ1_M: 1,
|
||||
IQ1_S: 1,
|
||||
IQ1_XS: 1,
|
||||
IQ1_XXS: 1,
|
||||
IQ2_M: 2,
|
||||
IQ2_S: 2,
|
||||
IQ2_XS: 2,
|
||||
IQ2_XXS: 2,
|
||||
IQ3_M: 3,
|
||||
IQ3_S: 3,
|
||||
IQ3_XS: 3,
|
||||
IQ3_XXS: 3,
|
||||
Q2_K: 2,
|
||||
Q2_K_M: 2,
|
||||
Q2_K_S: 2,
|
||||
Q3_K: 3,
|
||||
Q3_K_L: 3,
|
||||
Q3_K_M: 3,
|
||||
Q3_K_S: 3,
|
||||
Q4_0: 4,
|
||||
Q4_1: 4,
|
||||
Q4_K: 4,
|
||||
Q4_K_M: 4,
|
||||
Q4_K_S: 4,
|
||||
Q5_0: 5,
|
||||
Q5_1: 5,
|
||||
Q5_K: 5,
|
||||
Q5_K_M: 5,
|
||||
Q5_K_S: 5,
|
||||
Q6_K: 6,
|
||||
Q8_0: 8
|
||||
};
|
||||
|
||||
/**
|
||||
* Collapse split GGUF shard sets (`-00001-of-00015.gguf`, ...) to their first
|
||||
* shard, summing every shard's size so the kept entry reflects the whole
|
||||
* quant. Non-sharded files pass through unchanged. Downloads are tag-based
|
||||
* (`repo:quant`), so the first shard is enough to represent the set.
|
||||
*/
|
||||
static collapseGgufShards(siblings: HfModelSibling[]): HfModelSibling[] {
|
||||
const sizeByPath = new Map(siblings.map((f) => [f.path, f.size ?? 0]));
|
||||
const result: HfModelSibling[] = [];
|
||||
|
||||
for (const file of siblings) {
|
||||
const match = /-(\d{5})-of-(\d{5})\.gguf$/i.exec(file.path);
|
||||
|
||||
if (!match) {
|
||||
result.push(file);
|
||||
|
||||
continue;
|
||||
}
|
||||
|
||||
// Keep only the first shard; its size becomes the whole shard set's.
|
||||
if (match[1] !== '00001') continue;
|
||||
|
||||
const total = parseInt(match[2], 10);
|
||||
const stem = file.path.slice(0, file.path.length - match[0].length);
|
||||
|
||||
let size = 0;
|
||||
|
||||
for (let i = 1; i <= total; i++) {
|
||||
const shard = `${stem}-${String(i).padStart(5, '0')}-of-${String(total).padStart(5, '0')}.gguf`;
|
||||
|
||||
size += sizeByPath.get(shard) ?? 0;
|
||||
}
|
||||
|
||||
result.push({ ...file, size });
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
// GGUF Model Browsing
|
||||
|
||||
/**
|
||||
* Extract the GGUF quantization token (e.g. `Q4_K_M`) and any sidecar type
|
||||
* (`mtp`, `dflash`, `mmproj`, ...) from a `.gguf` filename. The sidecar token
|
||||
* shows up either as a sidecar prefix (`mtp-<name>.gguf`, `dflash-<name>.gguf`,
|
||||
* `mmproj-<name>.gguf`) or as the `-mtp` suffix when the draft model is
|
||||
* embedded in the same GGUF weight file.
|
||||
*
|
||||
* `sidecarForm` records which side of the filename the sidecar token sat
|
||||
* on so callers can render badges differently (e.g. prefix on the left of
|
||||
* the quant label, suffix appended to it).
|
||||
* `quant` is `null` for files that don't carry a bit-depth token
|
||||
* (e.g. `*-BF16.gguf`); `sidecar` is `null` if no sidecar flag is present.
|
||||
* Returns `null` only when the filename doesn't end in `.gguf`.
|
||||
*/
|
||||
static extractQuantMeta(filename: string): {
|
||||
quant: string | null;
|
||||
sidecar: ModelSidecar | null;
|
||||
sidecarForm: SidecarForm | null;
|
||||
} | null {
|
||||
if (!MODEL_ID.WEIGHT_EXTENSION_RE.test(filename)) return null;
|
||||
|
||||
let source = filename.replace(MODEL_ID.WEIGHT_EXTENSION_RE, '');
|
||||
let sidecar: ModelSidecar | null = null;
|
||||
let sidecarForm: SidecarForm | null = null;
|
||||
|
||||
const prefixMatch = source.match(MODEL_ID.SIDECAR_PREFIX_RE);
|
||||
|
||||
if (prefixMatch) {
|
||||
sidecar = sidecarFromFileToken(prefixMatch[1].toLowerCase());
|
||||
sidecarForm = 'prefix';
|
||||
source = prefixMatch[2];
|
||||
} else {
|
||||
const suffixMatch = source.match(MODEL_ID.SIDECAR_SUFFIX_RE);
|
||||
|
||||
if (suffixMatch) {
|
||||
const candidate = suffixMatch[1];
|
||||
const headSeg = candidate.split(MODEL_ID.SEGMENT_SEPARATOR).pop();
|
||||
|
||||
if (headSeg && MODEL_ID.QUANTIZATION_SEGMENT_RE.test(headSeg)) {
|
||||
sidecar = sidecarFromFileToken(suffixMatch[2].toLowerCase());
|
||||
sidecarForm = 'suffix';
|
||||
source = candidate;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Scan dash-separated segments left-to-right for the first quant match.
|
||||
// - For sidecars like `mtp-Q4_0-180MB.gguf` the quant is `Q4_0`.
|
||||
// - For embedded MTP like `Hy3-IQ1_M-mtp.gguf` we have `Hy3-IQ1_M` and `IQ1_M` matches.
|
||||
// - For main files like `Llama-3-8B-Q4_K_M.gguf` we land on the trailing quant.
|
||||
const segments = source.split(MODEL_ID.SEGMENT_SEPARATOR);
|
||||
const quantIdx = segments.findIndex((seg) => MODEL_ID.QUANTIZATION_SEGMENT_RE.test(seg));
|
||||
|
||||
let quant = quantIdx >= 0 ? segments[quantIdx].toUpperCase() : null;
|
||||
|
||||
// Recombine a `UD-` (Unsloth Dynamic) prefix, e.g. `...-UD-Q4_K_XL.gguf`.
|
||||
if (quant && quantIdx > 0 && segments[quantIdx - 1].toUpperCase() === 'UD') {
|
||||
quant = `UD-${quant}`;
|
||||
}
|
||||
|
||||
return { quant, sidecar, sidecarForm };
|
||||
}
|
||||
|
||||
/**
|
||||
* Filter raw siblings by file extension and sort by size descending.
|
||||
*/
|
||||
static filterByExtension(siblings: HfModelSibling[], ext: string): HfModelSibling[] {
|
||||
return siblings
|
||||
.filter((f) => f.path.toLowerCase().endsWith(ext.toLowerCase()) && (f.size ?? 0) > 0)
|
||||
.sort((a, b) => (b.size ?? 0) - (a.size ?? 0));
|
||||
}
|
||||
|
||||
/**
|
||||
* Format model downloads count with K/M/B suffix
|
||||
*/
|
||||
static formatDownloads(downloads: number): string {
|
||||
if (downloads >= 1_000_000) {
|
||||
return `${(downloads / 1_000_000).toFixed(1)}M`;
|
||||
}
|
||||
|
||||
if (downloads >= 1_000) {
|
||||
return `${(downloads / 1_000).toFixed(1)}K`;
|
||||
}
|
||||
|
||||
return downloads.toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Format file size in bytes to human-readable string
|
||||
*/
|
||||
static formatFileSize(bytes: number): string {
|
||||
if (bytes >= 1_000_000_000) {
|
||||
return `${(bytes / 1_000_000_000).toFixed(1)} GB`;
|
||||
}
|
||||
|
||||
if (bytes >= 1_000_000) {
|
||||
return `${(bytes / 1_000_000).toFixed(1)} MB`;
|
||||
}
|
||||
|
||||
if (bytes >= 1_000) {
|
||||
return `${(bytes / 1_000).toFixed(1)} KB`;
|
||||
}
|
||||
|
||||
return `${bytes} B`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format likes count with K suffix if applicable
|
||||
*/
|
||||
static formatLikes(likes: number): string {
|
||||
if (likes >= 1_000) {
|
||||
return `${(likes / 1_000).toFixed(1)}K`;
|
||||
}
|
||||
|
||||
return likes.toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Format timestamp to relative time
|
||||
*/
|
||||
static formatRelativeTime(timestamp: string): string {
|
||||
const date = new Date(timestamp);
|
||||
const now = new Date();
|
||||
const diffMs = now.getTime() - date.getTime();
|
||||
const diffDays = Math.floor(diffMs / (1000 * 60 * 60 * 24));
|
||||
|
||||
if (diffDays === 0) return 'Today';
|
||||
|
||||
if (diffDays === 1) return 'Yesterday';
|
||||
|
||||
if (diffDays < 7) return `${diffDays} days ago`;
|
||||
|
||||
if (diffDays < 30) return `${Math.floor(diffDays / 7)} weeks ago`;
|
||||
|
||||
if (diffDays < 365) return `${Math.floor(diffDays / 30)} months ago`;
|
||||
|
||||
return `${Math.floor(diffDays / 365)} years ago`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format a min-max size range with a single shared unit and no spaces
|
||||
* around the dash, e.g. `19.0-28.6 GB`.
|
||||
*/
|
||||
static formatSizeRange(min: number, max: number): string {
|
||||
const unit = max >= 1_000_000_000 ? 'GB' : max >= 1_000_000 ? 'MB' : max >= 1_000 ? 'KB' : 'B';
|
||||
const div =
|
||||
unit === 'GB' ? 1_000_000_000 : unit === 'MB' ? 1_000_000 : unit === 'KB' ? 1_000 : 1;
|
||||
const fmt = (n: number) => (div === 1 ? `${n}` : `${(n / div).toFixed(1)}`);
|
||||
|
||||
return `${fmt(min)}-${fmt(max)} ${unit}`;
|
||||
}
|
||||
|
||||
// Model Details & Files
|
||||
|
||||
/**
|
||||
* Avatar URL for an author (org or user). 404s when the author does not
|
||||
* exist, so callers should provide a fallback.
|
||||
*/
|
||||
static getAvatarUrl(author: string): string {
|
||||
return `https://huggingface.co/api/avatars/${author}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the original (non-GGUF) base model `{ org, name }` for a GGUF repo
|
||||
* from its HF card (`cardData.base_model`). Returns null when the card has no
|
||||
* base model. Results are cached per repo.
|
||||
*/
|
||||
static getBaseModel(repoId: string): Promise<{ org: string; name: string } | null> {
|
||||
const cached = this.baseModelCache.get(repoId);
|
||||
|
||||
if (cached !== undefined) return Promise.resolve(cached);
|
||||
|
||||
const pending = this.baseModelPending.get(repoId);
|
||||
|
||||
if (pending) return pending;
|
||||
|
||||
const promise = (async () => {
|
||||
const details = await this.getDetails(repoId);
|
||||
const base = this.getBaseModels(details)[0];
|
||||
|
||||
if (!base) return null;
|
||||
|
||||
const [org, ...rest] = base.split('/');
|
||||
|
||||
return { name: rest.join('/'), org };
|
||||
})();
|
||||
|
||||
this.baseModelPending.set(repoId, promise);
|
||||
|
||||
promise
|
||||
.then((result) => this.baseModelCache.set(repoId, result))
|
||||
.finally(() => this.baseModelPending.delete(repoId));
|
||||
|
||||
return promise;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract the original (non-GGUF) base model ids for a repo, from
|
||||
* `cardData.base_model` (string or list) and the `base_model:` tags.
|
||||
*/
|
||||
static getBaseModels(model: HfModelDetailInfo | null): string[] {
|
||||
if (!model) return [];
|
||||
|
||||
const cardBase = model.cardData?.base_model;
|
||||
const fromCard: string[] = Array.isArray(cardBase) ? cardBase : cardBase ? [cardBase] : [];
|
||||
const fromTags = (model.tags ?? [])
|
||||
.map((t) => /^base_model:(?:quantized:)?(.+)$/.exec(t)?.[1])
|
||||
.filter((v): v is string => Boolean(v));
|
||||
|
||||
return Array.from(new Set([...fromCard, ...fromTags]));
|
||||
}
|
||||
|
||||
/**
|
||||
* Look up the average bit-depth for a known GGUF quantization.
|
||||
* Returns `null` for unrecognized tokens.
|
||||
*/
|
||||
static getBitDepth(quant: string): number | null {
|
||||
// Strip a leading `UD-` (Unsloth Dynamic) prefix before lookup.
|
||||
const base = quant.replace(/^UD-/i, '');
|
||||
const direct = HuggingFaceService.QUANT_BIT_DEPTH[base];
|
||||
|
||||
if (direct !== undefined) return direct;
|
||||
|
||||
// Fall back to the leading precision digits for variants missing from the
|
||||
// map, e.g. `Q4_K_XL` -> 4, `IQ2_XXS` -> 2, `TQ1_0` -> 1, `BF16` -> 16.
|
||||
const match = /^(?:I?Q|TQ|BF|F|MXFP)?(\d+)/i.exec(base);
|
||||
|
||||
return match ? parseInt(match[1], 10) : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get GGUF models by pipeline task
|
||||
*/
|
||||
static async getByTask(
|
||||
pipelineTag: string,
|
||||
params: Omit<HfModelSearchParams, 'pipeline_tag'> = {}
|
||||
): Promise<HfModelInfo[]> {
|
||||
return this.search({
|
||||
...params,
|
||||
pipeline_tag: pipelineTag
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Get detailed information about a specific GGUF model
|
||||
*/
|
||||
/**
|
||||
* Fetch the llama.app model catalog (https://llama.app/v1/catalog.json).
|
||||
* Returns an empty array on failure so callers can fall back gracefully.
|
||||
*/
|
||||
static async getCatalog(): Promise<HfCatalogEntry[]> {
|
||||
const url = 'https://llama.app/v1/catalog.json';
|
||||
|
||||
try {
|
||||
const response = await fetch(url);
|
||||
|
||||
if (!response.ok) throw new Error(`Failed to fetch catalog: ${response.status}`);
|
||||
|
||||
return (await response.json()) as HfCatalogEntry[];
|
||||
} catch (error) {
|
||||
console.error('Error fetching catalog:', error);
|
||||
|
||||
return [];
|
||||
}
|
||||
}
|
||||
|
||||
static async getDetails(modelId: string): Promise<HfModelDetailInfo | null> {
|
||||
// Do not encode the modelId, it contains slashes for author/name.
|
||||
// `full=true` includes cardData (description, base_model) and safetensors.
|
||||
const url = `https://huggingface.co/api/models/${modelId}?full=true`;
|
||||
|
||||
try {
|
||||
const response = await fetch(url);
|
||||
|
||||
if (response.status === 404) return null;
|
||||
|
||||
if (!response.ok) throw new Error(`Failed to fetch model details: ${response.status}`);
|
||||
|
||||
const data = (await response.json()) as HfModelDetailInfo;
|
||||
|
||||
return data;
|
||||
} catch (error) {
|
||||
console.error(`Error fetching details for ${modelId}:`, error);
|
||||
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get model URL on Hugging Face Hub
|
||||
*/
|
||||
static getModelUrl(modelId: string): string {
|
||||
return `https://huggingface.co/${modelId}`;
|
||||
}
|
||||
|
||||
// Utility Methods
|
||||
|
||||
/**
|
||||
* Get most liked GGUF models
|
||||
*/
|
||||
static async getMostLiked(
|
||||
limit: number = HuggingFaceService.DEFAULT_LIMIT
|
||||
): Promise<HfModelInfo[]> {
|
||||
return this.search({ limit, sort: 'likes' });
|
||||
}
|
||||
|
||||
/**
|
||||
* Get newly released GGUF models
|
||||
*/
|
||||
static async getNew(limit: number = HuggingFaceService.DEFAULT_LIMIT): Promise<HfModelInfo[]> {
|
||||
return this.search({ limit, sort: 'createdAt' });
|
||||
}
|
||||
|
||||
/**
|
||||
* Get most popular GGUF models by downloads
|
||||
*/
|
||||
static async getPopular(
|
||||
limit: number = HuggingFaceService.DEFAULT_LIMIT
|
||||
): Promise<HfModelInfo[]> {
|
||||
return this.search({ limit, sort: 'downloads' });
|
||||
}
|
||||
|
||||
/**
|
||||
* Fetch the raw README.md for a repo, with the YAML frontmatter stripped.
|
||||
*/
|
||||
static async getReadme(modelId: string): Promise<string | null> {
|
||||
// Do not encode the modelId, it contains slashes for author/name
|
||||
const url = `https://huggingface.co/${modelId}/raw/main/README.md`;
|
||||
|
||||
try {
|
||||
const response = await fetch(url);
|
||||
|
||||
if (response.status === 404) return null;
|
||||
|
||||
if (!response.ok) throw new Error(`Failed to fetch README: ${response.status}`);
|
||||
|
||||
return HuggingFaceService.stripFrontmatter(await response.text());
|
||||
} catch (error) {
|
||||
console.error(`Error fetching README for ${modelId}:`, error);
|
||||
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get repository file tree to list available GGUF variants. Recursive so
|
||||
* repos that keep quants in per-quant subdirectories (e.g. `UD-Q4_K_XL/`)
|
||||
* are included; follows cursor pagination for repos over one page.
|
||||
*/
|
||||
static async getTree(modelId: string): Promise<HfModelSibling[]> {
|
||||
const files: HfModelSibling[] = [];
|
||||
|
||||
let url: string | null =
|
||||
`https://huggingface.co/api/models/${modelId}/tree/main?recursive=true`;
|
||||
|
||||
try {
|
||||
while (url) {
|
||||
const response: Response = await fetch(url);
|
||||
|
||||
if (!response.ok) return files;
|
||||
|
||||
const data = (await response.json()) as HfModelSibling[];
|
||||
|
||||
files.push(...data.filter((f) => f.type !== 'directory'));
|
||||
|
||||
url = HuggingFaceService.parseNextPageUrl(response.headers.get('Link'));
|
||||
}
|
||||
} catch {
|
||||
// Return whatever was fetched before the failure.
|
||||
}
|
||||
|
||||
return files;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get trending GGUF models
|
||||
*/
|
||||
static async getTrending(
|
||||
limit: number = HuggingFaceService.DEFAULT_LIMIT
|
||||
): Promise<HfModelInfo[]> {
|
||||
return this.search({ limit, sort: 'trendingScore' });
|
||||
}
|
||||
/**
|
||||
* Parse a local HF cache file path
|
||||
* (`.../models--<org>--<name>/snapshots/<sha>/<file>`) into its repo id and
|
||||
* repo-relative file path. Returns null when the path is not an HF cache path.
|
||||
*/
|
||||
static parseCachePath(path: string): { repo: string; file: string } | null {
|
||||
const match = /models--(.+?)\/snapshots\/[^/]+\/(.+)$/.exec(path);
|
||||
|
||||
if (!match) return null;
|
||||
|
||||
const parts = match[1].split('--');
|
||||
|
||||
if (parts.length < 2) return null;
|
||||
|
||||
return { file: match[2], repo: `${parts[0]}/${parts.slice(1).join('--')}` };
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort parameter count parsed from a model id/name, e.g. `27B` from
|
||||
* `Qwen3.8-27B-GGUF` or `300M` from `embeddinggemma-300M-GGUF`. Returns null
|
||||
* when no size token is present.
|
||||
*/
|
||||
static parseParamCount(name: string): string | null {
|
||||
const match = /(?:^|[^a-z0-9])(\d+(?:[._]\d+)?)\s*([bm])(?![a-z0-9])/i.exec(name);
|
||||
|
||||
if (!match) return null;
|
||||
|
||||
return `${match[1]}${match[2].toUpperCase()}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse model tags to extract useful information
|
||||
*/
|
||||
static parseTags(tags: string[]): {
|
||||
license: string | null;
|
||||
isGated: boolean;
|
||||
isGguf: boolean;
|
||||
isSafetensors: boolean;
|
||||
tasks: string[];
|
||||
} {
|
||||
const license = tags.find((tag) => tag.startsWith('license:'))?.replace('license:', '') || null;
|
||||
const isGated = tags.includes('gated');
|
||||
const isGguf = tags.includes('gguf');
|
||||
const isSafetensors = tags.includes('safetensors');
|
||||
const tasks = tags.filter((tag) => Object.keys(HuggingFaceService.TASKS).includes(tag));
|
||||
|
||||
return { isGated, isGguf, isSafetensors, license, tasks };
|
||||
}
|
||||
|
||||
/** Resolve a pipeline_tag to a lucide icon name, or null when unknown. */
|
||||
static pipelineTagIcon(tag: string | null | undefined): string | null {
|
||||
if (!tag) return null;
|
||||
|
||||
return pipelineTagIcon(tag);
|
||||
}
|
||||
|
||||
/** Resolve a pipeline_tag to a human-readable label. */
|
||||
static pipelineTagLabel(tag: string | null | undefined): string | null {
|
||||
if (!tag) return null;
|
||||
|
||||
return pipelineTagLabel(tag);
|
||||
}
|
||||
|
||||
/**
|
||||
* Search GGUF models with various filters and options
|
||||
*/
|
||||
static async search(params: HfModelSearchParams = {}): Promise<HfModelInfo[]> {
|
||||
const { limit = HuggingFaceService.DEFAULT_LIMIT, ...restParams } = params;
|
||||
const url = this.buildUrl({
|
||||
...restParams,
|
||||
filter: 'gguf',
|
||||
limit: Math.min(limit, HuggingFaceService.MAX_LIMIT)
|
||||
});
|
||||
|
||||
return this.fetchWithRetry(url);
|
||||
}
|
||||
|
||||
/**
|
||||
* Search models by query string
|
||||
*/
|
||||
static async searchByQuery(
|
||||
query: string,
|
||||
params: Omit<HfModelSearchParams, 'search'> = {}
|
||||
): Promise<HfModelInfo[]> {
|
||||
return this.search({
|
||||
...params,
|
||||
search: query
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Build API URL from search parameters
|
||||
*/
|
||||
private static buildUrl(params: HfModelSearchParams): string {
|
||||
const url = new URL(this.BASE_URL);
|
||||
|
||||
Object.entries(params).forEach(([key, value]) => {
|
||||
if (value !== undefined && value !== null && value !== '') {
|
||||
if (Array.isArray(value)) {
|
||||
value.forEach((v) => url.searchParams.append(key, v));
|
||||
} else {
|
||||
url.searchParams.set(key, String(value));
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
return url.toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Delay helper for retry logic
|
||||
*/
|
||||
private static delay(ms: number): Promise<void> {
|
||||
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
}
|
||||
|
||||
// Internal Methods
|
||||
|
||||
/**
|
||||
* Fetch data with retry logic for resilience
|
||||
*/
|
||||
private static async fetchWithRetry(url: string, attempt: number = 1): Promise<HfModelInfo[]> {
|
||||
const RETRY_ATTEMPTS = 3;
|
||||
const RETRY_DELAY_MS = 1000;
|
||||
|
||||
try {
|
||||
const response = await fetch(url);
|
||||
|
||||
if (!response.ok) {
|
||||
if (response.status === 404) {
|
||||
return [];
|
||||
}
|
||||
|
||||
if (response.status >= 500 && attempt < RETRY_ATTEMPTS) {
|
||||
await this.delay(RETRY_DELAY_MS * attempt);
|
||||
|
||||
return this.fetchWithRetry(url, attempt + 1);
|
||||
}
|
||||
|
||||
throw new Error(`API request failed: ${response.status} ${response.statusText}`);
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
|
||||
if (Array.isArray(data)) {
|
||||
return data as HfModelInfo[];
|
||||
}
|
||||
|
||||
if (data && Array.isArray(data.data)) {
|
||||
return data.data as HfModelInfo[];
|
||||
}
|
||||
|
||||
throw new Error('Unexpected API response format');
|
||||
} catch (error) {
|
||||
if (attempt < RETRY_ATTEMPTS) {
|
||||
await this.delay(RETRY_DELAY_MS * attempt);
|
||||
|
||||
return this.fetchWithRetry(url, attempt + 1);
|
||||
}
|
||||
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
/** Extract the `rel="next"` URL from an RFC 5988 `Link` header, if present. */
|
||||
private static parseNextPageUrl(linkHeader: string | null): string | null {
|
||||
if (!linkHeader) return null;
|
||||
|
||||
const match = /<([^>]+)>;\s*rel="next"/.exec(linkHeader);
|
||||
|
||||
return match ? match[1] : null;
|
||||
}
|
||||
|
||||
/** Strip a leading YAML frontmatter block (--- ... ---) from a markdown document. */
|
||||
private static stripFrontmatter(text: string): string {
|
||||
const match = text.match(/^---\r?\n[\s\S]*?\r?\n---\r?\n?/);
|
||||
|
||||
return match ? text.slice(match[0].length) : text;
|
||||
}
|
||||
}
|
||||
@@ -147,6 +147,16 @@ export { ConversationTransferService } from './conversation-transfer.service';
|
||||
*/
|
||||
export { ModelsService } from './models.service';
|
||||
|
||||
/**
|
||||
* **HuggingFaceService** - Hugging Face Hub browsing and searching
|
||||
*
|
||||
* Stateless HTTP client for the HF REST API (`/api/models`, `/tree`, raw
|
||||
* README) and the llama.app model catalog. Provides GGUF file analysis
|
||||
* (quant metadata, shard collapsing, size formatting) used by the models
|
||||
* discover UI.
|
||||
*/
|
||||
export { HuggingFaceService, type SidecarForm } from './huggingface.service';
|
||||
|
||||
/**
|
||||
* **PropsService** - Server properties and capabilities retrieval
|
||||
*
|
||||
|
||||
+221
@@ -0,0 +1,221 @@
|
||||
/**
|
||||
* HuggingFace Hub Model Browsing Types
|
||||
*
|
||||
* Types for the HuggingFace REST API (/api/models)
|
||||
* Reference: https://huggingface.co/docs/huggingface_hub/package_reference/hf_api
|
||||
*/
|
||||
|
||||
// Search Options
|
||||
|
||||
export interface HfModelSearchParams {
|
||||
/** Full-text search query */
|
||||
search?: string;
|
||||
/** Filter by pipeline task (e.g., "text-generation", "image-generation") */
|
||||
pipeline_tag?: string;
|
||||
/** Filter by library (e.g., "transformers", "diffusers", "gguf") */
|
||||
library_name?: string;
|
||||
/** Filter by tag (e.g., "gguf") */
|
||||
filter?: string;
|
||||
/** Filter by author or organization */
|
||||
author?: string;
|
||||
/** Sort field */
|
||||
sort?: HfModelSort;
|
||||
/** Results per page (1-100) */
|
||||
limit?: number;
|
||||
/** Pagination offset */
|
||||
offset?: number;
|
||||
/** Filter by model config */
|
||||
config?: string;
|
||||
/** Return full model info */
|
||||
full?: boolean;
|
||||
/** Filter by visibility */
|
||||
private?: boolean;
|
||||
/** Filter by gated status */
|
||||
gated?: boolean;
|
||||
}
|
||||
|
||||
export type HfModelSort = 'downloads' | 'likes' | 'createdAt' | 'lastModified' | 'trendingScore';
|
||||
|
||||
// Model Info (from /api/models)
|
||||
|
||||
export interface HfModelInfo {
|
||||
/** Unique document ID */
|
||||
_id: string;
|
||||
/** Model ID (e.g., "meta-llama/Llama-3.1-8B-Instruct") */
|
||||
id: string;
|
||||
/** Number of likes */
|
||||
likes: number;
|
||||
/** Trending score */
|
||||
trendingScore: number;
|
||||
/** Whether the model is private */
|
||||
private: boolean;
|
||||
/** Number of downloads */
|
||||
downloads: number;
|
||||
/** Model tags */
|
||||
tags: string[];
|
||||
/** Pipeline task (e.g., "text-generation") */
|
||||
pipeline_tag: string | null;
|
||||
/** Library name (e.g., "transformers", "diffusers") */
|
||||
library_name: string | null;
|
||||
/** Creation timestamp */
|
||||
createdAt: string;
|
||||
/** Model ID (alias for id) */
|
||||
modelId: string;
|
||||
/** Author / organization (present when full=true) */
|
||||
author?: string;
|
||||
/** Last modified timestamp (present when full=true) */
|
||||
lastModified?: string;
|
||||
/** Repository file listing (present when full=true) */
|
||||
siblings?: HfModelSiblingRef[];
|
||||
/** GGUF metadata (context length, architecture, etc.) */
|
||||
gguf?: HfModelGguf;
|
||||
}
|
||||
|
||||
// Model Details (with full=true)
|
||||
|
||||
export interface HfModelCardData {
|
||||
/** License identifier */
|
||||
license?: string;
|
||||
/** License URL */
|
||||
license_link?: string;
|
||||
/** Model description */
|
||||
description?: string;
|
||||
/** Model library */
|
||||
language?: string[];
|
||||
/** Tags */
|
||||
tags?: string[];
|
||||
/** Original (non-GGUF) model(s) this repo was converted from, e.g. `Qwen/Qwen3.8-27B`. The API returns a single string or a list. */
|
||||
base_model?: string | string[];
|
||||
/** Org that produced the quant, e.g. `bartowski` */
|
||||
quantized_by?: string;
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
/** GGUF metadata returned by /api/models/{id}?full=true for GGUF repos. */
|
||||
export interface HfModelGguf {
|
||||
/** Total parameter count */
|
||||
total?: number;
|
||||
/** Architecture, e.g. `gemma3`, `qwen3` */
|
||||
architecture?: string;
|
||||
/** Context length */
|
||||
context_length?: number;
|
||||
/** Chat template (Jinja) */
|
||||
chat_template?: string;
|
||||
bos_token?: string;
|
||||
eos_token?: string;
|
||||
/** Total size of all GGUF files in the repo, in bytes */
|
||||
totalFileSize?: number;
|
||||
}
|
||||
|
||||
export interface HfModelDetails {
|
||||
/** Model ID */
|
||||
id?: string;
|
||||
/** SHA256 digest */
|
||||
sha?: string;
|
||||
/** Last modified timestamp */
|
||||
lastModified?: string;
|
||||
/** Downloads count */
|
||||
downloads?: number;
|
||||
/** Number of likes */
|
||||
likes?: number;
|
||||
/** Whether the model is gated */
|
||||
gated?: boolean;
|
||||
/** Model card data */
|
||||
cardData?: HfModelCardData;
|
||||
/** Tags */
|
||||
tags?: string[];
|
||||
/** Pipeline tag */
|
||||
pipeline_tag?: string | null;
|
||||
/** Library name */
|
||||
library_name?: string | null;
|
||||
/** Safe tensors info */
|
||||
safetensors?: Record<string, unknown>;
|
||||
/** Model size in bytes */
|
||||
size?: number;
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
export interface HfModelDetailInfo extends HfModelInfo {
|
||||
/** Whether the model is gated (true/false/'auto') */
|
||||
gated?: boolean | string;
|
||||
/** Repository file listing mirrors of /api/models/{id}/tree/main */
|
||||
siblings?: HfModelSiblingRef[];
|
||||
/** Author / organization */
|
||||
author?: string;
|
||||
/** Last modified timestamp */
|
||||
lastModified?: string;
|
||||
/** Model card YAML data (only present when full=true) */
|
||||
cardData?: HfModelCardData;
|
||||
/** GGUF metadata (only present when full=true for GGUF repos) */
|
||||
gguf?: HfModelGguf;
|
||||
/** Model config (only present when full=true) */
|
||||
config?: Record<string, unknown>;
|
||||
/** Total repo storage in bytes (only present when full=true) */
|
||||
usedStorage?: number;
|
||||
/** Sample widget prompts */
|
||||
widgetData?: Array<{ text?: string }>;
|
||||
/** Related spaces */
|
||||
spaces?: string[];
|
||||
}
|
||||
|
||||
/** A single entry in a model repository's file tree (`/tree` responses) */
|
||||
export interface HfModelSibling {
|
||||
/** Relative path of the file or directory within the repo */
|
||||
path: string;
|
||||
/** Size in bytes (omitted for directories) */
|
||||
size?: number;
|
||||
/** Whether this entry is a directory */
|
||||
type?: 'file' | 'directory';
|
||||
/** OID/hash for the blob */
|
||||
oid?: string;
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
/**
|
||||
* A single file entry in a model's `siblings` list. List (`/api/models`) and
|
||||
* detail (`/api/models/{id}`) responses use `rfilename`, unlike `/tree`.
|
||||
*/
|
||||
export interface HfModelSiblingRef {
|
||||
/** Relative file name within the repo */
|
||||
rfilename: string;
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
// API Response
|
||||
|
||||
export interface HfModelApiResponse {
|
||||
/** List of models */
|
||||
data: HfModelInfo[];
|
||||
/** Total count (if available) */
|
||||
total?: number;
|
||||
}
|
||||
|
||||
// llama.app model catalog (https://llama.app/v1/catalog.json)
|
||||
|
||||
/** A single GGUF build/repo within a catalog size. */
|
||||
export interface HfCatalogBuild {
|
||||
quant: string;
|
||||
size: string;
|
||||
sizeBytes: number;
|
||||
repo: string;
|
||||
}
|
||||
|
||||
/** A size variant (e.g. `GPT-OSS 20B`) within a catalog entry. */
|
||||
export interface HfCatalogSize {
|
||||
name: string;
|
||||
params: string;
|
||||
builds: HfCatalogBuild[];
|
||||
}
|
||||
|
||||
/** A single model family in the catalog. `featured` marks the staff picks. */
|
||||
export interface HfCatalogEntry {
|
||||
name: string;
|
||||
brand: string;
|
||||
description: string;
|
||||
details: string;
|
||||
released: string;
|
||||
license: string;
|
||||
featured?: boolean;
|
||||
maxMemGb?: number;
|
||||
sizes: HfCatalogSize[];
|
||||
}
|
||||
@@ -36,6 +36,23 @@ export type {
|
||||
ApiStreamSession
|
||||
} from './api';
|
||||
|
||||
// HuggingFace types
|
||||
export type {
|
||||
HfCatalogBuild,
|
||||
HfCatalogEntry,
|
||||
HfCatalogSize,
|
||||
HfModelApiResponse,
|
||||
HfModelCardData,
|
||||
HfModelDetails,
|
||||
HfModelDetailInfo,
|
||||
HfModelGguf,
|
||||
HfModelInfo,
|
||||
HfModelSearchParams,
|
||||
HfModelSibling,
|
||||
HfModelSiblingRef,
|
||||
HfModelSort
|
||||
} from './huggingface';
|
||||
|
||||
// Chat types
|
||||
export type {
|
||||
AttachmentMenuItem,
|
||||
|
||||
Reference in New Issue
Block a user