ui : simplify model memory estimation

Replace the device-memory tier machinery with a plain file-size
estimate: required runtime memory is the model file size with
headroom for KV cache and allocator overhead (estimateModelMemoryBytes).
Callers present the requirement; there is no device detection and no
fit-versus-budget verdict.

Drops resolveDeviceMemoryGb, deviceMemoryBudgetMb,
computeFileCompatibilityTiers and the CompatibilityTier type, and the
barrel keeps only the new estimator.

Assisted-by: pi
This commit is contained in:
Aleksander Grygier
2026-09-04 20:07:36 +02:00
parent bd90ba5d7a
commit cd24a67c4a
2 changed files with 11 additions and 291 deletions
+2 -7
View File
@@ -344,13 +344,8 @@ export { executeBrowserInfoTool } from './browser-info';
// Tool-use support detection from a chat template
export { detectToolUseSupport } from './chat-template-tool-detector';
// Model hardware-compatibility estimation
export {
computeFileCompatibilityTiers,
deviceMemoryBudgetMb,
resolveDeviceMemoryGb,
type CompatibilityTier
} from './model-compatibility';
// Model memory estimation
export { estimateModelMemoryBytes } from './model-compatibility';
// Cryptography utilities
+9 -284
View File
@@ -1,294 +1,19 @@
/**
* Model hardware-compatibility estimation, ported from ggml-org/llama-macos
* (`Model+Compatibility.swift`, `HFRepoResolver.swift`, `SidecarPicker.swift`).
* Model memory estimation.
*
* A quant "fits" when its estimated runtime memory stays within the device
* budget. The budget mirrors llama.cpp's own fit target: the GPU working set
* (approximated from RAM) minus a fit slack, clamped by an OS floor so the
* desktop keeps enough to run. We cannot read Metal's working set from a
* browser, so the working set is approximated as 75% of RAM (Apple's ratio on
* the machines llama-macos targets).
*
* Compatibility tiers:
* - `full` -> fits at the model's native max context (green)
* - `limited` -> fits only at a reduced context (yellow)
* - `none` -> does not fit even at the minimum context (red)
* Runtime memory is approximated from the file size: the quantized weights
* plus KV cache/workspace overhead, rounded up to a GB. Context length and
* device-specific budgets are deliberately ignored - callers present the
* requirement and let the user judge.
*/
import { browser } from '$app/environment';
import { isAuxSidecar, type ModelSidecar } from '$lib/constants';
import { ModelAuxSidecar } from '$lib/enums';
import { HuggingFaceService } from '$lib/services';
import type { HfModelSibling } from '$lib/types/huggingface';
/** llama.cpp's default `--fit-target` margin, in MB. */
const FIT_SLACK_MB = 1024;
/** Physical RAM kept out of a model's reach for the OS and other apps, in MB. */
const OS_FLOOR_MB = 4096;
/** Overhead multiplier applied to the file size when estimating weight memory. */
const WEIGHT_OVERHEAD_MULTIPLIER = 1.05;
/** Fraction of RAM the GPU working set is approximated as (Apple's ~75%). */
const WORKING_SET_FRACTION = 0.75;
/** Minimum context a model must support to launch, matching llama.cpp's default. */
const MIN_CTX_TOKENS = 4096;
/** Standard context tiers, ascending, used to find the largest fitting one. */
const CTX_TIERS = [4096, 8192, 16384, 32768, 65536, 131072, 262144] as const;
export type CompatibilityTier = 'full' | 'limited' | 'none';
const MB = 1024 * 1024;
/**
* Resolve the device memory in GB: the user's settings override when set,
* else the browser's `navigator.deviceMemory` (Chrome/Edge only, capped at 8).
* Returns 0 when neither is available, which callers treat as "unknown".
* Estimated runtime memory (bytes) for a model of the given file size:
* file size with headroom for KV cache and allocator overhead.
*/
export function resolveDeviceMemoryGb(configuredGb: number): number {
if (configuredGb > 0) return configuredGb;
if (!browser) return 0;
const nav = navigator as Navigator & { deviceMemory?: number };
return typeof nav.deviceMemory === 'number' && nav.deviceMemory > 0 ? nav.deviceMemory : 0;
}
/**
* Memory a model may use, in MB: whichever of the GPU working set (less the
* fit slack) or the RAM-less-OS-floor binds first. Returns 0 when unknown.
*/
export function deviceMemoryBudgetMb(deviceMemoryGb: number): number {
if (deviceMemoryGb <= 0) return 0;
const physicalMb = deviceMemoryGb * 1024;
const gpuLimbMb = physicalMb * WORKING_SET_FRACTION - FIT_SLACK_MB;
const ramLimbMb = physicalMb - OS_FLOOR_MB;
return Math.max(Math.min(gpuLimbMb, ramLimbMb), 0);
}
/**
* Map every GGUF file in the repo to a compatibility tier. Main quants get
* their own tier; their shards and sidecars (mmproj + draft head) inherit the
* matched main quant's tier so the whole group reads consistently.
*/
export function computeFileCompatibilityTiers(
files: HfModelSibling[],
nativeCtxTokens: number,
deviceMemoryGb: number
): Map<string, CompatibilityTier> {
const tiers = new Map<string, CompatibilityTier>();
// Unknown device memory: leave every file untiered (neutral) rather than
// guessing a fit we cannot back up.
if (deviceMemoryGb <= 0) return tiers;
const allPaths = new Set(files.map((f) => f.path));
const sizeByPath = new Map(files.map((f) => [f.path, f.size ?? 0]));
const budgetMb = deviceMemoryBudgetMb(deviceMemoryGb);
// Candidate mains: skip draft heads, mmproj, imatrix, and non-first shards.
const mains = files.filter((f) => {
const meta = HuggingFaceService.extractQuantMeta(f.path);
if (!meta || meta.sidecar !== null) return false;
if ((f.path.split('/').pop() ?? f.path).toLowerCase().includes('imatrix')) return false;
return !isNonFirstShard(f.path);
});
// Track each main's quant bits + directory so draft sidecars can be matched
// to their closest-quant main for a tier below.
const mainInfos: MainQuantInfo[] = [];
for (const main of mains) {
// Aggregate size: main + shards + mmproj + quant-matched draft.
const picked = expandShards(main.path, allPaths);
const mmproj = pickSidecar(main.path, files, (v) => v === ModelAuxSidecar.MMPROJ);
const draft = pickSidecar(main.path, files, (v) => !isAuxSidecar(v));
if (mmproj) picked.push(mmproj.path);
const mainBytes = picked.reduce((sum, p) => sum + (sizeByPath.get(p) ?? 0), 0);
const draftBytes = draft ? (sizeByPath.get(draft.path) ?? 0) : 0;
const tier = compatibilityTier(mainBytes + draftBytes, nativeCtxTokens, budgetMb);
for (const p of picked) tiers.set(p, tier);
if (draft) tiers.set(draft.path, tier);
const meta = HuggingFaceService.extractQuantMeta(main.path);
const bits = meta?.quant ? (HuggingFaceService.getBitDepth(meta.quant) ?? 0) : 0;
mainInfos.push({ bits, dirs: dirComponents(main.path), tier });
}
// Assign every remaining draft sidecar (mtp, dflash, ...) the tier of its
// closest-quant main, so all sidecar badges show a fit icon - not just the
// single draft counted toward a main's size.
for (const file of files) {
if (tiers.has(file.path)) continue;
const meta = HuggingFaceService.extractQuantMeta(file.path);
if (!meta?.sidecar || isAuxSidecar(meta.sidecar)) continue;
const main = bestMainForSidecar(file.path, meta.quant, mainInfos);
if (main) tiers.set(file.path, main.tier);
}
return tiers;
}
interface MainQuantInfo {
bits: number;
dirs: string[];
tier: CompatibilityTier;
}
/**
* Find the main quant a draft sidecar pairs with: among mains in the
* sidecar's directory or a descendant of it, the one with the closest quant
* bit depth.
*/
function bestMainForSidecar(
sidecarPath: string,
sidecarQuant: string | null,
mains: MainQuantInfo[]
): MainQuantInfo | null {
const sidecarDirs = dirComponents(sidecarPath);
const sidecarBits = sidecarQuant ? (HuggingFaceService.getBitDepth(sidecarQuant) ?? 0) : 0;
let best: { diff: number; main: MainQuantInfo } | null = null;
for (const main of mains) {
// The sidecar's directory must be the main's directory or an ancestor.
if (sidecarDirs.length > main.dirs.length) continue;
if (!sidecarDirs.every((d, i) => main.dirs[i] === d)) continue;
const diff = Math.abs(main.bits - sidecarBits);
if (!best || diff < best.diff) best = { diff, main };
}
return best?.main ?? null;
}
/**
* Weight memory (MB) is the file size with overhead; context memory scales
* with the requested window. A quant is `full` when it fits at the native max
* context, `limited` when it only fits at a smaller standard tier, and `none`
* when it does not fit even at the minimum context.
*/
function compatibilityTier(
totalBytes: number,
nativeCtxTokens: number,
budgetMb: number
): CompatibilityTier {
// A known-but-tiny budget (<= 0) means nothing fits; unknown memory is
// handled by the caller returning no tiers at all.
const weightMb = (totalBytes / MB) * WEIGHT_OVERHEAD_MULTIPLIER;
const ctxBytesPer1k = ctxBytesPer1kTokens(nativeCtxTokens);
const fits = (ctxTokens: number) =>
weightMb + (ctxBytesPer1k * (ctxTokens / 1000)) / MB <= budgetMb;
if (nativeCtxTokens < MIN_CTX_TOKENS) return 'none';
if (fits(nativeCtxTokens)) return 'full';
// Find the largest standard tier that still fits within the native window.
const largestFitting = [...CTX_TIERS]
.filter((t) => t <= nativeCtxTokens)
.reverse()
.find((t) => fits(t));
return largestFitting !== undefined ? 'limited' : 'none';
}
/**
* Approximate KV-cache bytes per 1k tokens. Without a MemProfile probe (which
* only exists post-launch in llama-macos) we estimate from the native context
* window; ~0.1 MB per 1k tokens is a conservative mid-range for modern models.
*/
function ctxBytesPer1kTokens(_nativeCtxTokens: number): number {
return 0.1 * MB;
}
/** Directory components of a repo-relative path (`Q4_K_M/a.gguf` -> `["Q4_K_M"]`). */
function dirComponents(path: string): string[] {
return path.split('/').slice(0, -1);
}
/** True for a split-shard continuation (`-00002-of-00003.gguf`), not the first shard. */
function isNonFirstShard(path: string): boolean {
const match = /-(\d{5})-of-(\d{5})\.gguf$/i.exec(path);
return match !== null && match[1] !== '00001';
}
/** Expand a main GGUF to its full shard set; non-sharded files return `[main]`. */
function expandShards(main: string, allPaths: Set<string>): string[] {
const match = /-(\d{5})-of-(\d{5})\.gguf$/i.exec(main);
if (!match) return [main];
const total = parseInt(match[2], 10);
const stem = main.slice(0, main.length - match[0].length);
const shards: string[] = [];
for (let i = 1; i <= total; i++) {
const shard = `${stem}-${String(i).padStart(5, '0')}-of-${String(total).padStart(5, '0')}.gguf`;
if (allPaths.has(shard)) shards.push(shard);
}
return shards.length > 0 ? shards : [main];
}
/**
* Pick the best sidecar (mmproj or draft head) for a main file, mirroring
* llama.cpp's `find_best_sibling`: the candidate's directory must be the main
* file's directory or an ancestor; candidates rank by deepest directory, then
* exact quant-tag match, then closest quant-bit distance.
*/
function pickSidecar(
main: string,
files: HfModelSibling[],
isCandidate: (sidecar: ModelSidecar) => boolean
): HfModelSibling | null {
const mainDirs = dirComponents(main);
const mainMeta = HuggingFaceService.extractQuantMeta(main);
const mainBits = mainMeta?.quant ? (HuggingFaceService.getBitDepth(mainMeta.quant) ?? 0) : 0;
const mainTag = mainMeta?.quant?.toUpperCase();
let best: { depth: number; diff: number; exact: boolean; file: HfModelSibling } | null = null;
for (const file of files) {
const meta = HuggingFaceService.extractQuantMeta(file.path);
if (!meta || meta.sidecar === null || !isCandidate(meta.sidecar)) continue;
const dirs = dirComponents(file.path);
if (dirs.length > mainDirs.length || !dirs.every((d, i) => mainDirs[i] === d)) continue;
const depth = dirs.length;
const bits = meta.quant ? (HuggingFaceService.getBitDepth(meta.quant) ?? 0) : 0;
const diff = Math.abs(bits - mainBits);
const exact = mainTag ? file.path.toUpperCase().includes(`-${mainTag}.`) : false;
if (best) {
const better =
depth > best.depth ||
(depth === best.depth && exact && !best.exact) ||
(depth === best.depth && exact === best.exact && diff < best.diff);
if (!better) continue;
}
best = { depth, diff, exact, file };
}
return best?.file ?? null;
export function estimateModelMemoryBytes(sizeBytes: number): number {
return Math.round(sizeBytes * WEIGHT_OVERHEAD_MULTIPLIER);
}