mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-11 15:30:39 +02:00
ui : add the providers data layer
A provider is a server entry with a base url, an optional key, a protocol and the paths its API lives at; the local llama.cpp server is the built-in one. Backends persist in settings and the active one is restored on load. Requests resolve against the active backend, model ids become backend-qualified, and every backend's model list is fetched and cached in the background. The manager's helpers learn to read a model's drafts, context and the provider that serves it. Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
This commit is contained in:
Vendored
+2
@@ -12,6 +12,7 @@ import type {
|
|||||||
ApiChatCompletionStreamChunk,
|
ApiChatCompletionStreamChunk,
|
||||||
ApiChatCompletionToolCall,
|
ApiChatCompletionToolCall,
|
||||||
ApiChatCompletionToolCallDelta,
|
ApiChatCompletionToolCallDelta,
|
||||||
|
ApiChatCompletionUsage,
|
||||||
ApiChatMessageContentPart,
|
ApiChatMessageContentPart,
|
||||||
ApiChatMessageData,
|
ApiChatMessageData,
|
||||||
ApiContextSizeError,
|
ApiContextSizeError,
|
||||||
@@ -74,6 +75,7 @@ declare global {
|
|||||||
ApiChatCompletionResponse,
|
ApiChatCompletionResponse,
|
||||||
ApiChatCompletionStreamChunk,
|
ApiChatCompletionStreamChunk,
|
||||||
ApiChatCompletionToolCall,
|
ApiChatCompletionToolCall,
|
||||||
|
ApiChatCompletionUsage,
|
||||||
ApiChatCompletionToolCallDelta,
|
ApiChatCompletionToolCallDelta,
|
||||||
ApiChatMessageData,
|
ApiChatMessageData,
|
||||||
ApiChatMessageContentPart,
|
ApiChatMessageContentPart,
|
||||||
|
|||||||
@@ -109,6 +109,7 @@
|
|||||||
parsed.sidecar ||
|
parsed.sidecar ||
|
||||||
uniqueDraftKinds.length > 0 ||
|
uniqueDraftKinds.length > 0 ||
|
||||||
uniqueDraftSidecars.length > 0 ||
|
uniqueDraftSidecars.length > 0 ||
|
||||||
|
uniqueDraftKinds.length > 0 ||
|
||||||
(parsed.params && !hideParameters) ||
|
(parsed.params && !hideParameters) ||
|
||||||
(parsed.quantization && !resolvedHideQuantization) ||
|
(parsed.quantization && !resolvedHideQuantization) ||
|
||||||
primaryAlias ||
|
primaryAlias ||
|
||||||
|
|||||||
+1
-1
@@ -93,7 +93,7 @@
|
|||||||
/** Repos whose quants are folded away; the rest show them. */
|
/** Repos whose quants are folded away; the rest show them. */
|
||||||
const collapsedQuants = new SvelteSet<string>();
|
const collapsedQuants = new SvelteSet<string>();
|
||||||
/** Sections that list their models straight, without folding them into families. */
|
/** Sections that list their models straight, without folding them into families. */
|
||||||
const FLAT_SECTIONS = new Set<ModelsTableGroupKind>([
|
const FLAT_SECTIONS = new Set<ModelsTableGroup['kind']>([
|
||||||
ModelsTableGroupKind.DOWNLOADING,
|
ModelsTableGroupKind.DOWNLOADING,
|
||||||
ModelsTableGroupKind.FAVORITES,
|
ModelsTableGroupKind.FAVORITES,
|
||||||
ModelsTableGroupKind.LOADED
|
ModelsTableGroupKind.LOADED
|
||||||
|
|||||||
@@ -1,4 +1,11 @@
|
|||||||
import { LOCAL_BACKEND_ID, type ModalityKey } from '$lib/constants';
|
import {
|
||||||
|
LOCAL_BACKEND_ID,
|
||||||
|
type ModalityKey,
|
||||||
|
MODEL_OVERRIDES_LOCALSTORAGE_KEY,
|
||||||
|
type ModelSidecar,
|
||||||
|
SETTINGS_KEYS,
|
||||||
|
SPEC_TYPE
|
||||||
|
} from '$lib/constants';
|
||||||
import {
|
import {
|
||||||
ModelCapability,
|
ModelCapability,
|
||||||
ModelGroupKind,
|
ModelGroupKind,
|
||||||
@@ -6,9 +13,55 @@ import {
|
|||||||
ServerModelStatus
|
ServerModelStatus
|
||||||
} from '$lib/enums';
|
} from '$lib/enums';
|
||||||
import { HuggingFaceService, ModelsService } from '$lib/services';
|
import { HuggingFaceService, ModelsService } from '$lib/services';
|
||||||
import { modelsStore } from '$lib/stores';
|
import { backendsModelsStore, modelsStore, settingsStore } from '$lib/stores';
|
||||||
import type { ModelDownloadEntry, ModelDownloadProgress, ModelOption } from '$lib/types/models';
|
import type {
|
||||||
|
ModelDownloadEntry,
|
||||||
|
ModelDownloadProgress,
|
||||||
|
ModelLoadProgress,
|
||||||
|
ModelOption,
|
||||||
|
ModelSidecarFile
|
||||||
|
} from '$lib/types/models';
|
||||||
import { detectThinkingSupport, detectToolUseSupport, repoOf } from '$lib/utils';
|
import { detectThinkingSupport, detectToolUseSupport, repoOf } from '$lib/utils';
|
||||||
|
import { getBackend } from '$lib/utils/api-base';
|
||||||
|
import { getBackendCapabilities } from '$lib/utils/backend';
|
||||||
|
import { formatFileSize, formatParameters } from '$lib/utils/formatters';
|
||||||
|
import { rawModelId } from '$lib/utils/model-option-id';
|
||||||
|
import { SvelteMap } from 'svelte/reactivity';
|
||||||
|
|
||||||
|
/** Load parameters a model can override before it is loaded. */
|
||||||
|
export interface ModelLoadOverride {
|
||||||
|
batchSize?: number;
|
||||||
|
contextLength?: number;
|
||||||
|
cpuThreads?: number;
|
||||||
|
flashAttention?: boolean;
|
||||||
|
gpuOffload?: number;
|
||||||
|
keepInMemory?: boolean;
|
||||||
|
speculativeDecoding?: string;
|
||||||
|
ubatchSize?: number;
|
||||||
|
useMmap?: boolean;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Sampling parameters. A null value means the server default stays in charge. */
|
||||||
|
export interface ModelSamplingOverride {
|
||||||
|
minP?: number | null;
|
||||||
|
repeatPenalty?: number | null;
|
||||||
|
temperature?: number | null;
|
||||||
|
topK?: number | null;
|
||||||
|
topP?: number | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface ModelOverride {
|
||||||
|
load?: ModelLoadOverride;
|
||||||
|
reasoning?: { budget: string; enabled: boolean };
|
||||||
|
sampling?: ModelSamplingOverride;
|
||||||
|
stopStrings?: string[];
|
||||||
|
structuredOutput?: { enabled: boolean; schema: string };
|
||||||
|
systemPrompt?: string;
|
||||||
|
}
|
||||||
|
|
||||||
|
export type ModelOverrideMap = Record<string, ModelOverride>;
|
||||||
|
|
||||||
|
export type { ModelLoadProgress };
|
||||||
|
|
||||||
/** One repo of the table, with the rows it ships as. */
|
/** One repo of the table, with the rows it ships as. */
|
||||||
export interface ModelQuantGroup {
|
export interface ModelQuantGroup {
|
||||||
@@ -78,13 +131,201 @@ export function downloadGroups(
|
|||||||
|
|
||||||
/** One collapsible block of the manager's table. */
|
/** One collapsible block of the manager's table. */
|
||||||
export interface ModelsTableGroup {
|
export interface ModelsTableGroup {
|
||||||
|
/** Backend the block belongs to, when it is tied to one. */
|
||||||
|
backendId?: string | null;
|
||||||
/** One entry per repo, its quants hanging off it. */
|
/** One entry per repo, its quants hanging off it. */
|
||||||
items: ModelQuantGroup[];
|
items: ModelQuantGroup[];
|
||||||
|
isLocal?: boolean;
|
||||||
key: string;
|
key: string;
|
||||||
kind: ModelsTableGroupKind;
|
/** Manager sections use the kind constants, provider blocks their own kinds. */
|
||||||
|
kind: ModelsTableGroupKind | 'compat' | 'provider';
|
||||||
label: string;
|
label: string;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Values the load form falls back to when the server reports nothing. */
|
||||||
|
export const LOAD_DEFAULTS = {
|
||||||
|
batchSize: 2048,
|
||||||
|
contextLength: 8192,
|
||||||
|
cpuThreads: 13,
|
||||||
|
gpuOffload: 42,
|
||||||
|
speculativeDecoding: 'off',
|
||||||
|
ubatchSize: 512
|
||||||
|
};
|
||||||
|
|
||||||
|
export const SAMPLING_DEFAULTS = {
|
||||||
|
minP: 0.05,
|
||||||
|
repeatPenalty: 1.1,
|
||||||
|
temperature: 1,
|
||||||
|
topK: 64,
|
||||||
|
topP: 0.95
|
||||||
|
};
|
||||||
|
|
||||||
|
export const SPECULATIVE_OPTIONS = ['off', 'draft-model'];
|
||||||
|
|
||||||
|
/** True when the user saved anything for this model. */
|
||||||
|
export function isCustomized(override?: ModelOverride): boolean {
|
||||||
|
return override !== undefined && Object.keys(override).length > 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
export function loadOverrides(): ModelOverrideMap {
|
||||||
|
try {
|
||||||
|
const raw = localStorage.getItem(MODEL_OVERRIDES_LOCALSTORAGE_KEY);
|
||||||
|
|
||||||
|
if (!raw) return {};
|
||||||
|
|
||||||
|
const parsed = JSON.parse(raw) as unknown;
|
||||||
|
|
||||||
|
return parsed && typeof parsed === 'object' ? (parsed as ModelOverrideMap) : {};
|
||||||
|
} catch {
|
||||||
|
return {};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export function saveOverrides(overrides: ModelOverrideMap): void {
|
||||||
|
try {
|
||||||
|
localStorage.setItem(MODEL_OVERRIDES_LOCALSTORAGE_KEY, JSON.stringify(overrides));
|
||||||
|
} catch {
|
||||||
|
console.warn('[ModelsManager] Failed to persist model overrides');
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Backend a model is served by, the local server reads as "This server". */
|
||||||
|
/** A draft a model can speculate with, and whether a load would use it. */
|
||||||
|
export interface ModelDraft {
|
||||||
|
/** The draft a load would use, from the server's own arguments or from the settings. */
|
||||||
|
active: boolean;
|
||||||
|
kind: ModelSidecar | null;
|
||||||
|
/** Repo the draft comes from; null when the file sits in the model's own repo. */
|
||||||
|
model: string | null;
|
||||||
|
params: string | null;
|
||||||
|
quant: string | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Repo an id belongs to: the id without its backend prefix and quant tag. */
|
||||||
|
function draftRepoOf(modelId: string): string | null {
|
||||||
|
return repoOf(rawModelId(modelId)) || null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Sidecar a `--spec-type` value names, e.g. `draft-mtp` -> mtp. */
|
||||||
|
function sidecarFromSpecType(specType: string | null | undefined): ModelSidecar | null {
|
||||||
|
if (!specType) return null;
|
||||||
|
|
||||||
|
const entry = Object.entries(SPEC_TYPE).find(([, value]) => value === specType);
|
||||||
|
|
||||||
|
return (entry?.[0] as ModelSidecar | undefined) ?? null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Repo a draft path names. A Hub cache path carries it (`models--org--name`), a plain
|
||||||
|
* file next to the model does not, so that falls back to the file name.
|
||||||
|
*/
|
||||||
|
function repoFromDraftPath(path: string): string {
|
||||||
|
const cached = /models--([^/]+)[/]/.exec(path);
|
||||||
|
|
||||||
|
if (cached) {
|
||||||
|
const [org, ...rest] = cached[1].split('--');
|
||||||
|
|
||||||
|
if (rest.length > 0) return `${org}/${rest.join('--')}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
return path.split(/[/]/).pop() ?? path;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Draft the server's own launch arguments point at. The router reports the arguments a
|
||||||
|
* model loads with, so this is what a load would really speculate with.
|
||||||
|
*/
|
||||||
|
export function draftFromArgs(args: string[] | undefined, option: ModelOption): ModelDraft | null {
|
||||||
|
const flag = args?.indexOf('--model-draft') ?? -1;
|
||||||
|
const path = flag === -1 ? null : (args?.[flag + 1] ?? null);
|
||||||
|
|
||||||
|
if (!path) return null;
|
||||||
|
|
||||||
|
const parsed = ModelsService.parseModelId(path.split(/[/\\]/).pop() ?? path);
|
||||||
|
const repo = repoFromDraftPath(path);
|
||||||
|
|
||||||
|
return {
|
||||||
|
active: true,
|
||||||
|
kind: sidecarFromSpecType(args?.[(args?.indexOf('--spec-type') ?? -1) + 1]) ?? parsed.sidecar,
|
||||||
|
model: repo === draftRepoOf(option.model) ? null : repo,
|
||||||
|
params: parsed.params
|
||||||
|
? `${parsed.params}${parsed.activatedParams ? `-${parsed.activatedParams}` : ''}`
|
||||||
|
: null,
|
||||||
|
quant: parsed.quantization
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Draft the load settings name, resolved against the model's own repo. */
|
||||||
|
export function draftFromSetting(option: ModelOption, value?: string | null): ModelDraft | null {
|
||||||
|
const id = value?.trim();
|
||||||
|
|
||||||
|
if (!id || id === 'off') return null;
|
||||||
|
|
||||||
|
const parsed = ModelsService.parseModelId(id);
|
||||||
|
|
||||||
|
return {
|
||||||
|
active: true,
|
||||||
|
kind: parsed.sidecar,
|
||||||
|
model: draftRepoOf(id) === draftRepoOf(option.model) ? null : id,
|
||||||
|
params: parsed.params
|
||||||
|
? `${parsed.params}${parsed.activatedParams ? `-${parsed.activatedParams}` : ''}`
|
||||||
|
: null,
|
||||||
|
quant: parsed.quantization
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Drafts of a model, in the order they matter: what the server loads with, else what the
|
||||||
|
* settings name, then any other sidecar the model's own repo ships.
|
||||||
|
*/
|
||||||
|
export function modelDraftsFor(option: ModelOption, settingValue?: string | null): ModelDraft[] {
|
||||||
|
// speculative decoding is a llama.cpp feature
|
||||||
|
if (!getBackendCapabilities(getBackend(option.backendId)).loadUnload) return [];
|
||||||
|
|
||||||
|
const args = modelsStore.routerModels.find((model) => model.id === option.model)?.status?.args;
|
||||||
|
const configured = draftFromArgs(args, option) ?? draftFromSetting(option, settingValue);
|
||||||
|
|
||||||
|
return modelDrafts(option, sidecarFilesFor(option), configured);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Draft sidecars a listing reported for the model's repo. */
|
||||||
|
export function sidecarFilesFor(option: ModelOption): ModelSidecarFile[] {
|
||||||
|
const repo = option.model.split(':')[0] ?? '';
|
||||||
|
const state = backendsModelsStore.get(option.backendId ?? LOCAL_BACKEND_ID);
|
||||||
|
|
||||||
|
return state.drafts?.[repo] ?? [];
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Drafts to show for a model: the one the load settings name, plus any draft sidecar
|
||||||
|
* the model's own repo ships. The configured one is the active draft; a sidecar that
|
||||||
|
* is merely on disk stays visible but idle.
|
||||||
|
*/
|
||||||
|
export function modelDrafts(
|
||||||
|
option: ModelOption,
|
||||||
|
available: ModelSidecarFile[] = [],
|
||||||
|
configured?: ModelDraft | null
|
||||||
|
): ModelDraft[] {
|
||||||
|
const drafts: ModelDraft[] = [];
|
||||||
|
|
||||||
|
if (configured) drafts.push(configured);
|
||||||
|
|
||||||
|
for (const file of available) {
|
||||||
|
// a sidecar a load already points at is the active draft, not a second entry
|
||||||
|
if (drafts.some((draft) => draft.kind === file.kind)) continue;
|
||||||
|
|
||||||
|
drafts.push({
|
||||||
|
active: false,
|
||||||
|
kind: file.kind,
|
||||||
|
model: null,
|
||||||
|
params: file.params,
|
||||||
|
quant: file.quant
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
return drafts;
|
||||||
|
}
|
||||||
|
|
||||||
/** Context the model runs with: what a loaded model reports. */
|
/** Context the model runs with: what a loaded model reports. */
|
||||||
export function configuredContext(option: ModelOption): number | null {
|
export function configuredContext(option: ModelOption): number | null {
|
||||||
return modelsStore.isModelRunning(option.model)
|
return modelsStore.isModelRunning(option.model)
|
||||||
@@ -92,9 +333,11 @@ export function configuredContext(option: ModelOption): number | null {
|
|||||||
: null;
|
: null;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
/**
|
/**
|
||||||
* Capability a model reports: true or false once its listing or its chat template
|
* Capability a model reports: true or false once its listing or its chat template
|
||||||
* answers, null while the Hub record that carries the template is not read yet.
|
* answers, null while the Hub record that carries the template is not read yet. A
|
||||||
|
* listing that declares nothing is not a listing that lacks the capability.
|
||||||
*/
|
*/
|
||||||
export function modelCapability(option: ModelOption, capability: ModelCapability): boolean | null {
|
export function modelCapability(option: ModelOption, capability: ModelCapability): boolean | null {
|
||||||
if (option.capabilities.includes(capability)) return true;
|
if (option.capabilities.includes(capability)) return true;
|
||||||
@@ -111,6 +354,14 @@ export function modelCapability(option: ModelOption, capability: ModelCapability
|
|||||||
: detectThinkingSupport(template);
|
: detectThinkingSupport(template);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export function servedByLabel(option: ModelOption): string {
|
||||||
|
const backend = getBackend(option.backendId);
|
||||||
|
|
||||||
|
if (!backend || backend.id === LOCAL_BACKEND_ID) return 'This server';
|
||||||
|
|
||||||
|
return backend.name;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Context a model reports: the provider listing first, then the cached Hub record.
|
* Context a model reports: the provider listing first, then the cached Hub record.
|
||||||
* Null while neither of them has answered.
|
* Null while neither of them has answered.
|
||||||
@@ -123,12 +374,72 @@ export function modelContextLength(option: ModelOption): number | null {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
export function isLocalOption(option: ModelOption): boolean {
|
||||||
|
return (option.backendId ?? LOCAL_BACKEND_ID) === LOCAL_BACKEND_ID;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** File size of a local GGUF, when the router reported one. */
|
||||||
|
export function modelSizeLabel(option: ModelOption): string | null {
|
||||||
|
const bytes = option.meta?.size;
|
||||||
|
|
||||||
|
if (typeof bytes === 'number' && bytes > 0) return formatFileSize(bytes);
|
||||||
|
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export function modelParamsLabel(option: ModelOption): string | null {
|
||||||
|
if (option.parsedId?.params) return option.parsedId.params;
|
||||||
|
|
||||||
|
const params = option.meta?.n_params;
|
||||||
|
|
||||||
|
return typeof params === 'number' ? formatParameters(params) : null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export function modelQuantLabel(option: ModelOption): string | null {
|
||||||
|
return option.parsedId?.quantization ?? null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* File size of the model's own quant. The router reports one for some backends;
|
||||||
|
* otherwise a local GGUF reads it from its repo tree, the same source the
|
||||||
|
* discovery details use. Returns null when neither knows.
|
||||||
|
*/
|
||||||
|
export async function resolveModelSize(option: ModelOption): Promise<string | null> {
|
||||||
|
const reported = modelSizeLabel(option);
|
||||||
|
|
||||||
|
if (reported) return reported;
|
||||||
|
|
||||||
|
// the repo tree lookup only happens for installs that opted into the Hub
|
||||||
|
if (!isLocalOption(option) || !settingsStore.config[SETTINGS_KEYS.ENABLE_DISCOVER_MODELS]) {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
|
const [repo, quant] = option.model.split(':');
|
||||||
|
|
||||||
|
if (!repo || !quant) return null;
|
||||||
|
|
||||||
|
const tree = await HuggingFaceService.getTree(repo);
|
||||||
|
const file = HuggingFaceService.collapseGgufShards(
|
||||||
|
HuggingFaceService.filterByExtension(tree, '.gguf')
|
||||||
|
).find((entry) => {
|
||||||
|
const meta = HuggingFaceService.extractQuantMeta(entry.path);
|
||||||
|
|
||||||
|
return meta?.quant === quant && !meta.sidecar;
|
||||||
|
});
|
||||||
|
|
||||||
|
return file?.size ? formatFileSize(file.size) : null;
|
||||||
|
}
|
||||||
|
|
||||||
/** Fold the quants of one repo into a single entry, so the table shows one row per model. */
|
/** Fold the quants of one repo into a single entry, so the table shows one row per model. */
|
||||||
export function groupModelQuants(models: ModelOption[]): ModelQuantGroup[] {
|
export function groupModelQuants(models: ModelOption[], mergeProviders = false): ModelQuantGroup[] {
|
||||||
const groups = new Map<string, ModelQuantGroup>();
|
const groups = new SvelteMap<string, ModelQuantGroup>();
|
||||||
|
|
||||||
for (const option of models) {
|
for (const option of models) {
|
||||||
const key = repoOf(option.model);
|
const repo = repoOf(option.model);
|
||||||
|
// groups stay within one backend, so the same repo served by two providers
|
||||||
|
// is not read as two quants of one model. The OAI-compat block asks for the
|
||||||
|
// opposite: one repo, one row per provider that serves it.
|
||||||
|
const key = mergeProviders ? repo : `${option.backendId ?? ''}::${repo}`;
|
||||||
const group = groups.get(key);
|
const group = groups.get(key);
|
||||||
|
|
||||||
if (group) {
|
if (group) {
|
||||||
@@ -144,8 +455,13 @@ export function groupModelQuants(models: ModelOption[]): ModelQuantGroup[] {
|
|||||||
const kind = groupKind(group.quants);
|
const kind = groupKind(group.quants);
|
||||||
const modelIds = group.quants.map((option) => option.model);
|
const modelIds = group.quants.map((option) => option.model);
|
||||||
|
|
||||||
// the very same id twice is not a quant set; keep those rows apart
|
// the very same id twice is not a quant set; keep those rows apart, unless
|
||||||
if (modelIds.length > 1 && modelIds.every((model) => model === modelIds[0])) {
|
// the group exists to list the providers that serve it
|
||||||
|
if (
|
||||||
|
kind !== ModelGroupKind.PROVIDERS &&
|
||||||
|
modelIds.length > 1 &&
|
||||||
|
modelIds.every((model) => model === modelIds[0])
|
||||||
|
) {
|
||||||
return group.quants.map((option) => ({
|
return group.quants.map((option) => ({
|
||||||
...group,
|
...group,
|
||||||
base: option,
|
base: option,
|
||||||
@@ -159,8 +475,12 @@ export function groupModelQuants(models: ModelOption[]): ModelQuantGroup[] {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
/** What a group folds: quants or variants of one repo. */
|
/** What a group folds: providers, quants, or variants of one repo. */
|
||||||
function groupKind(quants: ModelOption[]): ModelGroupKind {
|
function groupKind(quants: ModelOption[]): ModelGroupKind {
|
||||||
|
const backends = new Set(quants.map((option) => option.backendId ?? ''));
|
||||||
|
|
||||||
|
if (backends.size > 1) return ModelGroupKind.PROVIDERS;
|
||||||
|
|
||||||
const isQuant = quants.every(
|
const isQuant = quants.every(
|
||||||
(option) => (option.parsedId ?? ModelsService.parseModelId(option.model)).quantization
|
(option) => (option.parsedId ?? ModelsService.parseModelId(option.model)).quantization
|
||||||
);
|
);
|
||||||
@@ -168,6 +488,50 @@ function groupKind(quants: ModelOption[]): ModelGroupKind {
|
|||||||
return isQuant ? ModelGroupKind.QUANTS : ModelGroupKind.VARIANTS;
|
return isQuant ? ModelGroupKind.QUANTS : ModelGroupKind.VARIANTS;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Compact "last used" label: minutes, hours, then days. */
|
||||||
|
export function formatLastUsed(timestamp?: number): string {
|
||||||
|
if (!timestamp) return '—';
|
||||||
|
|
||||||
|
const minutes = Math.floor((Date.now() - timestamp) / 60_000);
|
||||||
|
|
||||||
|
if (minutes < 1) return 'just now';
|
||||||
|
|
||||||
|
if (minutes < 60) return `${minutes}m`;
|
||||||
|
|
||||||
|
const hours = Math.floor(minutes / 60);
|
||||||
|
|
||||||
|
if (hours < 24) return `${hours}h`;
|
||||||
|
|
||||||
|
return `${Math.floor(hours / 24)}d`;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Extra args the router applies when this model is loaded. */
|
||||||
|
export function loadExtraArgs(override?: ModelOverride): string[] {
|
||||||
|
const load = override?.load;
|
||||||
|
|
||||||
|
if (!load) return [];
|
||||||
|
|
||||||
|
const args: string[] = [];
|
||||||
|
|
||||||
|
if (load.contextLength) args.push('--ctx-size', String(load.contextLength));
|
||||||
|
|
||||||
|
if (load.gpuOffload !== undefined) args.push('--n-gpu-layers', String(load.gpuOffload));
|
||||||
|
|
||||||
|
if (load.cpuThreads) args.push('--threads', String(load.cpuThreads));
|
||||||
|
|
||||||
|
if (load.batchSize) args.push('--batch-size', String(load.batchSize));
|
||||||
|
|
||||||
|
if (load.ubatchSize) args.push('--ubatch-size', String(load.ubatchSize));
|
||||||
|
|
||||||
|
if (load.flashAttention) args.push('--flash-attn', 'on');
|
||||||
|
|
||||||
|
if (load.useMmap === false) args.push('--no-mmap');
|
||||||
|
|
||||||
|
if (load.keepInMemory === false) args.push('--no-kv-offload');
|
||||||
|
|
||||||
|
return args;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Split the repos of a section into the quants that are hidden and the ones that
|
* Split the repos of a section into the quants that are hidden and the ones that
|
||||||
* are not, so a partly hidden repo lists in both blocks instead of dragging its
|
* are not, so a partly hidden repo lists in both blocks instead of dragging its
|
||||||
|
|||||||
@@ -1,2 +1,66 @@
|
|||||||
/** Marks the built-in local llama.cpp backend. */
|
import type { BackendCapabilities, BackendCompat, BackendProtocol } from '$lib/types';
|
||||||
|
|
||||||
|
/** Prefix for generated ids of user-added backends. */
|
||||||
|
export const BACKEND_ID_PREFIX = 'backend';
|
||||||
|
|
||||||
|
/** Protocols a configured backend can speak, in display order. */
|
||||||
|
export const BACKEND_PROTOCOLS: readonly BackendProtocol[] = ['llama.cpp', 'openai'];
|
||||||
|
|
||||||
|
/** Chat completions path used when a backend does not override it. */
|
||||||
|
export const DEFAULT_BACKEND_CHAT_PATH = '/v1/chat/completions';
|
||||||
|
|
||||||
|
/** Models listing path used when a backend does not override it. */
|
||||||
|
export const DEFAULT_BACKEND_MODELS_PATH = '/v1/models';
|
||||||
|
|
||||||
|
/** Id of the built-in backend that points at the server serving this UI. */
|
||||||
export const LOCAL_BACKEND_ID = 'local';
|
export const LOCAL_BACKEND_ID = 'local';
|
||||||
|
|
||||||
|
/** Capabilities of a full llama.cpp server. */
|
||||||
|
const LLAMA_CPP_CAPABILITIES: BackendCapabilities = {
|
||||||
|
corsProxy: true,
|
||||||
|
loadUnload: true,
|
||||||
|
props: true,
|
||||||
|
resumableStreams: true,
|
||||||
|
router: true,
|
||||||
|
slots: true,
|
||||||
|
statusFeed: true,
|
||||||
|
tools: true
|
||||||
|
};
|
||||||
|
/** Capabilities of a plain OpenAI-compatible endpoint. */
|
||||||
|
const COMPATIBLE_CAPABILITIES: BackendCapabilities = {
|
||||||
|
corsProxy: false,
|
||||||
|
loadUnload: false,
|
||||||
|
props: false,
|
||||||
|
resumableStreams: false,
|
||||||
|
router: false,
|
||||||
|
slots: false,
|
||||||
|
statusFeed: false,
|
||||||
|
tools: false
|
||||||
|
};
|
||||||
|
|
||||||
|
/** Capabilities per backend protocol. */
|
||||||
|
export const BACKEND_CAPABILITIES: Record<BackendProtocol, BackendCapabilities> = {
|
||||||
|
'llama.cpp': LLAMA_CPP_CAPABILITIES,
|
||||||
|
openai: COMPATIBLE_CAPABILITIES
|
||||||
|
};
|
||||||
|
|
||||||
|
/** Default wire quirks per protocol. */
|
||||||
|
export const BACKEND_COMPAT: Record<BackendProtocol, BackendCompat> = {
|
||||||
|
// llama-server reports its own timings, so it needs no usage chunk
|
||||||
|
'llama.cpp': { maxTokensField: 'max_tokens', supportsUsageInStreaming: false },
|
||||||
|
openai: { maxTokensField: 'max_tokens', supportsUsageInStreaming: true }
|
||||||
|
};
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Fields that may carry a model's context size in an OpenAI-compatible model
|
||||||
|
* listing. Providers pick their own name, and most report nothing at all.
|
||||||
|
*/
|
||||||
|
export const MODEL_CONTEXT_LENGTH_FIELDS = [
|
||||||
|
'context_length',
|
||||||
|
'context_window',
|
||||||
|
'max_context_length',
|
||||||
|
'max_position_embeddings'
|
||||||
|
] as const;
|
||||||
|
|
||||||
|
/** Favicon extract keyed by domain, for backends with no bundled mark. */
|
||||||
|
export const FAVICON_SERVICE_URL = 'https://www.google.com/s2/favicons?domain=';
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ export const SETTINGS_KEYS = {
|
|||||||
API_KEY: 'apiKey',
|
API_KEY: 'apiKey',
|
||||||
AUTO_MIC_ON_EMPTY: 'autoMicOnEmpty',
|
AUTO_MIC_ON_EMPTY: 'autoMicOnEmpty',
|
||||||
BACKEND_SAMPLING: 'backend_sampling',
|
BACKEND_SAMPLING: 'backend_sampling',
|
||||||
|
BACKENDS: 'backends',
|
||||||
CONVERSATION_TABS: 'conversationTabs',
|
CONVERSATION_TABS: 'conversationTabs',
|
||||||
COPY_TEXT_ATTACHMENTS_AS_PLAIN_TEXT: 'copyTextAttachmentsAsPlainText',
|
COPY_TEXT_ATTACHMENTS_AS_PLAIN_TEXT: 'copyTextAttachmentsAsPlainText',
|
||||||
CUSTOM_CSS: 'customCss',
|
CUSTOM_CSS: 'customCss',
|
||||||
@@ -32,6 +33,7 @@ export const SETTINGS_KEYS = {
|
|||||||
FULL_HEIGHT_CODE_BLOCKS: 'fullHeightCodeBlocks',
|
FULL_HEIGHT_CODE_BLOCKS: 'fullHeightCodeBlocks',
|
||||||
GROUP_MODELS_BY_FAMILY: 'groupModelsByFamily',
|
GROUP_MODELS_BY_FAMILY: 'groupModelsByFamily',
|
||||||
JS_SANDBOX_ENABLED: 'jsSandboxEnabled',
|
JS_SANDBOX_ENABLED: 'jsSandboxEnabled',
|
||||||
|
LOCAL_BACKEND_ENABLED: 'localBackendEnabled',
|
||||||
MAX_IMAGE_RESOLUTION: 'maxImageMPixels',
|
MAX_IMAGE_RESOLUTION: 'maxImageMPixels',
|
||||||
MAX_TOKENS: 'max_tokens',
|
MAX_TOKENS: 'max_tokens',
|
||||||
MCP_REQUEST_TIMEOUT_SECONDS: 'mcpRequestTimeoutSeconds',
|
MCP_REQUEST_TIMEOUT_SECONDS: 'mcpRequestTimeoutSeconds',
|
||||||
|
|||||||
@@ -16,6 +16,9 @@ export const DB_APP_NAME_DEPRECATED = 'LlamacppWebui';
|
|||||||
|
|
||||||
export const ALWAYS_ALLOWED_TOOLS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.alwaysAllowedTools`;
|
export const ALWAYS_ALLOWED_TOOLS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.alwaysAllowedTools`;
|
||||||
|
|
||||||
|
/** Id of the backend the selector and new requests target, restored on page load. */
|
||||||
|
export const ACTIVE_BACKEND_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.activeBackend`;
|
||||||
|
|
||||||
/** Paused model download ids (`<repo>:<tag>`), restored on the next page load. */
|
/** Paused model download ids (`<repo>:<tag>`), restored on the next page load. */
|
||||||
export const PAUSED_MODEL_DOWNLOADS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.pausedModelDownloads`;
|
export const PAUSED_MODEL_DOWNLOADS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.pausedModelDownloads`;
|
||||||
export const CONFIG_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.config`;
|
export const CONFIG_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.config`;
|
||||||
@@ -31,6 +34,12 @@ export const FAVORITE_MODELS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.favoriteMod
|
|||||||
/** Open state the user set for a model list section or one of its families, by id. */
|
/** Open state the user set for a model list section or one of its families, by id. */
|
||||||
export const MODEL_GROUP_OPEN_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.modelGroupOpen`;
|
export const MODEL_GROUP_OPEN_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.modelGroupOpen`;
|
||||||
|
|
||||||
|
/** Per-model load and inference overrides, keyed by backend-qualified model id. */
|
||||||
|
export const MODEL_OVERRIDES_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.modelOverrides`;
|
||||||
|
|
||||||
|
/** Model the user picked last, kept across reloads. Stores `{ id, model }`. */
|
||||||
|
export const SELECTED_MODEL_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.selectedModel`;
|
||||||
|
|
||||||
/** Recently used model ids, most recent first, backend-qualified. */
|
/** Recently used model ids, most recent first, backend-qualified. */
|
||||||
export const RECENT_MODELS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.recentModels`;
|
export const RECENT_MODELS_LOCALSTORAGE_KEY = `${STORAGE_APP_NAME}.recentModels`;
|
||||||
|
|
||||||
|
|||||||
@@ -2,6 +2,10 @@
|
|||||||
// while the tab was hidden. covers brief background pauses without thrashing live streams
|
// while the tab was hidden. covers brief background pauses without thrashing live streams
|
||||||
export const STREAM_VISIBILITY_KICK_MS = 3000;
|
export const STREAM_VISIBILITY_KICK_MS = 3000;
|
||||||
|
|
||||||
|
// minimum gap between synthesized live timing updates for backends that do not
|
||||||
|
// stream their own, keeps the per-chunk state updates cheap
|
||||||
|
export const STREAM_LIVE_TIMINGS_INTERVAL_MS = 500;
|
||||||
|
|
||||||
// separator joining a conversation id and its per-model stream identity
|
// separator joining a conversation id and its per-model stream identity
|
||||||
// suffix (conv::model) used by the server side replay buffer
|
// suffix (conv::model) used by the server side replay buffer
|
||||||
export const CONVERSATION_ID_SEPARATOR = '::';
|
export const CONVERSATION_ID_SEPARATOR = '::';
|
||||||
|
|||||||
@@ -66,8 +66,9 @@ export enum ModelDownloadConfirmAction {
|
|||||||
DELETE = 'delete'
|
DELETE = 'delete'
|
||||||
}
|
}
|
||||||
|
|
||||||
/** What a table row group folds: the quants of one repo, or its variants. */
|
/** What a table row group folds: the providers of one repo, its quants, or its variants. */
|
||||||
export enum ModelGroupKind {
|
export enum ModelGroupKind {
|
||||||
|
PROVIDERS = 'providers',
|
||||||
QUANTS = 'quants',
|
QUANTS = 'quants',
|
||||||
VARIANTS = 'variants'
|
VARIANTS = 'variants'
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,188 @@
|
|||||||
|
/**
|
||||||
|
* BackendsService - Stateless backend connectivity checks and model listing
|
||||||
|
*
|
||||||
|
* Probes a backend's models endpoint to validate its URL and credentials, and
|
||||||
|
* normalizes the response into the UI model shape. No reactive state;
|
||||||
|
* consumed by the backends settings UI and the per-backend model cache.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { API_MODELS, LOCAL_BACKEND_ID } from '$lib/constants';
|
||||||
|
import { ModelsService } from '$lib/services/models.service';
|
||||||
|
import type { ApiModelsListResponse, Backend, BackendProtocol, ModelOption } from '$lib/types';
|
||||||
|
import { isAbortError } from '$lib/utils/abort';
|
||||||
|
import { apiUrl } from '$lib/utils/api-base';
|
||||||
|
import { getAuthHeadersForBackend } from '$lib/utils/api-headers';
|
||||||
|
import { backendModelsUrl, readModelContextLength } from '$lib/utils/backend';
|
||||||
|
|
||||||
|
/** Models returned by a backend, plus the failure detail when the call fails. */
|
||||||
|
export interface BackendModelsResult {
|
||||||
|
error?: string;
|
||||||
|
models: ModelOption[];
|
||||||
|
ok: boolean;
|
||||||
|
status: number | null;
|
||||||
|
/** Untouched list payload of the local backend, kept so the router rows and their load statuses can be rebuilt without asking again. */
|
||||||
|
raw?: ApiModelsListResponse;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** What probing a backend's endpoint said about it. */
|
||||||
|
export interface BackendProbe {
|
||||||
|
/** The endpoint refused the request for want of a key, so it wants one. */
|
||||||
|
authRequired: boolean;
|
||||||
|
protocol: BackendProtocol;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Outcome of a backend connectivity check. */
|
||||||
|
export interface BackendTestResult {
|
||||||
|
error?: string;
|
||||||
|
modelCount?: number;
|
||||||
|
ok: boolean;
|
||||||
|
status: number | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
export class BackendsService {
|
||||||
|
/**
|
||||||
|
* List the models a backend exposes on its models endpoint.
|
||||||
|
*
|
||||||
|
* @param backend - Backend to query. Does not need to be registered yet.
|
||||||
|
* @param signal - Optional abort signal for a cancelled request.
|
||||||
|
*/
|
||||||
|
static async detectProtocol(backend: Backend): Promise<BackendProbe> {
|
||||||
|
const base = backend.baseUrl.trim().replace(/\/+$/, '');
|
||||||
|
|
||||||
|
if (!base) return { authRequired: false, protocol: 'openai' };
|
||||||
|
|
||||||
|
try {
|
||||||
|
// llama-server answers /props with its build and generation defaults; a
|
||||||
|
// plain OpenAI-compatible endpoint answers 404 there, or not at all
|
||||||
|
const response = await fetch(`${base}/props`, {
|
||||||
|
headers: getAuthHeadersForBackend(backend),
|
||||||
|
signal: AbortSignal.timeout(5000)
|
||||||
|
});
|
||||||
|
|
||||||
|
// a llama-server behind a key refuses before it says anything else, while
|
||||||
|
// an OpenAI-compatible endpoint has no /props to guard in the first place
|
||||||
|
if (response.status === 401) {
|
||||||
|
return { authRequired: true, protocol: 'llama.cpp' };
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!response.ok) return { authRequired: false, protocol: 'openai' };
|
||||||
|
|
||||||
|
const body = (await response.json()) as Record<string, unknown>;
|
||||||
|
const isLlamaCpp =
|
||||||
|
'default_generation_settings' in body || 'build_info' in body || body.role === 'router';
|
||||||
|
|
||||||
|
return {
|
||||||
|
authRequired: false,
|
||||||
|
protocol: isLlamaCpp ? 'llama.cpp' : 'openai'
|
||||||
|
};
|
||||||
|
} catch {
|
||||||
|
return { authRequired: false, protocol: 'openai' };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static async listModels(backend: Backend, signal?: AbortSignal): Promise<BackendModelsResult> {
|
||||||
|
// the local backend has no base URL; its models endpoint is base relative
|
||||||
|
const url = backend.baseUrl.trim()
|
||||||
|
? backendModelsUrl(backend)
|
||||||
|
: apiUrl(API_MODELS.LIST, LOCAL_BACKEND_ID);
|
||||||
|
|
||||||
|
if (!backend.baseUrl.trim() && backend.id !== LOCAL_BACKEND_ID) {
|
||||||
|
return { error: 'Backend URL is required', models: [], ok: false, status: null };
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
const response = await fetch(url, {
|
||||||
|
headers: getAuthHeadersForBackend(backend),
|
||||||
|
signal
|
||||||
|
});
|
||||||
|
|
||||||
|
if (!response.ok) {
|
||||||
|
return {
|
||||||
|
error: await describeFailure(response),
|
||||||
|
models: [],
|
||||||
|
ok: false,
|
||||||
|
status: response.status
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
const body = (await response.json()) as { data?: unknown };
|
||||||
|
const entries = Array.isArray(body?.data) ? body.data : [];
|
||||||
|
const models = entries.flatMap((entry) => normalizeBackendModel(entry));
|
||||||
|
// the local rows carry load status, external ones carry the context size
|
||||||
|
const raw = body as ApiModelsListResponse;
|
||||||
|
|
||||||
|
return { models, ok: true, raw, status: response.status };
|
||||||
|
} catch (error) {
|
||||||
|
if (isAbortError(error)) {
|
||||||
|
return { models: [], ok: false, status: null };
|
||||||
|
}
|
||||||
|
|
||||||
|
return {
|
||||||
|
error: error instanceof Error ? error.message : String(error),
|
||||||
|
models: [],
|
||||||
|
ok: false,
|
||||||
|
status: null
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Check that a backend answers on its models endpoint.
|
||||||
|
*
|
||||||
|
* @param backend - Backend to probe. Does not need to be registered yet.
|
||||||
|
* @param signal - Optional abort signal for a cancelled test.
|
||||||
|
*/
|
||||||
|
static async test(backend: Backend, signal?: AbortSignal): Promise<BackendTestResult> {
|
||||||
|
const result = await BackendsService.listModels(backend, signal);
|
||||||
|
|
||||||
|
return {
|
||||||
|
error: result.error,
|
||||||
|
modelCount: result.models.length,
|
||||||
|
ok: result.ok,
|
||||||
|
status: result.status
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Build a human-readable message from a non-OK response. */
|
||||||
|
async function describeFailure(response: Response): Promise<string> {
|
||||||
|
const status = `${response.status} ${response.statusText}`.trim();
|
||||||
|
|
||||||
|
try {
|
||||||
|
const body = (await response.json()) as { error?: { message?: string }; message?: string };
|
||||||
|
const message = body?.error?.message ?? body?.message;
|
||||||
|
|
||||||
|
if (message) return `${status}: ${message}`;
|
||||||
|
} catch {
|
||||||
|
// non-JSON error body, fall back to the status line
|
||||||
|
}
|
||||||
|
|
||||||
|
return status;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Normalize one entry of an OpenAI-compatible `/v1/models` response. External
|
||||||
|
* backends only guarantee an id, so that doubles as the display name.
|
||||||
|
*/
|
||||||
|
function normalizeBackendModel(entry: unknown): ModelOption[] {
|
||||||
|
if (!entry || typeof entry !== 'object') return [];
|
||||||
|
|
||||||
|
const raw = entry as Record<string, unknown>;
|
||||||
|
const id = typeof raw.id === 'string' ? raw.id.trim() : '';
|
||||||
|
|
||||||
|
if (!id) return [];
|
||||||
|
|
||||||
|
// a llama-compat server lists its projector and draft sidecars as models too
|
||||||
|
if (ModelsService.isSidecarEntry(id)) return [];
|
||||||
|
|
||||||
|
return [
|
||||||
|
{
|
||||||
|
capabilities: [],
|
||||||
|
contextLength: readModelContextLength(raw),
|
||||||
|
id,
|
||||||
|
model: id,
|
||||||
|
name: id,
|
||||||
|
status: raw.status as ApiModelDataEntry['status']
|
||||||
|
}
|
||||||
|
];
|
||||||
|
}
|
||||||
@@ -19,6 +19,14 @@
|
|||||||
*
|
*
|
||||||
*/
|
*/
|
||||||
|
|
||||||
|
/**
|
||||||
|
* **BackendsService** - Backend connectivity checks
|
||||||
|
*
|
||||||
|
* Probes an external backend's models endpoint to validate its URL and
|
||||||
|
* credentials before it is saved. Stateless.
|
||||||
|
*/
|
||||||
|
export { BackendsService } from './backends.service';
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* **ChatService** - Chat Completions API communication layer
|
* **ChatService** - Chat Completions API communication layer
|
||||||
*
|
*
|
||||||
|
|||||||
@@ -0,0 +1,113 @@
|
|||||||
|
/**
|
||||||
|
* backendsStore - API endpoints the UI can talk to.
|
||||||
|
*
|
||||||
|
* The built-in local backend is the llama-server serving this UI. External
|
||||||
|
* backends are user-configured endpoints persisted in settings. The store
|
||||||
|
* registers the resolved list with the api-base registry, which services use
|
||||||
|
* to build request URLs.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { browser } from '$app/environment';
|
||||||
|
import { ACTIVE_BACKEND_LOCALSTORAGE_KEY, LOCAL_BACKEND_ID, SETTINGS_KEYS } from '$lib/constants';
|
||||||
|
import { serverStore } from '$lib/stores/server.svelte';
|
||||||
|
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
||||||
|
import type { Backend } from '$lib/types';
|
||||||
|
import { setBackendsResolver } from '$lib/utils/api-base';
|
||||||
|
import { createLocalBackend, parseBackendsSettings } from '$lib/utils/backend';
|
||||||
|
|
||||||
|
function loadActiveBackendId(): string {
|
||||||
|
if (!browser) return LOCAL_BACKEND_ID;
|
||||||
|
|
||||||
|
try {
|
||||||
|
return localStorage.getItem(ACTIVE_BACKEND_LOCALSTORAGE_KEY) ?? LOCAL_BACKEND_ID;
|
||||||
|
} catch {
|
||||||
|
return LOCAL_BACKEND_ID;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function persistActiveBackendId(backendId: string): void {
|
||||||
|
if (!browser) return;
|
||||||
|
|
||||||
|
try {
|
||||||
|
localStorage.setItem(ACTIVE_BACKEND_LOCALSTORAGE_KEY, backendId);
|
||||||
|
} catch {
|
||||||
|
/* ignore */
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
class BackendsStore {
|
||||||
|
activeId = $state<string>(loadActiveBackendId());
|
||||||
|
|
||||||
|
get active(): Backend {
|
||||||
|
const active = this.enabled.find((backend) => backend.id === this.activeId);
|
||||||
|
|
||||||
|
return active ?? this.enabled[0] ?? this.local;
|
||||||
|
}
|
||||||
|
|
||||||
|
get enabled(): Backend[] {
|
||||||
|
return this.list.filter((backend) => backend.enabled && !this.isMissingLocal(backend));
|
||||||
|
}
|
||||||
|
|
||||||
|
get external(): Backend[] {
|
||||||
|
return parseBackendsSettings(settingsStore.config[SETTINGS_KEYS.BACKENDS]);
|
||||||
|
}
|
||||||
|
|
||||||
|
get list(): Backend[] {
|
||||||
|
return [this.local, ...this.external];
|
||||||
|
}
|
||||||
|
|
||||||
|
get local(): Backend {
|
||||||
|
return createLocalBackend(
|
||||||
|
settingsStore.config.apiKey?.toString().trim() || undefined,
|
||||||
|
settingsStore.config[SETTINGS_KEYS.LOCAL_BACKEND_ENABLED] !== false
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
addBackend(backend: Backend): void {
|
||||||
|
this.saveExternal([...this.external, backend]);
|
||||||
|
}
|
||||||
|
|
||||||
|
initialize(): void {
|
||||||
|
if (!browser) return;
|
||||||
|
|
||||||
|
setBackendsResolver(() => ({ activeId: this.active.id, backends: this.list }));
|
||||||
|
}
|
||||||
|
|
||||||
|
removeBackend(backendId: string): void {
|
||||||
|
this.saveExternal(this.external.filter((backend) => backend.id !== backendId));
|
||||||
|
|
||||||
|
if (this.activeId === backendId) {
|
||||||
|
this.setActive(LOCAL_BACKEND_ID);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
setActive(backendId: string): void {
|
||||||
|
const id = this.list.some((backend) => backend.id === backendId) ? backendId : LOCAL_BACKEND_ID;
|
||||||
|
|
||||||
|
this.activeId = id;
|
||||||
|
persistActiveBackendId(id);
|
||||||
|
}
|
||||||
|
|
||||||
|
setLocalEnabled(enabled: boolean): void {
|
||||||
|
settingsStore.updateConfig(SETTINGS_KEYS.LOCAL_BACKEND_ENABLED, enabled);
|
||||||
|
}
|
||||||
|
|
||||||
|
updateBackend(backendId: string, updates: Partial<Backend>): void {
|
||||||
|
this.saveExternal(
|
||||||
|
this.external.map((backend) =>
|
||||||
|
backend.id === backendId ? { ...backend, ...updates } : backend
|
||||||
|
)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The built-in backend counts only when a local server answered this session. */
|
||||||
|
private isMissingLocal(backend: Backend): boolean {
|
||||||
|
return backend.id === LOCAL_BACKEND_ID && serverStore.localServerMissing;
|
||||||
|
}
|
||||||
|
|
||||||
|
private saveExternal(backends: Backend[]): void {
|
||||||
|
settingsStore.updateConfig(SETTINGS_KEYS.BACKENDS, JSON.stringify(backends));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export const backendsStore = new BackendsStore();
|
||||||
@@ -0,0 +1,103 @@
|
|||||||
|
/**
|
||||||
|
* backendsModelsStore - Per-backend model catalog cache.
|
||||||
|
*
|
||||||
|
* Prefetches every enabled backend's model list, so switching backends is
|
||||||
|
* instant and the switcher can show load state. The active backend's list
|
||||||
|
* still lives in modelsStore, which owns selection and chat wiring; this
|
||||||
|
* cache is the prefetch layer the switches start from.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { BackendsService } from '$lib/services/backends.service';
|
||||||
|
import { ModelsService } from '$lib/services/models.service';
|
||||||
|
import { backendsStore } from '$lib/stores/backends.svelte';
|
||||||
|
import type { Backend } from '$lib/types';
|
||||||
|
import type { ApiModelsListResponse } from '$lib/types';
|
||||||
|
import type { ModelSidecarFile } from '$lib/types/models';
|
||||||
|
import type { ModelOption } from '$lib/types/models';
|
||||||
|
|
||||||
|
export interface BackendModelsState {
|
||||||
|
/** Draft sidecars the listing carries, keyed by the repo they belong to. */
|
||||||
|
drafts?: Record<string, ModelSidecarFile[]>;
|
||||||
|
error: string | null;
|
||||||
|
loaded: boolean;
|
||||||
|
loading: boolean;
|
||||||
|
models: ModelOption[];
|
||||||
|
/** Untouched list payload, kept for the local backend so its router rows survive a tab switch. */
|
||||||
|
raw?: ApiModelsListResponse;
|
||||||
|
}
|
||||||
|
|
||||||
|
const EMPTY_STATE: BackendModelsState = {
|
||||||
|
error: null,
|
||||||
|
loaded: false,
|
||||||
|
loading: false,
|
||||||
|
models: []
|
||||||
|
};
|
||||||
|
|
||||||
|
class BackendsModelsStore {
|
||||||
|
private states = $state<Record<string, BackendModelsState>>({});
|
||||||
|
|
||||||
|
clear(backendId: string): void {
|
||||||
|
delete this.states[backendId];
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Load a backend's models once.
|
||||||
|
*/
|
||||||
|
async ensureLoaded(backendId: string): Promise<void> {
|
||||||
|
const backend = backendsStore.enabled.find((candidate) => candidate.id === backendId);
|
||||||
|
|
||||||
|
if (!backend) return;
|
||||||
|
|
||||||
|
const state = this.states[backendId];
|
||||||
|
|
||||||
|
if (state?.loaded || state?.loading) return;
|
||||||
|
|
||||||
|
this.states[backendId] = { error: null, loaded: false, loading: true, models: [] };
|
||||||
|
await this.fetch(backend);
|
||||||
|
}
|
||||||
|
|
||||||
|
get(backendId: string): BackendModelsState {
|
||||||
|
return this.states[backendId] ?? EMPTY_STATE;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Prefetch every enabled backend's model list. */
|
||||||
|
async loadAll(): Promise<void> {
|
||||||
|
const enabled = backendsStore.enabled;
|
||||||
|
const ids = new Set(enabled.map((backend) => backend.id));
|
||||||
|
|
||||||
|
for (const id of Object.keys(this.states)) {
|
||||||
|
if (!ids.has(id)) {
|
||||||
|
delete this.states[id];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
await Promise.all(enabled.map((backend) => this.ensureLoaded(backend.id)));
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Ask a backend for its list again, keeping what is already known. Used while
|
||||||
|
* a remote load settles, since its status never reaches the local feed.
|
||||||
|
*/
|
||||||
|
async refresh(backendId: string): Promise<void> {
|
||||||
|
const backend = backendsStore.enabled.find((candidate) => candidate.id === backendId);
|
||||||
|
|
||||||
|
if (!backend) return;
|
||||||
|
|
||||||
|
await this.fetch(backend);
|
||||||
|
}
|
||||||
|
|
||||||
|
private async fetch(backend: Backend): Promise<void> {
|
||||||
|
const result = await BackendsService.listModels(backend);
|
||||||
|
|
||||||
|
this.states[backend.id] = {
|
||||||
|
drafts: result.raw ? ModelsService.draftSidecarsByRepo(result.raw) : undefined,
|
||||||
|
error: result.error ?? null,
|
||||||
|
loaded: result.ok,
|
||||||
|
loading: false,
|
||||||
|
models: result.models,
|
||||||
|
raw: result.raw
|
||||||
|
};
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export const backendsModelsStore = new BackendsModelsStore();
|
||||||
Vendored
+13
@@ -358,7 +358,9 @@ export interface ApiChatCompletionStreamChunk {
|
|||||||
metadata?: { model?: string };
|
metadata?: { model?: string };
|
||||||
delta: {
|
delta: {
|
||||||
content?: string;
|
content?: string;
|
||||||
|
reasoning?: string;
|
||||||
reasoning_content?: string;
|
reasoning_content?: string;
|
||||||
|
reasoning_text?: string;
|
||||||
model?: string;
|
model?: string;
|
||||||
tool_calls?: ApiChatCompletionToolCallDelta[];
|
tool_calls?: ApiChatCompletionToolCallDelta[];
|
||||||
};
|
};
|
||||||
@@ -372,6 +374,17 @@ export interface ApiChatCompletionStreamChunk {
|
|||||||
cache_n?: number;
|
cache_n?: number;
|
||||||
};
|
};
|
||||||
prompt_progress?: ChatMessagePromptProgress;
|
prompt_progress?: ChatMessagePromptProgress;
|
||||||
|
/** Token counts, sent by OpenAI-compatible servers on the final chunk. */
|
||||||
|
usage?: ApiChatCompletionUsage;
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface ApiChatCompletionUsage {
|
||||||
|
cached_tokens?: number;
|
||||||
|
completion_tokens?: number;
|
||||||
|
prompt_cache_hit_tokens?: number;
|
||||||
|
prompt_tokens?: number;
|
||||||
|
prompt_tokens_details?: { cached_tokens?: number; cache_write_tokens?: number };
|
||||||
|
total_tokens?: number;
|
||||||
}
|
}
|
||||||
|
|
||||||
export interface ApiChatCompletionResponse {
|
export interface ApiChatCompletionResponse {
|
||||||
|
|||||||
Vendored
+98
@@ -0,0 +1,98 @@
|
|||||||
|
/**
|
||||||
|
* Backend types.
|
||||||
|
*
|
||||||
|
* A backend is one API endpoint the UI can talk to. The built-in `local`
|
||||||
|
* backend is the llama-server serving the UI. External backends are
|
||||||
|
* user-configured endpoints that speak an OpenAI-compatible
|
||||||
|
* protocol.
|
||||||
|
*/
|
||||||
|
|
||||||
|
/** Request/response shape a backend speaks. */
|
||||||
|
export type BackendProtocol = 'llama.cpp' | 'openai';
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Wire-level quirks of a backend's protocol. Capabilities gate llama.cpp
|
||||||
|
* features; compat describes how the request and stream payloads differ.
|
||||||
|
*/
|
||||||
|
export interface BackendCompat {
|
||||||
|
/** Field carrying the output token cap. */
|
||||||
|
maxTokensField: 'max_completion_tokens' | 'max_tokens';
|
||||||
|
/** Whether the endpoint accepts stream_options.include_usage. */
|
||||||
|
supportsUsageInStreaming: boolean;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Features a backend supports. A llama.cpp server exposes extra endpoints on
|
||||||
|
* top of the OpenAI-compatible API; plain OpenAI-compatible
|
||||||
|
* endpoints only provide chat and model listing.
|
||||||
|
*/
|
||||||
|
export interface BackendCapabilities {
|
||||||
|
/** llama-server's /cors-proxy endpoint for cross-origin MCP requests. */
|
||||||
|
corsProxy: boolean;
|
||||||
|
/** Router-mode model load/unload. */
|
||||||
|
loadUnload: boolean;
|
||||||
|
/** The /props endpoint with server role and generation defaults. */
|
||||||
|
props: boolean;
|
||||||
|
/** Resumable stream sessions (/v1/stream, /v1/streams/lookup). */
|
||||||
|
resumableStreams: boolean;
|
||||||
|
/** Multi-model router mode. */
|
||||||
|
router: boolean;
|
||||||
|
/** The /slots introspection endpoint. */
|
||||||
|
slots: boolean;
|
||||||
|
/** The /models/sse load and download progress feed. */
|
||||||
|
statusFeed: boolean;
|
||||||
|
/** The /tools listing and execution endpoint. */
|
||||||
|
tools: boolean;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* One configured API endpoint.
|
||||||
|
*
|
||||||
|
* TODO: a backend paired by QR code is reached over WebRTC instead of plain
|
||||||
|
* HTTP, so it needs a transport discriminator and its peer description here.
|
||||||
|
*/
|
||||||
|
export interface Backend {
|
||||||
|
/** Bearer token / API key used for this backend. */
|
||||||
|
apiKey?: string;
|
||||||
|
/**
|
||||||
|
* API root the endpoint paths are appended to, e.g. https://api.example.com.
|
||||||
|
* Empty for the local backend, which resolves against the UI origin instead.
|
||||||
|
*/
|
||||||
|
baseUrl: string;
|
||||||
|
/** Chat completions path override, e.g. /v1/messages. */
|
||||||
|
chatPath?: string;
|
||||||
|
/** Wire quirks overriding the protocol defaults. */
|
||||||
|
compat?: Partial<BackendCompat>;
|
||||||
|
/** Disabled backends stay configured but are not queried. */
|
||||||
|
enabled: boolean;
|
||||||
|
/** Extra headers merged into every request to this backend. */
|
||||||
|
headers?: Record<string, string>;
|
||||||
|
/** Stable identity. The local backend id is reserved. */
|
||||||
|
id: string;
|
||||||
|
/** Models listing path override, e.g. /models. */
|
||||||
|
modelsPath?: string;
|
||||||
|
name: string;
|
||||||
|
protocol: BackendProtocol;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** A ready-made backend configuration offered when adding a backend. */
|
||||||
|
export interface BackendPreset {
|
||||||
|
/** Optional help text shown under the API key field. */
|
||||||
|
apiKeyHelp?: string;
|
||||||
|
baseUrl: string;
|
||||||
|
/** One line describing the endpoint, shown on the preset card. */
|
||||||
|
description?: string;
|
||||||
|
chatPath?: string;
|
||||||
|
/** Brand mark used in both themes, for logos that carry their own background. */
|
||||||
|
iconUrl?: string;
|
||||||
|
/** Brand mark for the dark theme. Preferred over `iconUrl` when paired with `iconUrlLight`. */
|
||||||
|
iconUrlDark?: string;
|
||||||
|
/** Brand mark for the light theme. Preferred over `iconUrl` when paired with `iconUrlDark`. */
|
||||||
|
iconUrlLight?: string;
|
||||||
|
/** Wire quirks this preset needs on top of the protocol defaults. */
|
||||||
|
compat?: Partial<BackendCompat>;
|
||||||
|
id: string;
|
||||||
|
modelsPath?: string;
|
||||||
|
name: string;
|
||||||
|
protocol: BackendProtocol;
|
||||||
|
}
|
||||||
@@ -25,6 +25,7 @@ export type {
|
|||||||
ApiChatCompletionToolCallDelta,
|
ApiChatCompletionToolCallDelta,
|
||||||
ApiChatCompletionToolCall,
|
ApiChatCompletionToolCall,
|
||||||
ApiChatCompletionStreamChunk,
|
ApiChatCompletionStreamChunk,
|
||||||
|
ApiChatCompletionUsage,
|
||||||
ApiChatCompletionResponse,
|
ApiChatCompletionResponse,
|
||||||
ApiSlotData,
|
ApiSlotData,
|
||||||
ApiProcessingState,
|
ApiProcessingState,
|
||||||
@@ -35,6 +36,15 @@ export type {
|
|||||||
ApiStreamSession
|
ApiStreamSession
|
||||||
} from './api';
|
} from './api';
|
||||||
|
|
||||||
|
// Backend types
|
||||||
|
export type {
|
||||||
|
Backend,
|
||||||
|
BackendCapabilities,
|
||||||
|
BackendCompat,
|
||||||
|
BackendPreset,
|
||||||
|
BackendProtocol
|
||||||
|
} from './backend';
|
||||||
|
|
||||||
// HuggingFace types
|
// HuggingFace types
|
||||||
export type {
|
export type {
|
||||||
HfCatalogBuild,
|
HfCatalogBuild,
|
||||||
|
|||||||
Vendored
+19
-12
@@ -16,6 +16,8 @@ export interface ModelOption {
|
|||||||
id: string;
|
id: string;
|
||||||
name: string;
|
name: string;
|
||||||
model: string;
|
model: string;
|
||||||
|
/** Backend that serves this model; set on the aggregated option list. */
|
||||||
|
backendId?: string;
|
||||||
description?: string;
|
description?: string;
|
||||||
capabilities: string[];
|
capabilities: string[];
|
||||||
/** Context size reported by the provider's model listing, when it reports one. */
|
/** Context size reported by the provider's model listing, when it reports one. */
|
||||||
@@ -25,6 +27,8 @@ export interface ModelOption {
|
|||||||
modalities?: ModelModalities;
|
modalities?: ModelModalities;
|
||||||
details?: ApiModelDetails['details'];
|
details?: ApiModelDetails['details'];
|
||||||
meta?: ApiModelDataEntry['meta'];
|
meta?: ApiModelDataEntry['meta'];
|
||||||
|
/** Load state the provider's own listing reports, when it reports one. */
|
||||||
|
status?: ApiModelDataEntry['status'];
|
||||||
parsedId?: ParsedModelId;
|
parsedId?: ParsedModelId;
|
||||||
aliases?: string[];
|
aliases?: string[];
|
||||||
tags?: string[];
|
tags?: string[];
|
||||||
@@ -58,6 +62,21 @@ export interface ModelDownloadEntry {
|
|||||||
repoWithTag: string;
|
repoWithTag: string;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A draft sidecar file a model listing reports as its own entry. The router lists a
|
||||||
|
* downloaded sidecar as a model, so this is what pairs it back with its model.
|
||||||
|
*/
|
||||||
|
export interface ModelSidecarFile {
|
||||||
|
id: string;
|
||||||
|
kind: ModelSidecar;
|
||||||
|
/** Repo the sidecar belongs to, e.g. `ggml-org/Qwen3.6-35B-A3B-GGUF`. */
|
||||||
|
model: string;
|
||||||
|
/** Parameter count the sidecar reports, e.g. `35B-A3B`. */
|
||||||
|
params: string | null;
|
||||||
|
/** Quantization of the sidecar file, e.g. `Q4_0`. */
|
||||||
|
quant: string | null;
|
||||||
|
}
|
||||||
|
|
||||||
export interface ParsedModelId {
|
export interface ParsedModelId {
|
||||||
raw: string;
|
raw: string;
|
||||||
orgName: string | null;
|
orgName: string | null;
|
||||||
@@ -84,15 +103,3 @@ export interface ModalityCapabilities {
|
|||||||
hasAudio: boolean;
|
hasAudio: boolean;
|
||||||
hasVideo: boolean;
|
hasVideo: boolean;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Sidecar file a listing reports as its own model entry, paired back to its model. */
|
|
||||||
export interface ModelSidecarFile {
|
|
||||||
id: string;
|
|
||||||
kind: ModelSidecar;
|
|
||||||
/** Repo the sidecar belongs to, e.g. `ggml-org/Qwen3.6-35B-A3B-GGUF`. */
|
|
||||||
model: string;
|
|
||||||
/** Parameter count the sidecar reports, e.g. `35B-A3B`. */
|
|
||||||
params: string | null;
|
|
||||||
/** Quantization of the sidecar file, e.g. `Q4_0`. */
|
|
||||||
quant: string | null;
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -0,0 +1,99 @@
|
|||||||
|
/**
|
||||||
|
* API base resolution for backends.
|
||||||
|
*
|
||||||
|
* The UI can talk to more than one backend endpoint. Services build request
|
||||||
|
* URLs through {@link apiUrl} so a request always targets the right backend.
|
||||||
|
* The backends store registers a resolver here; this module never imports the
|
||||||
|
* store, which keeps URL resolution free of store dependencies.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { backendChatUrl, backendModelsUrl } from './backend';
|
||||||
|
import { base } from '$app/paths';
|
||||||
|
import { API_ABSOLUTE_URL_PROTOCOLS, API_CHAT, API_MODELS } from '$lib/constants';
|
||||||
|
import type { Backend } from '$lib/types';
|
||||||
|
|
||||||
|
/** Backend list and active selection as exposed to URL resolution. */
|
||||||
|
export interface BackendsSnapshot {
|
||||||
|
activeId: string;
|
||||||
|
backends: Backend[];
|
||||||
|
}
|
||||||
|
|
||||||
|
type BackendsResolver = () => BackendsSnapshot;
|
||||||
|
|
||||||
|
let resolveBackends: BackendsResolver | null = null;
|
||||||
|
|
||||||
|
/** Registered once by the backends store. */
|
||||||
|
export function setBackendsResolver(resolver: BackendsResolver | null): void {
|
||||||
|
resolveBackends = resolver;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Look up a backend by id, defaulting to the active one.
|
||||||
|
*/
|
||||||
|
export function getBackend(backendId?: string): Backend | undefined {
|
||||||
|
const snapshot = resolveBackends?.();
|
||||||
|
|
||||||
|
if (!snapshot) return undefined;
|
||||||
|
|
||||||
|
const id = backendId ?? snapshot.activeId;
|
||||||
|
|
||||||
|
return snapshot.backends.find((backend) => backend.id === id);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** API root for a backend, or an empty string for the local backend. */
|
||||||
|
export function getBackendBaseUrl(backendId?: string): string {
|
||||||
|
return getBackend(backendId)?.baseUrl.trim() ?? '';
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Request target for a backend's chat completions endpoint. Returns a relative
|
||||||
|
* path for the local backend and an absolute URL for external ones, so callers
|
||||||
|
* can pass the result straight to `fetch` (or `apiFetch`, which resolves
|
||||||
|
* relative paths against the base).
|
||||||
|
*/
|
||||||
|
export function apiChatUrl(backendId?: string): string {
|
||||||
|
const backend = getBackend(backendId);
|
||||||
|
|
||||||
|
if (backend?.baseUrl.trim()) {
|
||||||
|
return backendChatUrl(backend);
|
||||||
|
}
|
||||||
|
|
||||||
|
return apiUrl(API_CHAT.COMPLETIONS, backendId);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Request target for a backend's models listing. Returns a plain path for the
|
||||||
|
* local backend (so `apiFetch` applies the base) and an absolute URL otherwise.
|
||||||
|
*/
|
||||||
|
export function apiModelsUrl(backendId?: string): string {
|
||||||
|
const backend = getBackend(backendId);
|
||||||
|
|
||||||
|
if (backend?.baseUrl.trim()) {
|
||||||
|
return backendModelsUrl(backend);
|
||||||
|
}
|
||||||
|
|
||||||
|
return API_MODELS.LIST;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Absolute URL for an API path on a backend.
|
||||||
|
*
|
||||||
|
* Absolute URLs pass through untouched. Paths on the local backend keep the
|
||||||
|
* existing base-path-relative form, so serving under a subpath still works.
|
||||||
|
* Paths on external backends resolve against the backend's API root.
|
||||||
|
*/
|
||||||
|
export function apiUrl(path: string, backendId?: string): string {
|
||||||
|
if (API_ABSOLUTE_URL_PROTOCOLS.some((protocol) => path.startsWith(protocol))) {
|
||||||
|
return path;
|
||||||
|
}
|
||||||
|
|
||||||
|
const baseUrl = getBackendBaseUrl(backendId);
|
||||||
|
|
||||||
|
if (!baseUrl) {
|
||||||
|
return `${base}${path}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
const root = baseUrl.endsWith('/') ? baseUrl : `${baseUrl}/`;
|
||||||
|
|
||||||
|
return new URL(path.replace(/^\.?\//, ''), root).toString();
|
||||||
|
}
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
|
import { apiUrl } from './api-base';
|
||||||
import { getAuthHeaders, getJsonHeaders } from './api-headers';
|
import { getAuthHeaders, getJsonHeaders } from './api-headers';
|
||||||
import { base } from '$app/paths';
|
import { ERROR_MESSAGES, HTTP_CODE_TO_STRING } from '$lib/constants';
|
||||||
import { API_ABSOLUTE_URL_PROTOCOLS, ERROR_MESSAGES, HTTP_CODE_TO_STRING } from '$lib/constants';
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* API Fetch Utilities
|
* API Fetch Utilities
|
||||||
@@ -32,6 +32,8 @@ export interface ApiFetchOptions extends Omit<RequestInit, 'headers'> {
|
|||||||
* Default: false (uses JSON headers with Content-Type: application/json)
|
* Default: false (uses JSON headers with Content-Type: application/json)
|
||||||
*/
|
*/
|
||||||
authOnly?: boolean;
|
authOnly?: boolean;
|
||||||
|
/** Backend to target; defaults to the active one. */
|
||||||
|
backendId?: string;
|
||||||
/**
|
/**
|
||||||
* Additional headers to merge with default headers.
|
* Additional headers to merge with default headers.
|
||||||
*/
|
*/
|
||||||
@@ -59,11 +61,10 @@ export interface ApiFetchOptions extends Omit<RequestInit, 'headers'> {
|
|||||||
* ```
|
* ```
|
||||||
*/
|
*/
|
||||||
export async function apiFetch<T>(path: string, options: ApiFetchOptions = {}): Promise<T> {
|
export async function apiFetch<T>(path: string, options: ApiFetchOptions = {}): Promise<T> {
|
||||||
const { authOnly = false, headers: customHeaders, ...fetchOptions } = options;
|
const { authOnly = false, backendId, headers: customHeaders, ...fetchOptions } = options;
|
||||||
const baseHeaders = authOnly ? getAuthHeaders() : getJsonHeaders();
|
const baseHeaders = authOnly ? getAuthHeaders(backendId) : getJsonHeaders(backendId);
|
||||||
const headers = { ...baseHeaders, ...customHeaders };
|
const headers = { ...baseHeaders, ...customHeaders };
|
||||||
// absolute URLs with an allowed protocol pass through untouched; relative paths get the base prefix
|
const url = apiUrl(path, backendId);
|
||||||
const url = API_ABSOLUTE_URL_PROTOCOLS.some((p) => path.startsWith(p)) ? path : `${base}${path}`;
|
|
||||||
|
|
||||||
let response;
|
let response;
|
||||||
|
|
||||||
@@ -106,7 +107,7 @@ export async function apiFetchWithParams<T>(
|
|||||||
params: Record<string, string>,
|
params: Record<string, string>,
|
||||||
options: ApiFetchOptions = {}
|
options: ApiFetchOptions = {}
|
||||||
): Promise<T> {
|
): Promise<T> {
|
||||||
const url = new URL(basePath, window.location.href);
|
const url = new URL(apiUrl(basePath, options.backendId), window.location.href);
|
||||||
|
|
||||||
for (const [key, value] of Object.entries(params)) {
|
for (const [key, value] of Object.entries(params)) {
|
||||||
if (value !== undefined && value !== null) {
|
if (value !== undefined && value !== null) {
|
||||||
|
|||||||
@@ -1,26 +1,42 @@
|
|||||||
|
import { getBackend } from './api-base';
|
||||||
import { redactValue } from './redact';
|
import { redactValue } from './redact';
|
||||||
import { CORS_PROXY, HEADERS } from '$lib/constants';
|
import { CORS_PROXY, HEADERS } from '$lib/constants';
|
||||||
import { MimeTypeApplication } from '$lib/enums';
|
import { MimeTypeApplication } from '$lib/enums';
|
||||||
|
import { getProtocolAdapter } from '$lib/services/protocols';
|
||||||
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
||||||
|
import type { Backend } from '$lib/types';
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Get authorization headers for API requests
|
* Get authorization headers for API requests to a backend.
|
||||||
* Includes Bearer token if API key is configured
|
|
||||||
*/
|
*/
|
||||||
export function getAuthHeaders(): Record<string, string> {
|
export function getAuthHeaders(backendId?: string): Record<string, string> {
|
||||||
const currentConfig = settingsStore.config;
|
const backend = getBackend(backendId);
|
||||||
const apiKey = currentConfig.apiKey?.toString().trim();
|
|
||||||
|
if (backend) return getAuthHeadersForBackend(backend);
|
||||||
|
|
||||||
|
// no backends resolver yet (early startup, or a non-browser call): keep the
|
||||||
|
// pre-backends behaviour and authenticate against the serving origin
|
||||||
|
const apiKey = settingsStore.config.apiKey?.toString().trim();
|
||||||
|
|
||||||
return apiKey ? { [HEADERS.AUTHORIZATION]: `${HEADERS.BEARER}${apiKey}` } : {};
|
return apiKey ? { [HEADERS.AUTHORIZATION]: `${HEADERS.BEARER}${apiKey}` } : {};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get authorization headers for a backend object, including one that is not
|
||||||
|
* registered yet (used by the connection test on the add-backend form).
|
||||||
|
* The protocol adapter owns the credential scheme and any required headers.
|
||||||
|
*/
|
||||||
|
export function getAuthHeadersForBackend(backend: Backend): Record<string, string> {
|
||||||
|
return getProtocolAdapter(backend).authHeaders(backend);
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Get standard JSON headers with optional authorization
|
* Get standard JSON headers with optional authorization
|
||||||
*/
|
*/
|
||||||
export function getJsonHeaders(): Record<string, string> {
|
export function getJsonHeaders(backendId?: string): Record<string, string> {
|
||||||
return {
|
return {
|
||||||
[HEADERS.CONTENT_TYPE]: MimeTypeApplication.JSON,
|
[HEADERS.CONTENT_TYPE]: MimeTypeApplication.JSON,
|
||||||
...getAuthHeaders()
|
...getAuthHeaders(backendId)
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,9 +1,10 @@
|
|||||||
import { error } from '@sveltejs/kit';
|
import { error } from '@sveltejs/kit';
|
||||||
import { browser } from '$app/environment';
|
import { browser } from '$app/environment';
|
||||||
import { base } from '$app/paths';
|
|
||||||
import { HEADERS } from '$lib/constants';
|
import { HEADERS } from '$lib/constants';
|
||||||
import { MimeTypeApplication } from '$lib/enums';
|
import { MimeTypeApplication } from '$lib/enums';
|
||||||
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
import { settingsStore } from '$lib/stores/settings/index.svelte';
|
||||||
|
import { apiUrl, getBackend } from '$lib/utils/api-base';
|
||||||
|
import { getBackendCapabilities } from '$lib/utils/backend';
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Validates API key by making a request to the server props endpoint
|
* Validates API key by making a request to the server props endpoint
|
||||||
@@ -14,6 +15,13 @@ export async function validateApiKey(fetch: typeof globalThis.fetch): Promise<vo
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// /props only exists on llama.cpp servers; external backends carry their own key
|
||||||
|
const backend = getBackend();
|
||||||
|
|
||||||
|
if (backend && !getBackendCapabilities(backend).props) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
const apiKey = settingsStore.config.apiKey;
|
const apiKey = settingsStore.config.apiKey;
|
||||||
|
|
||||||
try {
|
try {
|
||||||
@@ -28,7 +36,7 @@ export async function validateApiKey(fetch: typeof globalThis.fetch): Promise<vo
|
|||||||
headers[HEADERS.AUTHORIZATION] = `${HEADERS.BEARER}${apiKey}`;
|
headers[HEADERS.AUTHORIZATION] = `${HEADERS.BEARER}${apiKey}`;
|
||||||
}
|
}
|
||||||
|
|
||||||
const response = await fetch(`${base}/props`, { headers });
|
const response = await fetch(apiUrl('/props'), { headers });
|
||||||
|
|
||||||
if (!response.ok) {
|
if (!response.ok) {
|
||||||
if (response.status === 401 || response.status === 403) {
|
if (response.status === 401 || response.status === 403) {
|
||||||
|
|||||||
@@ -0,0 +1,243 @@
|
|||||||
|
/**
|
||||||
|
* Backend list parsing, defaults and endpoint URLs.
|
||||||
|
*
|
||||||
|
* External backends are persisted in settings as a JSON list. Malformed
|
||||||
|
* entries are dropped instead of throwing so a corrupted settings value can
|
||||||
|
* never break URL resolution.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import {
|
||||||
|
BACKEND_CAPABILITIES,
|
||||||
|
BACKEND_COMPAT,
|
||||||
|
BACKEND_ID_PREFIX,
|
||||||
|
BACKEND_PROTOCOLS,
|
||||||
|
DEFAULT_BACKEND_CHAT_PATH,
|
||||||
|
DEFAULT_BACKEND_MODELS_PATH,
|
||||||
|
FAVICON_SERVICE_URL,
|
||||||
|
LOCAL_BACKEND_ID,
|
||||||
|
MODEL_CONTEXT_LENGTH_FIELDS
|
||||||
|
} from '$lib/constants';
|
||||||
|
import type { Backend, BackendCapabilities, BackendCompat, BackendProtocol } from '$lib/types';
|
||||||
|
|
||||||
|
/** Absolute chat completions URL for a backend. */
|
||||||
|
export function backendChatUrl(backend: Backend): string {
|
||||||
|
return joinBackendUrl(backend.baseUrl, backend.chatPath ?? DEFAULT_BACKEND_CHAT_PATH);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Absolute models listing URL for a backend. */
|
||||||
|
export function backendModelsUrl(backend: Backend): string {
|
||||||
|
return joinBackendUrl(backend.baseUrl, backend.modelsPath ?? DEFAULT_BACKEND_MODELS_PATH);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Icon size requested from the favicon service, shown at 16px. */
|
||||||
|
const FAVICON_SIZE = 64;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Favicon of a backend's root domain, used when no bundled mark matches. API
|
||||||
|
* hosts rarely serve a favicon themselves, so the subdomain is dropped
|
||||||
|
* (`api.z.ai` -> `z.ai`) and the icon is requested through a favicon service.
|
||||||
|
* Returns null when the URL carries no usable host.
|
||||||
|
*/
|
||||||
|
export function backendFaviconUrl(baseUrl: string): string | null {
|
||||||
|
try {
|
||||||
|
const labels = new URL(baseUrl).hostname.split('.');
|
||||||
|
const root = labels.length > 2 ? labels.slice(-2).join('.') : labels.join('.');
|
||||||
|
|
||||||
|
return root ? `${FAVICON_SERVICE_URL}${root}&sz=${FAVICON_SIZE}` : null;
|
||||||
|
} catch {
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Features a backend supports, derived from its protocol. A missing backend
|
||||||
|
* (unknown model, early startup) gets the plain compatible defaults.
|
||||||
|
*/
|
||||||
|
export function getBackendCapabilities(backend?: Backend): BackendCapabilities {
|
||||||
|
return BACKEND_CAPABILITIES[backend?.protocol ?? 'openai'] ?? BACKEND_CAPABILITIES.openai;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Wire quirks for a backend: protocol defaults overridden by the backend. */
|
||||||
|
export function getBackendCompat(backend: Backend): BackendCompat {
|
||||||
|
return { ...(BACKEND_COMPAT[backend.protocol] ?? BACKEND_COMPAT.openai), ...backend.compat };
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The built-in backend pointing at the server that serves this UI. */
|
||||||
|
export function createLocalBackend(apiKey?: string, enabled = true): Backend {
|
||||||
|
return {
|
||||||
|
apiKey,
|
||||||
|
baseUrl: '',
|
||||||
|
enabled,
|
||||||
|
id: LOCAL_BACKEND_ID,
|
||||||
|
name: 'Local',
|
||||||
|
protocol: 'llama.cpp'
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Parse the persisted backends JSON into backend entries.
|
||||||
|
*/
|
||||||
|
export function parseBackendsSettings(rawBackends: unknown): Backend[] {
|
||||||
|
if (!rawBackends) return [];
|
||||||
|
|
||||||
|
let parsed: unknown;
|
||||||
|
|
||||||
|
if (typeof rawBackends === 'string') {
|
||||||
|
const trimmed = rawBackends.trim();
|
||||||
|
|
||||||
|
if (!trimmed) return [];
|
||||||
|
|
||||||
|
try {
|
||||||
|
parsed = JSON.parse(trimmed);
|
||||||
|
} catch (error) {
|
||||||
|
console.warn('[backends] Failed to parse backends JSON, ignoring value:', error);
|
||||||
|
|
||||||
|
return [];
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
parsed = rawBackends;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!Array.isArray(parsed)) return [];
|
||||||
|
|
||||||
|
return parsed.flatMap((entry, index) => {
|
||||||
|
const backend = parseBackendEntry(entry, index);
|
||||||
|
|
||||||
|
return backend ? [backend] : [];
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
function joinBackendUrl(baseUrl: string, path: string): string {
|
||||||
|
const base = baseUrl.replace(/\/+$/, '');
|
||||||
|
const suffix = path.startsWith('/') ? path : `/${path}`;
|
||||||
|
|
||||||
|
return `${base}${suffix}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
function parseBackendEntry(entry: unknown, index: number): Backend | null {
|
||||||
|
if (!entry || typeof entry !== 'object') return null;
|
||||||
|
|
||||||
|
const raw = entry as Record<string, unknown>;
|
||||||
|
const baseUrl = typeof raw.baseUrl === 'string' ? raw.baseUrl.trim() : '';
|
||||||
|
|
||||||
|
// the local backend is built in and never persisted
|
||||||
|
if (!baseUrl || raw.id === LOCAL_BACKEND_ID) return null;
|
||||||
|
|
||||||
|
const protocol = BACKEND_PROTOCOLS.includes(raw.protocol as BackendProtocol)
|
||||||
|
? (raw.protocol as BackendProtocol)
|
||||||
|
: 'openai';
|
||||||
|
const id =
|
||||||
|
typeof raw.id === 'string' && raw.id.trim()
|
||||||
|
? raw.id.trim()
|
||||||
|
: `${BACKEND_ID_PREFIX}-${index + 1}`;
|
||||||
|
const name = typeof raw.name === 'string' && raw.name.trim() ? raw.name.trim() : baseUrl;
|
||||||
|
const apiKey =
|
||||||
|
typeof raw.apiKey === 'string' && raw.apiKey.trim() ? raw.apiKey.trim() : undefined;
|
||||||
|
|
||||||
|
return {
|
||||||
|
apiKey,
|
||||||
|
baseUrl,
|
||||||
|
chatPath: parseOptionalPath(raw.chatPath),
|
||||||
|
compat: parseBackendCompat(raw.compat, protocol),
|
||||||
|
enabled: raw.enabled !== false,
|
||||||
|
headers: parseBackendHeaders(raw.headers),
|
||||||
|
id,
|
||||||
|
modelsPath: parseOptionalPath(raw.modelsPath),
|
||||||
|
name,
|
||||||
|
protocol
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// only keep the override keys the protocol understands, so a stale persisted
|
||||||
|
// value can never inject an unknown field into a request
|
||||||
|
function parseBackendCompat(
|
||||||
|
raw: unknown,
|
||||||
|
protocol: BackendProtocol
|
||||||
|
): Partial<BackendCompat> | undefined {
|
||||||
|
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) return undefined;
|
||||||
|
|
||||||
|
const entry = raw as Record<string, unknown>;
|
||||||
|
const defaults = BACKEND_COMPAT[protocol] ?? BACKEND_COMPAT.openai;
|
||||||
|
const overrides: Partial<BackendCompat> = {};
|
||||||
|
|
||||||
|
if (entry.maxTokensField === 'max_tokens' || entry.maxTokensField === 'max_completion_tokens') {
|
||||||
|
overrides.maxTokensField = entry.maxTokensField;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (typeof entry.supportsUsageInStreaming === 'boolean') {
|
||||||
|
overrides.supportsUsageInStreaming = entry.supportsUsageInStreaming;
|
||||||
|
}
|
||||||
|
|
||||||
|
// drop a no-op override so an unmodified backend stays undefined
|
||||||
|
const isDefault =
|
||||||
|
(overrides.maxTokensField === undefined ||
|
||||||
|
overrides.maxTokensField === defaults.maxTokensField) &&
|
||||||
|
(overrides.supportsUsageInStreaming === undefined ||
|
||||||
|
overrides.supportsUsageInStreaming === defaults.supportsUsageInStreaming);
|
||||||
|
|
||||||
|
return isDefault ? undefined : overrides;
|
||||||
|
}
|
||||||
|
|
||||||
|
function parseBackendHeaders(raw: unknown): Record<string, string> | undefined {
|
||||||
|
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) return undefined;
|
||||||
|
|
||||||
|
const entries = Object.entries(raw as Record<string, unknown>)
|
||||||
|
.filter(([, value]) => typeof value === 'string' && value.trim() !== '')
|
||||||
|
.map(([key, value]) => [key.trim(), (value as string).trim()] as const);
|
||||||
|
|
||||||
|
return entries.length > 0 ? Object.fromEntries(entries) : undefined;
|
||||||
|
}
|
||||||
|
|
||||||
|
function parseOptionalPath(raw: unknown): string | undefined {
|
||||||
|
return typeof raw === 'string' && raw.trim() ? raw.trim() : undefined;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Read a model's context size out of one `/v1/models` entry. Providers use
|
||||||
|
* different field names, and OpenRouter nests the authoritative value under
|
||||||
|
* `top_provider`, so try the flat fields first and the nested one after.
|
||||||
|
*/
|
||||||
|
export function readModelContextLength(value: unknown): number | undefined {
|
||||||
|
if (!value || typeof value !== 'object') return undefined;
|
||||||
|
|
||||||
|
const entry = value as Record<string, unknown>;
|
||||||
|
const flat = readContextField(entry);
|
||||||
|
|
||||||
|
if (flat !== undefined) return flat;
|
||||||
|
|
||||||
|
const topProvider = entry.top_provider;
|
||||||
|
|
||||||
|
if (topProvider && typeof topProvider === 'object') {
|
||||||
|
const nested = readContextField(topProvider as Record<string, unknown>);
|
||||||
|
|
||||||
|
if (nested !== undefined) return nested;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Hugging Face lists one entry per inference provider, and they disagree on
|
||||||
|
// the budget; take the largest so the gauge does not undersell the model.
|
||||||
|
const providers = entry.providers;
|
||||||
|
|
||||||
|
if (Array.isArray(providers)) {
|
||||||
|
const sizes = providers
|
||||||
|
.map((provider) =>
|
||||||
|
provider && typeof provider === 'object'
|
||||||
|
? readContextField(provider as Record<string, unknown>)
|
||||||
|
: undefined
|
||||||
|
)
|
||||||
|
.filter((size): size is number => size !== undefined);
|
||||||
|
|
||||||
|
if (sizes.length > 0) return Math.max(...sizes);
|
||||||
|
}
|
||||||
|
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
|
||||||
|
function readContextField(entry: Record<string, unknown>): number | undefined {
|
||||||
|
for (const field of MODEL_CONTEXT_LENGTH_FIELDS) {
|
||||||
|
const value = entry[field];
|
||||||
|
|
||||||
|
if (typeof value === 'number' && value > 0) return value;
|
||||||
|
}
|
||||||
|
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
@@ -2,7 +2,7 @@
|
|||||||
* CORS Proxy utility for routing requests through llama-server's CORS proxy.
|
* CORS Proxy utility for routing requests through llama-server's CORS proxy.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
import { base } from '$app/paths';
|
import { apiUrl } from './api-base';
|
||||||
import { CORS_PROXY, CORS_PROXY_ENDPOINT } from '$lib/constants';
|
import { CORS_PROXY, CORS_PROXY_ENDPOINT } from '$lib/constants';
|
||||||
import { UrlProtocol } from '$lib/enums';
|
import { UrlProtocol } from '$lib/enums';
|
||||||
|
|
||||||
@@ -12,7 +12,7 @@ import { UrlProtocol } from '$lib/enums';
|
|||||||
* @returns URL pointing to the CORS proxy with target encoded
|
* @returns URL pointing to the CORS proxy with target encoded
|
||||||
*/
|
*/
|
||||||
export function buildProxiedUrl(targetUrl: string): URL {
|
export function buildProxiedUrl(targetUrl: string): URL {
|
||||||
const proxyPath = `${base}${CORS_PROXY_ENDPOINT}`;
|
const proxyPath = apiUrl(CORS_PROXY_ENDPOINT);
|
||||||
const proxyUrl = new URL(proxyPath, window.location.origin);
|
const proxyUrl = new URL(proxyPath, window.location.origin);
|
||||||
|
|
||||||
proxyUrl.searchParams.set(CORS_PROXY.URL_PARAM, targetUrl);
|
proxyUrl.searchParams.set(CORS_PROXY.URL_PARAM, targetUrl);
|
||||||
|
|||||||
@@ -8,9 +8,30 @@
|
|||||||
*/
|
*/
|
||||||
|
|
||||||
// API utilities
|
// API utilities
|
||||||
export { getAuthHeaders, getJsonHeaders, sanitizeHeaders } from './api-headers';
|
export {
|
||||||
|
apiChatUrl,
|
||||||
|
apiModelsUrl,
|
||||||
|
apiUrl,
|
||||||
|
getBackend,
|
||||||
|
getBackendBaseUrl,
|
||||||
|
type BackendsSnapshot
|
||||||
|
} from './api-base';
|
||||||
|
export {
|
||||||
|
getAuthHeaders,
|
||||||
|
getAuthHeadersForBackend,
|
||||||
|
getJsonHeaders,
|
||||||
|
sanitizeHeaders
|
||||||
|
} from './api-headers';
|
||||||
export { ApiError, apiDelete, apiFetch, apiFetchWithParams, apiPost } from './api-fetch';
|
export { ApiError, apiDelete, apiFetch, apiFetchWithParams, apiPost } from './api-fetch';
|
||||||
export { validateApiKey } from './api-key-validation';
|
export { validateApiKey } from './api-key-validation';
|
||||||
|
export {
|
||||||
|
backendChatUrl,
|
||||||
|
backendFaviconUrl,
|
||||||
|
backendModelsUrl,
|
||||||
|
createLocalBackend,
|
||||||
|
getBackendCapabilities,
|
||||||
|
parseBackendsSettings
|
||||||
|
} from './backend';
|
||||||
|
|
||||||
// Attachment utilities
|
// Attachment utilities
|
||||||
export { getAttachmentDisplayItems, isMcpPrompt, isMcpResource } from './attachment-display';
|
export { getAttachmentDisplayItems, isMcpPrompt, isMcpResource } from './attachment-display';
|
||||||
@@ -118,6 +139,10 @@ export {
|
|||||||
// Model name utilities
|
// Model name utilities
|
||||||
export { isValidModelName, normalizeModelName, orgOf, repoOf } from './model-names';
|
export { isValidModelName, normalizeModelName, orgOf, repoOf } from './model-names';
|
||||||
|
|
||||||
|
// Backend-qualified model option ids
|
||||||
|
export { backendIdFromModelId, qualifyModelId, rawModelId } from './model-option-id';
|
||||||
|
export { groupModelFamilies, modelFamilyKey, type ModelFamilyGroup } from './model-families';
|
||||||
|
|
||||||
// Sidecar token utilities
|
// Sidecar token utilities
|
||||||
export { isAuxSidecar, isDraftSidecar, sidecarFromFileToken, sidecarFromTag } from './sidecars';
|
export { isAuxSidecar, isDraftSidecar, sidecarFromFileToken, sidecarFromTag } from './sidecars';
|
||||||
|
|
||||||
@@ -149,6 +174,8 @@ export { extractSseDataPayload, parseSseJsonStream, splitSseRecords } from './ss
|
|||||||
// Stream session identity (conversation-id based)
|
// Stream session identity (conversation-id based)
|
||||||
export { streamIdentity } from './stream-identity';
|
export { streamIdentity } from './stream-identity';
|
||||||
|
|
||||||
|
export { buildTimingsFromUsage, usageTokenCounts } from './timings';
|
||||||
|
|
||||||
// MCP utilities
|
// MCP utilities
|
||||||
export {
|
export {
|
||||||
detectMcpTransportFromUrl,
|
detectMcpTransportFromUrl,
|
||||||
@@ -371,7 +398,6 @@ export { remToPx } from './css';
|
|||||||
|
|
||||||
// Audio format helper (used by agentic store and chat service)
|
// Audio format helper (used by agentic store and chat service)
|
||||||
export { getAudioInputFormat } from './audio-format';
|
export { getAudioInputFormat } from './audio-format';
|
||||||
export { groupModelFamilies, modelFamilyKey, type ModelFamilyGroup } from './model-families';
|
|
||||||
|
|
||||||
// Svelte actions
|
// Svelte actions
|
||||||
export { nearViewport } from './near-viewport';
|
export { nearViewport } from './near-viewport';
|
||||||
|
|||||||
@@ -0,0 +1,28 @@
|
|||||||
|
/**
|
||||||
|
* Backend-qualified model option ids.
|
||||||
|
*
|
||||||
|
* The selector lists models from every enabled backend, so a bare model id can
|
||||||
|
* collide. Option ids are qualified as `<backendId>::<rawModelId>`; selection
|
||||||
|
* strips the prefix to find the row in the active backend's list.
|
||||||
|
*/
|
||||||
|
|
||||||
|
const MODEL_OPTION_ID_SEPARATOR = '::';
|
||||||
|
|
||||||
|
/** Prefix a raw model id with the backend that serves it. */
|
||||||
|
export function qualifyModelId(backendId: string, modelId: string): string {
|
||||||
|
return `${backendId}${MODEL_OPTION_ID_SEPARATOR}${modelId}`;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Backend id of a qualified model option id, or null when unqualified. */
|
||||||
|
export function backendIdFromModelId(qualifiedId: string): string | null {
|
||||||
|
const index = qualifiedId.indexOf(MODEL_OPTION_ID_SEPARATOR);
|
||||||
|
|
||||||
|
return index === -1 ? null : qualifiedId.slice(0, index);
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Strip the backend prefix from a qualified model option id. */
|
||||||
|
export function rawModelId(qualifiedId: string): string {
|
||||||
|
const index = qualifiedId.indexOf(MODEL_OPTION_ID_SEPARATOR);
|
||||||
|
|
||||||
|
return index === -1 ? qualifiedId : qualifiedId.slice(index + MODEL_OPTION_ID_SEPARATOR.length);
|
||||||
|
}
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
/**
|
||||||
|
* Client side timing fallback for backends that do not report their own.
|
||||||
|
*
|
||||||
|
* llama.cpp streams per-token timings; OpenAI-compatible servers
|
||||||
|
* do not. Token counts come from the usage block of the final chunk (or the
|
||||||
|
* count of streamed deltas as a fallback), times are measured locally: the wait
|
||||||
|
* for the first token is attributed to prompt processing, the rest to
|
||||||
|
* generation. Wall clock, so network and queueing are part of the numbers.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import type { ApiChatCompletionUsage } from '$lib/types/api';
|
||||||
|
import type { ChatMessageTimings } from '$lib/types/chat';
|
||||||
|
|
||||||
|
export interface StreamClock {
|
||||||
|
startedAt: number;
|
||||||
|
firstTokenAt: number | null;
|
||||||
|
lastTokenAt: number | null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Prompt/output/cache token counts. `promptTokens` excludes the cache read
|
||||||
|
* tokens, which are returned separately as `cacheTokens`, so the two always
|
||||||
|
* add up to the prompt size.
|
||||||
|
*/
|
||||||
|
export function usageTokenCounts(usage: ApiChatCompletionUsage | undefined): {
|
||||||
|
cacheTokens: number;
|
||||||
|
completionTokens: number;
|
||||||
|
promptTokens: number;
|
||||||
|
} {
|
||||||
|
// a total that includes the cache reads, which are reported separately
|
||||||
|
const cacheTokens =
|
||||||
|
usage?.prompt_tokens_details?.cached_tokens ??
|
||||||
|
usage?.prompt_cache_hit_tokens ??
|
||||||
|
usage?.cached_tokens ??
|
||||||
|
0;
|
||||||
|
const promptTotal = usage?.prompt_tokens ?? 0;
|
||||||
|
|
||||||
|
return {
|
||||||
|
cacheTokens,
|
||||||
|
completionTokens: usage?.completion_tokens ?? 0,
|
||||||
|
promptTokens: Math.max(0, promptTotal - cacheTokens)
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
export function buildTimingsFromUsage(
|
||||||
|
usage: ApiChatCompletionUsage | undefined,
|
||||||
|
clock: StreamClock,
|
||||||
|
fallbackTokens = 0
|
||||||
|
): ChatMessageTimings | null {
|
||||||
|
const { cacheTokens, completionTokens, promptTokens } = usageTokenCounts(usage);
|
||||||
|
const predictedN = completionTokens || fallbackTokens;
|
||||||
|
|
||||||
|
if (promptTokens === 0 && predictedN === 0) return null;
|
||||||
|
|
||||||
|
const { firstTokenAt, startedAt } = clock;
|
||||||
|
const lastTokenAt = clock.lastTokenAt ?? firstTokenAt;
|
||||||
|
|
||||||
|
return {
|
||||||
|
cache_n: cacheTokens,
|
||||||
|
// clamp so a one-token reply still reports a positive duration
|
||||||
|
predicted_ms: firstTokenAt && lastTokenAt ? Math.max(1, lastTokenAt - firstTokenAt) : undefined,
|
||||||
|
predicted_n: predictedN,
|
||||||
|
prompt_ms: firstTokenAt ? Math.max(1, firstTokenAt - startedAt) : undefined,
|
||||||
|
prompt_n: promptTokens
|
||||||
|
};
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user