mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-11 15:30:39 +02:00
A provider is a server entry with a base url, an optional key, a protocol and the paths its API lives at; the local llama.cpp server is the built-in one. Backends persist in settings and the active one is restored on load. Requests resolve against the active backend, model ids become backend-qualified, and every backend's model list is fetched and cached in the background. The manager's helpers learn to read a model's drafts, context and the provider that serves it. Assisted-by: pi:llama.cpp/DeepSeek-V4.1-Flash
67 lines
2.2 KiB
TypeScript
67 lines
2.2 KiB
TypeScript
/**
|
|
* Client side timing fallback for backends that do not report their own.
|
|
*
|
|
* llama.cpp streams per-token timings; OpenAI-compatible servers
|
|
* do not. Token counts come from the usage block of the final chunk (or the
|
|
* count of streamed deltas as a fallback), times are measured locally: the wait
|
|
* for the first token is attributed to prompt processing, the rest to
|
|
* generation. Wall clock, so network and queueing are part of the numbers.
|
|
*/
|
|
|
|
import type { ApiChatCompletionUsage } from '$lib/types/api';
|
|
import type { ChatMessageTimings } from '$lib/types/chat';
|
|
|
|
export interface StreamClock {
|
|
startedAt: number;
|
|
firstTokenAt: number | null;
|
|
lastTokenAt: number | null;
|
|
}
|
|
|
|
/**
|
|
* Prompt/output/cache token counts. `promptTokens` excludes the cache read
|
|
* tokens, which are returned separately as `cacheTokens`, so the two always
|
|
* add up to the prompt size.
|
|
*/
|
|
export function usageTokenCounts(usage: ApiChatCompletionUsage | undefined): {
|
|
cacheTokens: number;
|
|
completionTokens: number;
|
|
promptTokens: number;
|
|
} {
|
|
// a total that includes the cache reads, which are reported separately
|
|
const cacheTokens =
|
|
usage?.prompt_tokens_details?.cached_tokens ??
|
|
usage?.prompt_cache_hit_tokens ??
|
|
usage?.cached_tokens ??
|
|
0;
|
|
const promptTotal = usage?.prompt_tokens ?? 0;
|
|
|
|
return {
|
|
cacheTokens,
|
|
completionTokens: usage?.completion_tokens ?? 0,
|
|
promptTokens: Math.max(0, promptTotal - cacheTokens)
|
|
};
|
|
}
|
|
|
|
export function buildTimingsFromUsage(
|
|
usage: ApiChatCompletionUsage | undefined,
|
|
clock: StreamClock,
|
|
fallbackTokens = 0
|
|
): ChatMessageTimings | null {
|
|
const { cacheTokens, completionTokens, promptTokens } = usageTokenCounts(usage);
|
|
const predictedN = completionTokens || fallbackTokens;
|
|
|
|
if (promptTokens === 0 && predictedN === 0) return null;
|
|
|
|
const { firstTokenAt, startedAt } = clock;
|
|
const lastTokenAt = clock.lastTokenAt ?? firstTokenAt;
|
|
|
|
return {
|
|
cache_n: cacheTokens,
|
|
// clamp so a one-token reply still reports a positive duration
|
|
predicted_ms: firstTokenAt && lastTokenAt ? Math.max(1, lastTokenAt - firstTokenAt) : undefined,
|
|
predicted_n: predictedN,
|
|
prompt_ms: firstTokenAt ? Math.max(1, firstTokenAt - startedAt) : undefined,
|
|
prompt_n: promptTokens
|
|
};
|
|
}
|