feat: add first valid response timing and related metrics

- Introduced `first_valid_response_ms` and `ttfr_ms` to track the timing of the first valid response in LLM calls.
- Updated relevant interfaces and components to display and utilize the new metrics, including CallDetails, CallTable, and LatencyChart.
- Enhanced the database schema to accommodate the new timing fields.
- Implemented logic in the service layer to record the first valid response during model interactions.
This commit is contained in:
leookun
2026-08-28 22:56:39 +08:00
parent 5a0bc2e0e9
commit 5cc2401ce2
20 changed files with 461 additions and 52 deletions
+2 -1
View File
@@ -86,7 +86,7 @@ export interface LegacyModelImportResult {
export interface ModelConnectivityResult {
duration_ms: number;
first_text_ms: number | null;
first_valid_response_ms: number | null;
output_tokens: number;
tokens_per_second: number;
tokens_estimated: boolean;
@@ -197,6 +197,7 @@ export interface LlmCall {
created_at_ms: number;
ttfb_ms: number | null;
ttft_ms: number | null;
ttfr_ms: number | null;
duration_ms: number | null;
input_tokens: number | null;
output_tokens: number | null;
@@ -32,6 +32,7 @@ export function CallDetails({ detail }: { detail: CallDetail }) {
["Created At", `${call.created_at_ms} · ${new Date(call.created_at_ms).toLocaleString()}`],
[t("耗时"), timing(call.duration_ms)],
["TTFB", timing(call.ttfb_ms)],
["TTFR", timing(call.ttfr_ms)],
["TTFT", timing(call.ttft_ms)],
["Input Token", show(call.input_tokens)],
["Output Token", show(call.output_tokens)],
@@ -87,6 +87,11 @@ export function CallTable({ calls, onDetails }: { calls: LlmCall[]; onDetails: (
header: "TTFB",
render: (call) => milliseconds(call.ttfb_ms),
},
{
key: "ttfr",
header: "TTFR",
render: (call) => milliseconds(call.ttfr_ms),
},
{
key: "ttft",
header: "TTFT",
@@ -6,13 +6,14 @@ export function LatencyChart({ calls }: { calls: LlmCall[] }) {
const points = calls.slice(0, 20).reverse();
const option: EChartsCoreOption = {
animationDuration: 280,
color: ["#d79a62", "#ad84cf"],
color: ["#d79a62", "#ad84cf", "#72a8d8"],
grid: { top: 22, right: 18, bottom: 30, left: 48 },
legend: { top: 0, right: 8, textStyle: { color: "#999" } },
tooltip: { trigger: "axis" },
xAxis: { type: "category", data: points.map((call) => new Date(call.created_at_ms).toLocaleTimeString([], { hour: "2-digit", minute: "2-digit" })), axisLabel: { color: "#888" }, axisLine: { lineStyle: { color: "#5555" } } },
yAxis: { type: "value", axisLabel: { color: "#888", formatter: "{value} ms" }, splitLine: { lineStyle: { color: "#8882" } } },
series: [
{ name: "TTFR", type: "line", smooth: true, symbol: "none", data: points.map((call) => call.ttfr_ms ?? 0) },
{ name: "TTFT", type: "line", smooth: true, symbol: "none", data: points.map((call) => call.ttft_ms ?? 0) },
{ name: t("总耗时"), type: "line", smooth: true, symbol: "none", data: points.map((call) => call.duration_ms ?? 0) },
],
@@ -21,7 +21,7 @@ export function CursorModelTestResult({ state, testing = false }: { state?: Curs
const detail = success
? t("速度 {speed} tokens/s · 首字 {firstText} ms · 总耗时 {duration} ms · 输出 {tokens} tokens{estimated} · 返回:{output}", {
speed: formatSpeed(state.result.tokens_per_second),
firstText: state.result.first_text_ms ?? "--",
firstText: state.result.first_valid_response_ms ?? "--",
duration: state.result.duration_ms,
tokens: state.result.output_tokens,
estimated: state.result.tokens_estimated ? t("(估算)") : "",
+2 -1
View File
@@ -55,6 +55,7 @@ const calls: LlmCall[] = Array.from({ length: 24 }, (_, index) => {
finish_reason: failed ? null : "stop",
created_at_ms: FIXED_NOW - index * 3 * 60_000,
ttfb_ms: 210 + index * 13,
ttfr_ms: 290 + index * 15,
ttft_ms: 370 + index * 17,
duration_ms: failed ? 812 : 1_420 + index * 71,
input_tokens: 4_800 + index * 337,
@@ -116,7 +117,7 @@ export function installDemoApi() {
}
if (path === "/models/import-v0049") return json({ imported: 0, skipped: 0, total: 0 });
if (/^\/models\/[^/]+\/test\/[^/]+$/.test(path) && method === "POST") {
return json({ duration_ms: 1_284, first_text_ms: 418, output_tokens: 42, tokens_per_second: 38.6, tokens_estimated: false, output: "Mock connectivity test passed." });
return json({ duration_ms: 1_284, first_valid_response_ms: 418, output_tokens: 42, tokens_per_second: 38.6, tokens_estimated: false, output: "Mock connectivity test passed." });
}
if (/^\/models\/[^/]+\/test\/[^/]+$/.test(path) || /^\/models\/[^/]+$/.test(path)) {
return method === "DELETE" ? empty() : json(models[0]);