mirror of
https://wget.la/https://github.com/leookun/cursor-byok
synced 2026-10-03 18:23:51 +08:00
feat: add first valid response timing and related metrics
- Introduced `first_valid_response_ms` and `ttfr_ms` to track the timing of the first valid response in LLM calls. - Updated relevant interfaces and components to display and utilize the new metrics, including CallDetails, CallTable, and LatencyChart. - Enhanced the database schema to accommodate the new timing fields. - Implemented logic in the service layer to record the first valid response during model interactions.
This commit is contained in:
@@ -86,7 +86,7 @@ export interface LegacyModelImportResult {
|
||||
|
||||
export interface ModelConnectivityResult {
|
||||
duration_ms: number;
|
||||
first_text_ms: number | null;
|
||||
first_valid_response_ms: number | null;
|
||||
output_tokens: number;
|
||||
tokens_per_second: number;
|
||||
tokens_estimated: boolean;
|
||||
@@ -197,6 +197,7 @@ export interface LlmCall {
|
||||
created_at_ms: number;
|
||||
ttfb_ms: number | null;
|
||||
ttft_ms: number | null;
|
||||
ttfr_ms: number | null;
|
||||
duration_ms: number | null;
|
||||
input_tokens: number | null;
|
||||
output_tokens: number | null;
|
||||
|
||||
@@ -32,6 +32,7 @@ export function CallDetails({ detail }: { detail: CallDetail }) {
|
||||
["Created At", `${call.created_at_ms} · ${new Date(call.created_at_ms).toLocaleString()}`],
|
||||
[t("耗时"), timing(call.duration_ms)],
|
||||
["TTFB", timing(call.ttfb_ms)],
|
||||
["TTFR", timing(call.ttfr_ms)],
|
||||
["TTFT", timing(call.ttft_ms)],
|
||||
["Input Token", show(call.input_tokens)],
|
||||
["Output Token", show(call.output_tokens)],
|
||||
|
||||
@@ -87,6 +87,11 @@ export function CallTable({ calls, onDetails }: { calls: LlmCall[]; onDetails: (
|
||||
header: "TTFB",
|
||||
render: (call) => milliseconds(call.ttfb_ms),
|
||||
},
|
||||
{
|
||||
key: "ttfr",
|
||||
header: "TTFR",
|
||||
render: (call) => milliseconds(call.ttfr_ms),
|
||||
},
|
||||
{
|
||||
key: "ttft",
|
||||
header: "TTFT",
|
||||
|
||||
@@ -6,13 +6,14 @@ export function LatencyChart({ calls }: { calls: LlmCall[] }) {
|
||||
const points = calls.slice(0, 20).reverse();
|
||||
const option: EChartsCoreOption = {
|
||||
animationDuration: 280,
|
||||
color: ["#d79a62", "#ad84cf"],
|
||||
color: ["#d79a62", "#ad84cf", "#72a8d8"],
|
||||
grid: { top: 22, right: 18, bottom: 30, left: 48 },
|
||||
legend: { top: 0, right: 8, textStyle: { color: "#999" } },
|
||||
tooltip: { trigger: "axis" },
|
||||
xAxis: { type: "category", data: points.map((call) => new Date(call.created_at_ms).toLocaleTimeString([], { hour: "2-digit", minute: "2-digit" })), axisLabel: { color: "#888" }, axisLine: { lineStyle: { color: "#5555" } } },
|
||||
yAxis: { type: "value", axisLabel: { color: "#888", formatter: "{value} ms" }, splitLine: { lineStyle: { color: "#8882" } } },
|
||||
series: [
|
||||
{ name: "TTFR", type: "line", smooth: true, symbol: "none", data: points.map((call) => call.ttfr_ms ?? 0) },
|
||||
{ name: "TTFT", type: "line", smooth: true, symbol: "none", data: points.map((call) => call.ttft_ms ?? 0) },
|
||||
{ name: t("总耗时"), type: "line", smooth: true, symbol: "none", data: points.map((call) => call.duration_ms ?? 0) },
|
||||
],
|
||||
|
||||
@@ -21,7 +21,7 @@ export function CursorModelTestResult({ state, testing = false }: { state?: Curs
|
||||
const detail = success
|
||||
? t("速度 {speed} tokens/s · 首字 {firstText} ms · 总耗时 {duration} ms · 输出 {tokens} tokens{estimated} · 返回:{output}", {
|
||||
speed: formatSpeed(state.result.tokens_per_second),
|
||||
firstText: state.result.first_text_ms ?? "--",
|
||||
firstText: state.result.first_valid_response_ms ?? "--",
|
||||
duration: state.result.duration_ms,
|
||||
tokens: state.result.output_tokens,
|
||||
estimated: state.result.tokens_estimated ? t("(估算)") : "",
|
||||
|
||||
@@ -55,6 +55,7 @@ const calls: LlmCall[] = Array.from({ length: 24 }, (_, index) => {
|
||||
finish_reason: failed ? null : "stop",
|
||||
created_at_ms: FIXED_NOW - index * 3 * 60_000,
|
||||
ttfb_ms: 210 + index * 13,
|
||||
ttfr_ms: 290 + index * 15,
|
||||
ttft_ms: 370 + index * 17,
|
||||
duration_ms: failed ? 812 : 1_420 + index * 71,
|
||||
input_tokens: 4_800 + index * 337,
|
||||
@@ -116,7 +117,7 @@ export function installDemoApi() {
|
||||
}
|
||||
if (path === "/models/import-v0049") return json({ imported: 0, skipped: 0, total: 0 });
|
||||
if (/^\/models\/[^/]+\/test\/[^/]+$/.test(path) && method === "POST") {
|
||||
return json({ duration_ms: 1_284, first_text_ms: 418, output_tokens: 42, tokens_per_second: 38.6, tokens_estimated: false, output: "Mock connectivity test passed." });
|
||||
return json({ duration_ms: 1_284, first_valid_response_ms: 418, output_tokens: 42, tokens_per_second: 38.6, tokens_estimated: false, output: "Mock connectivity test passed." });
|
||||
}
|
||||
if (/^\/models\/[^/]+\/test\/[^/]+$/.test(path) || /^\/models\/[^/]+$/.test(path)) {
|
||||
return method === "DELETE" ? empty() : json(models[0]);
|
||||
|
||||
Reference in New Issue
Block a user