1. AI 聊天機器人的可觀察性堆疊
┌─────────── OBSERVABILITY ARCHITECTURE ───────────────┐
│ │
│ ┌──────────┐ ┌──────────┐ ┌──────────┐ │
│ │ Traces │ │ Metrics │ │ Logs │ 3 Pillars │
│ │(LangFuse)│ │(Prometheus│ │ (ELK / │ │
│ │ │ │ /Grafana) │ │ Loki) │ │
│ └────┬─────┘ └────┬─────┘ └────┬─────┘ │
│ │ │ │ │
│ └──────────────┼─────────────┘ │
│ │ │
│ ┌───────▼──────┐ │
│ │ Analytics │ │
│ │ Dashboard │ │
│ │ (Grafana / │ │
│ │ Custom) │ │
│ └──────────────┘ │
└───────────────────────────────────────────────────────┘
2. 對話分析-關鍵指標
| 公制 | 描述 | 目標 |
|---|---|---|
| 解決率 | % 無需人工解決的對話 | >70% |
| 電腦輔助能力測試 | 客戶滿意度(贊成/反對) | >4.0/5.0 |
| 平均匝數 | 每次對話的平均訊息數 | <8 |
| 首次回應時間 | 第一個令牌的時間 (TTFT) | <500ms |
| 升級率 | % 轉移給人工代理 | <30% |
| 遏制率 | % 完全由機器人處理 | >60% |
| 每次對話成本 | LLM + 每次轉換的基礎設施總成本。 | <$0.05 |
| 幻覺率 | 沒有支持的主張的答覆百分比 | <5% |
class ConversationAnalytics {
async recordConversationMetrics(
conversation: CompletedConversation,
): Promise<void> {
const metrics: ConversationMetric = {
tenantId: conversation.tenantId,
conversationId: conversation.id,
timestamp: new Date(),
// Conversation metrics
totalTurns: conversation.messages.filter(m => m.role === 'user').length,
duration: conversation.endedAt.getTime() - conversation.startedAt.getTime(),
resolved: conversation.status === 'resolved',
escalated: conversation.status === 'escalated',
// Satisfaction
userRating: conversation.feedback?.rating,
// LLM metrics (aggregated)
totalInputTokens: this.sumTokens(conversation, 'input'),
totalOutputTokens: this.sumTokens(conversation, 'output'),
totalCost: this.calculateCost(conversation),
avgLatency: this.avgLatency(conversation),
// Quality metrics
toolCallCount: this.countToolCalls(conversation),
ragRetrievalCount: this.countRagRetrievals(conversation),
guardrailTriggered: conversation.guardrailEvents?.length ?? 0,
// Classification
intent: conversation.detectedIntent,
sentiment: conversation.avgSentiment,
agentsUsed: conversation.agentHistory,
};
// Store in ClickHouse for fast OLAP queries
await this.clickhouse.insert('conversation_metrics', metrics);
// Also emit to Prometheus for real-time dashboards
this.prometheus.conversationDuration.observe(
{ tenant: conversation.tenantId, status: conversation.status },
metrics.duration / 1000,
);
this.prometheus.conversationCost.observe(
{ tenant: conversation.tenantId },
metrics.totalCost,
);
}
}
3. LLM 呼叫追蹤 — Langfuse 集成
class LLMTracer {
private langfuse: Langfuse;
async traceCompletion(
params: TracedCompletionParams,
): Promise<TracedCompletion> {
const trace = this.langfuse.trace({
name: 'chat-completion',
userId: params.userId,
metadata: {
tenantId: params.tenantId,
conversationId: params.conversationId,
agentId: params.agentId,
},
});
// Span for each pipeline stage
const ragSpan = trace.span({ name: 'rag-retrieval' });
const ragResults = await this.rag.retrieve(params.query);
ragSpan.end({
output: { documentCount: ragResults.length },
metadata: { topScore: ragResults[0]?.score },
});
const guardrailSpan = trace.span({ name: 'input-guardrail' });
const guardrailResult = await this.guardrails.checkInput(params.query);
guardrailSpan.end({
output: { passed: !guardrailResult.blocked },
});
const llmSpan = trace.span({ name: 'llm-generation' });
const startTime = Date.now();
const generation = trace.generation({
name: 'chat',
model: params.model,
input: params.messages,
modelParameters: { temperature: params.temperature },
});
const response = await this.llm.chat(params);
generation.end({
output: response.content,
usage: {
promptTokens: response.usage.promptTokens,
completionTokens: response.usage.completionTokens,
totalTokens: response.usage.totalTokens,
},
});
llmSpan.end({ metadata: { latencyMs: Date.now() - startTime } });
// Score the trace
trace.score({
name: 'cost',
value: this.calculateCost(params.model, response.usage),
});
return { response, traceId: trace.id };
}
}
4. 成本追蹤與最佳化
class CostTracker {
private modelPricing: Record<string, { input: number; output: number }> = {
'gpt-4o': { input: 2.50, output: 10.00 }, // per 1M tokens
'gpt-4o-mini': { input: 0.15, output: 0.60 },
'claude-3.5-sonnet': { input: 3.00, output: 15.00 },
'claude-3.5-haiku': { input: 0.80, output: 4.00 },
};
calculateCost(model: string, usage: TokenUsage): number {
const pricing = this.modelPricing[model];
if (!pricing) return 0;
return (
(usage.promptTokens / 1_000_000) * pricing.input +
(usage.completionTokens / 1_000_000) * pricing.output
);
}
async getDailyCostReport(tenantId: string, date: string): Promise<CostReport> {
const costs = await this.clickhouse.query(`
SELECT
model,
COUNT(*) as total_calls,
SUM(input_tokens) as total_input_tokens,
SUM(output_tokens) as total_output_tokens,
SUM(cost_usd) as total_cost
FROM llm_calls
WHERE tenant_id = {tenantId:String}
AND toDate(timestamp) = {date:Date}
GROUP BY model
ORDER BY total_cost DESC
`, { tenantId, date });
return {
date,
tenantId,
byModel: costs,
totalCost: costs.reduce((sum, c) => sum + c.total_cost, 0),
recommendations: this.generateCostRecommendations(costs),
};
}
private generateCostRecommendations(costs: ModelCost[]): string[] {
const recommendations: string[] = [];
for (const cost of costs) {
// Recommend cheaper model for simple tasks
if (cost.model === 'gpt-4o' && cost.total_calls > 1000) {
const potentialSaving = cost.total_cost * 0.85;
recommendations.push(
`Consider routing simple queries to gpt-4o-mini. Potential saving: $${potentialSaving.toFixed(2)}/day`,
);
}
}
return recommendations;
}
}
5. 警報和異常檢測
class AlertManager {
private rules: AlertRule[] = [
{
name: 'high_error_rate',
condition: (metrics) => metrics.errorRate > 0.05,
severity: 'critical',
message: 'Error rate exceeded 5%',
},
{
name: 'high_latency',
condition: (metrics) => metrics.p95LatencyMs > 5000,
severity: 'warning',
message: 'P95 latency exceeded 5s',
},
{
name: 'high_cost',
condition: (metrics) => metrics.dailyCostUsd > metrics.budget * 0.8,
severity: 'warning',
message: 'Daily cost approaching budget (80%)',
},
{
name: 'low_resolution_rate',
condition: (metrics) => metrics.resolutionRate < 0.5,
severity: 'warning',
message: 'Resolution rate below 50%',
},
{
name: 'hallucination_spike',
condition: (metrics) => metrics.hallucinationRate > 0.1,
severity: 'critical',
message: 'Hallucination rate exceeded 10%',
},
];
async evaluate(tenantId: string): Promise<Alert[]> {
const metrics = await this.getRecentMetrics(tenantId);
const alerts: Alert[] = [];
for (const rule of this.rules) {
if (rule.condition(metrics)) {
alerts.push({
tenantId,
rule: rule.name,
severity: rule.severity,
message: rule.message,
metrics: this.getRelevantMetrics(metrics, rule),
timestamp: new Date(),
});
}
}
// Send notifications
for (const alert of alerts) {
await this.notify(alert);
}
return alerts;
}
}
第 15 課總結
- 3 支柱:痕跡(Langfuse)、指標(Prometheus/Grafana)、日誌(ELK/Loki)
- 關鍵指標:解決率、CSAT、成本/對話、幻覺率、延遲
- 法學碩士追踪:Langfuse 追蹤每個呼叫的跨度(RAG、guardrail、LLM)+ 成本評分
- 成本追蹤:每個模型的成本、每日報告、最佳化建議
- 警報:錯誤率、延遲高峰、成本超支、品質下降的自動警報
下一篇: 多通路整合 — 全通路閘道、Facebook Messenger、Zalo OA、Web Widget、LINE、Slack、Teams。