1. AIチャットボットの可観測性スタック
┌─────────── OBSERVABILITY ARCHITECTURE ───────────────┐
│ │
│ ┌──────────┐ ┌──────────┐ ┌──────────┐ │
│ │ Traces │ │ Metrics │ │ Logs │ 3 Pillars │
│ │(LangFuse)│ │(Prometheus│ │ (ELK / │ │
│ │ │ │ /Grafana) │ │ Loki) │ │
│ └────┬─────┘ └────┬─────┘ └────┬─────┘ │
│ │ │ │ │
│ └──────────────┼─────────────┘ │
│ │ │
│ ┌───────▼──────┐ │
│ │ Analytics │ │
│ │ Dashboard │ │
│ │ (Grafana / │ │
│ │ Custom) │ │
│ └──────────────┘ │
└───────────────────────────────────────────────────────┘
2. 会話分析 — 主要な指標
| メトリック | 説明 | ターゲット |
|---|---|---|
| 解像度レート | 人間なしで解決された会話の割合 | >70% |
| CSAT | 顧客満足度 (高評価/低評価) | >4.0/5.0 |
| 平均回転数 | 会話ごとの平均メッセージ | <8 |
| 最初の応答時間 | 最初のトークンまでの時間 (TTFT) | <500ms |
| エスカレーション率 | 人間のエージェントに転送された割合 | <30% |
| 封じ込め率 | % は完全にボットによって処理されます | >60% |
| 会話あたりのコスト | LLM とコンバージョンあたりのインフラ費用の合計 | <$0.05 |
| 幻覚率 | サポートされていない主張を含む回答の割合 | <5% |
class ConversationAnalytics {
async recordConversationMetrics(
conversation: CompletedConversation,
): Promise<void> {
const metrics: ConversationMetric = {
tenantId: conversation.tenantId,
conversationId: conversation.id,
timestamp: new Date(),
// Conversation metrics
totalTurns: conversation.messages.filter(m => m.role === 'user').length,
duration: conversation.endedAt.getTime() - conversation.startedAt.getTime(),
resolved: conversation.status === 'resolved',
escalated: conversation.status === 'escalated',
// Satisfaction
userRating: conversation.feedback?.rating,
// LLM metrics (aggregated)
totalInputTokens: this.sumTokens(conversation, 'input'),
totalOutputTokens: this.sumTokens(conversation, 'output'),
totalCost: this.calculateCost(conversation),
avgLatency: this.avgLatency(conversation),
// Quality metrics
toolCallCount: this.countToolCalls(conversation),
ragRetrievalCount: this.countRagRetrievals(conversation),
guardrailTriggered: conversation.guardrailEvents?.length ?? 0,
// Classification
intent: conversation.detectedIntent,
sentiment: conversation.avgSentiment,
agentsUsed: conversation.agentHistory,
};
// Store in ClickHouse for fast OLAP queries
await this.clickhouse.insert('conversation_metrics', metrics);
// Also emit to Prometheus for real-time dashboards
this.prometheus.conversationDuration.observe(
{ tenant: conversation.tenantId, status: conversation.status },
metrics.duration / 1000,
);
this.prometheus.conversationCost.observe(
{ tenant: conversation.tenantId },
metrics.totalCost,
);
}
}
3. LLM コール トレーシング — Langfuse の統合
class LLMTracer {
private langfuse: Langfuse;
async traceCompletion(
params: TracedCompletionParams,
): Promise<TracedCompletion> {
const trace = this.langfuse.trace({
name: 'chat-completion',
userId: params.userId,
metadata: {
tenantId: params.tenantId,
conversationId: params.conversationId,
agentId: params.agentId,
},
});
// Span for each pipeline stage
const ragSpan = trace.span({ name: 'rag-retrieval' });
const ragResults = await this.rag.retrieve(params.query);
ragSpan.end({
output: { documentCount: ragResults.length },
metadata: { topScore: ragResults[0]?.score },
});
const guardrailSpan = trace.span({ name: 'input-guardrail' });
const guardrailResult = await this.guardrails.checkInput(params.query);
guardrailSpan.end({
output: { passed: !guardrailResult.blocked },
});
const llmSpan = trace.span({ name: 'llm-generation' });
const startTime = Date.now();
const generation = trace.generation({
name: 'chat',
model: params.model,
input: params.messages,
modelParameters: { temperature: params.temperature },
});
const response = await this.llm.chat(params);
generation.end({
output: response.content,
usage: {
promptTokens: response.usage.promptTokens,
completionTokens: response.usage.completionTokens,
totalTokens: response.usage.totalTokens,
},
});
llmSpan.end({ metadata: { latencyMs: Date.now() - startTime } });
// Score the trace
trace.score({
name: 'cost',
value: this.calculateCost(params.model, response.usage),
});
return { response, traceId: trace.id };
}
}
4. コストの追跡と最適化
class CostTracker {
private modelPricing: Record<string, { input: number; output: number }> = {
'gpt-4o': { input: 2.50, output: 10.00 }, // per 1M tokens
'gpt-4o-mini': { input: 0.15, output: 0.60 },
'claude-3.5-sonnet': { input: 3.00, output: 15.00 },
'claude-3.5-haiku': { input: 0.80, output: 4.00 },
};
calculateCost(model: string, usage: TokenUsage): number {
const pricing = this.modelPricing[model];
if (!pricing) return 0;
return (
(usage.promptTokens / 1_000_000) * pricing.input +
(usage.completionTokens / 1_000_000) * pricing.output
);
}
async getDailyCostReport(tenantId: string, date: string): Promise<CostReport> {
const costs = await this.clickhouse.query(`
SELECT
model,
COUNT(*) as total_calls,
SUM(input_tokens) as total_input_tokens,
SUM(output_tokens) as total_output_tokens,
SUM(cost_usd) as total_cost
FROM llm_calls
WHERE tenant_id = {tenantId:String}
AND toDate(timestamp) = {date:Date}
GROUP BY model
ORDER BY total_cost DESC
`, { tenantId, date });
return {
date,
tenantId,
byModel: costs,
totalCost: costs.reduce((sum, c) => sum + c.total_cost, 0),
recommendations: this.generateCostRecommendations(costs),
};
}
private generateCostRecommendations(costs: ModelCost[]): string[] {
const recommendations: string[] = [];
for (const cost of costs) {
// Recommend cheaper model for simple tasks
if (cost.model === 'gpt-4o' && cost.total_calls > 1000) {
const potentialSaving = cost.total_cost * 0.85;
recommendations.push(
`Consider routing simple queries to gpt-4o-mini. Potential saving: $${potentialSaving.toFixed(2)}/day`,
);
}
}
return recommendations;
}
}
5. アラートと異常検出
class AlertManager {
private rules: AlertRule[] = [
{
name: 'high_error_rate',
condition: (metrics) => metrics.errorRate > 0.05,
severity: 'critical',
message: 'Error rate exceeded 5%',
},
{
name: 'high_latency',
condition: (metrics) => metrics.p95LatencyMs > 5000,
severity: 'warning',
message: 'P95 latency exceeded 5s',
},
{
name: 'high_cost',
condition: (metrics) => metrics.dailyCostUsd > metrics.budget * 0.8,
severity: 'warning',
message: 'Daily cost approaching budget (80%)',
},
{
name: 'low_resolution_rate',
condition: (metrics) => metrics.resolutionRate < 0.5,
severity: 'warning',
message: 'Resolution rate below 50%',
},
{
name: 'hallucination_spike',
condition: (metrics) => metrics.hallucinationRate > 0.1,
severity: 'critical',
message: 'Hallucination rate exceeded 10%',
},
];
async evaluate(tenantId: string): Promise<Alert[]> {
const metrics = await this.getRecentMetrics(tenantId);
const alerts: Alert[] = [];
for (const rule of this.rules) {
if (rule.condition(metrics)) {
alerts.push({
tenantId,
rule: rule.name,
severity: rule.severity,
message: rule.message,
metrics: this.getRelevantMetrics(metrics, rule),
timestamp: new Date(),
});
}
}
// Send notifications
for (const alert of alerts) {
await this.notify(alert);
}
return alerts;
}
}
レッスン 15 のまとめ
- 3本の柱: トレース (Langfuse)、メトリクス (Prometheus/Grafana)、ログ (ELK/Loki)
- 主要な指標:解決率、CSAT、コスト/会話、幻覚率、待ち時間
- LLM トレース: スパン (RAG、ガードレール、LLM) + コスト スコアリングを使用したコールごとの Langfuse トレース
- コストの追跡: モデルごとのコスト、日次レポート、最適化の推奨事項
- 警告中: エラー率、遅延のスパイク、コストの超過、品質の低下に関する自動アラート
次の記事: マルチチャネル統合 — オムニチャネル ゲートウェイ、Facebook メッセンジャー、Zalo OA、Web ウィジェット、LINE、Slack、Teams。