1. 人工切換架構
沒有一個聊天機器人能夠達到 100% 的解決率。企業需求 無縫切換 從機器人→人類代理,與 完整的上下文傳輸 這樣代理就不必從頭開始再詢問。
┌───────── HANDOFF FLOW ─────────────────────────────────┐
│ │
│ ┌────────┐ ┌───────────┐ ┌──────────────────┐ │
│ │ User │───▶│ AI Bot │───▶│ Escalation │ │
│ │ │ │ │ │ Decision Engine │ │
│ └────────┘ └───────────┘ └────────┬─────────┘ │
│ │ │
│ ┌───────────▼──────────┐ │
│ │ HANDOFF MANAGER │ │
│ │ • Context packaging │ │
│ │ • Agent matching │ │
│ │ • Queue management │ │
│ └───────────┬──────────┘ │
│ │ │
│ ┌────────┐ ┌───────────┐ ┌────────▼─────────┐ │
│ │ User │◀──▶│ Human │◀───│ Agent Workspace │ │
│ │ │ │ Agent │ │ (AI-assisted) │ │
│ └────────┘ └───────────┘ └──────────────────┘ │
└─────────────────────────────────────────────────────────┘
2.升級決策引擎
interface EscalationTrigger {
type: 'explicit' | 'sentiment' | 'confidence' | 'topic' | 'loop' | 'vip';
check: (context: ConversationContext) => EscalationDecision;
}
class EscalationEngine {
private triggers: EscalationTrigger[] = [
// User explicitly asks for human
{
type: 'explicit',
check: (ctx) => {
const keywords = ['nói chuyện với người', 'gặp nhân viên', 'hỗ trợ viên',
'talk to agent', 'human agent', 'real person'];
const lastMessage = ctx.lastUserMessage.toLowerCase();
const triggered = keywords.some(kw => lastMessage.includes(kw));
return { shouldEscalate: triggered, reason: 'User requested human agent' };
},
},
// Negative sentiment detected
{
type: 'sentiment',
check: (ctx) => {
const triggered = ctx.sentimentScore < -0.6;
return {
shouldEscalate: triggered,
reason: `Negative sentiment: ${ctx.sentimentScore.toFixed(2)}`,
priority: 'high',
};
},
},
// Low AI confidence
{
type: 'confidence',
check: (ctx) => {
const triggered = ctx.lastConfidenceScore < 0.3;
return {
shouldEscalate: triggered,
reason: `Low confidence: ${ctx.lastConfidenceScore.toFixed(2)}`,
};
},
},
// Restricted topic
{
type: 'topic',
check: (ctx) => {
const escalationTopics = ['complaint', 'refund', 'legal', 'billing_dispute'];
const triggered = escalationTopics.includes(ctx.detectedIntent);
return {
shouldEscalate: triggered,
reason: `Sensitive topic: ${ctx.detectedIntent}`,
priority: 'high',
targetSkill: ctx.detectedIntent,
};
},
},
// Conversation loop (bot can't help)
{
type: 'loop',
check: (ctx) => {
const triggered = ctx.turnCount > 8 && !ctx.isProgressing;
return {
shouldEscalate: triggered,
reason: 'Conversation not progressing after 8 turns',
};
},
},
// VIP customer
{
type: 'vip',
check: (ctx) => {
const triggered = ctx.customerSegment === 'vip' || ctx.customerSegment === 'enterprise';
return {
shouldEscalate: triggered,
reason: 'VIP customer — prioritize human support',
priority: 'urgent',
};
},
},
];
evaluate(context: ConversationContext): EscalationDecision {
for (const trigger of this.triggers) {
const decision = trigger.check(context);
if (decision.shouldEscalate) {
return decision;
}
}
return { shouldEscalate: false };
}
}
3. Handoff Manager — 上下文打包和代理匹配
class HandoffManager {
async initiateHandoff(
conversationId: string,
decision: EscalationDecision,
): Promise<HandoffResult> {
const conversation = await this.conversationService.get(conversationId);
// 1. Package context for human agent
const handoffContext = await this.packageContext(conversation);
// 2. Find best available agent
const agent = await this.routeToAgent(decision, conversation.tenantId);
// 3. Create handoff record
const handoff = await this.db.handoff.create({
conversationId,
agentId: agent?.id,
status: agent ? 'assigned' : 'queued',
priority: decision.priority ?? 'normal',
context: handoffContext,
queuedAt: new Date(),
});
// 4. Notify user
await this.notifyUser(conversationId, agent);
// 5. Notify agent (if assigned)
if (agent) {
await this.notifyAgent(agent.id, handoff);
}
return {
handoffId: handoff.id,
status: handoff.status,
estimatedWaitTime: agent ? 0 : await this.estimateWaitTime(conversation.tenantId),
queuePosition: agent ? 0 : await this.getQueuePosition(handoff.id),
};
}
private async packageContext(conversation: Conversation): Promise<HandoffContext> {
// Generate AI summary of the conversation
const summary = await this.llm.chat({
messages: [{
role: 'system',
content: `Summarize this customer conversation for a human support agent. Include:
1. Customer's main issue/question
2. What the bot tried to do
3. Why it couldn't resolve
4. Customer sentiment
5. Any actions already taken
Be concise - max 200 words.`,
}, {
role: 'user',
content: JSON.stringify(conversation.messages.slice(-20)),
}],
model: 'gpt-4o-mini',
});
return {
summary: summary.content,
customerInfo: conversation.customerInfo,
conversationHistory: conversation.messages,
detectedIntent: conversation.intent,
sentiment: conversation.sentimentScore,
toolActionsPerformed: conversation.toolResults,
ragSourcesUsed: conversation.ragSources,
suggestedNextActions: await this.suggestActions(conversation),
};
}
private async routeToAgent(
decision: EscalationDecision,
tenantId: string,
): Promise<HumanAgent | null> {
// Find available agents with matching skills
const availableAgents = await this.agentService.getAvailable(tenantId, {
skill: decision.targetSkill,
language: decision.language,
});
if (availableAgents.length === 0) return null;
// Route based on: skill match + availability + current load
return availableAgents.sort((a, b) => {
const scoreA = this.routingScore(a, decision);
const scoreB = this.routingScore(b, decision);
return scoreB - scoreA;
})[0];
}
}
4. Agent Assist-人類代理的人工智慧副駕駛
class AgentAssistService {
// Real-time suggestions as customer types
async getSuggestions(
conversationId: string,
customerMessage: string,
): Promise<AgentAssistSuggestion> {
const conversation = await this.conversationService.get(conversationId);
// 1. Generate response suggestions
const suggestions = await this.llm.chat({
messages: [{
role: 'system',
content: `You are an AI assistant helping a human support agent.
Based on the conversation and customer's latest message, suggest 3 response options:
1. A direct answer (if you know it)
2. A clarifying question
3. An empathetic response + next steps
Also list any relevant knowledge base articles.
Output JSON.`,
}, {
role: 'user',
content: `Conversation:\n${JSON.stringify(conversation.messages.slice(-10))}\n\nLatest customer message: ${customerMessage}`,
}],
response_format: { type: 'json_object' },
});
const parsed = JSON.parse(suggestions.content);
// 2. Search knowledge base
const articles = await this.rag.retrieve(customerMessage, { topK: 3 });
// 3. Check for macros/templates
const macros = await this.macroService.findRelevant(
conversation.tenantId,
customerMessage,
);
return {
suggestedResponses: parsed.suggestions,
knowledgeArticles: articles.map(a => ({
title: a.title,
snippet: a.content.slice(0, 200),
url: a.sourceUrl,
})),
macros,
customerSentiment: parsed.sentiment,
};
}
// Auto-summarize when agent closes conversation
async generateWrapUp(conversationId: string): Promise<WrapUpSummary> {
const conversation = await this.conversationService.get(conversationId);
const summary = await this.llm.chat({
messages: [{
role: 'system',
content: `Generate a wrap-up summary for this support conversation:
- Issue category
- Root cause
- Resolution
- Follow-up actions needed
- Customer satisfaction assessment
Output JSON.`,
}, {
role: 'user',
content: JSON.stringify(conversation.messages),
}],
response_format: { type: 'json_object' },
});
return JSON.parse(summary.content);
}
}
5. 佇列管理和 SLA 追蹤
class QueueManager {
async getQueueStatus(tenantId: string): Promise<QueueStatus> {
const queued = await this.db.handoff.count({
where: { tenantId, status: 'queued' },
});
const avgWaitTime = await this.db.handoff.aggregate({
where: {
tenantId,
status: 'assigned',
assignedAt: { gte: new Date(Date.now() - 3600 * 1000) },
},
_avg: { waitTimeMs: true },
});
const availableAgents = await this.agentService.countAvailable(tenantId);
return {
queuedConversations: queued,
avgWaitTimeMs: avgWaitTime._avg.waitTimeMs ?? 0,
availableAgents,
estimatedWaitForNew: queued > 0
? (avgWaitTime._avg.waitTimeMs ?? 120_000) * (queued / Math.max(availableAgents, 1))
: 0,
};
}
// SLA monitoring
async checkSLABreaches(tenantId: string): Promise<SLABreach[]> {
const slaConfig = await this.getSLAConfig(tenantId);
const breaches: SLABreach[] = [];
// Check first response time SLA
const pendingHandoffs = await this.db.handoff.findMany({
where: {
tenantId,
status: 'queued',
queuedAt: { lt: new Date(Date.now() - slaConfig.firstResponseTimeMs) },
},
});
for (const handoff of pendingHandoffs) {
breaches.push({
type: 'first_response_time',
handoffId: handoff.id,
slaTarget: slaConfig.firstResponseTimeMs,
actualMs: Date.now() - handoff.queuedAt.getTime(),
priority: handoff.priority,
});
}
return breaches;
}
}
第 17 課總結
- Escalation Triggers:6 種類型 — 明確請求、情緒、置信度、主題、循環、VIP
- 上下文包裝:人工智慧產生的摘要+對話歷史記錄+客戶資訊+建議的操作
- 代理路由:基於技能+可用性+負載平衡
- 代理協助:AI副駕駛建議回覆、尋找知識文章、自動總結
- SLA 追蹤:監控首次回應時間、佇列長度、違規警報
下一篇: 聊天機器人評估和測試——法學碩士作為法官、自動化測試套件、回歸測試、評估指標、紅隊。