1.企業聊天機器人中的多模態人工智慧
企業用戶需要聊天機器人不僅能理解文本,還能理解文本 圖像、文件、圖表、螢幕截圖。多模式人工智慧將聊天機器人從「純文字助理」轉變為「視覺感知智慧代理」。
┌─────────── MULTIMODAL PIPELINE ──────────────────────┐
│ │
│ Input Types: │
│ ┌────────┐ ┌────────┐ ┌────────┐ ┌────────────┐ │
│ │ Photo │ │ PDF │ │ Chart │ │ Screenshot │ │
│ │ │ │ Scan │ │ Graph │ │ │ │
│ └───┬────┘ └───┬────┘ └───┬────┘ └─────┬──────┘ │
│ │ │ │ │ │
│ ┌───▼──────────▼──────────▼────────────▼──────┐ │
│ │ MULTIMODAL ROUTER │ │
│ │ (Detect content type → route to pipeline) │ │
│ └───┬──────────┬──────────┬────────────┬──────┘ │
│ │ │ │ │ │
│ ▼ ▼ ▼ ▼ │
│ ┌──────┐ ┌──────┐ ┌──────┐ ┌──────────┐ │
│ │Vision│ │ OCR │ │Chart │ │ Screen │ │
│ │Model │ │Engine│ │Parser│ │ Understanding│ │
│ └──┬───┘ └──┬───┘ └──┬───┘ └─────┬────┘ │
│ └─────────┴─────────┴──────────────┘ │
│ │ │
│ ┌──────▼──────┐ │
│ │ Unified │ │
│ │ Context │──▶ LLM │
│ └─────────────┘ │
└───────────────────────────────────────────────────────┘
2. 視覺模型集成
class VisionProcessor {
async processImage(
image: ImageInput,
query: string,
context: VisionContext,
): Promise<VisionResult> {
// 1. Optimize image for vision model
const optimized = await this.optimizeForVision(image);
// 2. Route to appropriate vision pipeline
const contentType = await this.classifyImageContent(optimized);
switch (contentType) {
case 'document':
return this.processDocument(optimized, query);
case 'chart':
return this.processChart(optimized, query);
case 'product_photo':
return this.processProductPhoto(optimized, query);
case 'screenshot':
return this.processScreenshot(optimized, query);
default:
return this.processGeneral(optimized, query);
}
}
private async classifyImageContent(image: OptimizedImage): Promise<string> {
const response = await this.llm.chat({
messages: [{
role: 'user',
content: [
{ type: 'text', text: 'Classify this image into one of: document, chart, product_photo, screenshot, general. Output only the category.' },
{
type: 'image_url',
image_url: {
url: `data:${image.mimeType};base64,${image.base64}`,
detail: 'low', // Low detail for classification (cheaper)
},
},
],
}],
model: 'gpt-4o-mini',
maxTokens: 20,
});
return response.content.trim().toLowerCase();
}
private async processGeneral(
image: OptimizedImage,
query: string,
): Promise<VisionResult> {
const response = await this.llm.chat({
messages: [{
role: 'user',
content: [
{ type: 'text', text: query || 'Describe this image in detail.' },
{
type: 'image_url',
image_url: {
url: `data:${image.mimeType};base64,${image.base64}`,
detail: 'high',
},
},
],
}],
model: 'gpt-4o',
});
return {
type: 'general',
description: response.content,
extractedText: null,
structuredData: null,
};
}
}
3. 文檔 OCR 管道
class DocumentOCRPipeline {
async process(document: DocumentInput): Promise<OCRResult> {
const pages: PageResult[] = [];
// 1. Convert document to images (if PDF)
const images = document.type === 'pdf'
? await this.pdfToImages(document.data)
: [{ data: document.data, mimeType: document.mimeType }];
// 2. OCR each page
for (let i = 0; i < images.length; i++) {
// Strategy A: Vision model OCR (higher accuracy, slower)
const visionOCR = await this.visionModelOCR(images[i]);
// Strategy B: Tesseract OCR (faster, good for clear text)
const tesseractOCR = await this.tesseractOCR(images[i]);
// 3. Merge results (use vision for complex layouts, Tesseract for simple)
const mergedText = this.mergeOCRResults(visionOCR, tesseractOCR);
// 4. Structure extraction (tables, forms, lists)
const structured = await this.extractStructure(images[i], mergedText);
pages.push({
pageNumber: i + 1,
text: mergedText,
tables: structured.tables,
formFields: structured.formFields,
confidence: structured.confidence,
});
}
return {
totalPages: pages.length,
fullText: pages.map(p => p.text).join('\n\n---\n\n'),
pages,
language: await this.detectLanguage(pages[0].text),
};
}
private async visionModelOCR(image: ImageData): Promise<string> {
const response = await this.llm.chat({
messages: [{
role: 'user',
content: [
{
type: 'text',
text: `Extract ALL text from this document image.
Preserve the original formatting and structure.
For tables, use markdown table format.
For forms, extract field labels and values as "Label: Value".
Output the raw extracted text only.`,
},
{
type: 'image_url',
image_url: {
url: `data:${image.mimeType};base64,${Buffer.from(image.data).toString('base64')}`,
detail: 'high',
},
},
],
}],
model: 'gpt-4o',
maxTokens: 4096,
});
return response.content;
}
private async extractStructure(
image: ImageData,
text: string,
): Promise<StructuredContent> {
const response = await this.llm.chat({
messages: [{
role: 'user',
content: [
{
type: 'text',
text: `Analyze this document and extract structured data.
Return JSON with: tables (as arrays), formFields (key-value pairs), lists.`,
},
{
type: 'image_url',
image_url: {
url: `data:${image.mimeType};base64,${Buffer.from(image.data).toString('base64')}`,
detail: 'high',
},
},
],
}],
response_format: { type: 'json_object' },
model: 'gpt-4o',
});
return JSON.parse(response.content);
}
}
4. 圖表分析
class ChartAnalyzer {
async analyzeChart(
chartImage: ImageInput,
question: string,
): Promise<ChartAnalysis> {
const response = await this.llm.chat({
messages: [{
role: 'user',
content: [
{
type: 'text',
text: `Analyze this chart/graph and answer the question.
1. Identify chart type (bar, line, pie, scatter, etc.)
2. Extract data points and labels
3. Identify trends and insights
4. Answer the specific question
Question: ${question || 'What are the key insights from this chart?'}
Output JSON:
{
"chartType": "...",
"title": "...",
"dataPoints": [{"label": "...", "value": N}],
"trends": ["..."],
"insights": ["..."],
"answer": "..."
}`,
},
{
type: 'image_url',
image_url: {
url: `data:${chartImage.mimeType};base64,${chartImage.base64}`,
detail: 'high',
},
},
],
}],
response_format: { type: 'json_object' },
model: 'gpt-4o',
});
return JSON.parse(response.content);
}
}
5. Multimodal RAG — 在知識庫中索引影像
class MultimodalRAG {
// Index images alongside text in knowledge base
async indexDocumentWithImages(
document: DocumentWithImages,
tenantId: string,
): Promise<void> {
// 1. Index text chunks as usual
for (const chunk of document.textChunks) {
const embedding = await this.textEmbedder.embed(chunk.content);
await this.vectorStore.upsert({
id: `${document.id}:text:${chunk.index}`,
vector: embedding,
metadata: {
tenantId,
documentId: document.id,
type: 'text',
content: chunk.content,
},
});
}
// 2. Generate descriptions for images → index as text
for (const image of document.images) {
const description = await this.describeImage(image);
const embedding = await this.textEmbedder.embed(description);
await this.vectorStore.upsert({
id: `${document.id}:image:${image.index}`,
vector: embedding,
metadata: {
tenantId,
documentId: document.id,
type: 'image',
content: description,
imageUrl: image.url,
pageNumber: image.pageNumber,
},
});
}
}
// Retrieve with image context
async retrieve(
query: string,
tenantId: string,
): Promise<MultimodalSearchResult[]> {
const results = await this.vectorStore.search({
vector: await this.textEmbedder.embed(query),
filter: { tenantId },
topK: 10,
});
return results.map(r => ({
content: r.metadata.content as string,
type: r.metadata.type as 'text' | 'image',
imageUrl: r.metadata.type === 'image' ? r.metadata.imageUrl as string : undefined,
score: r.score,
documentId: r.metadata.documentId as string,
}));
}
}
第 21 課總結
- 視覺路由器:自動分類影像類型(文件、圖表、照片、螢幕截圖)→路由到管道
- 文檔OCR:視覺模型+Tesseract後備、表格/表單提取、多頁面支持
- 圖表分析:識別圖表類型、擷取資料點、偵測趨勢、回答問題
- 多模式RAG:將圖像索引為文字描述 → 可與文字文件一起搜尋
- 成本最佳化:使用
細節:“低”用於分類,詳細資訊:“高”用於提取
下一篇: 工作流程自動化 — 聊天機器人觸發的工作流程、審核流程、與 n8n/Temporal 整合、事件驅動的自動化。