1.知識庫管理-超越RAG
RAG 管道(第 5 課)解決 檢索。檢索。但企業需要管理 生命週期 知識-從多個來源取得、版本控制、存取控制、過時內容偵測、多租戶隔離。
┌──────────── KNOWLEDGE BASE MANAGEMENT ────────────────┐
│ │
│ ┌──────────┐ ┌──────────┐ ┌──────────┐ ┌──────────┐ │
│ │Confluence│ │SharePoint│ │ Google │ │ Upload │ │
│ │ │ │ │ │ Drive │ │ (API) │ │
│ └────┬─────┘ └────┬─────┘ └────┬─────┘ └────┬─────┘ │
│ └─────────────┼───────────┘ │ │
│ ▼ │ │
│ ┌──────────────────────────────────────────┐ │ │
│ │ CONNECTOR FRAMEWORK │◀┘ │
│ │ (Sync, Transform, Deduplicate) │ │
│ └─────────────────┬────────────────────────┘ │
│ │ │
│ ┌─────────────────▼────────────────────────┐ │
│ │ PROCESSING PIPELINE │ │
│ │ Parse → Clean → Chunk → Embed → Index │ │
│ └─────────────────┬────────────────────────┘ │
│ │ │
│ ┌─────────────────▼────────────────────────┐ │
│ │ KNOWLEDGE STORE │ │
│ │ PostgreSQL (metadata) + Vector DB │ │
│ │ + Access Control + Version History │ │
│ └──────────────────────────────────────────┘ │
└────────────────────────────────────────────────────────┘
2. 連接器框架-多源攝取
interface KnowledgeConnector {
id: string;
name: string;
type: 'confluence' | 'sharepoint' | 'gdrive' | 'notion' | 's3' | 'api';
sync(config: ConnectorConfig): AsyncIterable<RawDocument>;
getChanges(since: Date): AsyncIterable<DocumentChange>;
}
class ConfluenceConnector implements KnowledgeConnector {
id = 'confluence';
name = 'Atlassian Confluence';
type = 'confluence' as const;
async *sync(config: ConnectorConfig): AsyncIterable<RawDocument> {
const spaces = config.spaces ?? ['ALL'];
for (const space of spaces) {
let start = 0;
const limit = 50;
while (true) {
const response = await this.client.get(`/wiki/api/v2/spaces/${space}/pages`, {
params: { start, limit, body_format: 'storage' },
});
for (const page of response.data.results) {
yield {
sourceId: `confluence:${page.id}`,
title: page.title,
content: page.body.storage.value,
contentType: 'html',
metadata: {
space: space,
author: page.version.authorId,
lastModified: page.version.createdAt,
version: page.version.number,
labels: page.labels?.results?.map(l => l.name) ?? [],
url: `${config.baseUrl}/wiki${page._links.webui}`,
},
};
}
if (response.data.results.length < limit) break;
start += limit;
}
}
}
async *getChanges(since: Date): AsyncIterable<DocumentChange> {
// Use Confluence audit log for incremental sync
const auditRecords = await this.client.get('/wiki/rest/api/audit', {
params: {
startDate: since.toISOString(),
searchString: 'page',
},
});
for (const record of auditRecords.data.results) {
yield {
type: record.action === 'page_removed' ? 'deleted' : 'updated',
sourceId: `confluence:${record.affectedObject.objectId}`,
timestamp: new Date(record.creationDate),
};
}
}
}
class ConnectorOrchestrator {
async syncAll(tenantId: string): Promise<SyncReport> {
const connectors = await this.getConnectors(tenantId);
const report: SyncReport = { processed: 0, created: 0, updated: 0, deleted: 0, errors: 0 };
for (const connector of connectors) {
const lastSync = await this.getLastSyncTime(tenantId, connector.id);
if (lastSync) {
// Incremental sync
for await (const change of connector.getChanges(lastSync)) {
try {
if (change.type === 'deleted') {
await this.knowledgeStore.softDelete(change.sourceId);
report.deleted++;
} else {
await this.processDocument(tenantId, change.sourceId, connector);
report.updated++;
}
report.processed++;
} catch (error) {
report.errors++;
}
}
} else {
// Full sync
for await (const doc of connector.sync(connector.config)) {
await this.processDocument(tenantId, doc.sourceId, connector, doc);
report.created++;
report.processed++;
}
}
await this.setLastSyncTime(tenantId, connector.id, new Date());
}
return report;
}
}
3. 多格式文件處理器
class DocumentProcessor {
private parsers = new Map<string, DocumentParser>([
['pdf', new PDFParser()],
['docx', new DocxParser()],
['html', new HTMLParser()],
['md', new MarkdownParser()],
['csv', new CSVParser()],
['xlsx', new ExcelParser()],
]);
async process(rawDoc: RawDocument, tenantId: string): Promise<ProcessedDocument> {
// 1. Parse content
const parser = this.parsers.get(rawDoc.contentType);
if (!parser) throw new Error(`Unsupported format: ${rawDoc.contentType}`);
const parsed = await parser.parse(rawDoc.content);
// 2. Clean & normalize
const cleaned = this.cleanContent(parsed);
// 3. Extract metadata via LLM
const enrichedMetadata = await this.enrichMetadata(cleaned, rawDoc.metadata);
// 4. Chunk with metadata preservation
const chunks = await this.chunker.chunk(cleaned, {
strategy: 'recursive',
chunkSize: 512,
overlap: 50,
preserveMetadata: true,
});
// 5. Generate embeddings
const embeddings = await this.embedder.embedBatch(
chunks.map(c => c.content),
);
// 6. Create document record
return {
id: crypto.randomUUID(),
tenantId,
sourceId: rawDoc.sourceId,
title: rawDoc.title,
content: cleaned,
chunks: chunks.map((chunk, i) => ({
id: `${rawDoc.sourceId}:chunk:${i}`,
content: chunk.content,
embedding: embeddings[i],
metadata: { ...enrichedMetadata, chunkIndex: i },
})),
metadata: enrichedMetadata,
status: 'active',
version: 1,
};
}
private async enrichMetadata(
content: string,
existingMetadata: Record<string, unknown>,
): Promise<EnrichedMetadata> {
const response = await this.llm.chat({
messages: [{
role: 'system',
content: `Extract metadata from this document content:
- summary (1-2 sentences)
- topics (list of main topics)
- documentType (policy, procedure, faq, guide, reference)
- language
- targetAudience
Output JSON.`,
}, {
role: 'user',
content: content.slice(0, 2000), // First 2000 chars
}],
response_format: { type: 'json_object' },
model: 'gpt-4o-mini',
});
return { ...existingMetadata, ...JSON.parse(response.content) };
}
}
4. 文件級存取控制
interface DocumentACL {
documentId: string;
rules: ACLRule[];
}
interface ACLRule {
principal: { type: 'user' | 'group' | 'role'; id: string };
permission: 'read' | 'write' | 'admin';
}
class KnowledgeAccessControl {
async filterSearchResults(
results: SearchResult[],
userId: string,
tenantId: string,
): Promise<SearchResult[]> {
// Get user's groups and roles
const userContext = await this.getUserContext(userId, tenantId);
return results.filter(result => {
const acl = result.metadata.acl as DocumentACL | undefined;
// No ACL = public within tenant
if (!acl?.rules.length) return true;
return acl.rules.some(rule => {
if (rule.permission !== 'read') return false;
switch (rule.principal.type) {
case 'user':
return rule.principal.id === userId;
case 'group':
return userContext.groups.includes(rule.principal.id);
case 'role':
return userContext.roles.includes(rule.principal.id);
default:
return false;
}
});
});
}
// Sync ACLs from source system (Confluence, SharePoint)
async syncACLs(connector: KnowledgeConnector, documentId: string): Promise<DocumentACL> {
const sourceACL = await connector.getPermissions(documentId);
return {
documentId,
rules: sourceACL.map(acl => ({
principal: { type: acl.principalType, id: acl.principalId },
permission: this.mapPermission(acl.sourcePermission),
})),
};
}
}
5. 過時內容偵測與生命週期管理
class KnowledgeLifecycleManager {
async detectStaleContent(tenantId: string): Promise<StaleReport> {
const staleDocuments: StaleDocument[] = [];
// Strategy 1: Time-based staleness
const oldDocs = await this.db.document.findMany({
where: {
tenantId,
updatedAt: { lt: new Date(Date.now() - 90 * 24 * 3600 * 1000) }, // 90 days
status: 'active',
},
});
staleDocuments.push(...oldDocs.map(d => ({
...d, reason: 'not_updated_90_days',
})));
// Strategy 2: Source deleted
for (const doc of await this.db.document.findActive(tenantId)) {
const sourceExists = await this.checkSourceExists(doc.sourceId);
if (!sourceExists) {
staleDocuments.push({ ...doc, reason: 'source_deleted' });
}
}
// Strategy 3: Low retrieval score (nobody finds this useful)
const lowScoreDocs = await this.analytics.getDocumentsWithLowRetrievalScore(
tenantId,
{ minQueries: 100, maxAvgScore: 0.3 },
);
staleDocuments.push(...lowScoreDocs.map(d => ({
...d, reason: 'low_retrieval_relevance',
})));
return {
staleDocuments,
recommendations: this.generateRecommendations(staleDocuments),
};
}
async archiveDocument(documentId: string): Promise<void> {
// Soft delete — keep history
await this.db.document.update(documentId, { status: 'archived' });
// Remove from vector store
await this.vectorStore.deleteByFilter({ documentId });
// Log for audit
await this.auditLog.log({
action: 'document_archived',
documentId,
timestamp: new Date(),
});
}
}
第 13 課總結
- 連接器框架:Confluence、SharePoint、Google Drive — 完全同步 + 增量同步
- 文件處理:多重格式解析→清除→分塊→嵌入→索引
- 存取控制:文檔級ACL(使用者/群組/角色),從來源系統同步
- 生命週期管理:過時檢測(基於時間、來源刪除、低相關性)→ 存檔
- 增量同步:僅處理更改,無需每次同步時完全重新索引
下一篇: 多租戶架構-租戶隔離、每個租戶的配置、資源配額、計費計量。