はじめに
RAG (検索拡張生成) はドキュメント処理から始まり、埋め込みと検索のためにドキュメントを意味のある小さなチャンクに分割します。この記事では、Document Processor を最初から実装します。
1. ドキュメントタイプとパーサー
// packages/core/src/rag/document-processor.ts
export interface DocumentChunk {
id: string;
content: string;
metadata: {
source: string;
sourceType: 'file' | 'url' | 'text';
title?: string;
pageNumber?: number;
chunkIndex: number;
totalChunks: number;
};
}
export class DocumentProcessor {
private parsers: Map<string, DocumentParser> = new Map();
constructor() {
this.parsers.set('text/plain', new TextParser());
this.parsers.set('text/markdown', new MarkdownParser());
this.parsers.set('text/csv', new CsvParser());
this.parsers.set('application/pdf', new PdfParser());
this.parsers.set('text/html', new HtmlParser());
}
async process(
input: string | Buffer,
mimeType: string,
options: ChunkOptions = {},
): Promise<DocumentChunk[]> {
const parser = this.parsers.get(mimeType);
if (!parser) throw new Error(`Unsupported type: ${mimeType}`);
// Step 1: Parse to plain text
const text = await parser.parse(input);
// Step 2: Clean text
const cleaned = this.cleanText(text);
// Step 3: Chunk
const chunks = this.chunk(cleaned, options);
// Step 4: Add metadata
return chunks.map((content, i) => ({
id: crypto.randomUUID(),
content,
metadata: {
source: options.source || 'unknown',
sourceType: options.sourceType || 'text',
chunkIndex: i,
totalChunks: chunks.length,
},
}));
}
private cleanText(text: string): string {
return text
.replace(/\r\n/g, '\n') // Normalize line endings
.replace(/\n{3,}/g, '\n\n') // Max 2 newlines
.replace(/[ \t]+/g, ' ') // Normalize whitespace
.trim();
}
}
2. チャンク戦略
2.1 固定サイズのチャンク化
interface ChunkOptions {
strategy?: 'fixed' | 'semantic' | 'recursive';
chunkSize?: number; // characters
chunkOverlap?: number; // overlap characters
source?: string;
sourceType?: 'file' | 'url' | 'text';
}
private chunkFixed(text: string, size: number, overlap: number): string[] {
const chunks: string[] = [];
let start = 0;
while (start < text.length) {
let end = start + size;
// Don't break mid-word — find nearest sentence/paragraph boundary
if (end < text.length) {
const boundarySearch = text.slice(end - 50, end + 50);
const sentenceEnd = boundarySearch.search(/[.!?]\s/);
if (sentenceEnd !== -1) {
end = end - 50 + sentenceEnd + 2;
}
}
chunks.push(text.slice(start, end).trim());
start = end - overlap;
}
return chunks.filter(c => c.length > 50); // Skip tiny chunks
}
2.2 再帰的チャンク化
private chunkRecursive(text: string, size: number): string[] {
// Try splitting by largest separator first
const separators = ['\n\n', '\n', '. ', ', ', ' '];
for (const sep of separators) {
const parts = text.split(sep);
if (parts.every(p => p.length <= size)) {
return this.mergeParts(parts, size, sep);
}
}
// Fallback: fixed-size
return this.chunkFixed(text, size, Math.floor(size * 0.1));
}
private mergeParts(parts: string[], maxSize: number, sep: string): string[] {
const chunks: string[] = [];
let current = '';
for (const part of parts) {
if ((current + sep + part).length > maxSize && current) {
chunks.push(current.trim());
current = part;
} else {
current = current ? current + sep + part : part;
}
}
if (current.trim()) chunks.push(current.trim());
return chunks;
}
3. Web クローラー
// packages/core/src/rag/web-crawler.ts
export class WebCrawler {
async crawl(url: string, options: CrawlOptions = {}): Promise<string> {
const response = await fetch(url, {
headers: { 'User-Agent': 'xClaw-Bot/1.0' },
signal: AbortSignal.timeout(30_000),
});
if (!response.ok) throw new Error(`HTTP ${response.status}`);
const html = await response.text();
return this.extractText(html);
}
private extractText(html: string): string {
// Remove scripts, styles, nav, footer
const cleaned = html
.replace(/<script[\s\S]*?<\/script>/gi, '')
.replace(/<style[\s\S]*?<\/style>/gi, '')
.replace(/<nav[\s\S]*?<\/nav>/gi, '')
.replace(/<footer[\s\S]*?<\/footer>/gi, '')
.replace(/<[^>]+>/g, ' ') // Strip remaining tags
.replace(/ /g, ' ')
.replace(/\s+/g, ' ');
return cleaned.trim();
}
}
4. まとめ
| 戦略 | いつ使用するか |
|---|---|
| 固定サイズ | シンプルで予測可能なチャンク — 適切なデフォルト |
| 再帰的 | 構造化ドキュメント (マークダウン、コード) — 構造を保持 |
| セマンティック | 研究論文 - 埋め込みモデルが必要 |
次の記事: 埋め込みとベクター ストア — テキストをベクターに変換します。