はじめに
LLM API 呼び出しは高価です。 1 日あたり 10,000 人のユーザーにサービスを提供する RAG チャットボット アプリケーションの費用は、API コストだけで 月額 500 ~ 5000 ドル かかります。このレッスンでは、品質を犠牲にすることなくコストを削減するテクニックを学びます。
🎯 目標: 品質を維持しながら、LLM コストを 50 ~ 80% 削減します。
1. LLM コストの内訳
Chi phí LLM call = Input Tokens × Input Price + Output Tokens × Output Price
Ví dụ GPT-4o (per 1M tokens):
Input: $2.50
Output: $10.00
10K requests/ngày × 2000 tokens/request:
= 20M tokens/ngày
= $50-200/ngày
= $1,500-6,000/tháng 😱
| モデル | インプット ($/1M) | 生産高 ($/100 万) | スピード |
|---|---|---|---|
| GPT-4o | $2.50 | $10.00 | 中 |
| GPT-4o-mini | $0.15 | $0.60 | 速い |
| クロード 3.5 ソネット | $3.00 | $15.00 | 中 |
| クロード 3.5 俳句 | $0.25 | $1.25 | 速い |
| ジェミニ 1.5 フラッシュ | $0.075 | $0.30 | 非常に速い |
| ラマ 3.1 70B (セルフ) | ~$0.00 | ~$0.00 | GPU に依存 |
2. セマンティック キャッシング
2.1 完全一致キャッシュ
"""Exact match caching — đơn giản nhất"""
import hashlib
import json
import redis
class ExactCache:
def __init__(self):
self.redis = redis.Redis(host='localhost', port=6379)
self.ttl = 3600 * 24 # 24h
def _key(self, messages, model):
content = json.dumps({"messages": messages, "model": model}, sort_keys=True)
return f"llm:exact:{hashlib.md5(content.encode()).hexdigest()}"
def get(self, messages, model):
key = self._key(messages, model)
cached = self.redis.get(key)
if cached:
return json.loads(cached)
return None
def set(self, messages, model, response):
key = self._key(messages, model)
self.redis.setex(key, self.ttl, json.dumps(response))
# Usage
cache = ExactCache()
def cached_llm_call(messages, model="gpt-4o-mini"):
# Check cache
cached = cache.get(messages, model)
if cached:
print("💾 Cache hit!")
return cached
# API call
response = client.chat.completions.create(model=model, messages=messages)
result = response.choices[0].message.content
# Cache result
cache.set(messages, model, result)
return result
2.2 セマンティック キャッシュ
"""Semantic caching — cache câu hỏi tương tự"""
import numpy as np
from openai import OpenAI
class SemanticCache:
def __init__(self, similarity_threshold=0.95):
self.client = OpenAI()
self.threshold = similarity_threshold
self.cache = [] # [{embedding, query, response}]
def _get_embedding(self, text):
response = self.client.embeddings.create(
model="text-embedding-3-small",
input=text,
)
return np.array(response.data[0].embedding)
def _cosine_similarity(self, a, b):
return np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b))
def get(self, query):
"""Find similar cached query"""
query_embedding = self._get_embedding(query)
best_score = 0
best_response = None
for item in self.cache:
score = self._cosine_similarity(query_embedding, item["embedding"])
if score > best_score:
best_score = score
best_response = item["response"]
if best_score >= self.threshold:
print(f"💾 Semantic cache hit! (similarity: {best_score:.3f})")
return best_response
return None
def set(self, query, response):
self.cache.append({
"embedding": self._get_embedding(query),
"query": query,
"response": response,
})
# Usage
sem_cache = SemanticCache(similarity_threshold=0.92)
# Query 1: "MLOps là gì?"
response1 = llm_call("MLOps là gì?")
sem_cache.set("MLOps là gì?", response1)
# Query 2: "ML Ops nghĩa là gì?" → Cache HIT (semantic similar)
cached = sem_cache.get("ML Ops nghĩa là gì?")
2.3 GPTCache (実稼働対応)
"""GPTCache — production semantic caching"""
# pip install gptcache
from gptcache import cache
from gptcache.adapter import openai as gptcache_openai
from gptcache.embedding import Onnx
from gptcache.manager import get_data_manager, CacheBase
from gptcache.similarity_evaluation.distance import SearchDistanceEvaluation
# Setup
onnx = Onnx()
cache_base = CacheBase('sqlite')
data_manager = get_data_manager(cache_base, vector_base=None)
cache.init(
embedding_func=onnx.to_embeddings,
data_manager=data_manager,
similarity_evaluation=SearchDistanceEvaluation(),
)
cache.set_openai_key()
# Use (drop-in replacement)
response = gptcache_openai.ChatCompletion.create(
model="gpt-4o-mini",
messages=[{"role": "user", "content": "What is MLOps?"}],
)
# → First call: API, subsequent similar calls: cache
3. モデルのルーティング
"""Smart model routing — dùng model rẻ khi có thể"""
class ModelRouter:
def __init__(self):
self.client = OpenAI()
def classify_complexity(self, query):
"""Phân loại độ phức tạp của query"""
# Rule-based (nhanh, free)
simple_patterns = [
"dịch", "translate", "define", "what is",
"list", "yes or no", "true or false",
]
complex_patterns = [
"analyze", "compare", "explain why",
"design", "architect", "optimize",
"code review", "debug",
]
query_lower = query.lower()
for pattern in simple_patterns:
if pattern in query_lower:
return "simple"
for pattern in complex_patterns:
if pattern in query_lower:
return "complex"
return "medium"
def route(self, query, messages):
"""Route to appropriate model"""
complexity = self.classify_complexity(query)
model_map = {
"simple": "gpt-4o-mini", # $0.15/M input — rẻ
"medium": "gpt-4o-mini", # $0.15/M input
"complex": "gpt-4o", # $2.50/M input — đắt nhưng mạnh
}
model = model_map[complexity]
print(f"🔀 Routing [{complexity}] → {model}")
response = self.client.chat.completions.create(
model=model,
messages=messages,
)
return response
router = ModelRouter()
# Simple → gpt-4o-mini (rẻ 16x)
router.route("Dịch sang tiếng Anh: Xin chào", messages=[...])
# Complex → gpt-4o (mạnh)
router.route("Analyze this system architecture and suggest improvements", messages=[...])
カスケード戦略
"""Cascading: thử model rẻ trước, fallback model đắt"""
class CascadingRouter:
def __init__(self):
self.client = OpenAI()
self.models = [
{"name": "gpt-4o-mini", "max_attempts": 1},
{"name": "gpt-4o", "max_attempts": 1},
]
def call(self, messages, quality_threshold=0.7):
"""Try cheap model first, escalate if quality is low"""
for model_config in self.models:
model = model_config["name"]
response = self.client.chat.completions.create(
model=model,
messages=messages,
)
output = response.choices[0].message.content
# Quick quality check
quality = self.assess_quality(messages[-1]["content"], output)
if quality >= quality_threshold:
print(f"✅ {model} passed quality check ({quality:.2f})")
return output, model
else:
print(f"⚠️ {model} failed quality ({quality:.2f}), escalating...")
return output, model # Return last attempt
def assess_quality(self, question, answer):
"""Quick quality assessment (rule-based for speed)"""
score = 0.5
if len(answer) > 50: score += 0.1
if "?" not in answer: score += 0.1 # Not just asking back
if len(answer) < 2000: score += 0.1 # Not too verbose
# Add more heuristics
return min(score, 1.0)
4. トークンの最適化
"""Giảm token usage"""
# Technique 1: Prompt compression
def compress_prompt(system_prompt, max_words=200):
"""Rút gọn prompt giữ ý chính"""
words = system_prompt.split()
if len(words) <= max_words:
return system_prompt
# Remove filler words
fillers = {"the", "a", "an", "is", "are", "was", "were",
"will", "would", "could", "should", "very", "really"}
compressed = [w for w in words if w.lower() not in fillers]
return " ".join(compressed[:max_words])
# Technique 2: Context pruning cho RAG
def prune_context(documents, max_tokens=2000):
"""Giữ docs quan trọng nhất, cắt phần dư"""
import tiktoken
enc = tiktoken.encoding_for_model("gpt-4o-mini")
pruned = []
total_tokens = 0
for doc in sorted(documents, key=lambda d: d["score"], reverse=True):
doc_tokens = len(enc.encode(doc["content"]))
if total_tokens + doc_tokens > max_tokens:
# Truncate last doc
remaining = max_tokens - total_tokens
if remaining > 100:
truncated = enc.decode(enc.encode(doc["content"])[:remaining])
pruned.append({**doc, "content": truncated})
break
pruned.append(doc)
total_tokens += doc_tokens
return pruned
# Technique 3: Structured output (giảm output tokens)
# Thay vì: "The sentiment of this text is positive because..."
# Dùng: {"sentiment": "positive", "confidence": 0.95}
def get_structured_output(text):
response = client.chat.completions.create(
model="gpt-4o-mini",
messages=[
{"role": "system", "content": "Classify sentiment. Return JSON only: {\"sentiment\": \"positive|negative|neutral\", \"confidence\": 0.0-1.0}"},
{"role": "user", "content": text},
],
response_format={"type": "json_object"},
max_tokens=50, # Giới hạn output
)
return response
5. セルフホストモデル
5.1 vLLM — 高スループットのサービス提供
"""vLLM: Serve LLM self-hosted, rất nhanh"""
# pip install vllm
# Start server
# python -m vllm.entrypoints.openai.api_server \
# --model meta-llama/Llama-3.1-8B-Instruct \
# --dtype float16 \
# --max-model-len 8192 \
# --gpu-memory-utilization 0.9
# Client (OpenAI-compatible API)
from openai import OpenAI
client = OpenAI(
base_url="http://localhost:8000/v1",
api_key="not-needed",
)
response = client.chat.completions.create(
model="meta-llama/Llama-3.1-8B-Instruct",
messages=[{"role": "user", "content": "MLOps là gì?"}],
)
print(response.choices[0].message.content)
# Cost: $0 per token (chỉ trả tiền GPU)
5.2 オラマ — 地域開発
"""Ollama: Chạy LLM trên laptop"""
# brew install ollama
# ollama pull llama3.1:8b
from openai import OpenAI
client = OpenAI(
base_url="http://localhost:11434/v1",
api_key="ollama",
)
response = client.chat.completions.create(
model="llama3.1:8b",
messages=[{"role": "user", "content": "Explain Docker"}],
)
# Free, offline, private!
5.3 コストの比較: API とセルフホスト型
Scenario: 1M requests/tháng, 1000 tokens/request
API (GPT-4o-mini):
1B tokens × $0.375/M = $375/tháng
Self-hosted (Llama 3.1 8B on A100):
GPU: ~$1.50/hour × 24 × 30 = $1,080/tháng
But: faster, private, no rate limits
Self-hosted (Llama 3.1 8B on A10G):
GPU: ~$0.75/hour × 24 × 30 = $540/tháng
Break-even point:
API rẻ hơn khi: < 2.5M requests/tháng (GPT-4o-mini)
Self-host rẻ hơn khi: > 2.5M requests/tháng
6. コストの監視と予算編成
"""Cost monitoring system"""
import time
from collections import defaultdict
from datetime import datetime, timedelta
class CostMonitor:
def __init__(self, daily_budget=50.0):
self.daily_budget = daily_budget
self.costs = defaultdict(float) # date → cost
self.model_costs = defaultdict(float) # model → cost
def log_cost(self, model, input_tokens, output_tokens):
"""Log cost for a call"""
prices = {
"gpt-4o": {"input": 2.50/1e6, "output": 10.0/1e6},
"gpt-4o-mini": {"input": 0.15/1e6, "output": 0.60/1e6},
}
p = prices.get(model, prices["gpt-4o-mini"])
cost = input_tokens * p["input"] + output_tokens * p["output"]
today = datetime.now().strftime("%Y-%m-%d")
self.costs[today] += cost
self.model_costs[model] += cost
# Budget check
if self.costs[today] > self.daily_budget:
self._alert(f"🚨 Daily budget exceeded: ${self.costs[today]:.2f} > ${self.daily_budget}")
if self.costs[today] > self.daily_budget * 0.8:
self._alert(f"⚠️ 80% of daily budget used: ${self.costs[today]:.2f}")
return cost
def get_report(self):
"""Cost report"""
today = datetime.now().strftime("%Y-%m-%d")
return {
"today_cost": f"${self.costs[today]:.2f}",
"today_budget_usage": f"{self.costs[today]/self.daily_budget:.0%}",
"cost_by_model": {k: f"${v:.2f}" for k, v in self.model_costs.items()},
"total_cost": f"${sum(self.costs.values()):.2f}",
}
def _alert(self, message):
print(message)
# Send to Slack/PagerDuty/etc.
# Usage
monitor = CostMonitor(daily_budget=50.0)
cost = monitor.log_cost("gpt-4o-mini", input_tokens=500, output_tokens=200)
print(monitor.get_report())
概要
| テクニック | 節約 | 努力 | トレードオフ |
|---|---|---|---|
| セマンティック キャッシング | 30-60% | 🔨🔨 | キャッシュの古さ |
| モデル ルーティング | 40-70% | 🔨🔨 | エッジケースの品質 |
| トークンの最適化 | 20-40% | 🔨 | プロンプトの長さの制限 |
| 構造化された出力 | 30-50% | 🔨 | 冗長ではない |
| セルフホスト | 50-90% | 🔨🔨🔨🔨 | インフラ管理 |
| カスケード | 40-60% | 🔨🔨 | レイテンシの追加 |
演習
- セマンティック キャッシュ: セマンティック キャッシュを実装し、キャッシュ ヒット率と正確なキャッシュを比較します。
- ルーター: ルーター モデルを構築します (シンプル→ミニ、複雑→4o)。 100 件のクエリでテスト済み。
- コスト ダッシュボード: コスト モニターを構築し、1000 回の LLM 呼び出しを実行し、レポートを生成します。
- セルフホスト: Ollama 経由で Llama 3.1 8B を展開します。ベンチマークと GPT-4o-mini を比較します。
次の記事: ガードレール、安全性、コンプライアンス。