Chuyển đến nội dung chính

レッスン 10: コストの最適化 — キャッシュ、ルーティング、量子化

LLM コストの最適化: セマンティック キャッシュ、プロンプト圧縮、モデル ルーティング、トークンの最適化。セルフホスト モデル: vLLM、Ollama。推論のための量子化。コストの監視と予算編成。

🧠 AI と ML — レッスン 9 レッスン 10: コストの最適化 — キャッシュ、 ルーティングと量子化

MLOps と LLMOps: AI を本番環境に導入する

パート 4: 生産とガバナンス

xdev.asia

はじめに

LLM API 呼び出しは高価です。 1 日あたり 10,000 人のユーザーにサービスを提供する RAG チャットボット アプリケーションの費用は、API コストだけで 月額 500 ~ 5000 ドル かかります。このレッスンでは、品質を犠牲にすることなくコストを削減するテクニックを学びます。

🎯 目標: 品質を維持しながら、LLM コストを 50 ~ 80% 削減します。


1. LLM コストの内訳

Chi phí LLM call = Input Tokens × Input Price + Output Tokens × Output Price

Ví dụ GPT-4o (per 1M tokens):
  Input:  $2.50
  Output: $10.00

10K requests/ngày × 2000 tokens/request:
  = 20M tokens/ngày
  = $50-200/ngày
  = $1,500-6,000/tháng 😱
モデルインプット ($/1M)生産高 ($/100 万)スピード
GPT-4o$2.50$10.00中
GPT-4o-mini$0.15$0.60速い
クロード 3.5 ソネット$3.00$15.00中
クロード 3.5 俳句$0.25$1.25速い
ジェミニ 1.5 フラッシュ$0.075$0.30非常に速い
ラマ 3.1 70B (セルフ)~$0.00~$0.00GPU に依存

2. セマンティック キャッシング

2.1 完全一致キャッシュ

"""Exact match caching — đơn giản nhất"""
import hashlib
import json
import redis

class ExactCache:
    def __init__(self):
        self.redis = redis.Redis(host='localhost', port=6379)
        self.ttl = 3600 * 24  # 24h

    def _key(self, messages, model):
        content = json.dumps({"messages": messages, "model": model}, sort_keys=True)
        return f"llm:exact:{hashlib.md5(content.encode()).hexdigest()}"

    def get(self, messages, model):
        key = self._key(messages, model)
        cached = self.redis.get(key)
        if cached:
            return json.loads(cached)
        return None

    def set(self, messages, model, response):
        key = self._key(messages, model)
        self.redis.setex(key, self.ttl, json.dumps(response))

# Usage
cache = ExactCache()

def cached_llm_call(messages, model="gpt-4o-mini"):
    # Check cache
    cached = cache.get(messages, model)
    if cached:
        print("💾 Cache hit!")
        return cached

    # API call
    response = client.chat.completions.create(model=model, messages=messages)
    result = response.choices[0].message.content

    # Cache result
    cache.set(messages, model, result)
    return result

2.2 セマンティック キャッシュ

"""Semantic caching — cache câu hỏi tương tự"""
import numpy as np
from openai import OpenAI

class SemanticCache:
    def __init__(self, similarity_threshold=0.95):
        self.client = OpenAI()
        self.threshold = similarity_threshold
        self.cache = []  # [{embedding, query, response}]

    def _get_embedding(self, text):
        response = self.client.embeddings.create(
            model="text-embedding-3-small",
            input=text,
        )
        return np.array(response.data[0].embedding)

    def _cosine_similarity(self, a, b):
        return np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b))

    def get(self, query):
        """Find similar cached query"""
        query_embedding = self._get_embedding(query)

        best_score = 0
        best_response = None

        for item in self.cache:
            score = self._cosine_similarity(query_embedding, item["embedding"])
            if score > best_score:
                best_score = score
                best_response = item["response"]

        if best_score >= self.threshold:
            print(f"💾 Semantic cache hit! (similarity: {best_score:.3f})")
            return best_response
        return None

    def set(self, query, response):
        self.cache.append({
            "embedding": self._get_embedding(query),
            "query": query,
            "response": response,
        })

# Usage
sem_cache = SemanticCache(similarity_threshold=0.92)

# Query 1: "MLOps là gì?"
response1 = llm_call("MLOps là gì?")
sem_cache.set("MLOps là gì?", response1)

# Query 2: "ML Ops nghĩa là gì?" → Cache HIT (semantic similar)
cached = sem_cache.get("ML Ops nghĩa là gì?")

2.3 GPTCache (実稼働対応)

"""GPTCache — production semantic caching"""
# pip install gptcache
from gptcache import cache
from gptcache.adapter import openai as gptcache_openai
from gptcache.embedding import Onnx
from gptcache.manager import get_data_manager, CacheBase
from gptcache.similarity_evaluation.distance import SearchDistanceEvaluation

# Setup
onnx = Onnx()
cache_base = CacheBase('sqlite')
data_manager = get_data_manager(cache_base, vector_base=None)

cache.init(
    embedding_func=onnx.to_embeddings,
    data_manager=data_manager,
    similarity_evaluation=SearchDistanceEvaluation(),
)
cache.set_openai_key()

# Use (drop-in replacement)
response = gptcache_openai.ChatCompletion.create(
    model="gpt-4o-mini",
    messages=[{"role": "user", "content": "What is MLOps?"}],
)
# → First call: API, subsequent similar calls: cache

3. モデルのルーティング

"""Smart model routing — dùng model rẻ khi có thể"""

class ModelRouter:
    def __init__(self):
        self.client = OpenAI()

    def classify_complexity(self, query):
        """Phân loại độ phức tạp của query"""
        # Rule-based (nhanh, free)
        simple_patterns = [
            "dịch", "translate", "define", "what is",
            "list", "yes or no", "true or false",
        ]
        complex_patterns = [
            "analyze", "compare", "explain why",
            "design", "architect", "optimize",
            "code review", "debug",
        ]

        query_lower = query.lower()
        for pattern in simple_patterns:
            if pattern in query_lower:
                return "simple"
        for pattern in complex_patterns:
            if pattern in query_lower:
                return "complex"
        return "medium"

    def route(self, query, messages):
        """Route to appropriate model"""
        complexity = self.classify_complexity(query)

        model_map = {
            "simple": "gpt-4o-mini",      # $0.15/M input — rẻ
            "medium": "gpt-4o-mini",       # $0.15/M input
            "complex": "gpt-4o",           # $2.50/M input — đắt nhưng mạnh
        }

        model = model_map[complexity]
        print(f"🔀 Routing [{complexity}] → {model}")

        response = self.client.chat.completions.create(
            model=model,
            messages=messages,
        )
        return response

router = ModelRouter()

# Simple → gpt-4o-mini (rẻ 16x)
router.route("Dịch sang tiếng Anh: Xin chào", messages=[...])

# Complex → gpt-4o (mạnh)
router.route("Analyze this system architecture and suggest improvements", messages=[...])

カスケード戦略

"""Cascading: thử model rẻ trước, fallback model đắt"""

class CascadingRouter:
    def __init__(self):
        self.client = OpenAI()
        self.models = [
            {"name": "gpt-4o-mini", "max_attempts": 1},
            {"name": "gpt-4o", "max_attempts": 1},
        ]

    def call(self, messages, quality_threshold=0.7):
        """Try cheap model first, escalate if quality is low"""
        for model_config in self.models:
            model = model_config["name"]
            response = self.client.chat.completions.create(
                model=model,
                messages=messages,
            )

            output = response.choices[0].message.content

            # Quick quality check
            quality = self.assess_quality(messages[-1]["content"], output)

            if quality >= quality_threshold:
                print(f"✅ {model} passed quality check ({quality:.2f})")
                return output, model
            else:
                print(f"⚠️ {model} failed quality ({quality:.2f}), escalating...")

        return output, model  # Return last attempt

    def assess_quality(self, question, answer):
        """Quick quality assessment (rule-based for speed)"""
        score = 0.5
        if len(answer) > 50:  score += 0.1
        if "?" not in answer:  score += 0.1  # Not just asking back
        if len(answer) < 2000: score += 0.1  # Not too verbose
        # Add more heuristics
        return min(score, 1.0)

4. トークンの最適化

"""Giảm token usage"""

# Technique 1: Prompt compression
def compress_prompt(system_prompt, max_words=200):
    """Rút gọn prompt giữ ý chính"""
    words = system_prompt.split()
    if len(words) <= max_words:
        return system_prompt

    # Remove filler words
    fillers = {"the", "a", "an", "is", "are", "was", "were",
               "will", "would", "could", "should", "very", "really"}
    compressed = [w for w in words if w.lower() not in fillers]
    return " ".join(compressed[:max_words])

# Technique 2: Context pruning cho RAG
def prune_context(documents, max_tokens=2000):
    """Giữ docs quan trọng nhất, cắt phần dư"""
    import tiktoken
    enc = tiktoken.encoding_for_model("gpt-4o-mini")

    pruned = []
    total_tokens = 0

    for doc in sorted(documents, key=lambda d: d["score"], reverse=True):
        doc_tokens = len(enc.encode(doc["content"]))
        if total_tokens + doc_tokens > max_tokens:
            # Truncate last doc
            remaining = max_tokens - total_tokens
            if remaining > 100:
                truncated = enc.decode(enc.encode(doc["content"])[:remaining])
                pruned.append({**doc, "content": truncated})
            break
        pruned.append(doc)
        total_tokens += doc_tokens

    return pruned

# Technique 3: Structured output (giảm output tokens)
# Thay vì: "The sentiment of this text is positive because..."
# Dùng: {"sentiment": "positive", "confidence": 0.95}
def get_structured_output(text):
    response = client.chat.completions.create(
        model="gpt-4o-mini",
        messages=[
            {"role": "system", "content": "Classify sentiment. Return JSON only: {\"sentiment\": \"positive|negative|neutral\", \"confidence\": 0.0-1.0}"},
            {"role": "user", "content": text},
        ],
        response_format={"type": "json_object"},
        max_tokens=50,  # Giới hạn output
    )
    return response

5. セルフホストモデル

5.1 vLLM — 高スループットのサービス提供

"""vLLM: Serve LLM self-hosted, rất nhanh"""
# pip install vllm

# Start server
# python -m vllm.entrypoints.openai.api_server \
#   --model meta-llama/Llama-3.1-8B-Instruct \
#   --dtype float16 \
#   --max-model-len 8192 \
#   --gpu-memory-utilization 0.9

# Client (OpenAI-compatible API)
from openai import OpenAI

client = OpenAI(
    base_url="http://localhost:8000/v1",
    api_key="not-needed",
)

response = client.chat.completions.create(
    model="meta-llama/Llama-3.1-8B-Instruct",
    messages=[{"role": "user", "content": "MLOps là gì?"}],
)
print(response.choices[0].message.content)
# Cost: $0 per token (chỉ trả tiền GPU)

5.2 オラマ — 地域開発

"""Ollama: Chạy LLM trên laptop"""
# brew install ollama
# ollama pull llama3.1:8b

from openai import OpenAI

client = OpenAI(
    base_url="http://localhost:11434/v1",
    api_key="ollama",
)

response = client.chat.completions.create(
    model="llama3.1:8b",
    messages=[{"role": "user", "content": "Explain Docker"}],
)
# Free, offline, private!

5.3 コストの比較: API とセルフホスト型

Scenario: 1M requests/tháng, 1000 tokens/request

API (GPT-4o-mini):
  1B tokens × $0.375/M = $375/tháng

Self-hosted (Llama 3.1 8B on A100):
  GPU: ~$1.50/hour × 24 × 30 = $1,080/tháng
  But: faster, private, no rate limits

Self-hosted (Llama 3.1 8B on A10G):
  GPU: ~$0.75/hour × 24 × 30 = $540/tháng

Break-even point:
  API rẻ hơn khi: < 2.5M requests/tháng (GPT-4o-mini)
  Self-host rẻ hơn khi: > 2.5M requests/tháng

6. コストの監視と予算編成

"""Cost monitoring system"""
import time
from collections import defaultdict
from datetime import datetime, timedelta

class CostMonitor:
    def __init__(self, daily_budget=50.0):
        self.daily_budget = daily_budget
        self.costs = defaultdict(float)  # date → cost
        self.model_costs = defaultdict(float)  # model → cost

    def log_cost(self, model, input_tokens, output_tokens):
        """Log cost for a call"""
        prices = {
            "gpt-4o": {"input": 2.50/1e6, "output": 10.0/1e6},
            "gpt-4o-mini": {"input": 0.15/1e6, "output": 0.60/1e6},
        }

        p = prices.get(model, prices["gpt-4o-mini"])
        cost = input_tokens * p["input"] + output_tokens * p["output"]

        today = datetime.now().strftime("%Y-%m-%d")
        self.costs[today] += cost
        self.model_costs[model] += cost

        # Budget check
        if self.costs[today] > self.daily_budget:
            self._alert(f"🚨 Daily budget exceeded: ${self.costs[today]:.2f} > ${self.daily_budget}")

        if self.costs[today] > self.daily_budget * 0.8:
            self._alert(f"⚠️ 80% of daily budget used: ${self.costs[today]:.2f}")

        return cost

    def get_report(self):
        """Cost report"""
        today = datetime.now().strftime("%Y-%m-%d")
        return {
            "today_cost": f"${self.costs[today]:.2f}",
            "today_budget_usage": f"{self.costs[today]/self.daily_budget:.0%}",
            "cost_by_model": {k: f"${v:.2f}" for k, v in self.model_costs.items()},
            "total_cost": f"${sum(self.costs.values()):.2f}",
        }

    def _alert(self, message):
        print(message)
        # Send to Slack/PagerDuty/etc.

# Usage
monitor = CostMonitor(daily_budget=50.0)
cost = monitor.log_cost("gpt-4o-mini", input_tokens=500, output_tokens=200)
print(monitor.get_report())

概要

テクニック節約努力トレードオフ
セマンティック キャッシング30-60%🔨🔨キャッシュの古さ
モデル ルーティング40-70%🔨🔨エッジケースの品質
トークンの最適化20-40%🔨プロンプトの長さの制限
構造化された出力30-50%🔨冗長ではない
セルフホスト50-90%🔨🔨🔨🔨インフラ管理
カスケード40-60%🔨🔨レイテンシの追加

演習

  1. セマンティック キャッシュ: セマンティック キャッシュを実装し、キャッシュ ヒット率と正確なキャッシュを比較します。
  2. ルーター: ルーター モデルを構築します (シンプル→ミニ、複雑→4o)。 100 件のクエリでテスト済み。
  3. コスト ダッシュボード: コスト モニターを構築し、1000 回の LLM 呼び出しを実行し、レポートを生成します。
  4. セルフホスト: Ollama 経由で Llama 3.1 8B を展開します。ベンチマークと GPT-4o-mini を比較します。

次の記事: ガードレール、安全性、コンプライアンス。