Chuyển đến nội dung chính

レッスン 23: GPU インフラストラクチャとモデルの提供 — セルフホスト型 LLM デプロイメント

vLLM/TGI を使用したセルフホスト型 LLM 導入、Kubernetes での GPU クラスター管理、モデルのキャッシュと量子化、自動スケーリング推論、コストの最適化。

🏗️ アーキテクチャ — レッスン 23 レッスン 23: GPU インフラストラクチャとモデルの提供 — 自己ホスト型 LLM 導入

エンタープライズ AI チャットボット プラットフォームのアーキテクチャ — プロトタイプから本番まで

パート 7: インフラストラクチャ、セキュリティ、および生産

xdev.asia

1. セルフホスト型 LLM を使用する理由

API ベースの LLM (OpenAI、Claude) はプロトタイピングに適していますが、企業は次の場合にセルフホスト型を必要とします。

因子 APIベース 自己ホスト型
データ主権 送信されたデータ データはオンプレミスにある
規模に応じたコスト $50,000+/月 @ 1,000 万トークン/日 月額 8,000 ~ 15,000 ドルの GPU レンタル
レイテンシ 200-800ms TTFT 50~200ミリ秒のTTFT
カスタマイズ 微調整制限あり フルコントロール微調整 + LoRA
レート制限 プロバイダーが課す GPUによってのみ制限される
コンプライアンス プロバイダーに依存 フルコントロール監査

┌────────── GPU INFRASTRUCTURE STACK ───────────────────┐
│                                                       │
│  ┌─────────────────────────────────────────────────┐   │
│  │             INFERENCE GATEWAY                   │   │
│  │  (Load balancer, auth, rate limiting, routing)  │   │
│  └────────────────┬────────────────────────────────┘   │
│                   │                                    │
│     ┌─────────────┼─────────────┐                      │
│     ▼             ▼             ▼                      │
│  ┌──────┐    ┌──────┐    ┌──────────┐                  │
│  │ vLLM │    │ vLLM │    │   TGI    │                  │
│  │ Llama│    │ Qwen │    │ Mixtral  │                  │
│  │ 3.1  │    │ 2.5  │    │ 8x22B   │                  │
│  │ 70B  │    │ 72B  │    │         │                  │
│  └──┬───┘    └──┬───┘    └────┬─────┘                  │
│     │           │             │                        │
│  ┌──▼───┐    ┌──▼───┐    ┌────▼─────┐                  │
│  │ GPU  │    │ GPU  │    │   GPU    │                  │
│  │ A100 │    │ A100 │    │  H100   │                  │
│  │ 80GB │    │ 80GB │    │  80GB   │                  │
│  └──────┘    └──────┘    └──────────┘                  │
│                                                       │
│  ┌─────────────────────────────────────────────────┐   │
│  │  SHARED INFRASTRUCTURE                          │   │
│  │  • Model Cache (NFS / S3)                       │   │
│  │  • KV Cache (GPU Memory + CPU Offload)          │   │
│  │  • Monitoring (Prometheus + Grafana)             │   │
│  └─────────────────────────────────────────────────┘   │
└───────────────────────────────────────────────────────┘

2. Kubernetes での vLLM のデプロイメント


# vLLM Deployment with GPU
apiVersion: apps/v1
kind: Deployment
metadata:
  name: vllm-llama-70b
  namespace: ai-inference
spec:
  replicas: 2
  selector:
    matchLabels:
      app: vllm-llama-70b
  template:
    metadata:
      labels:
        app: vllm-llama-70b
    spec:
      containers:
        - name: vllm
          image: vllm/vllm-openai:v0.7.0
          args:
            - "--model"
            - "/models/Meta-Llama-3.1-70B-Instruct"
            - "--tensor-parallel-size"
            - "2"  # Split across 2 GPUs
            - "--max-model-len"
            - "8192"
            - "--gpu-memory-utilization"
            - "0.90"
            - "--enable-prefix-caching"  # KV cache reuse
            - "--max-num-seqs"
            - "256"  # Max concurrent sequences
            - "--quantization"
            - "awq"  # 4-bit quantization
            - "--served-model-name"
            - "llama-70b"
          ports:
            - containerPort: 8000
          resources:
            limits:
              nvidia.com/gpu: "2"  # 2x A100 80GB
              memory: "200Gi"
            requests:
              nvidia.com/gpu: "2"
              memory: "160Gi"
          volumeMounts:
            - name: model-cache
              mountPath: /models
            - name: shm
              mountPath: /dev/shm
          env:
            - name: NCCL_P2P_DISABLE
              value: "0"
            - name: CUDA_VISIBLE_DEVICES
              value: "0,1"
          readinessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 120
            periodSeconds: 10
          livenessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 180
            periodSeconds: 30
      volumes:
        - name: model-cache
          persistentVolumeClaim:
            claimName: model-cache-pvc
        - name: shm
          emptyDir:
            medium: Memory
            sizeLimit: 16Gi
      nodeSelector:
        nvidia.com/gpu.product: "NVIDIA-A100-SXM4-80GB"
      tolerations:
        - key: nvidia.com/gpu
          operator: Exists
          effect: NoSchedule
---
# HPA for GPU-based scaling
apiVersion: autoscaling/v2
kind: HorizontalPodAutoscaler
metadata:
  name: vllm-llama-70b-hpa
  namespace: ai-inference
spec:
  scaleTargetRef:
    apiVersion: apps/v1
    kind: Deployment
    name: vllm-llama-70b
  minReplicas: 1
  maxReplicas: 4
  metrics:
    - type: Pods
      pods:
        metric:
          name: vllm_num_requests_waiting
        target:
          type: AverageValue
          averageValue: "50"  # Scale up when queue > 50
    - type: Pods
      pods:
        metric:
          name: vllm_gpu_cache_usage_perc
        target:
          type: AverageValue
          averageValue: "85"  # Scale up when KV cache > 85%

3. 推論ゲートウェイ — ルーティングと負荷分散


class InferenceGateway {
  private backends: Map<string, InferenceBackend[]> = new Map();

  constructor(
    private readonly config: GatewayConfig,
    private readonly metrics: MetricsCollector,
  ) {
    this.initializeBackends();
  }

  async route(request: InferenceRequest): Promise<InferenceResponse> {
    const model = request.model;
    const backends = this.backends.get(model);
    if (!backends?.length) {
      throw new Error(`No backend available for model: ${model}`);
    }

    // Select backend based on strategy
    const backend = this.selectBackend(backends, request);

    const startTime = Date.now();
    try {
      const response = await backend.infer(request);

      // Record metrics
      this.metrics.recordInference({
        model,
        backend: backend.id,
        latencyMs: Date.now() - startTime,
        inputTokens: response.usage.promptTokens,
        outputTokens: response.usage.completionTokens,
        queueDepth: backend.queueDepth,
      });

      return response;
    } catch (error) {
      // Fallback to API provider if self-hosted fails
      if (this.config.fallbackToAPI) {
        return this.fallbackToAPI(request, model);
      }
      throw error;
    }
  }

  private selectBackend(
    backends: InferenceBackend[],
    request: InferenceRequest,
  ): InferenceBackend {
    // Filter healthy backends
    const healthy = backends.filter(b => b.isHealthy());
    if (healthy.length === 0) {
      throw new Error('No healthy backends available');
    }

    // Least-connections with queue awareness
    return healthy.reduce((best, current) => {
      const bestScore = best.activeRequests + best.queueDepth * 2;
      const currentScore = current.activeRequests + current.queueDepth * 2;
      return currentScore < bestScore ? current : best;
    });
  }

  // Graceful fallback to API providers
  private async fallbackToAPI(
    request: InferenceRequest,
    failedModel: string,
  ): Promise<InferenceResponse> {
    const modelMap: Record<string, { provider: string; model: string }> = {
      'llama-70b': { provider: 'together', model: 'meta-llama/Llama-3.1-70B-Instruct' },
      'qwen-72b': { provider: 'fireworks', model: 'accounts/fireworks/models/qwen2p5-72b' },
    };

    const fallback = modelMap[failedModel];
    if (!fallback) throw new Error(`No API fallback for model: ${failedModel}`);

    this.metrics.recordFallback({ model: failedModel, reason: 'backend_unavailable' });

    return this.apiProviders.get(fallback.provider)!.infer({
      ...request,
      model: fallback.model,
    });
  }
}

4. モデルの最適化 — 量子化とキャッシュ


class ModelOptimizer {
  // AWQ Quantization script
  async quantizeModel(config: QuantizationConfig): Promise<void> {
    // AWQ: 4-bit quantization with minimal quality loss
    const command = [
      'python', '-m', 'awq.entry',
      '--model_path', config.inputPath,
      '--w_bit', String(config.bits),       // 4
      '--q_group_size', String(config.groupSize), // 128
      '--output_path', config.outputPath,
    ];
    await this.exec(command);
  }
}

// Semantic Cache — avoid redundant inference
class SemanticInferenceCache {
  constructor(
    private readonly vectorStore: VectorStore,
    private readonly cache: Redis,
  ) {}

  async getCachedResponse(
    prompt: string,
    model: string,
    threshold = 0.95,
  ): Promise<CachedInference | null> {
    const embedding = await this.embedder.embed(prompt);

    const results = await this.vectorStore.search({
      vector: embedding,
      filter: { model },
      topK: 1,
    });

    if (results.length > 0 && results[0].score >= threshold) {
      const cached = await this.cache.get(
        `inference:${results[0].id}`,
      );
      if (cached) {
        return JSON.parse(cached);
      }
    }

    return null;
  }

  async cacheResponse(
    prompt: string,
    model: string,
    response: InferenceResponse,
    ttl = 3600,
  ): Promise<void> {
    const embedding = await this.embedder.embed(prompt);
    const id = crypto.randomUUID();

    await this.vectorStore.upsert({
      id,
      vector: embedding,
      metadata: { model },
    });

    await this.cache.setex(
      `inference:${id}`,
      ttl,
      JSON.stringify({ prompt, response, cachedAt: new Date() }),
    );
  }
}

5. GPU モニタリングと自動スケーリング


# Prometheus rules for GPU monitoring
groups:
  - name: gpu-inference-alerts
    rules:
      # GPU utilization below threshold — consider scaling down
      - alert: GPUUnderutilized
        expr: avg(DCGM_FI_DEV_GPU_UTIL{namespace="ai-inference"}) < 30
        for: 30m
        labels:
          severity: info
        annotations:
          summary: "GPU utilization below 30% for 30m — consider scaling down"

      # KV cache pressure — scale up
      - alert: KVCachePressure
        expr: vllm_gpu_cache_usage_perc > 90
        for: 5m
        labels:
          severity: warning
        annotations:
          summary: "KV cache usage > 90% — inference quality may degrade"

      # Request queue growing — scale up
      - alert: InferenceQueueBacklog
        expr: vllm_num_requests_waiting > 100
        for: 2m
        labels:
          severity: critical
        annotations:
          summary: "Inference queue backlog > 100 requests"

      # GPU memory OOM risk
      - alert: GPUMemoryHigh
        expr: DCGM_FI_DEV_FB_USED / DCGM_FI_DEV_FB_FREE > 9
        for: 5m
        labels:
          severity: critical
        annotations:
          summary: "GPU memory > 90% used — OOM risk"

// Cost Tracker for GPU vs API comparison
class GPUCostTracker {
  calculateMonthlyCost(config: GPUClusterConfig): GPUCostBreakdown {
    const gpuHourlyRate: Record<string, number> = {
      'A100-80GB': 3.50,  // Cloud hourly rate
      'H100-80GB': 5.50,
      'L40S-48GB': 1.80,
    };

    const gpuCost = config.gpuCount
      * (gpuHourlyRate[config.gpuType] ?? 3.50)
      * 24 * 30; // 24/7 operation

    const storageCost = config.modelStorageGB * 0.10; // per GB/month
    const networkCost = config.egressGB * 0.08;

    return {
      gpuCompute: gpuCost,
      storage: storageCost,
      network: networkCost,
      total: gpuCost + storageCost + networkCost,
      costPerMillionTokens: this.estimateCostPerToken(config, gpuCost),
    };
  }

  private estimateCostPerToken(
    config: GPUClusterConfig,
    monthlyCost: number,
  ): number {
    // Estimate throughput based on model and GPU
    const tokensPerSecond: Record<string, number> = {
      'llama-70b-A100': 35,
      'llama-70b-H100': 80,
      'qwen-72b-A100': 30,
    };

    const key = `${config.modelName}-${config.gpuType.split('-')[0]}`;
    const tps = tokensPerSecond[key] ?? 30;
    const monthlyTokens = tps * 60 * 60 * 24 * 30 * config.gpuCount;

    return (monthlyCost / monthlyTokens) * 1_000_000;
  }
}

レッスン 23 のまとめ

  • 自己ホスト型 LLM: データ主権、規模に応じたコスト、低遅延のために必要
  • K8 上の vLLM: テンソル並列、AWQ 量子化、プレフィックス キャッシュ、GPU ノード セレクター
  • 推論ゲートウェイ: 最小接続ルーティング、ヘルスチェック、API フォールバック
  • 最適化: AWQ 4 ビット量子化、セマンティック キャッシュ (95% 類似度しきい値)
  • モニタリング: GPU 使用率、KV キャッシュ プレッシャー、キュー バックログ、OOM アラート

次の記事: セキュリティとコンプライアンス — エンドツーエンドの暗号化、監査ログ、GDPR/HIPAA 準拠、AI システムの侵入テスト。