Chuyển đến nội dung chính

第 23 課:GPU 基礎架構與模型服務 — 自託管 LLM 部署

使用 vLLM/TGI 進行自架 LLM 部署、Kubernetes 上的 GPU 叢集管理、模型快取和量化、自動縮放推理、成本最佳化。

🏗️ 建築 — 第 23 課 第 23 課:GPU 基礎設施與模型服務 — 自託管 LLM 部署

企業人工智慧聊天機器人平台架構-從原型到生產

第 7 部分:基礎設施、安全與生產

亞洲開發網

1. 為什麼選擇自辦法學碩士?

基於API的LLM(OpenAI,Claude)適合原型設計,但企業在以下情況下需要自架:

因素 基於API 自託管
數據主權 數據發送出去 資料是本地的
規模成本 $50K+/月 @ 10M 代幣/天 GPU 租金為 8K-15K 美元/月
延遲 200-800ms TFT 50-200ms TTFT
客製化 微調有限 全控微調+LoRA
速率限制 提供者強加的 僅受 GPU 限制
合規性 依賴提供者 全面控制審核

┌────────── GPU INFRASTRUCTURE STACK ───────────────────┐
│                                                       │
│  ┌─────────────────────────────────────────────────┐   │
│  │             INFERENCE GATEWAY                   │   │
│  │  (Load balancer, auth, rate limiting, routing)  │   │
│  └────────────────┬────────────────────────────────┘   │
│                   │                                    │
│     ┌─────────────┼─────────────┐                      │
│     ▼             ▼             ▼                      │
│  ┌──────┐    ┌──────┐    ┌──────────┐                  │
│  │ vLLM │    │ vLLM │    │   TGI    │                  │
│  │ Llama│    │ Qwen │    │ Mixtral  │                  │
│  │ 3.1  │    │ 2.5  │    │ 8x22B   │                  │
│  │ 70B  │    │ 72B  │    │         │                  │
│  └──┬───┘    └──┬───┘    └────┬─────┘                  │
│     │           │             │                        │
│  ┌──▼───┐    ┌──▼───┐    ┌────▼─────┐                  │
│  │ GPU  │    │ GPU  │    │   GPU    │                  │
│  │ A100 │    │ A100 │    │  H100   │                  │
│  │ 80GB │    │ 80GB │    │  80GB   │                  │
│  └──────┘    └──────┘    └──────────┘                  │
│                                                       │
│  ┌─────────────────────────────────────────────────┐   │
│  │  SHARED INFRASTRUCTURE                          │   │
│  │  • Model Cache (NFS / S3)                       │   │
│  │  • KV Cache (GPU Memory + CPU Offload)          │   │
│  │  • Monitoring (Prometheus + Grafana)             │   │
│  └─────────────────────────────────────────────────┘   │
└───────────────────────────────────────────────────────┘

2.vLLM在Kubernetes上的部署


# vLLM Deployment with GPU
apiVersion: apps/v1
kind: Deployment
metadata:
  name: vllm-llama-70b
  namespace: ai-inference
spec:
  replicas: 2
  selector:
    matchLabels:
      app: vllm-llama-70b
  template:
    metadata:
      labels:
        app: vllm-llama-70b
    spec:
      containers:
        - name: vllm
          image: vllm/vllm-openai:v0.7.0
          args:
            - "--model"
            - "/models/Meta-Llama-3.1-70B-Instruct"
            - "--tensor-parallel-size"
            - "2"  # Split across 2 GPUs
            - "--max-model-len"
            - "8192"
            - "--gpu-memory-utilization"
            - "0.90"
            - "--enable-prefix-caching"  # KV cache reuse
            - "--max-num-seqs"
            - "256"  # Max concurrent sequences
            - "--quantization"
            - "awq"  # 4-bit quantization
            - "--served-model-name"
            - "llama-70b"
          ports:
            - containerPort: 8000
          resources:
            limits:
              nvidia.com/gpu: "2"  # 2x A100 80GB
              memory: "200Gi"
            requests:
              nvidia.com/gpu: "2"
              memory: "160Gi"
          volumeMounts:
            - name: model-cache
              mountPath: /models
            - name: shm
              mountPath: /dev/shm
          env:
            - name: NCCL_P2P_DISABLE
              value: "0"
            - name: CUDA_VISIBLE_DEVICES
              value: "0,1"
          readinessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 120
            periodSeconds: 10
          livenessProbe:
            httpGet:
              path: /health
              port: 8000
            initialDelaySeconds: 180
            periodSeconds: 30
      volumes:
        - name: model-cache
          persistentVolumeClaim:
            claimName: model-cache-pvc
        - name: shm
          emptyDir:
            medium: Memory
            sizeLimit: 16Gi
      nodeSelector:
        nvidia.com/gpu.product: "NVIDIA-A100-SXM4-80GB"
      tolerations:
        - key: nvidia.com/gpu
          operator: Exists
          effect: NoSchedule
---
# HPA for GPU-based scaling
apiVersion: autoscaling/v2
kind: HorizontalPodAutoscaler
metadata:
  name: vllm-llama-70b-hpa
  namespace: ai-inference
spec:
  scaleTargetRef:
    apiVersion: apps/v1
    kind: Deployment
    name: vllm-llama-70b
  minReplicas: 1
  maxReplicas: 4
  metrics:
    - type: Pods
      pods:
        metric:
          name: vllm_num_requests_waiting
        target:
          type: AverageValue
          averageValue: "50"  # Scale up when queue > 50
    - type: Pods
      pods:
        metric:
          name: vllm_gpu_cache_usage_perc
        target:
          type: AverageValue
          averageValue: "85"  # Scale up when KV cache > 85%

3. 推理網關-路由與負載平衡


class InferenceGateway {
  private backends: Map<string, InferenceBackend[]> = new Map();

  constructor(
    private readonly config: GatewayConfig,
    private readonly metrics: MetricsCollector,
  ) {
    this.initializeBackends();
  }

  async route(request: InferenceRequest): Promise<InferenceResponse> {
    const model = request.model;
    const backends = this.backends.get(model);
    if (!backends?.length) {
      throw new Error(`No backend available for model: ${model}`);
    }

    // Select backend based on strategy
    const backend = this.selectBackend(backends, request);

    const startTime = Date.now();
    try {
      const response = await backend.infer(request);

      // Record metrics
      this.metrics.recordInference({
        model,
        backend: backend.id,
        latencyMs: Date.now() - startTime,
        inputTokens: response.usage.promptTokens,
        outputTokens: response.usage.completionTokens,
        queueDepth: backend.queueDepth,
      });

      return response;
    } catch (error) {
      // Fallback to API provider if self-hosted fails
      if (this.config.fallbackToAPI) {
        return this.fallbackToAPI(request, model);
      }
      throw error;
    }
  }

  private selectBackend(
    backends: InferenceBackend[],
    request: InferenceRequest,
  ): InferenceBackend {
    // Filter healthy backends
    const healthy = backends.filter(b => b.isHealthy());
    if (healthy.length === 0) {
      throw new Error('No healthy backends available');
    }

    // Least-connections with queue awareness
    return healthy.reduce((best, current) => {
      const bestScore = best.activeRequests + best.queueDepth * 2;
      const currentScore = current.activeRequests + current.queueDepth * 2;
      return currentScore < bestScore ? current : best;
    });
  }

  // Graceful fallback to API providers
  private async fallbackToAPI(
    request: InferenceRequest,
    failedModel: string,
  ): Promise<InferenceResponse> {
    const modelMap: Record<string, { provider: string; model: string }> = {
      'llama-70b': { provider: 'together', model: 'meta-llama/Llama-3.1-70B-Instruct' },
      'qwen-72b': { provider: 'fireworks', model: 'accounts/fireworks/models/qwen2p5-72b' },
    };

    const fallback = modelMap[failedModel];
    if (!fallback) throw new Error(`No API fallback for model: ${failedModel}`);

    this.metrics.recordFallback({ model: failedModel, reason: 'backend_unavailable' });

    return this.apiProviders.get(fallback.provider)!.infer({
      ...request,
      model: fallback.model,
    });
  }
}

4. 模型最佳化-量化與緩存


class ModelOptimizer {
  // AWQ Quantization script
  async quantizeModel(config: QuantizationConfig): Promise<void> {
    // AWQ: 4-bit quantization with minimal quality loss
    const command = [
      'python', '-m', 'awq.entry',
      '--model_path', config.inputPath,
      '--w_bit', String(config.bits),       // 4
      '--q_group_size', String(config.groupSize), // 128
      '--output_path', config.outputPath,
    ];
    await this.exec(command);
  }
}

// Semantic Cache — avoid redundant inference
class SemanticInferenceCache {
  constructor(
    private readonly vectorStore: VectorStore,
    private readonly cache: Redis,
  ) {}

  async getCachedResponse(
    prompt: string,
    model: string,
    threshold = 0.95,
  ): Promise<CachedInference | null> {
    const embedding = await this.embedder.embed(prompt);

    const results = await this.vectorStore.search({
      vector: embedding,
      filter: { model },
      topK: 1,
    });

    if (results.length > 0 && results[0].score >= threshold) {
      const cached = await this.cache.get(
        `inference:${results[0].id}`,
      );
      if (cached) {
        return JSON.parse(cached);
      }
    }

    return null;
  }

  async cacheResponse(
    prompt: string,
    model: string,
    response: InferenceResponse,
    ttl = 3600,
  ): Promise<void> {
    const embedding = await this.embedder.embed(prompt);
    const id = crypto.randomUUID();

    await this.vectorStore.upsert({
      id,
      vector: embedding,
      metadata: { model },
    });

    await this.cache.setex(
      `inference:${id}`,
      ttl,
      JSON.stringify({ prompt, response, cachedAt: new Date() }),
    );
  }
}

5.GPU監控與自動縮放


# Prometheus rules for GPU monitoring
groups:
  - name: gpu-inference-alerts
    rules:
      # GPU utilization below threshold — consider scaling down
      - alert: GPUUnderutilized
        expr: avg(DCGM_FI_DEV_GPU_UTIL{namespace="ai-inference"}) < 30
        for: 30m
        labels:
          severity: info
        annotations:
          summary: "GPU utilization below 30% for 30m — consider scaling down"

      # KV cache pressure — scale up
      - alert: KVCachePressure
        expr: vllm_gpu_cache_usage_perc > 90
        for: 5m
        labels:
          severity: warning
        annotations:
          summary: "KV cache usage > 90% — inference quality may degrade"

      # Request queue growing — scale up
      - alert: InferenceQueueBacklog
        expr: vllm_num_requests_waiting > 100
        for: 2m
        labels:
          severity: critical
        annotations:
          summary: "Inference queue backlog > 100 requests"

      # GPU memory OOM risk
      - alert: GPUMemoryHigh
        expr: DCGM_FI_DEV_FB_USED / DCGM_FI_DEV_FB_FREE > 9
        for: 5m
        labels:
          severity: critical
        annotations:
          summary: "GPU memory > 90% used — OOM risk"

// Cost Tracker for GPU vs API comparison
class GPUCostTracker {
  calculateMonthlyCost(config: GPUClusterConfig): GPUCostBreakdown {
    const gpuHourlyRate: Record<string, number> = {
      'A100-80GB': 3.50,  // Cloud hourly rate
      'H100-80GB': 5.50,
      'L40S-48GB': 1.80,
    };

    const gpuCost = config.gpuCount
      * (gpuHourlyRate[config.gpuType] ?? 3.50)
      * 24 * 30; // 24/7 operation

    const storageCost = config.modelStorageGB * 0.10; // per GB/month
    const networkCost = config.egressGB * 0.08;

    return {
      gpuCompute: gpuCost,
      storage: storageCost,
      network: networkCost,
      total: gpuCost + storageCost + networkCost,
      costPerMillionTokens: this.estimateCostPerToken(config, gpuCost),
    };
  }

  private estimateCostPerToken(
    config: GPUClusterConfig,
    monthlyCost: number,
  ): number {
    // Estimate throughput based on model and GPU
    const tokensPerSecond: Record<string, number> = {
      'llama-70b-A100': 35,
      'llama-70b-H100': 80,
      'qwen-72b-A100': 30,
    };

    const key = `${config.modelName}-${config.gpuType.split('-')[0]}`;
    const tps = tokensPerSecond[key] ?? 30;
    const monthlyTokens = tps * 60 * 60 * 24 * 30 * config.gpuCount;

    return (monthlyCost / monthlyTokens) * 1_000_000;
  }
}

第 23 課總結

  • 自主辦法學碩士:資料主權、規模成本和低延遲所需
  • K8s 上的 vLLM:張量並行、AWQ量化、前綴快取、GPU節點選擇器
  • 推理網關:最少連線路由、健康檢查、API 回退
  • 最佳化:AWQ 4位元量化,語意緩存(95%相似度閾值)
  • 監控:GPU利用率、KV緩存壓力、佇列積壓、OOM警報

下一篇: 安全性與合規性 — 端對端加密、審核日誌記錄、GDPR/HIPAA 合規性、AI 系統滲透測試。