1. 為什麼選擇自辦法學碩士?
基於API的LLM(OpenAI,Claude)適合原型設計,但企業在以下情況下需要自架:
| 因素 | 基於API | 自託管 |
|---|---|---|
| 數據主權 | 數據發送出去 | 資料是本地的 |
| 規模成本 | $50K+/月 @ 10M 代幣/天 | GPU 租金為 8K-15K 美元/月 |
| 延遲 | 200-800ms TFT | 50-200ms TTFT |
| 客製化 | 微調有限 | 全控微調+LoRA |
| 速率限制 | 提供者強加的 | 僅受 GPU 限制 |
| 合規性 | 依賴提供者 | 全面控制審核 |
┌────────── GPU INFRASTRUCTURE STACK ───────────────────┐
│ │
│ ┌─────────────────────────────────────────────────┐ │
│ │ INFERENCE GATEWAY │ │
│ │ (Load balancer, auth, rate limiting, routing) │ │
│ └────────────────┬────────────────────────────────┘ │
│ │ │
│ ┌─────────────┼─────────────┐ │
│ ▼ ▼ ▼ │
│ ┌──────┐ ┌──────┐ ┌──────────┐ │
│ │ vLLM │ │ vLLM │ │ TGI │ │
│ │ Llama│ │ Qwen │ │ Mixtral │ │
│ │ 3.1 │ │ 2.5 │ │ 8x22B │ │
│ │ 70B │ │ 72B │ │ │ │
│ └──┬───┘ └──┬───┘ └────┬─────┘ │
│ │ │ │ │
│ ┌──▼───┐ ┌──▼───┐ ┌────▼─────┐ │
│ │ GPU │ │ GPU │ │ GPU │ │
│ │ A100 │ │ A100 │ │ H100 │ │
│ │ 80GB │ │ 80GB │ │ 80GB │ │
│ └──────┘ └──────┘ └──────────┘ │
│ │
│ ┌─────────────────────────────────────────────────┐ │
│ │ SHARED INFRASTRUCTURE │ │
│ │ • Model Cache (NFS / S3) │ │
│ │ • KV Cache (GPU Memory + CPU Offload) │ │
│ │ • Monitoring (Prometheus + Grafana) │ │
│ └─────────────────────────────────────────────────┘ │
└───────────────────────────────────────────────────────┘
2.vLLM在Kubernetes上的部署
# vLLM Deployment with GPU
apiVersion: apps/v1
kind: Deployment
metadata:
name: vllm-llama-70b
namespace: ai-inference
spec:
replicas: 2
selector:
matchLabels:
app: vllm-llama-70b
template:
metadata:
labels:
app: vllm-llama-70b
spec:
containers:
- name: vllm
image: vllm/vllm-openai:v0.7.0
args:
- "--model"
- "/models/Meta-Llama-3.1-70B-Instruct"
- "--tensor-parallel-size"
- "2" # Split across 2 GPUs
- "--max-model-len"
- "8192"
- "--gpu-memory-utilization"
- "0.90"
- "--enable-prefix-caching" # KV cache reuse
- "--max-num-seqs"
- "256" # Max concurrent sequences
- "--quantization"
- "awq" # 4-bit quantization
- "--served-model-name"
- "llama-70b"
ports:
- containerPort: 8000
resources:
limits:
nvidia.com/gpu: "2" # 2x A100 80GB
memory: "200Gi"
requests:
nvidia.com/gpu: "2"
memory: "160Gi"
volumeMounts:
- name: model-cache
mountPath: /models
- name: shm
mountPath: /dev/shm
env:
- name: NCCL_P2P_DISABLE
value: "0"
- name: CUDA_VISIBLE_DEVICES
value: "0,1"
readinessProbe:
httpGet:
path: /health
port: 8000
initialDelaySeconds: 120
periodSeconds: 10
livenessProbe:
httpGet:
path: /health
port: 8000
initialDelaySeconds: 180
periodSeconds: 30
volumes:
- name: model-cache
persistentVolumeClaim:
claimName: model-cache-pvc
- name: shm
emptyDir:
medium: Memory
sizeLimit: 16Gi
nodeSelector:
nvidia.com/gpu.product: "NVIDIA-A100-SXM4-80GB"
tolerations:
- key: nvidia.com/gpu
operator: Exists
effect: NoSchedule
---
# HPA for GPU-based scaling
apiVersion: autoscaling/v2
kind: HorizontalPodAutoscaler
metadata:
name: vllm-llama-70b-hpa
namespace: ai-inference
spec:
scaleTargetRef:
apiVersion: apps/v1
kind: Deployment
name: vllm-llama-70b
minReplicas: 1
maxReplicas: 4
metrics:
- type: Pods
pods:
metric:
name: vllm_num_requests_waiting
target:
type: AverageValue
averageValue: "50" # Scale up when queue > 50
- type: Pods
pods:
metric:
name: vllm_gpu_cache_usage_perc
target:
type: AverageValue
averageValue: "85" # Scale up when KV cache > 85%
3. 推理網關-路由與負載平衡
class InferenceGateway {
private backends: Map<string, InferenceBackend[]> = new Map();
constructor(
private readonly config: GatewayConfig,
private readonly metrics: MetricsCollector,
) {
this.initializeBackends();
}
async route(request: InferenceRequest): Promise<InferenceResponse> {
const model = request.model;
const backends = this.backends.get(model);
if (!backends?.length) {
throw new Error(`No backend available for model: ${model}`);
}
// Select backend based on strategy
const backend = this.selectBackend(backends, request);
const startTime = Date.now();
try {
const response = await backend.infer(request);
// Record metrics
this.metrics.recordInference({
model,
backend: backend.id,
latencyMs: Date.now() - startTime,
inputTokens: response.usage.promptTokens,
outputTokens: response.usage.completionTokens,
queueDepth: backend.queueDepth,
});
return response;
} catch (error) {
// Fallback to API provider if self-hosted fails
if (this.config.fallbackToAPI) {
return this.fallbackToAPI(request, model);
}
throw error;
}
}
private selectBackend(
backends: InferenceBackend[],
request: InferenceRequest,
): InferenceBackend {
// Filter healthy backends
const healthy = backends.filter(b => b.isHealthy());
if (healthy.length === 0) {
throw new Error('No healthy backends available');
}
// Least-connections with queue awareness
return healthy.reduce((best, current) => {
const bestScore = best.activeRequests + best.queueDepth * 2;
const currentScore = current.activeRequests + current.queueDepth * 2;
return currentScore < bestScore ? current : best;
});
}
// Graceful fallback to API providers
private async fallbackToAPI(
request: InferenceRequest,
failedModel: string,
): Promise<InferenceResponse> {
const modelMap: Record<string, { provider: string; model: string }> = {
'llama-70b': { provider: 'together', model: 'meta-llama/Llama-3.1-70B-Instruct' },
'qwen-72b': { provider: 'fireworks', model: 'accounts/fireworks/models/qwen2p5-72b' },
};
const fallback = modelMap[failedModel];
if (!fallback) throw new Error(`No API fallback for model: ${failedModel}`);
this.metrics.recordFallback({ model: failedModel, reason: 'backend_unavailable' });
return this.apiProviders.get(fallback.provider)!.infer({
...request,
model: fallback.model,
});
}
}
4. 模型最佳化-量化與緩存
class ModelOptimizer {
// AWQ Quantization script
async quantizeModel(config: QuantizationConfig): Promise<void> {
// AWQ: 4-bit quantization with minimal quality loss
const command = [
'python', '-m', 'awq.entry',
'--model_path', config.inputPath,
'--w_bit', String(config.bits), // 4
'--q_group_size', String(config.groupSize), // 128
'--output_path', config.outputPath,
];
await this.exec(command);
}
}
// Semantic Cache — avoid redundant inference
class SemanticInferenceCache {
constructor(
private readonly vectorStore: VectorStore,
private readonly cache: Redis,
) {}
async getCachedResponse(
prompt: string,
model: string,
threshold = 0.95,
): Promise<CachedInference | null> {
const embedding = await this.embedder.embed(prompt);
const results = await this.vectorStore.search({
vector: embedding,
filter: { model },
topK: 1,
});
if (results.length > 0 && results[0].score >= threshold) {
const cached = await this.cache.get(
`inference:${results[0].id}`,
);
if (cached) {
return JSON.parse(cached);
}
}
return null;
}
async cacheResponse(
prompt: string,
model: string,
response: InferenceResponse,
ttl = 3600,
): Promise<void> {
const embedding = await this.embedder.embed(prompt);
const id = crypto.randomUUID();
await this.vectorStore.upsert({
id,
vector: embedding,
metadata: { model },
});
await this.cache.setex(
`inference:${id}`,
ttl,
JSON.stringify({ prompt, response, cachedAt: new Date() }),
);
}
}
5.GPU監控與自動縮放
# Prometheus rules for GPU monitoring
groups:
- name: gpu-inference-alerts
rules:
# GPU utilization below threshold — consider scaling down
- alert: GPUUnderutilized
expr: avg(DCGM_FI_DEV_GPU_UTIL{namespace="ai-inference"}) < 30
for: 30m
labels:
severity: info
annotations:
summary: "GPU utilization below 30% for 30m — consider scaling down"
# KV cache pressure — scale up
- alert: KVCachePressure
expr: vllm_gpu_cache_usage_perc > 90
for: 5m
labels:
severity: warning
annotations:
summary: "KV cache usage > 90% — inference quality may degrade"
# Request queue growing — scale up
- alert: InferenceQueueBacklog
expr: vllm_num_requests_waiting > 100
for: 2m
labels:
severity: critical
annotations:
summary: "Inference queue backlog > 100 requests"
# GPU memory OOM risk
- alert: GPUMemoryHigh
expr: DCGM_FI_DEV_FB_USED / DCGM_FI_DEV_FB_FREE > 9
for: 5m
labels:
severity: critical
annotations:
summary: "GPU memory > 90% used — OOM risk"
// Cost Tracker for GPU vs API comparison
class GPUCostTracker {
calculateMonthlyCost(config: GPUClusterConfig): GPUCostBreakdown {
const gpuHourlyRate: Record<string, number> = {
'A100-80GB': 3.50, // Cloud hourly rate
'H100-80GB': 5.50,
'L40S-48GB': 1.80,
};
const gpuCost = config.gpuCount
* (gpuHourlyRate[config.gpuType] ?? 3.50)
* 24 * 30; // 24/7 operation
const storageCost = config.modelStorageGB * 0.10; // per GB/month
const networkCost = config.egressGB * 0.08;
return {
gpuCompute: gpuCost,
storage: storageCost,
network: networkCost,
total: gpuCost + storageCost + networkCost,
costPerMillionTokens: this.estimateCostPerToken(config, gpuCost),
};
}
private estimateCostPerToken(
config: GPUClusterConfig,
monthlyCost: number,
): number {
// Estimate throughput based on model and GPU
const tokensPerSecond: Record<string, number> = {
'llama-70b-A100': 35,
'llama-70b-H100': 80,
'qwen-72b-A100': 30,
};
const key = `${config.modelName}-${config.gpuType.split('-')[0]}`;
const tps = tokensPerSecond[key] ?? 30;
const monthlyTokens = tps * 60 * 60 * 24 * 30 * config.gpuCount;
return (monthlyCost / monthlyTokens) * 1_000_000;
}
}
第 23 課總結
- 自主辦法學碩士:資料主權、規模成本和低延遲所需
- K8s 上的 vLLM:張量並行、AWQ量化、前綴快取、GPU節點選擇器
- 推理網關:最少連線路由、健康檢查、API 回退
- 最佳化:AWQ 4位元量化,語意緩存(95%相似度閾值)
- 監控:GPU利用率、KV緩存壓力、佇列積壓、OOM警報
下一篇: 安全性與合規性 — 端對端加密、審核日誌記錄、GDPR/HIPAA 合規性、AI 系統滲透測試。