Tổng quan
Thay vì tự deploy model, bạn thường sẽ gọi LLM qua API. Bài này hướng dẫn thực tế với 3 provider lớn nhất: OpenAI, Anthropic và Google Gemini — covering streaming, vision, structured output và cost optimization.
1. So sánh Providers (2025)
| Provider | Model | Context | Điểm mạnh |
|---|---|---|---|
| OpenAI | GPT-4o, o3 | 128K | Tool use, coding, ecosystem lớn |
| Anthropic | Claude Opus/Sonnet | 200K | Long context, safety, reasoning |
| Gemini 2.0 | 1M | Multimodal, free tier, long context | |
| Meta (via API) | LLaMA 3.3 | 128K | Open weights, self-hostable |
| Mistral | Mistral Large | 128K | EU data, affordable |
2. OpenAI API
Setup
pip install openai
export OPENAI_API_KEY="sk-..."
Chat Completions
from openai import OpenAI
client = OpenAI() # tự đọc OPENAI_API_KEY từ env
# Basic call
response = client.chat.completions.create(
model="gpt-4o",
messages=[
{"role": "system", "content": "Bạn là assistant lập trình Python."},
{"role": "user", "content": "Giải thích list comprehension với ví dụ"}
],
max_tokens=500,
temperature=0.3
)
print(response.choices[0].message.content)
print(f"Tokens dùng: {response.usage.total_tokens}")
Streaming
# Streaming — in ra từng token ngay khi nhận được
with client.chat.completions.stream(
model="gpt-4o",
messages=[{"role": "user", "content": "Viết bài thơ về Hà Nội"}]
) as stream:
for text in stream.text_stream:
print(text, end="", flush=True)
print() # Newline cuối
Structured Output với Pydantic
from pydantic import BaseModel
from typing import List
class CodeReview(BaseModel):
bugs: List[str]
improvements: List[str]
security_issues: List[str]
overall_score: int # 1-10
response = client.beta.chat.completions.parse(
model="gpt-4o",
messages=[
{"role": "user", "content": """Review code này:
```python
import os
password = os.environ.get('PASSWORD', 'admin123')
query = f"SELECT * FROM users WHERE id = {user_id}"
```"""}
],
response_format=CodeReview
)
review = response.choices[0].message.parsed
print(f"Bugs: {review.bugs}")
print(f"Security: {review.security_issues}")
print(f"Score: {review.overall_score}/10")
Vision (Image Input)
import base64
from pathlib import Path
def analyze_image(image_path: str, question: str) -> str:
image_data = base64.b64encode(Path(image_path).read_bytes()).decode()
response = client.chat.completions.create(
model="gpt-4o",
messages=[{
"role": "user",
"content": [
{"type": "text", "text": question},
{
"type": "image_url",
"image_url": {
"url": f"data:image/jpeg;base64,{image_data}",
"detail": "high" # "low" tốn ít tokens hơn
}
}
]
}]
)
return response.choices[0].message.content
# Hoặc dùng URL trực tiếp
response = client.chat.completions.create(
model="gpt-4o",
messages=[{
"role": "user",
"content": [
{"type": "text", "text": "Mô tả ảnh này"},
{"type": "image_url", "image_url": {"url": "https://example.com/image.jpg"}}
]
}]
)
3. Anthropic Claude API
Setup
pip install anthropic
export ANTHROPIC_API_KEY="sk-ant-..."
Messages API
import anthropic
client = anthropic.Anthropic()
# Basic
message = client.messages.create(
model="claude-opus-4-5",
max_tokens=1024,
system="Bạn là expert về distributed systems.",
messages=[
{"role": "user", "content": "Giải thích CAP theorem"}
]
)
print(message.content[0].text)
print(f"Input tokens: {message.usage.input_tokens}")
print(f"Output tokens: {message.usage.output_tokens}")
Streaming
with client.messages.stream(
model="claude-opus-4-5",
max_tokens=1024,
messages=[{"role": "user", "content": "Viết class Python cho binary search tree"}]
) as stream:
for text in stream.text_stream:
print(text, end="", flush=True)
Extended Thinking (Claude 3.7+)
# Claude có thể "suy nghĩ" trước khi trả lời (như chain-of-thought nội bộ)
response = client.messages.create(
model="claude-opus-4-5",
max_tokens=16000,
thinking={
"type": "enabled",
"budget_tokens": 10000 # Tokens dành cho thinking
},
messages=[{
"role": "user",
"content": "Giải bài toán: 17 × 19 - sqrt(144) + 2^8"
}]
)
for block in response.content:
if block.type == "thinking":
print(f"[Thinking]: {block.thinking[:200]}...")
elif block.type == "text":
print(f"[Answer]: {block.text}")
Long Document Analysis
# Claude hỗ trợ 200K context — phân tích tài liệu dài
with open("contract.txt") as f:
document = f.read()
response = client.messages.create(
model="claude-opus-4-5",
max_tokens=2048,
messages=[{
"role": "user",
"content": f"""Phân tích hợp đồng sau và trả lời:
1. Các điều khoản quan trọng nhất
2. Rủi ro tiềm ẩn
3. Ngày hết hạn (nếu có)
<document>
{document}
</document>"""
}]
)
4. Google Gemini API
Setup
pip install google-generativeai
export GOOGLE_API_KEY="AIza..."
Basic Usage
import google.generativeai as genai
import os
genai.configure(api_key=os.environ["GOOGLE_API_KEY"])
model = genai.GenerativeModel(
model_name="gemini-2.0-flash",
system_instruction="Bạn là tutor Python cho người mới học."
)
response = model.generate_content("Decorators trong Python là gì?")
print(response.text)
Multi-turn Chat
chat = model.start_chat(history=[])
while True:
user_input = input("Bạn: ")
if user_input.lower() == "quit":
break
response = chat.send_message(user_input)
print(f"Gemini: {response.text}")
Multimodal (Image + Video + Audio)
import PIL.Image
# Image analysis
image = PIL.Image.open("chart.png")
response = model.generate_content([
"Phân tích biểu đồ này và đưa ra insights",
image
])
print(response.text)
# Gemini 1.5 Pro — 1M context với video
video_file = genai.upload_file("presentation.mp4")
response = model.generate_content([
"Tóm tắt nội dung video này",
video_file
])
5. Unified Client Pattern
from abc import ABC, abstractmethod
from dataclasses import dataclass
from typing import Iterator
@dataclass
class LLMResponse:
content: str
input_tokens: int
output_tokens: int
class LLMClient(ABC):
@abstractmethod
def complete(self, prompt: str, system: str = "") -> LLMResponse:
pass
@abstractmethod
def stream(self, prompt: str, system: str = "") -> Iterator[str]:
pass
class OpenAIClient(LLMClient):
def __init__(self, model="gpt-4o"):
from openai import OpenAI
self.client = OpenAI()
self.model = model
def complete(self, prompt: str, system: str = "") -> LLMResponse:
messages = []
if system:
messages.append({"role": "system", "content": system})
messages.append({"role": "user", "content": prompt})
r = self.client.chat.completions.create(model=self.model, messages=messages)
return LLMResponse(
content=r.choices[0].message.content,
input_tokens=r.usage.prompt_tokens,
output_tokens=r.usage.completion_tokens
)
def stream(self, prompt: str, system: str = "") -> Iterator[str]:
messages = [{"role": "user", "content": prompt}]
with self.client.chat.completions.stream(
model=self.model, messages=messages
) as s:
yield from s.text_stream
class AnthropicClient(LLMClient):
def __init__(self, model="claude-opus-4-5"):
import anthropic
self.client = anthropic.Anthropic()
self.model = model
def complete(self, prompt: str, system: str = "") -> LLMResponse:
r = self.client.messages.create(
model=self.model, max_tokens=2048,
system=system or "You are a helpful assistant.",
messages=[{"role": "user", "content": prompt}]
)
return LLMResponse(
content=r.content[0].text,
input_tokens=r.usage.input_tokens,
output_tokens=r.usage.output_tokens
)
def stream(self, prompt: str, system: str = "") -> Iterator[str]:
with self.client.messages.stream(
model=self.model, max_tokens=2048,
messages=[{"role": "user", "content": prompt}]
) as s:
yield from s.text_stream
# Dùng
def get_client(provider: str = "openai") -> LLMClient:
if provider == "openai":
return OpenAIClient()
elif provider == "anthropic":
return AnthropicClient()
raise ValueError(f"Unknown provider: {provider}")
llm = get_client("openai")
response = llm.complete("Giải thích REST API cho người mới")
print(response.content)
6. Cost Optimization
# Pricing tham khảo (USD per 1M tokens, 2025)
PRICING = {
"gpt-4o": {"input": 2.50, "output": 10.00},
"gpt-4o-mini": {"input": 0.15, "output": 0.60},
"claude-opus-4-5": {"input": 3.00, "output": 15.00},
"claude-haiku-4-5": {"input": 0.80, "output": 4.00},
"gemini-2.0-flash": {"input": 0.10, "output": 0.40},
}
def estimate_cost(model: str, input_tokens: int, output_tokens: int) -> float:
p = PRICING.get(model, {"input": 1.0, "output": 1.0})
return (input_tokens * p["input"] + output_tokens * p["output"]) / 1_000_000
# Chiến lược tiết kiệm
tips = """
1. Model routing: dùng cheap model (gpt-4o-mini) cho tasks đơn giản,
expensive model chỉ khi cần
2. Prompt caching: Anthropic cache system prompts dài
→ tiết kiệm 90% input cost cho repeated calls
3. Batch API: OpenAI Batch API 50% cheaper (async, 24h delay)
4. Context window management: trim conversation history
→ tránh truyền lại toàn bộ history mỗi call
5. Response caching: cache kết quả cho cùng input
(đặc biệt hữu ích cho RAG retrieval)
"""
Tổng kết
| Task | Gợi ý |
|---|---|
| Coding, agents | GPT-4o hoặc Claude Opus |
| Cheap, fast | GPT-4o-mini, Claude Haiku, Gemini Flash |
| Long document (100K+) | Claude Opus (200K), Gemini Pro (1M) |
| Multimodal | GPT-4o, Gemini, Claude (images) |
| Structured output | OpenAI JSON mode + Pydantic |
| Streaming UI | Tất cả đều hỗ trợ |
Bài tiếp theo (cuối series): Deploying LLMs tự host với Ollama và vLLM, evaluation và production checklist.