from transformers import pipeline
summarizer = pipeline("summarization", model="facebook/bart-large-cnn")
article = """
Natural Language Processing (NLP) is a subfield of artificial intelligence
that focuses on enabling computers to understand, interpret, and generate
human language. NLP combines computational linguistics, machine learning,
and deep learning to process and analyze large amounts of natural language
data. Key applications include machine translation, sentiment analysis,
chatbots, and text summarization. Recent advances in transformer-based
models like BERT and GPT have significantly improved NLP capabilities.
"""
summary = summarizer(article, max_length=60, min_length=20, do_sample=False)
print(summary[0]["summary_text"])
1.3 評估:ROUGE 指標
from rouge_score import rouge_scorer
scorer = rouge_scorer.RougeScorer(['rouge1', 'rouge2', 'rougeL'], use_stemmer=True)
reference = "NLP enables computers to understand human language using AI and deep learning."
hypothesis = "NLP is an AI subfield that helps computers process natural language."
scores = scorer.score(reference, hypothesis)
for key, value in scores.items():
print(f" {key}: P={value.precision:.3f} R={value.recall:.3f} F1={value.fmeasure:.3f}")
指標
測量什麼
胭脂-1
一元重疊
胭脂-2
二元組重疊
胭脂-L
最長公共子序列
2. 機器翻譯
2.1 抱臉翻譯
from transformers import pipeline
# English → Vietnamese
translator = pipeline("translation", model="Helsinki-NLP/opus-mt-en-vi")
result = translator("Natural Language Processing is a fascinating field of AI")
print(result[0]["translation_text"])
# Multilingual translation với NLLB
from transformers import AutoModelForSeq2SeqLM, AutoTokenizer
model_name = "facebook/nllb-200-distilled-600M"
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModelForSeq2SeqLM.from_pretrained(model_name)
# Dịch EN → VI
text = "Machine learning is transforming every industry."
inputs = tokenizer(text, return_tensors="pt")
translated = model.generate(
**inputs,
forced_bos_token_id=tokenizer.convert_tokens_to_ids("vie_Latn"),
max_length=128,
)
print(tokenizer.decode(translated[0], skip_special_tokens=True))
2.2 評估:BLEU 分數
from sacrebleu import corpus_bleu
references = [["Xử lý ngôn ngữ tự nhiên là lĩnh vực hấp dẫn của AI"]]
hypotheses = ["Xử lý ngôn ngữ tự nhiên là lĩnh vực thú vị của trí tuệ nhân tạo"]
bleu = corpus_bleu(hypotheses, references)
print(f"BLEU: {bleu.score:.2f}")
指標
測量什麼
範圍
藍色
N 元語法精度
0-100
chrF
角色 F 分數
0-100
彗星
學習指標(神經)
0-1
3. ViT5 越南語總結
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM
model_name = "VietAI/vit5-base-vietnews-summarization"
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModelForSeq2SeqLM.from_pretrained(model_name)
article = """Trí tuệ nhân tạo đang thay đổi mọi lĩnh vực trong cuộc sống.
Từ y tế, giáo dục đến tài chính, AI mang lại nhiều lợi ích to lớn.
Tuy nhiên, việc phát triển AI cũng đặt ra nhiều thách thức về đạo đức
và quyền riêng tư cần được giải quyết."""
inputs = tokenizer(article, return_tensors="pt", max_length=512, truncation=True)
outputs = model.generate(**inputs, max_length=100)
summary = tokenizer.decode(outputs[0], skip_special_tokens=True)
print(summary)