from transformers import pipeline
summarizer = pipeline("summarization", model="facebook/bart-large-cnn")
article = """
Natural Language Processing (NLP) is a subfield of artificial intelligence
that focuses on enabling computers to understand, interpret, and generate
human language. NLP combines computational linguistics, machine learning,
and deep learning to process and analyze large amounts of natural language
data. Key applications include machine translation, sentiment analysis,
chatbots, and text summarization. Recent advances in transformer-based
models like BERT and GPT have significantly improved NLP capabilities.
"""
summary = summarizer(article, max_length=60, min_length=20, do_sample=False)
print(summary[0]["summary_text"])
1.3 評価: ROUGE メトリクス
from rouge_score import rouge_scorer
scorer = rouge_scorer.RougeScorer(['rouge1', 'rouge2', 'rougeL'], use_stemmer=True)
reference = "NLP enables computers to understand human language using AI and deep learning."
hypothesis = "NLP is an AI subfield that helps computers process natural language."
scores = scorer.score(reference, hypothesis)
for key, value in scores.items():
print(f" {key}: P={value.precision:.3f} R={value.recall:.3f} F1={value.fmeasure:.3f}")
メトリクス
何を測定するか
ルージュ-1
ユニグラムオーバーラップ
ルージュ2
バイグラムの重複
ルージュエル
最長共通部分列
2. 機械翻訳
2.1 顔を抱きしめながら翻訳する
from transformers import pipeline
# English → Vietnamese
translator = pipeline("translation", model="Helsinki-NLP/opus-mt-en-vi")
result = translator("Natural Language Processing is a fascinating field of AI")
print(result[0]["translation_text"])
# Multilingual translation với NLLB
from transformers import AutoModelForSeq2SeqLM, AutoTokenizer
model_name = "facebook/nllb-200-distilled-600M"
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModelForSeq2SeqLM.from_pretrained(model_name)
# Dịch EN → VI
text = "Machine learning is transforming every industry."
inputs = tokenizer(text, return_tensors="pt")
translated = model.generate(
**inputs,
forced_bos_token_id=tokenizer.convert_tokens_to_ids("vie_Latn"),
max_length=128,
)
print(tokenizer.decode(translated[0], skip_special_tokens=True))
2.2 評価: BLEU スコア
from sacrebleu import corpus_bleu
references = [["Xử lý ngôn ngữ tự nhiên là lĩnh vực hấp dẫn của AI"]]
hypotheses = ["Xử lý ngôn ngữ tự nhiên là lĩnh vực thú vị của trí tuệ nhân tạo"]
bleu = corpus_bleu(hypotheses, references)
print(f"BLEU: {bleu.score:.2f}")
メトリクス
何を測定するか
範囲
ブルー
N グラムの精度
0-100
chrF
文字の F スコア
0-100
コメット
学習済みメトリクス (ニューラル)
0-1
3. ViT5 によるベトナム語のまとめ
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM
model_name = "VietAI/vit5-base-vietnews-summarization"
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModelForSeq2SeqLM.from_pretrained(model_name)
article = """Trí tuệ nhân tạo đang thay đổi mọi lĩnh vực trong cuộc sống.
Từ y tế, giáo dục đến tài chính, AI mang lại nhiều lợi ích to lớn.
Tuy nhiên, việc phát triển AI cũng đặt ra nhiều thách thức về đạo đức
và quyền riêng tư cần được giải quyết."""
inputs = tokenizer(article, return_tensors="pt", max_length=512, truncation=True)
outputs = model.generate(**inputs, max_length=100)
summary = tokenizer.decode(outputs[0], skip_special_tokens=True)
print(summary)