Tiếng Anh: "machine learning" → ["machine", "learning"] ← Rõ ràng
Tiếng Việt: "học sinh học sinh học" → ???
- "học_sinh / học / sinh_học" ← học sinh ĐI học môn sinh học
- "học / sinh_học / sinh_học" ← ???
from underthesea import word_tokenize
text = "Trường đại học Bách Khoa Hà Nội là trường đại học kỹ thuật hàng đầu"
tokens = word_tokenize(text)
print(tokens)
# ['Trường', 'đại_học', 'Bách_Khoa', 'Hà_Nội', 'là', 'trường',
# 'đại_học', 'kỹ_thuật', 'hàng_đầu']
1.2 聲調標記
# 6 thanh điệu: ngang, sắc, huyền, hỏi, ngã, nặng
# "ma", "má", "mà", "mả", "mã", "mạ" — 6 từ hoàn toàn khác nghĩa!
# Vấn đề: user thường gõ không dấu
text_no_accent = "hoc sinh hoc sinh hoc"
# Cần accent restoration trước khi xử lý NLP
1.3 挑戰比較表
挑戰
英語
越南語
字邊界
空間
需要分詞
形態
詞形變化(跑/跑/跑)
無屈折變化
語氣/口音
沒有
6 種聲音
資源
很多
少很多
分詞器效率
~1 個標記/單字
~1.5-2 個令牌/字(法學碩士)
2.越南語NLP工具
2.1 海底
from underthesea import (
word_tokenize,
pos_tag,
ner,
classify,
sentiment,
)
text = "Nguyễn Phú Trọng làm việc tại Hà Nội, Việt Nam"
# Word segmentation
print(word_tokenize(text))
# POS Tagging
print(pos_tag(text))
# [('Nguyễn_Phú_Trọng', 'Np'), ('làm_việc', 'V'), ('tại', 'E'),
# ('Hà_Nội', 'Np'), (',', 'CH'), ('Việt_Nam', 'Np')]
# NER
print(ner(text))
# [('Nguyễn_Phú_Trọng', 'B-PER'), ..., ('Hà_Nội', 'B-LOC'), ...]
# Sentiment
print(sentiment("Sản phẩm này rất tốt"))
# positive
2.2 VnCoreNLP
from vncorenlp import VnCoreNLP
annotator = VnCoreNLP("VnCoreNLP-1.2.jar", annotators="wseg,pos,ner", max_heap_size='-Xmx2g')
text = "Trường Đại học Bách Khoa Hà Nội tuyển sinh năm 2026"
result = annotator.annotate(text)
for sentence in result['sentences']:
for word_info in sentence:
print(f" {word_info['form']:20s} | {word_info['posTag']:5s} | {word_info['nerLabel']}")
3. 越南語預訓練模型
型號
類型
基地
任務
菲伯特
編碼器
羅伯塔
分類、NER、QA
巴特佛
編碼-十二月
巴特
總結、生成
ViT5
編碼-十二月
T5
總結、翻譯
XLM-羅伯塔
編碼器
羅伯塔
多語言任務
BGE-M3
編碼器
—
多語言嵌入
用於文字分類的 PhoBERT
from transformers import AutoTokenizer, AutoModelForSequenceClassification
# PhoBERT yêu cầu word segmentation TRƯỚC khi tokenize
from underthesea import word_tokenize
text = "Sản phẩm rất tốt và giao hàng nhanh"
segmented = word_tokenize(text, format="text")
# "Sản_phẩm rất tốt và giao_hàng nhanh"
tokenizer = AutoTokenizer.from_pretrained("vinai/phobert-base-v2")
model = AutoModelForSequenceClassification.from_pretrained(
"vinai/phobert-base-v2", num_labels=3
)
inputs = tokenizer(segmented, return_tensors="pt")
outputs = model(**inputs)