Downloads · 30 days
0
undertheseanlp/sonar_core_1
sonar_core_1 is a text classification model from undertheseanlp. Use it when you need a label for a piece of text. It is set up for scikit-learn. The card lists the license as apache-2.0.
A machine learning-based text classification model designed for Vietnamese language processing. Built on TF-IDF feature extraction pipeline combined with Support Vector Classification (SVC) and Logistic Regression, ac…
Downloads · 30 days
0
Access
Public
Updated Sep 28, 2025
Repo size
8.4 MB
Likes
0
Public
Click a slice to open those files.
.joblib10.4 MB · 99%
From the Hugging Face model README
A machine learning-based text classification model designed for Vietnamese language processing. Built on TF-IDF feature extraction pipeline combined with Support Vector Classification (SVC) and Logistic Regression, achieving 92.80% accuracy on VNTC (news) and 72.47% accuracy on UTS2017_Bank (banking) datasets with SVC.
📋 View Detailed System Card for comprehensive model documentation, performance analysis, and limitations.
Sonar Core 1 is a Vietnamese text classification model that supports multiple domains including news categorization and banking text classification. The model is specifically designed for Vietnamese news article classification, banking text categorization, content categorization for Vietnamese text, and document organization and tagging.
pip install scikit-learn>=1.6 joblib
# Default training with VNTC dataset
python train.py --dataset vntc --model logistic
# With specific parameters
python train.py --dataset vntc --model logistic --max-features 20000 --ngram-min 1 --ngram-max 2
# Train with UTS2017_Bank dataset (SVC recommended)
python train.py --dataset uts2017 --model svc_linear
# Train with Logistic Regression
python train.py --dataset uts2017 --model logistic
# With specific parameters (SVC)
python train.py --dataset uts2017 --model svc_linear --max-features 20000 --ngram-min 1 --ngram-max 2
# Compare multiple configurations
python train.py --dataset uts2017 --compare
from train import train_notebook
# Train VNTC model
vntc_results = train_notebook(
dataset="vntc",
model_name="logistic",
max_features=20000,
ngram_min=1,
ngram_max=2
)
# Train UTS2017_Bank model
bank_results = train_notebook(
dataset="uts2017",
model_name="logistic",
max_features=20000,
ngram_min=1,
ngram_max=2
)
from huggingface_hub import hf_hub_download
import joblib
# Download and load VNTC model
vntc_model = joblib.load(
hf_hub_download("undertheseanlp/sonar_core_1", "vntc_classifier_20250927_161550.joblib")
)
# Enhanced prediction function
def predict_text(model, text):
probabilities = model.predict_proba([text])[0]
# Get top 3 predictions sorted by probability
top_indices = probabilities.argsort()[-3:][::-1]
top_predictions = []
for idx in top_indices:
category = model.classes_[idx]
prob = probabilities[idx]
top_predictions.append((category, prob))
# The prediction should be the top category
prediction = top_predictions[0][0]
confidence = top_predictions[0][1]
return prediction, confidence, top_predictions
# Make prediction on news text
news_text = "Đội tuyển bóng đá Việt Nam giành chiến thắng"
prediction, confidence, top_predictions = predict_text(vntc_model, news_text)
print(f"News category: {prediction}")
print(f"Confidence: {confidence:.3f}")
print("Top 3 predictions:")
for i, (category, prob) in enumerate(top_predictions, 1):
print(f" {i}. {category}: {prob:.3f}")
from huggingface_hub import hf_hub_download
import joblib
# Download and load UTS2017_Bank model (latest SVC model)
bank_model = joblib.load(
hf_hub_download("undertheseanlp/sonar_core_1", "uts2017_bank_classifier_20250928_060819.joblib")
)
# Enhanced prediction function (same as above)
def predict_text(model, text):
probabilities = model.predict_proba([text])[0]
# Get top 3 predictions sorted by probability
top_indices = probabilities.argsort()[-3:][::-1]
top_predictions = []
for idx in top_indices:
category = model.classes_[idx]
prob = probabilities[idx]
top_predictions.append((category, prob))
# The prediction should be the top category
prediction = top_predictions[0][0]
confidence = top_predictions[0][1]
return prediction, confidence, top_predictions
# Make prediction on banking text
bank_text = "Tôi muốn mở tài khoản tiết kiệm"
prediction, confidence, top_predictions = predict_text(bank_model, bank_text)
print(f"Banking category: {prediction}")
print(f"Confidence: {confidence:.3f}")
print("Top 3 predictions:")
for i, (category, prob) in enumerate(top_predictions, 1):
print(f" {i}. {category}: {prob:.3f}")
from huggingface_hub import hf_hub_download
import joblib
# Load both models
vntc_model = joblib.load(
hf_hub_download("undertheseanlp/sonar_core_1", "vntc_classifier_20250927_161550.joblib")
)
bank_model = joblib.load(
hf_hub_download("undertheseanlp/sonar_core_1", "uts2017_bank_classifier_20250928_060819.joblib")
)
# Enhanced prediction function for both models
def predict_text(model, text):
probabilities = model.predict_proba([text])[0]
# Get top 3 predictions sorted by probability
top_indices = probabilities.argsort()[-3:][::-1]
top_predictions = []
for idx in top_indices:
category = model.classes_[idx]
prob = probabilities[idx]
top_predictions.append((category, prob))
# The prediction should be the top category
prediction = top_predictions[0][0]
confidence = top_predictions[0][1]
return prediction, confidence, top_predictions
# Function to classify any Vietnamese text
def classify_vietnamese_text(text, domain="auto"):
"""
Classify Vietnamese text using appropriate model with detailed predictions
Args:
text: Vietnamese text to classify
domain: "news", "banking", or "auto" to detect domain
Returns:
tuple: (prediction, confidence, top_predictions, domain_used)
"""
if domain == "news":
prediction, confidence, top_predictions = predict_text(vntc_model, text)
return prediction, confidence, top_predictions, "news"
elif domain == "banking":
prediction, confidence, top_predictions = predict_text(bank_model, text)
return prediction, confidence, top_predictions, "banking"
else:
# Try both models and return higher confidence
news_pred, news_conf, news_top = predict_text(vntc_model, text)
bank_pred, bank_conf, bank_top = predict_text(bank_model, text)
if news_conf > bank_conf:
return f"NEWS: {news_pred}", news_conf, news_top, "news"
else:
return f"BANKING: {bank_pred}", bank_conf, bank_top, "banking"
# Examples
examples = [
"Đội tuyển bóng đá Việt Nam thắng 2-0",
"Tôi muốn vay tiền mua nhà",
"Chính phủ thông qua luật mới"
]
for text in examples:
category, confidence, top_predictions, domain = classify_vietnamese_text(text)
print(f"Text: {text}")
print(f"Category: {category}")
print(f"Confidence: {confidence:.3f}")
print(f"Domain: {domain}")
print("Top 3 predictions:")
for i, (cat, prob) in enumerate(top_predictions, 1):
print(f" {i}. {cat}: {prob:.3f}")
print()
dataset: Dataset to use ("vntc" or "uts2017")model: Model type ("logistic" or "svc" - SVC recommended for best performance)max_features: Maximum number of TF-IDF features (default: 20000)ngram_min/max: N-gram range (default: 1-2)split_ratio: Train/test split ratio for UTS2017 (default: 0.2)n_samples: Optional sample limit for quick testingIf you use this model, please cite:
@misc{undertheseanlp_2025,
author = { undertheseanlp },
title = { Sonar Core 1 - Vietnamese Text Classification Model },
year = 2025,
url = { https://huggingface.co/undertheseanlp/sonar_core_1 },
doi = { 10.57967/hf/6599 },
publisher = { Hugging Face }
}