Downloads · 30 days
26
100% of all-time downloads
echoboi/electric_vehicles-distilbert-classifier
electric_vehicles-distilbert-classifier is a text classification model from echoboi. Use it when you need a label for a piece of text. The card lists the license as mit.
This model classifies content related to electric vehicles on climate change subreddits.
Downloads · 30 days
26
100% of all-time downloads
All-time downloads
26
Public
Repo size
266 MB
Likes
0
Public
Click a slice to open those files.
.pt266 MB · 100%
From the Hugging Face model README
This model classifies content related to electric vehicles on climate change subreddits.
The model predicts 7 labels simultaneously:
Note: Label order in predictions matches the order above.
import torch, sys, os, tempfile
from transformers import DistilBertTokenizer
from huggingface_hub import snapshot_download
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
def print_sorted_label_scores(label_scores):
# Sort label_scores dict by score descending
sorted_items = sorted(label_scores.items(), key=lambda x: x[1], reverse=True)
for label, score in sorted_items:
print(f" {label}: {score:.6f}")
# Model link and examples for this specific model
model_link = 'sanchow/electric_vehicles-distilbert-classifier'
examples = [
"Switching to electric cars can cut down on smog and carbon output."
]
print(f"\n{'='*60}")
print("MODEL: ELECTRIC VEHICLES SECTOR")
print(f"{'='*60}")
print(f"Downloading model: {model_link}")
with tempfile.TemporaryDirectory() as temp_dir:
snapshot_download(
repo_id=model_link,
local_dir=temp_dir,
local_dir_use_symlinks=False
)
model_class_path = os.path.join(temp_dir, 'model_class.py')
if not os.path.exists(model_class_path):
print(f"model_class.py not found in downloaded files")
print(f" Available files: {os.listdir(temp_dir)}")
else:
sys.path.insert(0, temp_dir)
from model_class import MultilabelClassifier
tokenizer = DistilBertTokenizer.from_pretrained(temp_dir)
checkpoint = torch.load(os.path.join(temp_dir, 'model.pt'), map_location='cpu', weights_only=False)
model = MultilabelClassifier(checkpoint['model_name'], len(checkpoint['label_names']))
model.load_state_dict(checkpoint['model_state_dict'])
model.to(device)
model.eval()
print("Model loaded successfully")
print(f" Labels: {checkpoint['label_names']}")
print("\nElectric Vehicles classifier results:\n")
for i, test_text in enumerate(examples):
inputs = tokenizer(
test_text,
return_tensors="pt",
truncation=True,
max_length=512,
padding=True
).to(device)
with torch.no_grad():
outputs = model(**inputs)
predictions = outputs.cpu().numpy() if isinstance(outputs, (tuple, list)) else outputs.cpu().numpy()
label_scores = {label: float(score) for label, score in zip(checkpoint['label_names'], predictions[0])}
print(f"Example {i+1}: '{test_text}'")
print("Predictions (all label scores, highest first):")
print_sorted_label_scores(label_scores)
print("-" * 40)
Best model performance:
Dataset: ~900 GPT-labeled samples per sector (600 train, 150 validation, 150 test)
optimal_thresholds = {'Alternative Modes': 0.28427787391225384, 'Charging Infrastructure': 0.3619448731592626, 'Environmental Benefit': 0.4029443119613918, 'Grid Impact And Energy Mix': 0.29907076386497516, 'Mineral Supply Chain': 0.2987419331439881, 'Policy And Mandates': 0.36899998622725905, 'Purchase Price': 0.3463644004166977}
for label, score in zip(label_names, predictions[0]):
threshold = optimal_thresholds.get(label, 0.5)
if score > threshold:
print(f"{label}: {score:.3f}")
Trained on GPT-labeled Reddit data:
If you use this model in your research, please cite:
@misc{electric_vehicles_distilbert_classifier,
title={Electric Vehicles Classifier for Climate Change Analysis},
author={Sandeep Chowdhary},
year={2025},
publisher={Hugging Face},
journal={Hugging Face Hub},
howpublished={\url{https://huggingface.co/echoboi/electric_vehicles-distilbert-classifier}},
}