Downloads · 30 days
14
12% of all-time downloads
egor2014/mini-russian-bert
mini-russian-bert is a feature extraction model from egor2014. Use it when you need embeddings to search or compare text. The card lists the license as apache-2.0.
Downloads · 30 days
14
12% of all-time downloads
All-time downloads
116
Public
Parameters
12M
47.8 MB on disk
Likes
1
Public
Click a slice to open those files.
.safetensors47.8 MB · 97%
From the Hugging Face model README
from transformers import AutoTokenizer, AutoModel
import torch
model = AutoModel.from_pretrained("egor2014/mini-russian-bert")
tokenizer = AutoTokenizer.from_pretrained("egor2014/mini-russian-bert")
device = "cuda"
class CustomSentenceTransformer(torch.nn.Module):
def __init__(self, model):
super().__init__()
self.model = model
self.tokenizer = tokenizer
def encode(self, sentences, batch_size=2, convert_to_numpy=True, **kwargs):
all_embeddings = []
for i in range(0, len(sentences), batch_size):
batch = sentences[i:i + batch_size]
inputs = self.tokenizer(batch, return_tensors="pt", padding=True, truncation=True, max_length=512)
inputs = {key: value.to(device) for key, value in inputs.items()}
with torch.no_grad():
outputs = self.model(**inputs, output_hidden_states=True)
last_hidden_state = outputs.hidden_states[-1]
batch_embeddings = last_hidden_state.mean(dim=1)
all_embeddings.append(batch_embeddings.cpu())
return torch.cat(all_embeddings).numpy() if convert_to_numpy else torch.cat(all_embeddings)
model.to(device)
model.eval()
custom_model = CustomSentenceTransformer(model)
custom_model.encode(["Привет, мир"])