Downloads · 30 days
30
3% of all-time downloads
Lowin/chinese-bigbird-tiny-1024
chinese-bigbird-tiny-1024 is a feature extraction model from Lowin. Use it when you need embeddings to search or compare text. It is set up for transformers. The card lists the license as apache-2.0.
https://github.com/LowinLi/chinese-bigbird
Downloads · 30 days
30
3% of all-time downloads
All-time downloads
1.1K
Public
Repo size
87.1 MB
Likes
2
Public
Click a slice to open those files.
.bin43.5 MB · 100%
From the Hugging Face model README
import jieba_fast
from transformers import BertTokenizer
from transformers import BigBirdModel
class JiebaTokenizer(BertTokenizer):
def __init__(
self, pre_tokenizer=lambda x: jieba_fast.cut(x, HMM=False), *args, **kwargs
):
super().__init__(*args, **kwargs)
self.pre_tokenizer = pre_tokenizer
def _tokenize(self, text, *arg, **kwargs):
split_tokens = []
for text in self.pre_tokenizer(text):
if text in self.vocab:
split_tokens.append(text)
else:
split_tokens.extend(super()._tokenize(text))
return split_tokens
model = BigBirdModel.from_pretrained('Lowin/chinese-bigbird-tiny-1024')
tokenizer = JiebaTokenizer.from_pretrained('Lowin/chinese-bigbird-tiny-1024')