baichuan-inc
/

Baichuan2-13B-Base

Text Generation

text-generation-inference

Model card Files Files and versions

GradientGuru commited on Dec 24, 2023

Commit

c6f590c

·

1 Parent(s): e2a280b

Update tokenization_baichuan.py

Files changed (1) hide show

tokenization_baichuan.py +7 -5

tokenization_baichuan.py CHANGED Viewed

@@ -68,6 +68,13 @@ class BaichuanTokenizer(PreTrainedTokenizer):
             if isinstance(pad_token, str)
             else pad_token
         )
         super().__init__(
             bos_token=bos_token,
             eos_token=eos_token,
@@ -79,11 +86,6 @@ class BaichuanTokenizer(PreTrainedTokenizer):
             clean_up_tokenization_spaces=clean_up_tokenization_spaces,
             **kwargs,
         )
-        self.vocab_file = vocab_file
-        self.add_bos_token = add_bos_token
-        self.add_eos_token = add_eos_token
-        self.sp_model = spm.SentencePieceProcessor(**self.sp_model_kwargs)
-        self.sp_model.Load(vocab_file)
     def __getstate__(self):
         state = self.__dict__.copy()

             if isinstance(pad_token, str)
             else pad_token
         )
+        self.vocab_file = vocab_file
+        self.add_bos_token = add_bos_token
+        self.add_eos_token = add_eos_token
+        self.sp_model = spm.SentencePieceProcessor(**self.sp_model_kwargs)
+        self.sp_model.Load(vocab_file)
         super().__init__(
             bos_token=bos_token,
             eos_token=eos_token,
             clean_up_tokenization_spaces=clean_up_tokenization_spaces,
             **kwargs,
         )
     def __getstate__(self):
         state = self.__dict__.copy()