3b-september-2025 upload

Browse files

Files changed (7) hide show

1_Pooling/config.json +8 -8
README.md +102 -99
config.json +212 -90
config_sentence_transformers.json +8 -4
modeling_gigarembed.py +2 -1
sentence_bert_config.json +2 -2
tokenizer_config.json +1 -1

1_Pooling/config.json CHANGED Viewed

@@ -1,10 +1,10 @@
 {
-  "word_embedding_dimension": 2048,
-  "pooling_mode_cls_token": false,
-  "pooling_mode_mean_tokens": true,
-  "pooling_mode_max_tokens": false,
-  "pooling_mode_mean_sqrt_len_tokens": false,
-  "pooling_mode_weightedmean_tokens": false,
-  "pooling_mode_lasttoken": false,
-  "include_prompt": true
 }

 {
+    "word_embedding_dimension": 2048,
+    "pooling_mode_cls_token": false,
+    "pooling_mode_mean_tokens": true,
+    "pooling_mode_max_tokens": false,
+    "pooling_mode_mean_sqrt_len_tokens": false,
+    "pooling_mode_weightedmean_tokens": false,
+    "pooling_mode_lasttoken": false,
+    "include_prompt": true
 }

README.md CHANGED Viewed

@@ -1,138 +1,141 @@
 ---
-license: mit
-language:
-- ru
-- en
-pipeline_tag: feature-extraction
 tags:
-- MTEB
 ---
-## Giga-Embeddings-instruct
-- Base Decoder-only LLM: GigaChat-3b
-- Pooling Type: Latent-Attention
-- Embedding Dimension: 2048
-## Использование
-Ниже приведен пример кодирования запросов и текстов.
-### Requirements
-```bash
-pip install -q transformers==4.46.3 sentence-transformers==3.3.1 datasets langchain_community langchain_huggingface langchain_gigachat
-```
-### Transformers
-```python
-import os
-import torch
-import torch.nn.functional as F
-from transformers import AutoTokenizer, AutoModel
-# Each query needs to be accompanied by an corresponding instruction describing the task.
-task_name_to_instruct = {"example": "Given a question, retrieve passages that answer the question",}
-query_prefix = task_name_to_instruct["example"] + "\nquestion: "
-queries = [
-    'are judo throws allowed in wrestling?',
-    'how to become a radiology technician in michigan?'
-]
-# No instruction needed for retrieval passages
-passage_prefix = ""
-passages = [
-    "Since you're reading this, you are probably someone from a judo background or someone who is just wondering how judo techniques can be applied under wrestling rules. So without further ado, let's get to the question. Are Judo throws allowed in wrestling? Yes, judo throws are allowed in freestyle and folkstyle wrestling. You only need to be careful to follow the slam rules when executing judo throws. In wrestling, a slam is lifting and returning an opponent to the mat with unnecessary force.",
-    "Below are the basic steps to becoming a radiologic technologist in Michigan:Earn a high school diploma. As with most careers in health care, a high school education is the first step to finding entry-level employment. Taking classes in math and science, such as anatomy, biology, chemistry, physiology, and physics, can help prepare students for their college studies and future careers.Earn an associate degree. Entry-level radiologic positions typically require at least an Associate of Applied Science. Before enrolling in one of these degree programs, students should make sure it has been properly accredited by the Joint Review Committee on Education in Radiologic Technology (JRCERT).Get licensed or certified in the state of Michigan."
-]
-# load model with tokenizer
-model = AutoModel.from_pretrained('ai-sage/Giga-Embeddings-instruct', trust_remote_code=True)
-# get the embeddings
-query_embeddings = model.encode(queries, instruction=query_prefix)
-passage_embeddings = model.encode(passages, instruction=passage_prefix)
-scores = (query_embeddings @ passage_embeddings.T) * 100
-print(scores.tolist())
-```
-### LangChain
 ```python
-import torch
-from langchain_huggingface import HuggingFaceEmbeddings
-# Load model
-embeddings = HuggingFaceEmbeddings(
-    model_name='ai-sage/Giga-Embeddings-instruct',
-    encode_kwargs={},
-    model_kwargs={
-        'device': 'cuda', # or 'cpu'
-        'trust_remote_code': True,
-        'model_kwargs': {'torch_dtype': torch.bfloat16},
-        'prompts': {'query': 'Given a question, retrieve passages that answer the question\nquestion: '}
-    }
-)
-# Tokenizer
-embeddings._client.tokenizer.tokenize("Hello world! I am GigaChat")
-# Query embeddings
-query_embeddings = embeddings.embed_query("Hello world!")
-print(f"Your embeddings: {query_embeddings[0:20]}...")
-print(f"Vector size: {len(query_embeddings)}")
-# Document embeddings
-documents = ["foo bar", "bar foo"]
-documents_embeddings = embeddings.embed_documents(documents)
-print(f"Vector size: {len(documents_embeddings)} x {len(documents_embeddings[0])}")
-```
-## Инструктивность
-**Использование инструкций для улучшения качества эмбеддингов**
-Для достижения более точных результатов при работе с эмбеддингами, особенно в задачах поиска и извлечения информации (retrieval), рекомендуется добавлять инструкцию на естественном языке перед текстовым запросом (query). Это помогает модели лучше понять контекст и цель запроса, что положительно сказывается на качестве результатов. Важно отметить, что инструкцию нужно добавлять только перед запросом, а не перед документом.
-Для **симметричных задач**, таких как классификация (classification) или семантическое сравнение текстов (semantic text similarity), инструкцию необходимо добавлять перед каждым запросом. Это связано с тем, что такие задачи требуют одинакового контекста для всех входных данных, чтобы модель могла корректно сравнивать или классифицировать их.
-**Примеры инструкций для симметричных задач:**
-- `"Retrieve semantically similar text \ntext: {query}"`
-- `"Given a text, retrieve semantically similar text \ntext: {query}"`
-- `"Дано предложение, необходимо найти его парафраз \nпредложение: {query}"`
-- `"Классифицируй отзыв на товар как положительный, отрицательный или нейтральный \nотзыв: {query}"`
-- `"Классифицируй чувствительную тему по запросу \nзапрос: {query}"`
-Для **retrieval-задач** (например, поиск ответа в тексте) можно использовать инструкцию:
-`'Дан вопрос, необходимо найти абзац текста с ответом \nвопрос: {query}'`.
-Такой подход особенно эффективен для задач поиска и извлечения информации, таких как поиск релевантных документов или извлечение ответов из текста.
-**Примеры инструкций для retrieval-задач:**
-- `'Дан вопрос, необходимо найти абзац текста с ответом \nвопрос: {query}'`
-- `'Given the question, find a paragraph with the answer \nquestion: {query}'`
-Использование инструкций позволяет значительно улучшить качество поиска и релевантность результатов, что подтверждается тестами на бенчмарках, таких как RuBQ. Для симметричных задач добавление инструкции перед каждым запросом обеспечивает согласованность и повышает точность модели.
-## Поддерживаемые языки
-Эта модель инициализирована pretrain моделью GigaChat и дополнительно обучена на смеси английских и русских данных. Однако, поскольку pretrain GigaChat'a делался в основном на русскоязычных данных, мы рекомендуем использовать эту модель только для русского языка.
-## FAQ
-1. Нужно ли добавлять инструкции к запросу?
-Да, именно так модель обучалась, иначе вы увидите снижение качества. Определение задачи должно быть инструкцией в одном предложении, которая описывает задачу. Это способ настройки текстовых эмбеддингов для разных сценариев с помощью инструкций на естественном языке.
-С другой стороны, добавлять инструкции на сторону документа не требуется.
-2. Почему мои воспроизведённые результаты немного отличаются от указанных в карточке модели?
-Разные версии библиотек transformers и pytorch могут вызывать незначительные, но ненулевые различия в результатах.
-## Ограничения
-Использование этой модели для входных данных, содержащих более 4096 токенов, невозможно.

 ---
 tags:
+- sentence-transformers
+- sentence-similarity
+- feature-extraction
+- dense
+pipeline_tag: sentence-similarity
+library_name: sentence-transformers
 ---
+# SentenceTransformer
+This is a [sentence-transformers](https://www.SBERT.net) model trained. It maps sentences & paragraphs to a 2048-dimensional dense vector space and can be used for semantic textual similarity, semantic search, paraphrase mining, text classification, clustering, and more.
+## Model Details
+### Model Description
+- **Model Type:** Sentence Transformer
+<!-- - **Base model:** [Unknown](https://huggingface.co/unknown) -->
+- **Maximum Sequence Length:** None tokens
+- **Output Dimensionality:** 2048 dimensions
+- **Similarity Function:** Cosine Similarity
+<!-- - **Training Dataset:** Unknown -->
+<!-- - **Language:** Unknown -->
+<!-- - **License:** Unknown -->
+### Model Sources
+- **Documentation:** [Sentence Transformers Documentation](https://sbert.net)
+- **Repository:** [Sentence Transformers on GitHub](https://github.com/UKPLab/sentence-transformers)
+- **Hugging Face:** [Sentence Transformers on Hugging Face](https://huggingface.co/models?library=sentence-transformers)
+### Full Model Architecture
+```
+SentenceTransformer(
+  (0): Transformer({'max_seq_length': None, 'do_lower_case': False, 'architecture': 'GigarEmbedModel'})
+  (1): Pooling({'word_embedding_dimension': 2048, 'pooling_mode_cls_token': False, 'pooling_mode_mean_tokens': True, 'pooling_mode_max_tokens': False, 'pooling_mode_mean_sqrt_len_tokens': False, 'pooling_mode_weightedmean_tokens': False, 'pooling_mode_lasttoken': False, 'include_prompt': True})
+)
+```
+## Usage
+### Direct Usage (Sentence Transformers)
+First install the Sentence Transformers library:
+```bash
+pip install -U sentence-transformers
+```
+Then you can load this model and run inference.
 ```python
+from sentence_transformers import SentenceTransformer
+# Download from the 🤗 Hub
+model = SentenceTransformer("sentence_transformers_model_id")
+# Run inference
+sentences = [
+    'The weather is lovely today.',
+    "It's so sunny outside!",
+    'He drove to the stadium.',
+]
+embeddings = model.encode(sentences)
+print(embeddings.shape)
+# [3, 2048]
+# Get the similarity scores for the embeddings
+similarities = model.similarity(embeddings, embeddings)
+print(similarities.shape)
+# [3, 3]
+```
+<!--
+### Direct Usage (Transformers)
+<details><summary>Click to see the direct usage in Transformers</summary>
+</details>
+-->
+<!--
+### Downstream Usage (Sentence Transformers)
+You can finetune this model on your own dataset.
+<details><summary>Click to expand</summary>
+</details>
+-->
+<!--
+### Out-of-Scope Use
+*List how the model may foreseeably be misused and address what users ought not to do with the model.*
+-->
+<!--
+## Bias, Risks and Limitations
+*What are the known or foreseeable issues stemming from this model? You could also flag here known failure cases or weaknesses of the model.*
+-->
+<!--
+### Recommendations
+*What are recommendations with respect to the foreseeable issues? For example, filtering explicit content.*
+-->
+## Training Details
+### Framework Versions
+- Python: 3.10.12
+- Sentence Transformers: 5.1.1
+- Transformers: 4.51.0
+- PyTorch: 2.5.1+cu124
+- Accelerate: 1.2.1
+- Datasets: 2.21.0
+- Tokenizers: 0.21.4
+## Citation
+### BibTeX
+<!--
+## Glossary
+*Clearly define terms in order to be accessible across audiences.*
+-->
+<!--
+## Model Card Authors
+*Lists the people who create the model card, providing recognition and accountability for the detailed work that goes into its construction.*
+-->
+<!--
+## Model Card Contact
+*Provides a way for people who have updates to the Model Card, suggestions, or questions, to contact the Model Card authors.*
+-->

config.json CHANGED Viewed

@@ -1,96 +1,218 @@
 {
-    "_name_or_path": "ai-sage/Giga-Embeddings-instruct",
-    "_non_freeze_layers_idxs": null,
-    "activation_checkpoint_layers_num": null,
-    "apply_torch_compile_to_projections": true,
-    "architectures": [
-        "GigarEmbedModel"
-    ],
-    "auto_map": {
-        "AutoConfig": "configuration_gigarembed.GigarEmbedConfig",
-        "AutoModel": "modeling_gigarembed.GigarEmbedModel"
     },
-    "latent_attention_config": {
-        "model_type": "latent_attention",
-        "num_latents_value": 512,
-        "num_cross_heads": 8,
-        "cross_dim_head": 2048,
-        "hidden_dim": 2048,
-        "latent_dim": 2048,
-        "mult": 4
     },
     "hidden_size": 2048,
-    "text_config": {
-        "_name_or_path": "ai-sage/Giga-Embeddings-instruct",
-        "apply_qk_norm": true,
-        "attention_bias": false,
-        "attention_dropout": 0.0,
-        "attention_hidden_size": null,
-        "attention_type": "LlamaLatentAttention",
-        "bos_token_id": 1,
-        "delete_logits": true,
-        "deterministic_attention": false,
-        "enable_async_tp": false,
-        "eos_token_id": 2,
-        "freeze_non_embed": false,
-        "fused_mlp": true,
-        "fused_mlp_checkpoint_lvl": 3,
-        "head_dim": 64,
-        "hidden_act": "silu",
-        "hidden_size": 2048,
-        "ignore_index": -100,
-        "init_device": "meta",
-        "initializer_range": 0.02,
-        "intermediate_size": 11008,
-        "kv_lora_rank": 1024,
-        "lora_alpha": null,
-        "lora_r": null,
-        "loss_inplace_backward": false,
-        "max_position_embeddings": 4096,
-        "max_window_layers": 36,
-        "mla_config": {
-            "kv_lora_rank": 1024,
-            "q_lora_rank": 0,
-            "qk_nope_head_dim": 64,
-            "qk_rope_head_dim": 64,
-            "v_head_dim": 128
-        },
-        "mlp_bias": false,
-        "model_type": "gigar",
-        "mtp_loss_weight": 0.1,
-        "mtp_predictor_num": 1,
-        "norm_type": "LlamaRMSNorm",
-        "num_attention_heads": 16,
-        "num_hidden_layers": 36,
-        "num_key_value_heads": 16,
-        "pad_token_id": 2,
-        "parallel_embedding_type": "EmbeddingParallelEmbedding",
-        "pretraining_tp": 1,
-        "q_lora_rank": 0,
-        "qk_nope_head_dim": 64,
-        "qk_rope_head_dim": 64,
-        "rms_norm_eps": 1e-06,
-        "rope_scaling": null,
-        "rope_theta": 100000.0,
-        "skip_init_tp_modules": true,
-        "sliding_window": null,
-        "sp_split_type": "equal",
-        "tie_word_embeddings": false,
-        "tp_group": null,
-        "tp_size": 1,
-        "unk_token_id": 0,
-        "use_cache": false,
-        "use_cache_force": false,
-        "use_custom_rotary_kernel": false,
-        "use_liger": false,
-        "use_mrope": false,
-        "use_mtp": true,
-        "use_sliding_window": false,
-        "v_head_dim": 128,
-        "varlen_input": true,
-        "vocab_size": 128256,
-        "z_loss_eps": 5e-05
     },
-    "torch_dtype": "bfloat16",
-    "transformers_version": "4.48.0"
 }

 {
+  "_non_freeze_layers_idxs": null,
+  "activation_checkpoint_layers_num": null,
+  "add_eos": true,
+  "add_pad_token": true,
+  "apply_torch_compile_to_projections": true,
+  "architectures": [
+    "GigarEmbedModel"
+  ],
+  "auto_map": {
+    "AutoConfig": "configuration_gigarembed.GigarEmbedConfig",
+    "AutoModel": "modeling_gigarembed.GigarEmbedModel"
+  },
+  "hidden_size": 2048,
+  "is_mask_instruction": true,
+  "latent_attention_config": {
+    "_attn_implementation_autoset": false,
+    "_name_or_path": "",
+    "add_cross_attention": false,
+    "architectures": null,
+    "bad_words_ids": null,
+    "begin_suppress_tokens": null,
+    "bos_token_id": null,
+    "chunk_size_feed_forward": 0,
+    "cross_attention_hidden_size": null,
+    "cross_dim_head": 2048,
+    "decoder_start_token_id": null,
+    "diversity_penalty": 0.0,
+    "do_sample": false,
+    "early_stopping": false,
+    "encoder_no_repeat_ngram_size": 0,
+    "eos_token_id": null,
+    "exponential_decay_length_penalty": null,
+    "finetuning_task": null,
+    "forced_bos_token_id": null,
+    "forced_eos_token_id": null,
+    "hidden_dim": 2048,
+    "id2label": {
+      "0": "LABEL_0",
+      "1": "LABEL_1"
     },
+    "is_decoder": false,
+    "is_encoder_decoder": false,
+    "label2id": {
+      "LABEL_0": 0,
+      "LABEL_1": 1
     },
+    "latent_dim": 2048,
+    "length_penalty": 1.0,
+    "max_length": 20,
+    "min_length": 0,
+    "model_type": "latent_attention",
+    "mult": 4,
+    "no_repeat_ngram_size": 0,
+    "num_beam_groups": 1,
+    "num_beams": 1,
+    "num_cross_heads": 8,
+    "num_latents_value": 512,
+    "num_return_sequences": 1,
+    "output_attentions": false,
+    "output_hidden_states": false,
+    "output_scores": false,
+    "pad_token_id": null,
+    "prefix": null,
+    "problem_type": null,
+    "pruned_heads": {},
+    "remove_invalid_values": false,
+    "repetition_penalty": 1.0,
+    "return_dict": true,
+    "return_dict_in_generate": false,
+    "sep_token_id": null,
+    "suppress_tokens": null,
+    "task_specific_params": null,
+    "temperature": 1.0,
+    "tf_legacy_loss": false,
+    "tie_encoder_decoder": false,
+    "tie_word_embeddings": true,
+    "tokenizer_class": null,
+    "top_k": 50,
+    "top_p": 1.0,
+    "torch_dtype": null,
+    "torchscript": false,
+    "typical_p": 1.0,
+    "use_bfloat16": false
+  },
+  "mask_type": "b",
+  "model_type": "gigarembed",
+  "padding_side": "right",
+  "text_config": {
+    "_attn_implementation_autoset": false,
+    "_name_or_path": "ai-sage/Giga-Embeddings-instruct",
+    "add_cross_attention": false,
+    "apply_qk_norm": true,
+    "architectures": null,
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "attention_hidden_size": null,
+    "attention_type": "LlamaLatentAttention",
+    "bad_words_ids": null,
+    "begin_suppress_tokens": null,
+    "bos_token_id": 1,
+    "chunk_size_feed_forward": 0,
+    "cross_attention_hidden_size": null,
+    "decoder_start_token_id": null,
+    "delete_logits": true,
+    "deterministic_attention": false,
+    "diversity_penalty": 0.0,
+    "do_sample": false,
+    "early_stopping": false,
+    "enable_async_tp": false,
+    "encoder_no_repeat_ngram_size": 0,
+    "eos_token_id": 2,
+    "exponential_decay_length_penalty": null,
+    "finetuning_task": null,
+    "forced_bos_token_id": null,
+    "forced_eos_token_id": null,
+    "freeze_non_embed": false,
+    "fused_mlp": true,
+    "fused_mlp_checkpoint_lvl": 3,
+    "head_dim": 64,
+    "hidden_act": "silu",
     "hidden_size": 2048,
+    "id2label": {
+      "0": "LABEL_0",
+      "1": "LABEL_1"
+    },
+    "ignore_index": -100,
+    "init_device": "meta",
+    "initializer_range": 0.02,
+    "intermediate_size": 11008,
+    "is_decoder": false,
+    "is_encoder_decoder": false,
+    "kv_lora_rank": 1024,
+    "label2id": {
+      "LABEL_0": 0,
+      "LABEL_1": 1
+    },
+    "length_penalty": 1.0,
+    "lora_alpha": null,
+    "lora_r": null,
+    "loss_inplace_backward": false,
+    "max_length": 20,
+    "max_position_embeddings": 4096,
+    "max_window_layers": 36,
+    "min_length": 0,
+    "mla_config": {
+      "kv_lora_rank": 1024,
+      "q_lora_rank": 0,
+      "qk_nope_head_dim": 64,
+      "qk_rope_head_dim": 64,
+      "v_head_dim": 128
     },
+    "mlp_bias": false,
+    "model_type": "gigar",
+    "mtp_loss_weight": 0.1,
+    "mtp_predictor_num": 1,
+    "no_repeat_ngram_size": 0,
+    "norm_type": "LlamaRMSNorm",
+    "num_attention_heads": 16,
+    "num_beam_groups": 1,
+    "num_beams": 1,
+    "num_hidden_layers": 36,
+    "num_key_value_heads": 16,
+    "num_return_sequences": 1,
+    "output_attentions": false,
+    "output_hidden_states": false,
+    "output_scores": false,
+    "pad_token_id": 2,
+    "parallel_embedding_type": "EmbeddingParallelEmbedding",
+    "prefix": null,
+    "pretraining_tp": 1,
+    "problem_type": null,
+    "pruned_heads": {},
+    "q_lora_rank": 0,
+    "qk_nope_head_dim": 64,
+    "qk_rope_head_dim": 64,
+    "remove_invalid_values": false,
+    "repetition_penalty": 1.0,
+    "return_dict": true,
+    "return_dict_in_generate": false,
+    "rms_norm_eps": 1e-06,
+    "rope_scaling": null,
+    "rope_theta": 100000.0,
+    "sep_token_id": null,
+    "skip_init_tp_modules": true,
+    "sliding_window": null,
+    "sp_split_type": "equal",
+    "suppress_tokens": null,
+    "task_specific_params": null,
+    "temperature": 1.0,
+    "tf_legacy_loss": false,
+    "tie_encoder_decoder": false,
+    "tie_word_embeddings": false,
+    "tokenizer_class": null,
+    "top_k": 50,
+    "top_p": 1.0,
+    "torch_dtype": null,
+    "torchscript": false,
+    "tp_group": null,
+    "tp_size": 1,
+    "typical_p": 1.0,
+    "unk_token_id": 0,
+    "use_bfloat16": false,
+    "use_cache": false,
+    "use_cache_force": false,
+    "use_custom_rotary_kernel": false,
+    "use_liger": false,
+    "use_mrope": false,
+    "use_mtp": true,
+    "use_sliding_window": false,
+    "v_head_dim": 128,
+    "varlen_input": true,
+    "vocab_size": 128256,
+    "z_loss_eps": 5e-05
+  },
+  "torch_dtype": "float32",
+  "transformers_version": "4.51.0"
 }

config_sentence_transformers.json CHANGED Viewed

@@ -1,10 +1,14 @@
 {
   "__version__": {
-    "sentence_transformers": "3.3.1",
-    "transformers": "4.48.0",
-    "pytorch": "2.1.1+cu121"
   },
-  "prompts": {},
   "default_prompt_name": null,
   "similarity_fn_name": "cosine"
 }

 {
+  "model_type": "SentenceTransformer",
   "__version__": {
+    "sentence_transformers": "5.1.1",
+    "transformers": "4.51.0",
+    "pytorch": "2.5.1+cu124"
+  },
+  "prompts": {
+    "query": "",
+    "document": ""
   },
   "default_prompt_name": null,
   "similarity_fn_name": "cosine"
 }

modeling_gigarembed.py CHANGED Viewed

@@ -1135,7 +1135,8 @@ class GigarEmbedModel(PreTrainedModel):
         if return_embeddings:
             return self.mean_pool(last_hidden, attention_mask)
-        return last_hidden
     def mean_pool(self, last_hidden: torch.Tensor, attention_mask: torch.Tensor):
         last_hidden = last_hidden.masked_fill(~attention_mask[..., None].bool(), 0.0)

         if return_embeddings:
             return self.mean_pool(last_hidden, attention_mask)
+        # return last_hidden
+        return BaseModelOutputWithPast(last_hidden_state=last_hidden)
     def mean_pool(self, last_hidden: torch.Tensor, attention_mask: torch.Tensor):
         last_hidden = last_hidden.masked_fill(~attention_mask[..., None].bool(), 0.0)

sentence_bert_config.json CHANGED Viewed

@@ -1,4 +1,4 @@
 {
-  "max_seq_length": null,
-  "do_lower_case": false
 }

 {
+    "max_seq_length": null,
+    "do_lower_case": false
 }

tokenizer_config.json CHANGED Viewed

@@ -2086,7 +2086,7 @@
   "padding_side": "right",
   "sep_token": "<unk>",
   "stride": 0,
-  "tokenizer_class": "PreTrainedTokenizerFast",
   "truncation_side": "right",
   "truncation_strategy": "longest_first",
   "unk_token": "<unk>"

   "padding_side": "right",
   "sep_token": "<unk>",
   "stride": 0,
+  "tokenizer_class": "PreTrainedTokenizer",
   "truncation_side": "right",
   "truncation_strategy": "longest_first",
   "unk_token": "<unk>"