Upload 6 files

Model Description
This model is an extended version of XLM-R (XLM-RoBERTa), specifically adapted for Geez script languages. Leveraging the robust architecture of XLM-RoBERTa, it incorporates additional tokens and vocabulary to handle the unique linguistic features and characters of Geez script languages. This extension enhances the model’s ability to perform a variety of NLP tasks, including classification, sentiment analysis, and more, within the Geez script context.

Tokenizer Description
The tokenizer for this extended XLM-R model has been customized to include new tokens that represent the characters and symbols specific to Geez script languages. It is designed to preprocess text data by converting Geez script characters into tokens compatible with the extended model. This ensures that text in the Geez script is accurately and efficiently processed for downstream tasks.

Files changed (6) hide show

added_tokens.json +0 -0
config.json +28 -0
model.safetensors +3 -0
sentencepiece.bpe.model +3 -0
special_tokens_map.json +15 -0
tokenizer_config.json +0 -0

added_tokens.json ADDED Viewed

The diff for this file is too large to render. See raw diff

config.json ADDED Viewed

	@@ -0,0 +1,28 @@

+{
+  "_name_or_path": "xlm-roberta-base",
+  "architectures": [
+    "XLMRobertaForSequenceClassification"
+  ],
+  "attention_probs_dropout_prob": 0.1,
+  "bos_token_id": 0,
+  "classifier_dropout": null,
+  "eos_token_id": 2,
+  "hidden_act": "gelu",
+  "hidden_dropout_prob": 0.1,
+  "hidden_size": 768,
+  "initializer_range": 0.02,
+  "intermediate_size": 3072,
+  "layer_norm_eps": 1e-05,
+  "max_position_embeddings": 514,
+  "model_type": "xlm-roberta",
+  "num_attention_heads": 12,
+  "num_hidden_layers": 12,
+  "output_past": true,
+  "pad_token_id": 1,
+  "position_embedding_type": "absolute",
+  "torch_dtype": "float32",
+  "transformers_version": "4.41.2",
+  "type_vocab_size": 1,
+  "use_cache": true,
+  "vocab_size": 280147
+}

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f4fdaffa07400cdbbc4552fe7cd470fe36eac02bb9e4f03d8923d6bc07491357
+size 1204810552

sentencepiece.bpe.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:cfc8146abe2a0488e9e2a0c56de7952f7c11ab059eca145a0a727afce0db2865
+size 5069051

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "bos_token": "<s>",
+  "cls_token": "<s>",
+  "eos_token": "</s>",
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": true,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": "<pad>",
+  "sep_token": "</s>",
+  "unk_token": "<unk>"
+}

tokenizer_config.json ADDED Viewed

The diff for this file is too large to render. See raw diff