codebyzeb commited on 30 days ago

Commit

a93858a

verified ·

1 Parent(s): d6f07d9

Upload folder using huggingface_hub

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

Cantonese/config.json +31 -0
Cantonese/generation_config.json +6 -0
Cantonese/model.safetensors +3 -0
Cantonese/special_tokens_map.json +30 -0
Cantonese/tokenizer.json +269 -0
Cantonese/tokenizer_config.json +44 -0
Cantonese/training_args.bin +3 -0
Cantonese/vocab.json +1 -0
Dutch/config.json +31 -0
Dutch/generation_config.json +6 -0
Dutch/model.safetensors +3 -0
Dutch/special_tokens_map.json +30 -0
Dutch/tokenizer.json +167 -0
Dutch/tokenizer_config.json +44 -0
Dutch/training_args.bin +3 -0
Dutch/vocab.json +1 -0
EnglishNA/config.json +31 -0
EnglishNA/generation_config.json +6 -0
EnglishNA/model.safetensors +3 -0
EnglishNA/special_tokens_map.json +30 -0
EnglishNA/tokenizer.json +164 -0
EnglishNA/tokenizer_config.json +44 -0
EnglishNA/training_args.bin +3 -0
EnglishNA/vocab.json +1 -0
EnglishUK/config.json +31 -0
EnglishUK/generation_config.json +6 -0
EnglishUK/model.safetensors +3 -0
EnglishUK/special_tokens_map.json +30 -0
EnglishUK/tokenizer.json +168 -0
EnglishUK/tokenizer_config.json +44 -0
EnglishUK/training_args.bin +3 -0
EnglishUK/vocab.json +1 -0
Estonian/config.json +31 -0
Estonian/generation_config.json +6 -0
Estonian/model.safetensors +3 -0
Estonian/special_tokens_map.json +30 -0
Estonian/tokenizer.json +185 -0
Estonian/tokenizer_config.json +44 -0
Estonian/training_args.bin +3 -0
Estonian/vocab.json +1 -0
French/config.json +31 -0
French/generation_config.json +6 -0
French/model.safetensors +3 -0
French/special_tokens_map.json +30 -0
French/tokenizer.json +156 -0
French/tokenizer_config.json +44 -0
French/training_args.bin +3 -0
French/vocab.json +1 -0
German/config.json +31 -0
German/generation_config.json +6 -0

Cantonese/config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.3,
+  "bos_token_id": 3,
+  "embd_pdrop": 0.3,
+  "eos_token_id": 3,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_embd": 128,
+  "n_head": 4,
+  "n_inner": 512,
+  "n_layer": 4,
+  "n_positions": 256,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.3,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.44.2",
+  "use_cache": true,
+  "vocab_size": 152
+}

Cantonese/generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 3,
+  "eos_token_id": 3,
+  "transformers_version": "4.44.2"
+}

Cantonese/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e80e5a76e879718c865ae642eac461076dbe1cd2c3245f386ffe6f9109d2a987
+size 3387304

Cantonese/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "PAD",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "UNK",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

Cantonese/tokenizer.json ADDED Viewed

	@@ -0,0 +1,269 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 0,
+      "content": "UNK",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 1,
+      "content": "PAD",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 2,
+      "content": "WORD_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 3,
+      "content": "UTT_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "WhitespaceSplit"
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "UTT_BOUNDARY": {
+        "id": "UTT_BOUNDARY",
+        "ids": [
+          3
+        ],
+        "tokens": [
+          "UTT_BOUNDARY"
+        ]
+      }
+    }
+  },
+  "decoder": null,
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "UNK": 0,
+      "PAD": 1,
+      "WORD_BOUNDARY": 2,
+      "UTT_BOUNDARY": 3,
+      "j": 4,
+      "ɐ˥": 5,
+      "t": 6,
+      "k": 7,
+      "ɐu˧˥": 8,
+      "i˨": 9,
+      "n": 10,
+      "i˧˩̰": 11,
+      "y˨": 12,
+      "s": 13,
+      "ɐ˨": 14,
+      "p": 15,
+      "ts": 16,
+      "ɐu˥": 17,
+      "ɪ̞˧˥": 18,
+      "ŋ": 19,
+      "ɵ˧": 20,
+      "a̞˧": 21,
+      "l": 22,
+      "ʊ̟˥": 23,
+      "a̞˧˩̰": 24,
+      "ɛ˥": 25,
+      "ei˩˧": 26,
+      "w": 27,
+      "a̞˨": 28,
+      "ɐi˧˥": 29,
+      "a̞˧˥": 30,
+      "m̩˧˥": 31,
+      "m": 32,
+      "ou˥": 33,
+      "ei˥": 34,
+      "i˧": 35,
+      "ɔ̽˧˥": 36,
+      "tʰ": 37,
+      "i˥": 38,
+      "f": 39,
+      "aːĭ˧": 40,
+      "h": 41,
+      "ɵy˧": 42,
+      "a̞˥": 43,
+      "ei˧˩̰": 44,
+      "ou˨": 45,
+      "ɔ̽˧": 46,
+      "ɐi˧˩̰": 47,
+      "u˧": 48,
+      "ɔːĭ˥": 49,
+      "ɐu˨": 50,
+      "ei˧˥": 51,
+      "ɐi˨": 52,
+      "ʊ̟˧˩̰": 53,
+      "ʊ̟˨": 54,
+      "a̞˩˧": 55,
+      "ou˧˥": 56,
+      "aːĭ˧˥": 57,
+      "ɔ̽˨": 58,
+      "ɛ˩˧": 59,
+      "ɪ̞˨": 60,
+      "iːŭ˧": 61,
+      "ɛ˧˩̰": 62,
+      "m̩˧˩̰": 63,
+      "ɵ˧˥": 64,
+      "ei˧": 65,
+      "ɐu˧˩̰": 66,
+      "m̩˧": 67,
+      "ɐ˧˥": 68,
+      "ɐu˩˧": 69,
+      "ɐi˥": 70,
+      "ɔ̽˥": 71,
+      "ɔ̽˧˩̰": 72,
+      "ɔːĭ˧": 73,
+      "ou˩˧": 74,
+      "m̩˥": 75,
+      "ɐ˧": 76,
+      "tsʰ": 77,
+      "ɛ˧˥": 78,
+      "i˧˥": 79,
+      "ɔ̽˩˧": 80,
+      "kʰ": 81,
+      "ɐ˧˩̰": 82,
+      "aːŭ˧˥": 83,
+      "pʰ": 84,
+      "aːĭ˧˩̰": 85,
+      "ɵy˩˧": 86,
+      "ɛ˧": 87,
+      "u˧˥": 88,
+      "ɛ˨": 89,
+      "ʊ̟˧": 90,
+      "u˥": 91,
+      "m̩˩˧": 92,
+      "aːŭ˧": 93,
+      "œ̞˩˧": 94,
+      "i˩˧": 95,
+      "ɪ̞˧˩̰": 96,
+      "u˨": 97,
+      "ɪ̞˥": 98,
+      "iːŭ˧˩̰": 99,
+      "œ̞˧˥": 100,
+      "y˧": 101,
+      "uːĭ˩˧": 102,
+      "uːĭ˥": 103,
+      "ɵy˧˥": 104,
+      "y˧˩̰": 105,
+      "ɔːĭ˧˥": 106,
+      "ɛ": 107,
+      "ou˧": 108,
+      "ei˨": 109,
+      "ɵ˥": 110,
+      "u˧˩̰": 111,
+      "y˥": 112,
+      "œ̞˥": 113,
+      "œ̞˧˩̰": 114,
+      "aːĭ˨": 115,
+      "ɐ˩˧": 116,
+      "œ̞˧": 117,
+      "uːĭ˧˥": 118,
+      "ɐu˧": 119,
+      "ɐi˩˧": 120,
+      "ɐi˧": 121,
+      "ou˧˩̰": 122,
+      "aːĭ˥": 123,
+      "aːŭ˥": 124,
+      "ŋ˩˧": 125,
+      "y˧˥": 126,
+      "iːŭ˥": 127,
+      "ɔːĭ˨": 128,
+      "ʊ̟˧˥": 129,
+      "iːŭ˧˥": 130,
+      "ɵy˥": 131,
+      "ɔːĭ˧˩̰": 132,
+      "uːĭ˧": 133,
+      "ɵy˧˩̰": 134,
+      "œ̞˨": 135,
+      "m̩˨": 136,
+      "aːŭ˧˩̰": 137,
+      "y˩˧": 138,
+      "aːŭ˩˧": 139,
+      "aːĭ˩˧": 140,
+      "uːĭ˨": 141,
+      "ɵy˨": 142,
+      "aːŭ˨": 143,
+      "ɪ̞˧": 144,
+      "ɵ˨": 145,
+      "iːŭ˩˧": 146,
+      "iːŭ˨": 147,
+      "ɵ˧˩̰": 148,
+      "uːĭ˧˩̰": 149,
+      "u˩˧": 150,
+      "ŋ˧˩̰": 151
+    },
+    "unk_token": "UNK"
+  }
+}

Cantonese/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "UNK",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "PAD",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "WORD_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "UTT_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "UTT_BOUNDARY",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "UTT_BOUNDARY",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "PAD",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "UNK"
+}

Cantonese/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a3b843b2d3e207797e1396680aa7485a6a0bdcef4077450466fa42f9c477b48e
+size 5368

Cantonese/vocab.json ADDED Viewed

	@@ -0,0 +1 @@

+ {"UNK":0,"PAD":1,"WORD_BOUNDARY":2,"UTT_BOUNDARY":3,"j":4,"ɐ˥":5,"t":6,"k":7,"ɐu˧˥":8,"i˨":9,"n":10,"i˧˩̰":11,"y˨":12,"s":13,"ɐ˨":14,"p":15,"ts":16,"ɐu˥":17,"ɪ̞˧˥":18,"ŋ":19,"ɵ˧":20,"a̞˧":21,"l":22,"ʊ̟˥":23,"a̞˧˩̰":24,"ɛ˥":25,"ei˩˧":26,"w":27,"a̞˨":28,"ɐi˧˥":29,"a̞˧˥":30,"m̩˧˥":31,"m":32,"ou˥":33,"ei˥":34,"i˧":35,"ɔ̽˧˥":36,"tʰ":37,"i˥":38,"f":39,"aːĭ˧":40,"h":41,"ɵy˧":42,"a̞˥":43,"ei˧˩̰":44,"ou˨":45,"ɔ̽˧":46,"ɐi˧˩̰":47,"u˧":48,"ɔːĭ˥":49,"ɐu˨":50,"ei˧˥":51,"ɐi˨":52,"ʊ̟˧˩̰":53,"ʊ̟˨":54,"a̞˩˧":55,"ou˧˥":56,"aːĭ˧˥":57,"ɔ̽˨":58,"ɛ˩˧":59,"ɪ̞˨":60,"iːŭ˧":61,"ɛ˧˩̰":62,"m̩˧˩̰":63,"ɵ˧˥":64,"ei˧":65,"ɐu˧˩̰":66,"m̩˧":67,"ɐ˧˥":68,"ɐu˩˧":69,"ɐi˥":70,"ɔ̽˥":71,"ɔ̽˧˩̰":72,"ɔːĭ˧":73,"ou˩˧":74,"m̩˥":75,"ɐ˧":76,"tsʰ":77,"ɛ˧˥":78,"i˧˥":79,"ɔ̽˩˧":80,"kʰ":81,"ɐ˧˩̰":82,"aːŭ˧˥":83,"pʰ":84,"aːĭ˧˩̰":85,"ɵy˩˧":86,"ɛ˧":87,"u˧˥":88,"ɛ˨":89,"ʊ̟˧":90,"u˥":91,"m̩˩˧":92,"aːŭ˧":93,"œ̞˩˧":94,"i˩˧":95,"ɪ̞˧˩̰":96,"u˨":97,"ɪ̞˥":98,"iːŭ˧˩̰":99,"œ̞˧˥":100,"y˧":101,"uːĭ˩˧":102,"uːĭ˥":103,"ɵy˧˥":104,"y˧˩̰":105,"ɔːĭ˧˥":106,"ɛ":107,"ou˧":108,"ei˨":109,"ɵ˥":110,"u˧˩̰":111,"y˥":112,"œ̞˥":113,"œ̞˧˩̰":114,"aːĭ˨":115,"ɐ˩˧":116,"œ̞˧":117,"uːĭ˧˥":118,"ɐu˧":119,"ɐi˩˧":120,"ɐi˧":121,"ou˧˩̰":122,"aːĭ˥":123,"aːŭ˥":124,"ŋ˩˧":125,"y˧˥":126,"iːŭ˥":127,"ɔːĭ˨":128,"ʊ̟˧˥":129,"iːŭ˧˥":130,"ɵy˥":131,"ɔːĭ˧˩̰":132,"uːĭ˧":133,"ɵy˧˩̰":134,"œ̞˨":135,"m̩˨":136,"aːŭ˧˩̰":137,"y˩˧":138,"aːŭ˩˧":139,"aːĭ˩˧":140,"uːĭ˨":141,"ɵy˨":142,"aːŭ˨":143,"ɪ̞˧":144,"ɵ˨":145,"iːŭ˩˧":146,"iːŭ˨":147,"ɵ˧˩̰":148,"uːĭ˧˩̰":149,"u˩˧":150,"ŋ˧˩̰":151}

Dutch/config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.3,
+  "bos_token_id": 3,
+  "embd_pdrop": 0.3,
+  "eos_token_id": 3,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_embd": 128,
+  "n_head": 4,
+  "n_inner": 512,
+  "n_layer": 4,
+  "n_positions": 256,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.3,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.44.2",
+  "use_cache": true,
+  "vocab_size": 50
+}

Dutch/generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 3,
+  "eos_token_id": 3,
+  "transformers_version": "4.44.2"
+}

Dutch/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d997b37f113b5e7583ce581f013938c4cb1f2d94d081bc4d7d1cfca9f09b1acc
+size 3335080

Dutch/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "PAD",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "UNK",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

Dutch/tokenizer.json ADDED Viewed

	@@ -0,0 +1,167 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 0,
+      "content": "UNK",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 1,
+      "content": "PAD",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 2,
+      "content": "WORD_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 3,
+      "content": "UTT_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Whitespace"
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "UTT_BOUNDARY": {
+        "id": "UTT_BOUNDARY",
+        "ids": [
+          3
+        ],
+        "tokens": [
+          "UTT_BOUNDARY"
+        ]
+      }
+    }
+  },
+  "decoder": null,
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "UNK": 0,
+      "PAD": 1,
+      "WORD_BOUNDARY": 2,
+      "UTT_BOUNDARY": 3,
+      "z": 4,
+      "oː": 5,
+      "j": 6,
+      "ãː": 7,
+      "ɦ": 8,
+      "ɾ": 9,
+      "d": 10,
+      "i": 11,
+      "ɛ": 12,
+      "p": 13,
+      "ɪ": 14,
+      "k": 15,
+      "ɑ": 16,
+      "l": 17,
+      "ɛː": 18,
+      "n": 19,
+      "s": 20,
+      "v": 21,
+      "ə": 22,
+      "ɛi": 23,
+      "ʋ": 24,
+      "t": 25,
+      "m": 26,
+      "ɣ": 27,
+      "ʏ": 28,
+      "ɔ": 29,
+      "x": 30,
+      "u": 31,
+      "f": 32,
+      "ŋ": 33,
+      "øː": 34,
+      "b": 35,
+      "ɔː": 36,
+      "ʌu": 37,
+      "y": 38,
+      "œy": 39,
+      "tʲ": 40,
+      "w": 41,
+      "ʃ": 42,
+      "t̠ʃ": 43,
+      "ɲ": 44,
+      "ʒ": 45,
+      "iː": 46,
+      "ɡ": 47,
+      "d̠ʒ": 48,
+      "ã": 49
+    },
+    "unk_token": "UNK"
+  }
+}

Dutch/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "UNK",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "PAD",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "WORD_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "UTT_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "UTT_BOUNDARY",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "UTT_BOUNDARY",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "PAD",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "UNK"
+}

Dutch/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3079aa2abf5484337ccf8ff4e6bddef13c05a4e0558bf07f5790c65d85b7e185
+size 5368

Dutch/vocab.json ADDED Viewed

	@@ -0,0 +1 @@

+ {"UNK":0,"PAD":1,"WORD_BOUNDARY":2,"UTT_BOUNDARY":3,"z":4,"oː":5,"j":6,"ãː":7,"ɦ":8,"ɾ":9,"d":10,"i":11,"ɛ":12,"p":13,"ɪ":14,"k":15,"ɑ":16,"l":17,"ɛː":18,"n":19,"s":20,"v":21,"ə":22,"ɛi":23,"ʋ":24,"t":25,"m":26,"ɣ":27,"ʏ":28,"ɔ":29,"x":30,"u":31,"f":32,"ŋ":33,"øː":34,"b":35,"ɔː":36,"ʌu":37,"y":38,"œy":39,"tʲ":40,"w":41,"ʃ":42,"t̠ʃ":43,"ɲ":44,"ʒ":45,"iː":46,"ɡ":47,"d̠ʒ":48,"ã":49}

EnglishNA/config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.3,
+  "bos_token_id": 3,
+  "embd_pdrop": 0.3,
+  "eos_token_id": 3,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_embd": 128,
+  "n_head": 4,
+  "n_inner": 512,
+  "n_layer": 4,
+  "n_positions": 256,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.3,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.44.2",
+  "use_cache": true,
+  "vocab_size": 47
+}

EnglishNA/generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 3,
+  "eos_token_id": 3,
+  "transformers_version": "4.44.2"
+}

EnglishNA/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:273fc2318bc0e4e81c4a414587514493ce5f67caf190c1c9cd3e8967a3eade00
+size 3333544

EnglishNA/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "PAD",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "UNK",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

EnglishNA/tokenizer.json ADDED Viewed

	@@ -0,0 +1,164 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 0,
+      "content": "UNK",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 1,
+      "content": "PAD",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 2,
+      "content": "WORD_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 3,
+      "content": "UTT_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Whitespace"
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "UTT_BOUNDARY": {
+        "id": "UTT_BOUNDARY",
+        "ids": [
+          3
+        ],
+        "tokens": [
+          "UTT_BOUNDARY"
+        ]
+      }
+    }
+  },
+  "decoder": null,
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "UNK": 0,
+      "PAD": 1,
+      "WORD_BOUNDARY": 2,
+      "UTT_BOUNDARY": 3,
+      "d̠ʒ": 4,
+      "ʌ": 5,
+      "s": 6,
+      "t": 7,
+      "l": 8,
+      "aɪ": 9,
+      "k": 10,
+      "j": 11,
+      "ʊ": 12,
+      "ɹ": 13,
+      "b": 14,
+      "æ": 15,
+      "h": 16,
+      "oʊ": 17,
+      "m": 18,
+      "iː": 19,
+      "ð": 20,
+      "ɛ": 21,
+      "z": 22,
+      "f": 23,
+      "eɪ": 24,
+      "w": 25,
+      "ɪ": 26,
+      "ɡ": 27,
+      "ɑ": 28,
+      "ə": 29,
+      "p": 30,
+      "uː": 31,
+      "i": 32,
+      "θ": 33,
+      "ŋ": 34,
+      "ɔ": 35,
+      "ɔɪ": 36,
+      "n": 37,
+      "d": 38,
+      "aʊ": 39,
+      "v": 40,
+      "ɜː": 41,
+      "t̠ʃ": 42,
+      "ʃ": 43,
+      "iə": 44,
+      "ʒ": 45,
+      "x": 46
+    },
+    "unk_token": "UNK"
+  }
+}

EnglishNA/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "UNK",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "PAD",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "WORD_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "UTT_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "UTT_BOUNDARY",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "UTT_BOUNDARY",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "PAD",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "UNK"
+}

EnglishNA/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7ee949648c57d5e1171f9887784b0ac9d35ee1311228155f45271dc332aaa7df
+size 5368

EnglishNA/vocab.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"UNK":0,"PAD":1,"WORD_BOUNDARY":2,"UTT_BOUNDARY":3,"d̠ʒ":4,"ʌ":5,"s":6,"t":7,"l":8,"aɪ":9,"k":10,"j":11,"ʊ":12,"ɹ":13,"b":14,"æ":15,"h":16,"oʊ":17,"m":18,"iː":19,"ð":20,"ɛ":21,"z":22,"f":23,"eɪ":24,"w":25,"ɪ":26,"ɡ":27,"ɑ":28,"ə":29,"p":30,"uː":31,"i":32,"θ":33,"ŋ":34,"ɔ":35,"ɔɪ":36,"n":37,"d":38,"aʊ":39,"v":40,"ɜː":41,"t̠ʃ":42,"ʃ":43,"iə":44,"ʒ":45,"x":46}

EnglishUK/config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.3,
+  "bos_token_id": 3,
+  "embd_pdrop": 0.3,
+  "eos_token_id": 3,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_embd": 128,
+  "n_head": 4,
+  "n_inner": 512,
+  "n_layer": 4,
+  "n_positions": 256,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.3,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.44.2",
+  "use_cache": true,
+  "vocab_size": 51
+}

EnglishUK/generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 3,
+  "eos_token_id": 3,
+  "transformers_version": "4.44.2"
+}

EnglishUK/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:5e4c9057fadb4c7c4b6f32215fb526998cb95f5fd1011bd83a683f6f8a103cca
+size 3335592

EnglishUK/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "PAD",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "UNK",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

EnglishUK/tokenizer.json ADDED Viewed

	@@ -0,0 +1,168 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 0,
+      "content": "UNK",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 1,
+      "content": "PAD",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 2,
+      "content": "WORD_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 3,
+      "content": "UTT_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Whitespace"
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "UTT_BOUNDARY": {
+        "id": "UTT_BOUNDARY",
+        "ids": [
+          3
+        ],
+        "tokens": [
+          "UTT_BOUNDARY"
+        ]
+      }
+    }
+  },
+  "decoder": null,
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "UNK": 0,
+      "PAD": 1,
+      "WORD_BOUNDARY": 2,
+      "UTT_BOUNDARY": 3,
+      "ð": 4,
+      "æ": 5,
+      "tʰ": 6,
+      "ɡ": 7,
+      "ʊ": 8,
+      "d": 9,
+      "ɑː": 10,
+      "l": 11,
+      "ɪ": 12,
+      "n": 13,
+      "eɪ": 14,
+      "t̠ʃ": 15,
+      "w": 16,
+      "ɒ": 17,
+      "ʌ": 18,
+      "z": 19,
+      "m": 20,
+      "iː": 21,
+      "aɪ": 22,
+      "h": 23,
+      "e": 24,
+      "kʰ": 25,
+      "s": 26,
+      "ə": 27,
+      "ɔː": 28,
+      "ɹ": 29,
+      "i": 30,
+      "əʊ": 31,
+      "uː": 32,
+      "j": 33,
+      "ɪə": 34,
+      "ɔɪ": 35,
+      "v": 36,
+      "f": 37,
+      "ɜː": 38,
+      "b": 39,
+      "pʰ": 40,
+      "d̠ʒ": 41,
+      "ɐ": 42,
+      "eə": 43,
+      "ʃ": 44,
+      "θ": 45,
+      "ŋ": 46,
+      "aʊ": 47,
+      "ʊə": 48,
+      "n̩": 49,
+      "ʒ": 50
+    },
+    "unk_token": "UNK"
+  }
+}

EnglishUK/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "UNK",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "PAD",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "WORD_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "UTT_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "UTT_BOUNDARY",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "UTT_BOUNDARY",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "PAD",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "UNK"
+}

EnglishUK/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:383bf2e423c7742c010ffe46113f10f008fc82558484de49464279c0dcf24a23
+size 5368

EnglishUK/vocab.json ADDED Viewed

	@@ -0,0 +1 @@

+ {"UNK":0,"PAD":1,"WORD_BOUNDARY":2,"UTT_BOUNDARY":3,"ð":4,"æ":5,"tʰ":6,"ɡ":7,"ʊ":8,"d":9,"ɑː":10,"l":11,"ɪ":12,"n":13,"eɪ":14,"t̠ʃ":15,"w":16,"ɒ":17,"ʌ":18,"z":19,"m":20,"iː":21,"aɪ":22,"h":23,"e":24,"kʰ":25,"s":26,"ə":27,"ɔː":28,"ɹ":29,"i":30,"əʊ":31,"uː":32,"j":33,"ɪə":34,"ɔɪ":35,"v":36,"f":37,"ɜː":38,"b":39,"pʰ":40,"d̠ʒ":41,"ɐ":42,"eə":43,"ʃ":44,"θ":45,"ŋ":46,"aʊ":47,"ʊə":48,"n̩":49,"ʒ":50}

Estonian/config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.3,
+  "bos_token_id": 3,
+  "embd_pdrop": 0.3,
+  "eos_token_id": 3,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_embd": 128,
+  "n_head": 4,
+  "n_inner": 512,
+  "n_layer": 4,
+  "n_positions": 256,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.3,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.44.2",
+  "use_cache": true,
+  "vocab_size": 68
+}

Estonian/generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 3,
+  "eos_token_id": 3,
+  "transformers_version": "4.44.2"
+}

Estonian/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:afee2c8ad18b21e611292f658db3801ecafae03015168233fc9a039f2175e0c8
+size 3344296

Estonian/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "PAD",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "UNK",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

Estonian/tokenizer.json ADDED Viewed

	@@ -0,0 +1,185 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 0,
+      "content": "UNK",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 1,
+      "content": "PAD",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 2,
+      "content": "WORD_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 3,
+      "content": "UTT_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Whitespace"
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "UTT_BOUNDARY": {
+        "id": "UTT_BOUNDARY",
+        "ids": [
+          3
+        ],
+        "tokens": [
+          "UTT_BOUNDARY"
+        ]
+      }
+    }
+  },
+  "decoder": null,
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "UNK": 0,
+      "PAD": 1,
+      "WORD_BOUNDARY": 2,
+      "UTT_BOUNDARY": 3,
+      "n": 4,
+      "o": 5,
+      "t": 6,
+      "ʃ": 7,
+      "a": 8,
+      "uː": 9,
+      "m": 10,
+      "u": 11,
+      "tʲ": 12,
+      "i": 13,
+      "s": 14,
+      "eː": 15,
+      "d": 16,
+      "iː": 17,
+      "k": 18,
+      "ɡ": 19,
+      "ɑ": 20,
+      "ɤ": 21,
+      "ʊ": 22,
+      "sʲ": 23,
+      "j": 24,
+      "aː": 25,
+      "h": 26,
+      "v": 27,
+      "æi": 28,
+      "kː": 29,
+      "e": 30,
+      "ɪ": 31,
+      "tː": 32,
+      "r": 33,
+      "ɛ": 34,
+      "mː": 35,
+      "p": 36,
+      "sː": 37,
+      "æ": 38,
+      "l": 39,
+      "pː": 40,
+      "yː": 41,
+      "æː": 42,
+      "b": 43,
+      "ɔ": 44,
+      "ɤː": 45,
+      "lː": 46,
+      "ø": 47,
+      "øː": 48,
+      "ŋ": 49,
+      "y": 50,
+      "oː": 51,
+      "rː": 52,
+      "ɲ": 53,
+      "nː": 54,
+      "w": 55,
+      "tʲː": 56,
+      "øɪ̯": 57,
+      "f": 58,
+      "dʲ": 59,
+      "sʲː": 60,
+      "t̠ʃ": 61,
+      "ʃː": 62,
+      "ʒ": 63,
+      "z": 64,
+      "fː": 65,
+      "dː": 66,
+      "yi": 67
+    },
+    "unk_token": "UNK"
+  }
+}

Estonian/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "UNK",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "PAD",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "WORD_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "UTT_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "UTT_BOUNDARY",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "UTT_BOUNDARY",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "PAD",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "UNK"
+}

Estonian/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9d97e5ae5878d9838daf01a6d58ddfa4f068f3912a2fe181a8f6bf2e3d1465a3
+size 5368

Estonian/vocab.json ADDED Viewed

	@@ -0,0 +1 @@

+ {"UNK":0,"PAD":1,"WORD_BOUNDARY":2,"UTT_BOUNDARY":3,"n":4,"o":5,"t":6,"ʃ":7,"a":8,"uː":9,"m":10,"u":11,"tʲ":12,"i":13,"s":14,"eː":15,"d":16,"iː":17,"k":18,"ɡ":19,"ɑ":20,"ɤ":21,"ʊ":22,"sʲ":23,"j":24,"aː":25,"h":26,"v":27,"æi":28,"kː":29,"e":30,"ɪ":31,"tː":32,"r":33,"ɛ":34,"mː":35,"p":36,"sː":37,"æ":38,"l":39,"pː":40,"yː":41,"æː":42,"b":43,"ɔ":44,"ɤː":45,"lː":46,"ø":47,"øː":48,"ŋ":49,"y":50,"oː":51,"rː":52,"ɲ":53,"nː":54,"w":55,"tʲː":56,"øɪ̯":57,"f":58,"dʲ":59,"sʲː":60,"t̠ʃ":61,"ʃː":62,"ʒ":63,"z":64,"fː":65,"dː":66,"yi":67}

French/config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.3,
+  "bos_token_id": 3,
+  "embd_pdrop": 0.3,
+  "eos_token_id": 3,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_embd": 128,
+  "n_head": 4,
+  "n_inner": 512,
+  "n_layer": 4,
+  "n_positions": 256,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.3,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.44.2",
+  "use_cache": true,
+  "vocab_size": 39
+}

French/generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 3,
+  "eos_token_id": 3,
+  "transformers_version": "4.44.2"
+}

French/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6990ba97ea1ba903cc98ed8d71dcbf40cb7016357e45128a75f085d79837922a
+size 3329448

French/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "UTT_BOUNDARY",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "PAD",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "UNK",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

French/tokenizer.json ADDED Viewed

	@@ -0,0 +1,156 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 0,
+      "content": "UNK",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 1,
+      "content": "PAD",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 2,
+      "content": "WORD_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 3,
+      "content": "UTT_BOUNDARY",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": {
+    "type": "Sequence",
+    "normalizers": [
+      {
+        "type": "Strip",
+        "strip_left": true,
+        "strip_right": true
+      }
+    ]
+  },
+  "pre_tokenizer": {
+    "type": "Whitespace"
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "SpecialToken": {
+          "id": "UTT_BOUNDARY",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "UTT_BOUNDARY": {
+        "id": "UTT_BOUNDARY",
+        "ids": [
+          3
+        ],
+        "tokens": [
+          "UTT_BOUNDARY"
+        ]
+      }
+    }
+  },
+  "decoder": null,
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "UNK": 0,
+      "PAD": 1,
+      "WORD_BOUNDARY": 2,
+      "UTT_BOUNDARY": 3,
+      "m": 4,
+      "a": 5,
+      "ɑ̃": 6,
+      "d": 7,
+      "ɔ": 8,
+      "n": 9,
+      "b": 10,
+      "ʁ": 11,
+      "ə": 12,
+      "ɡ": 13,
+      "ʒ": 14,
+      "i": 15,
+      "v": 16,
+      "t": 17,
+      "k": 18,
+      "o": 19,
+      "ɛ̃": 20,
+      "w": 21,
+      "y": 22,
+      "j": 23,
+      "e": 24,
+      "ɔ̃": 25,
+      "p": 26,
+      "ɛ": 27,
+      "f": 28,
+      "s": 29,
+      "z": 30,
+      "l": 31,
+      "u": 32,
+      "ʃ": 33,
+      "œ": 34,
+      "ø": 35,
+      "ɲ": 36,
+      "t̠ʃ": 37,
+      "d̠ʒ": 38
+    },
+    "unk_token": "UNK"
+  }
+}

French/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,44 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "UNK",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "PAD",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "WORD_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "UTT_BOUNDARY",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "UTT_BOUNDARY",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "UTT_BOUNDARY",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "PAD",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "UNK"
+}

French/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:19db843d5235caa94bee0fa15a7d2f0b4c32bb2fdb0e11a2352df8221d9016b9
+size 5368

French/vocab.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"UNK":0,"PAD":1,"WORD_BOUNDARY":2,"UTT_BOUNDARY":3,"m":4,"a":5,"ɑ̃":6,"d":7,"ɔ":8,"n":9,"b":10,"ʁ":11,"ə":12,"ɡ":13,"ʒ":14,"i":15,"v":16,"t":17,"k":18,"o":19,"ɛ̃":20,"w":21,"y":22,"j":23,"e":24,"ɔ̃":25,"p":26,"ɛ":27,"f":28,"s":29,"z":30,"l":31,"u":32,"ʃ":33,"œ":34,"ø":35,"ɲ":36,"t̠ʃ":37,"d̠ʒ":38}

German/config.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.3,
+  "bos_token_id": 3,
+  "embd_pdrop": 0.3,
+  "eos_token_id": 3,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_embd": 128,
+  "n_head": 4,
+  "n_inner": 512,
+  "n_layer": 4,
+  "n_positions": 256,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.3,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.44.2",
+  "use_cache": true,
+  "vocab_size": 45
+}

German/generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 3,
+  "eos_token_id": 3,
+  "transformers_version": "4.44.2"
+}