Spaces:
Running
on
CPU Upgrade
Running
on
CPU Upgrade
File size: 33,394 Bytes
559bd6b 7084d46 559bd6b 7084d46 559bd6b 7084d46 559bd6b 7084d46 559bd6b 7084d46 559bd6b 7084d46 559bd6b 0266144 559bd6b 7084d46 559bd6b 7084d46 559bd6b 7084d46 559bd6b 7084d46 |
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 |
// 1- base models <=7B
{"model": "TinyLlama/TinyLlama-1.1B-intermediate-step-1431k-3T", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "meta-llama/Llama-2-7b-hf", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "mistralai/Mistral-7B-v0.1", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "huggyllama/llama-7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "openlm-research/open_llama_3b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "openlm-research/open_llama_3b_v2", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "openlm-research/open_llama_7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "openlm-research/open_llama_7b_v2", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
// 2 - Larger base models <= 13B
{"model": "meta-llama/Llama-2-13b-hf", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "huggyllama/llama-13b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "openlm-research/open_llama_13b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "upstage/SOLAR-10.7B-v1.0", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
// 3 - portuguese models
{"model": "maritaca-ai/sabia-7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "dominguesm/canarim-7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "22h/open-cabrita3b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "recogna-nlp/bode-7b-alpaca-pt-br", "base_model": "meta-llama/Llama-2-7b-chat-hf", "revision": "main", "precision": "float16", "weight_type": "Adapter", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
{"model": "recogna-nlp/bode-13b-alpaca-pt-br", "base_model": "meta-llama/Llama-2-13b-chat-hf", "revision": "main", "precision": "float16", "weight_type": "Adapter", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
{"model": "22h/cabrita_7b_pt_850000", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "22h/cabrita-lora-v0-1", "base_model": "huggyllama/llama-7b", "revision": "main", "precision": "float16", "weight_type": "Adapter", "model_type": "πΆ : fine-tuned"}
{"model": "wandgibaut/periquito-3B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "nicolasdec/Cabra", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
{"model": "nicolasdec/cabra13b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
{"model": "lrds-code/samba-1.1B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
{"model": "lrds-code/boana-7b-instruct", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
{"model": "nicholasKluge/Aira-2-portuguese-124M", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
{"model": "nicholasKluge/Aira-2-portuguese-560M", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
{"model": "nicholasKluge/Aira-2-portuguese-1B7", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
// other must-have <=7B
{"model": "dynamofl/dynamo-8B-v0.1", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "01-ai/Yi-6B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "Unbabel/TowerBase-7B-v0.1", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "tiiuae/falcon-7b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "bigscience/bloom-560m", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "bigscience/bloom-1b7", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "bigscience/bloom-3b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "bigscience/bloom-7b1", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "stabilityai/stablelm-2-1_6b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "stabilityai/stablelm-3b-4e1t", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
// Larger base models >13B
{"model": "mistralai/Mixtral-8x7B-v0.1", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "huggyllama/llama-30b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "01-ai/Yi-34B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "meta-llama/Llama-2-70b-hf", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "huggyllama/llama-65b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
// minors must
{"model": "togethercomputer/RedPajama-INCITE-Base-3B-v1", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "togethercomputer/RedPajama-INCITE-7B-Base", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "DAMO-NLP-MT/polylm-1.7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "DAMO-NLP-MT/polylm-13b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "Deci/DeciLM-6b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
{"model": "Deci/DeciLM-7B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
// multiple (ch-jp)/en bi/multi lingual models
{"model": "internlm/internlm2-7b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
{"model": "internlm/internlm2-base-7b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "internlm/internlm-7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "internlm/internlm2-20b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
{"model": "internlm/internlm2-base-20b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "internlm/internlm-20b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "Qwen/Qwen-1_8B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "Qwen/Qwen-7B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "Qwen/Qwen-14B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "xverse/XVERSE-7B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "xverse/XVERSE-13B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "xverse/XVERSE-13B-256K", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "Skywork/Skywork-13B-base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "baichuan-inc/Baichuan-7B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "baichuan-inc/Baichuan-13B-Base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "baichuan-inc/Baichuan2-7B-Base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "baichuan-inc/Baichuan2-13B-Base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "OrionStarAI/Orion-14B-Base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "deepseek-ai/deepseek-llm-7b-base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "deepseek-ai/deepseek-moe-16b-base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "BAAI/Aquila-7B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "BAAI/Aquila2-7B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "THUDM/chatglm3-6b-base", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "THUDM/glm-2b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "THUDM/glm-10b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "fnlp/moss-moon-003-base", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "fnlp/moss-base-7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
// multiple chinese/jp large
{"model": "Qwen/Qwen-72B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "xverse/XVERSE-65B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "xverse/XVERSE-65B-2", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "deepseek-ai/deepseek-llm-67b-base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "BAAI/Aquila2-34B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "BAAI/Aquila2-70B-Expr", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
// minors must 2
{"model": "gpt2", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "t5-small", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "t5-base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "t5-large", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/mt5-small", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/mt5-base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/mt5-large", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
//others
{"model": "NucleusAI/nucleus-22B-token-500B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-14m", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-70m-deduped", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-160m-deduped", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-410m-deduped", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-1b-deduped", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-2.8b-deduped", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-6.9b-deduped", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-12b-deduped", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/gpt-neo-125m", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/gpt-neo-1.3B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/gpt-neo-2.7B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/gpt-j-6b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/gpt-neox-20b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/opt-125m", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/opt-350m", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/opt-1.3b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/opt-2.7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/opt-6.7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/opt-13b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/opt-30b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
//other large
{"model": "facebook/opt-66b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "tiiuae/falcon-40b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
// minors portuguese
{"model": "pierreguillou/gpt2-small-portuguese", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "pucpr/gpt2-bio-pt", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "unicamp-dl/ptt5-small-portuguese-vocab", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "unicamp-dl/ptt5-base-portuguese-vocab", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "unicamp-dl/ptt5-large-portuguese-vocab", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "unicamp-dl/ptt5-small-t5-vocab", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "unicamp-dl/ptt5-base-t5-vocab", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "unicamp-dl/ptt5-large-t5-vocab", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "josu/gpt-neo-pt-br", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "josu/gpt-neo-pt-1.3B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "monilouise/opt125M_portuguese", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "HeyLucasLeao/gpt-neo-small-portuguese", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
// other langs (es/Ko/Jp/nordic)
{"model": "projecte-aina/FLOR-760M", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "projecte-aina/FLOR-1.3B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "projecte-aina/FLOR-6.3B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "projecte-aina/aguila-7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "EleutherAI/polyglot-ko-12.8b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "matsuo-lab/weblab-10b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "pfnet/plamo-13b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "AI-Sweden-Models/gpt-sw3-6.7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "AI-Sweden-Models/gpt-sw3-6.7b-v2", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "AI-Sweden-Models/gpt-sw3-20b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "AI-Sweden-Models/gpt-sw3-40b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "OpenLLM-France/Claire-Mistral-7B-0.1", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
{"model": "OpenLLM-France/Claire-7B-0.1", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π : language adapted models (FP, FT, ...)"}
// huge models:
//{"model": "bigscience/bloom", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
//{"model": "tiiuae/falcon-180B", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
//{"model": "facebook/galactica-120b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
//random chat models
{"model": "openchat/openchat-3.5-0106", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π¬ : chat models (RLHF, DPO, IFT, ...)"}
//other 2
{"model": "stabilityai/stablelm-base-alpha-3b-v2", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "stabilityai/stablelm-base-alpha-7b-v2", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "stabilityai/stablelm-base-alpha-3b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "stabilityai/stablelm-base-alpha-7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "openai-community/openai-gpt", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "openai-community/gpt2-medium", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "openai-community/gpt2-large", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "openai-community/gpt2-xl", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "microsoft/phi-1", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "microsoft/phi-1_5", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "microsoft/phi-2", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "mosaicml/mpt-7b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "mosaicml/mpt-30b", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "mosaicml/mpt-7b-8k", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "01-ai/Yi-6B-200K", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "01-ai/Yi-34B-200K", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/t5-v1_1-base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/t5-v1_1-small", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/t5-v1_1-large", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/t5-v1_1-xl", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/t5-v1_1-xxl", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/mt5-xl", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/mt5-xxl", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/umt5-small", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/umt5-base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/umt5-xl", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "google/umt5-xxl", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "AdaptLLM/law-LLM", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
{"model": "AdaptLLM/medicine-LLM", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
{"model": "AdaptLLM/finance-LLM", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
{"model": "AdaptLLM/law-LLM-13B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
{"model": "AdaptLLM/medicine-LLM-13B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
{"model": "AdaptLLM/finance-LLM-13B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "πΆ : fine-tuned"}
{"model": "cerebras/Cerebras-GPT-111M", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "cerebras/Cerebras-GPT-256M", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "cerebras/Cerebras-GPT-590M", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "cerebras/Cerebras-GPT-1.3B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "cerebras/Cerebras-GPT-2.7B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "cerebras/Cerebras-GPT-6.7B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "cerebras/Cerebras-GPT-13B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "cerebras/btlm-3b-8k-base", "base_model": "", "revision": "main", "precision": "bfloat16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "ai-forever/mGPT-13B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "ai-forever/mGPT", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-70m", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-160m", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-410m", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-1b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-2.8b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-6.9b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "EleutherAI/pythia-12b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/galactica-125m", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/galactica-1.3b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/galactica-6.7b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/galactica-30b", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/xglm-564M", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/xglm-1.7B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/xglm-2.9B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/xglm-4.5B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
{"model": "facebook/xglm-7.5B", "base_model": "", "revision": "main", "precision": "float16", "weight_type": "Original", "model_type": "π’ : pretrained"}
|