lrl-modelcloud commited on Jun 22

Commit

d809459

•

1 Parent(s): 8c89a11

Upload folder using huggingface_hub (#2)

- 105f1178f9d8b66bfdb8d6aeace71d87a44882c93dc1a364951edf3bed5b7fed (4469e6cebb072c49afb19891bb1a39938256c32c)
- 8e0288f8b29a72b2b789f8ee9ef0f1d37afed293ac673c71dd23de9fefb803d6 (6522af42bbc4f9ddec800957f5040b4b37234f8a)
- 7560fe9b264ad4b595838c002a9b7055e3681319cfeb309ac10b11266196080a (6c92710166b0a0c0d4da211136419aaf8b9dd645)
- 3f2c528bc240a676a1082f61de1324301c08251cfbb6acc8eab451682b2182f1 (e959a697ce2dde9767fb6e0c9b5a9184b4813467)
- 113d4616586d81e5b29178fea579ad5a9f9d2a93b4fad2fb6c2bd68a000e16b2 (c7cdb1262588cf8298197381834df203b788bea2)
- 51bbd95fb593c1e8a136ac9d4f19bb4b0717ef9c1405cba883c02c3c836e8a94 (50ab2cb8524ccdd797e658c35307bb72335640cd)
- 8ad6e0f2cf13660450ce44366813e4e9945122f741e988d88b5f63e24c1e780a (7061eae46fbdff8f296a0f8cec7ac1dd6d8f5731)
- 249fcafe317fe688216accde549c582c77256d95a00c8c860e39ee5dcdc1ec3c (fb1517e3001b1393d15c76e2aeee22a3df121784)
- bd798f3f39314cc2ddf4e84ba55da3f5a3c486f3217b6b877356f2209f973899 (4a3e2e6f8e798df20c97083ad2cba535a0d04900)
- 694356148ecfeecf35a16d28f789fe7d35c36f5c292757e27038b35cf3529790 (e419bfb933ceb9c4843b6d6b250f29af140047d0)
- c2fdeaa8fd250a6d83594539757c2c2ef70dcea353326d00effa0bd73d744bac (d3f0bd86195bbc44a0317760036534287864eb53)
- f82470d211e7ccd1cee116bf69ddc54748ec338cc4f642915d994c5861c98eef (704ba3399b346689ee33412c9ef1f664fea895ed)
- 5843f1d77eb70e28c96bd10f0e459b584dca6ecb8b442b8e9a8e592097d54dcb (ee2a1c1088422d6019d5dcd0387e45d477898e58)
- b81e0ecaab94a01095456e12799428eaea7e1fd28cb49c49f920d498b1cfb83f (c26051d373a19b1448a583581681353e45d3f9a1)
- ddcba34b76bc2752e40fa7b6efaf7bf588a401576dfa3aacc7c1c09ebc042aa0 (9bb2a601a0addc0252039b6d01da2098eb669c95)
- 4b7b956dc227b6ec28aa950d8648bef9b7395958aee2541acd4d104352de276e (1484d0d14c5ba4330ca031cd17bc700b9e66c381)
- da6c84c653e90f4366a59652fa2e56d108506e803a235fc5882c9c474298ec63 (7edd69e4f7e5225fcaedbba6933597893642ba5c)
- bd65ef470d916351b344f1e9352a55ec75f41dddcea743ecf2c99e9955a69c99 (e91acc7cbf6d80ea24fa318976628c1828974050)
- 4c70ce4344fabec195450a1705006e2e7bd2bce13e257709040244038c1d819a (cb39d975475b4c126f25c0e4e36cf5e22f6f717d)
- edcbe4b8923b070db9ea4d8c0b9e95aed56d00125c84145b12786806fd2e5cb3 (dcb366a0191bab119f62347bf9cc56dc5460af2e)
- 4960442ece6d6e86446161206bb506f80d00e1cc78cca059ea44e10fba1f1c8b (e807d24ac21156e7358c5ad13bbb3fda7a77f51e)
- c56b2d7ea34183445d32ceb34148ca2f25f579f204a0f5dd1de064edb500e3d7 (9f2990d137daaef900d8e284dde2b75b952184c9)
- a3845ff13533bc9656e226d3da4e0d12ce0951f864a5e42b8af7688d0cad6394 (661176789bd9c32bf4a552c14ce4a7dc34c6c494)
- 219e9050e5931b3f8f618075c87feadb4c00bf6192d06a47b75f6134edda40ff (97fcd3c1834000c7821e6f80104a7d9a2c927784)
- 32823684c8d23182a644e0fabcb4494bc0432a1ce8decc43796a151ce3461da6 (e1a1e6362ffc8b821203ad9711e3f80fc5d22846)
- 357c32f90fe9da08fb4ca1c8d243636766551358918ea669ab5f8f58926b567c (6581ab2859647ded6fa2675b8c1e449e0146a2f8)
- ceb186e57582dd86b2924ebbd130c7a86ec5f602806a871d33bbddef7c62992e (3b9551b7f3376458760001eb04e26a390de496a9)
- de14ddb3914ab39b44cd0fce8ef69912ed6c6e70b0a7de0877fecd12500ed98d (7fb926d9a252f056e7ea59ad3bf6bdf7e1866830)
- 87e869d7f1bc09eb692757974f8e1bbad935c49a6b912b516b7fab6635c2737d (b4d4fc88d2693b0916c74f14724058a3b78e345f)
- 3bf9b1e66836776b69f108b671807028d16566bb6bf11e73cd38cd18615247a5 (521b06c76977ceb60b10a5540fce75781902731f)
- 9cc9da92b7486056bf80bc841ce876c7d769ba68d43e2a25faf973487681c142 (0446b17469d1395100829059897095fd4843cd3c)
- e1fd7e2e15394223d44185a2c131b19acd5f2b25708ad5c4b1ca7f336e15bb09 (f8f75c6ce93d222953187ba2fdc7e4c9e16a70c6)
- 45fec8464d770d8ad8c8b525d692b4e5b8129223742b133cdd0a8dfd14ecccce (5ca196be9ebaf3a7ee62a8ff7d6582886d4626d5)
- d9a94592bd4cd790a51687de8cb052b569edbb5e1727477862234bb7d1a384e6 (88e98173a67e43100eeb91df6fb1f007a0c9321d)
- eaf75a1d942c468d1f429c5b6843107d13f0d27d66a884ae92234beb33e3edd5 (24991f171e6927cc81a37145f43226b17176218e)
- b0a2041f568db86f9cf2f8693db136d60d9e353d348fbc1f8d7e25d9cade12e4 (b5f145f0fcdb0337566557d17ba2a4c2a5f9cbea)
- 9be833f4a5691563507c997a7d7dcb5642bbb1ca5a2c6e215f7373e113e0f5c4 (990675a96d7209eb708747a1dca4583cd9d951e7)
- bee0bd188d8ccde632602cbd4fb165df91bf989ab6ac91756ab5c53c3fb94051 (ee66ccba1ed473a4319a71559ff0d95bcf23c2a2)
- 669f4b11ad6d583d6e639bcc24779e000805bece6bc909682543963a6cfeaf72 (3a329541262ba0ace6b8a4d201fb5efd68081c48)
- 5bb94bb9877c7c0e04b01572b68fd672607a066bd60bf549f5d673b61d24dcf9 (a114c6c7cafc369945040caca12a679aa4156268)
- 17ec2e9f474d62291e04b7177028b293760d2af6ec8f1b1d626cb4da86cd8f03 (36bcb4b1bff73206c13756d5f525c64b289e342b)
- 4881b32cae3096b54069d56857b10d2a615e136be77f8146f2665df2bd68aefe (3b72b6040a59fdc1aba74e4add3e1a29db66b047)
- f82fa822b269247bad7a8881e5e1fd0b073c25b04b5087ae74ae8bc9431db94a (ccee8d8f420ade950a07aea70dbc045c63643f7f)
- 404e08513b373c06ccdaa759b540ddbcd77d998eb9f3431dfdc84b89fb429c4f (c970c653765031965c1772cc2f92090604ebe1d1)
- 1892254fff70c78383b164d6ac9eff9cf432b8685c8f19e68ddbad86a6adbcc3 (24be0f8dd98b61603e1491dba5a590b3423bca66)
- 6d8f664c864cfa0b9f3844ac43d6030ed16bd4f42ab2b42b468ff0cf98246ef0 (270c547307591bff0613e866c4870985c2e3a09e)
- 382478dfd90a1f893c87fcae2dcd308b4459bf3bc51a1093edab543b1b0a4dd0 (bbbcfab9292ea8ca43d1fa769a25e5ed97cfb553)
- 77273d0a1c6d503e54737c79a4b60f39eb4fc28b86b4916da46f086ffee5260a (a61a83183842a52de53d9f4cfb4879b51dff52ab)
- e02cc9d20667e37018e0bf72f77cd696e03ed422e8fc34d5f0d8cea39b756558 (12a1afff85f609cd659507aee452010a86374646)
- 119556f42f9d9b485e426f0156df2b4bfe5a4cdb2d2eb6015f5ff9a4aa59bde5 (d8e8a3b0ab52f038a041752c12704ef697278ee7)
- 6806b3935aacd07f05c5b7cba5967f7545fa6d8d2bee1e4e75f93ff6f51550be (44926ffd2b7050a306725fbe0e11061ad35b9134)
- dd04a1956d17542de4f27dc0a35579a52c48b3e5e794afad60549c39269dd3d3 (4a1b4a2bda1e74e1083718121029a81a3a1ff1e3)
- e64f1ed46e0006b804989983e6ea55e886253266f315ef746613e2a9dd9ece23 (7e3700d11739ee3aa5bc10c6e8696e28b7adae88)
- a412e94a4cd272b33678ade13e4b04a30330159d9f970b59bc685af003e5811a (228911a525eeccfa6bc460a7ca2624b246b8a982)
- d811614d6b7e24f7a12d39c93a6c5237201fcb3151e15a638e283c84b490e37f (21e0452b1d9673ef54626844f22bd7c69b7639f5)
- e76e612bcca95d39ac6d7d8c328bcb08493293a7985deb4f157ce2cac9135f49 (3e66aae99088e0c5a3c150f62607c8054d97f51b)
- a45781161db45b4e965e38302e8bb3f9a67750fbb0e6cfc0f614bf90af52679f (f0a0b82eea0e984c8bb5c5e49ed014026ab7f2aa)
- 5db6743cb290ec2073ac7245bb6f3dfb39e285f835d9edcc43fdc4e63f02358f (05fc3bf1893a16f1b39d0deae1cae049d3f9889f)
- 372664a11a6d5225663ebbd784460491c9b3eceebc736a04668ed87687a0723a (21a0a5e1757e10c3f9e5f2b2ee40a017e63d78ba)
- 6a715d21a640564947ec47d36590def3600e37451ad1c7e17c5794e866b9d25d (b888c36438b9192a414db2cff373a0ba4368d594)
- 2ba83c8624b6f4894287e981390ac00418eae754d8944a6ca40c77f0b1e4ed4c (7cdf1b06af4b4961fb35d600557eab93dc3c0cbb)
- f90ce0b8fdbca174da1b904d6f72ba4c2624cd89bbe3461d64e899f5b7f65f03 (f66f59442fa825a2a993beda0c7793ed0be05729)

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

LICENSE.txt +176 -0
NOTICE.txt +1 -0
README.md +172 -3
config.json +38 -0
configuration_dbrx.py +264 -0
generation_config.json +7 -0
model-00001-of-00061.safetensors +3 -0
model-00002-of-00061.safetensors +3 -0
model-00003-of-00061.safetensors +3 -0
model-00004-of-00061.safetensors +3 -0
model-00005-of-00061.safetensors +3 -0
model-00006-of-00061.safetensors +3 -0
model-00007-of-00061.safetensors +3 -0
model-00008-of-00061.safetensors +3 -0
model-00009-of-00061.safetensors +3 -0
model-00010-of-00061.safetensors +3 -0
model-00011-of-00061.safetensors +3 -0
model-00012-of-00061.safetensors +3 -0
model-00013-of-00061.safetensors +3 -0
model-00014-of-00061.safetensors +3 -0
model-00015-of-00061.safetensors +3 -0
model-00016-of-00061.safetensors +3 -0
model-00017-of-00061.safetensors +3 -0
model-00018-of-00061.safetensors +3 -0
model-00019-of-00061.safetensors +3 -0
model-00020-of-00061.safetensors +3 -0
model-00021-of-00061.safetensors +3 -0
model-00022-of-00061.safetensors +3 -0
model-00023-of-00061.safetensors +3 -0
model-00024-of-00061.safetensors +3 -0
model-00025-of-00061.safetensors +3 -0
model-00026-of-00061.safetensors +3 -0
model-00027-of-00061.safetensors +3 -0
model-00028-of-00061.safetensors +3 -0
model-00029-of-00061.safetensors +3 -0
model-00030-of-00061.safetensors +3 -0
model-00031-of-00061.safetensors +3 -0
model-00032-of-00061.safetensors +3 -0
model-00033-of-00061.safetensors +3 -0
model-00034-of-00061.safetensors +3 -0
model-00035-of-00061.safetensors +3 -0
model-00036-of-00061.safetensors +3 -0
model-00037-of-00061.safetensors +3 -0
model-00038-of-00061.safetensors +3 -0
model-00039-of-00061.safetensors +3 -0
model-00040-of-00061.safetensors +3 -0
model-00041-of-00061.safetensors +3 -0
model-00042-of-00061.safetensors +3 -0
model-00043-of-00061.safetensors +3 -0
model-00044-of-00061.safetensors +3 -0

LICENSE.txt ADDED Viewed

	@@ -0,0 +1,176 @@

+Databricks Open Model License
+By using, reproducing, modifying, distributing, performing or displaying
+any portion or element of DBRX or DBRX Derivatives, or otherwise accepting
+the terms of this Agreement, you agree to be bound by this Agreement.
+Version Release Date: March 27, 2024
+Section 1: Definitions
+“Agreement” means these terms and conditions that govern the use, reproduction,
+modification, distribution, performance or display of DBRX and/or DBRX
+Derivatives and any terms and conditions incorporated by reference.
+“Databricks” or “we” means Databricks, Inc.
+“Licensee” or “you” means you, or your employer or any other person or entity
+(if you are entering into this Agreement on such person or entity’s behalf),
+of the age required under applicable laws, rules or regulations to provide
+legal consent and that has legal authority to bind your employer or such other
+person or entity if you are entering in this Agreement on their behalf.
+“DBRX Derivatives” means all (i) modifications to DBRX, (ii) works based on
+DBRX and (iii) any other derivative works thereof. Outputs are not deemed DBRX
+Derivatives.
+“DBRX” means the foundational large language models and software and
+algorithms, including machine-learning model code, trained model weights,
+inference-enabling code, training-enabling code, fine-tuning enabling code,
+documentation and other elements of the foregoing identified by Databricks at
+https://github.com/databricks/dbrx, regardless of the source that you obtained
+it from.
+“Output” means the results of operating DBRX or DBRX Derivatives.
+As used in this Agreement, “including” means “including without limitation.”
+Section 2: License Rights and Conditions on Use and Distribution
+2.1 Grant of Rights
+You are granted a non-exclusive, worldwide, non-transferable and royalty-free
+limited license under Databricks’ intellectual property or other rights owned
+by Databricks embodied in DBRX to use, reproduce, distribute, copy, modify,
+and create derivative works of DBRX in accordance with the terms of this
+Agreement.
+2.2 Reproduction and Distribution
+	1. All distributions of DBRX or DBRX Derivatives must be accompanied by a
+    "Notice" text file that contains the following notice: "DBRX is provided
+    under and subject to the Databricks Open Model License, Copyright ©
+    Databricks, Inc. All rights reserved."
+	2. If you distribute or make DBRX or DBRX Derivatives available to a third
+    party, you must provide a copy of this Agreement to such third party.
+	3. You must cause any modified files that you distribute to carry prominent
+    notices stating that you modified the files.
+You may add your own intellectual property statement to your modifications of
+DBRX and, except as set forth in this Section, may provide additional or
+different terms and conditions for use, reproduction, or distribution of DBRX
+or DBRX Derivatives as a whole, provided your use, reproduction, modification,
+distribution, performance, and display of DBRX or DBRX Derivatives otherwise
+complies with the terms and conditions of this Agreement. Any additional or
+different terms and conditions you impose must not conflict with the terms of
+this Agreement and in the event of a conflict, the terms and conditions of this
+Agreement shall govern over any such additional or different terms and conditions.
+2.3 Use Restrictions
+You will not use DBRX or DBRX Derivatives or any Output to improve any other
+large language model (excluding DBRX or DBRX Derivatives).
+You will not use DBRX or DBRX Derivatives:
+	1. for any restricted use set forth in the Databricks Open Model Acceptable
+    Use Policy identified at
+    https://www.databricks.com/legal/acceptable-use-policy-open-model
+    ("Acceptable Use Policy"), which is hereby incorporated by reference into
+    this Agreement; or
+	2. in violation of applicable laws and regulations.
+To the maximum extent permitted by law, Databricks reserves the right to
+restrict (remotely or otherwise) usage of DBRX or DBRX Derivatives that
+Databricks reasonably believes are in violation of this Agreement.
+Section 3: Additional Commercial Terms
+If, on the DBRX version release date, the monthly active users of the products
+or services made available by or for Licensee, or Licensee’s affiliates, is
+greater than 700 million monthly active users in the preceding calendar month,
+you must request a license from Databricks, which we may grant to you in our
+sole discretion, and you are not authorized to exercise any of the rights under
+this Agreement unless or until Databricks otherwise expressly grants you such
+rights.
+If you receive DBRX or DBRX Derivatives from a direct or indirect licensee as
+part of an integrated end user product, then this section (Section 3) of the
+Agreement will not apply to you.
+Section 4: Additional Provisions
+4.1 Updates
+Databricks may update DBRX from time to time, and you must make reasonable
+efforts to use the latest version of DBRX.
+4.2 Intellectual Property
+a. No trademark licenses are granted under this Agreement, and in connection
+with DBRX or DBRX Derivatives, neither Databricks nor Licensee may use any name
+or mark owned by or associated with the other or any of its affiliates, except
+as required for reasonable and customary use in describing and redistributing
+DBRX or DBRX Derivatives.
+b. Subject to Databricks’ ownership of DBRX and DRBX Derivatives made by or for
+Databricks, with respect to any DBRX Derivatives that are made by you, as
+between you and Databricks, you are and will be the owner of such DBRX
+Derivatives.
+c. Databricks claims no ownership rights in Outputs. You are responsible for
+Outputs and their subsequent uses.
+d. If you institute litigation or other proceedings against Databricks or any
+entity (including a cross-claim or counterclaim in a lawsuit) alleging that
+DBRX or Outputs or results therefrom, or any portion of any of the foregoing,
+constitutes infringement of intellectual property or other rights owned or
+licensable by you, then any licenses granted to you under this Agreement shall
+terminate as of the date such litigation or claim is filed or instituted. You
+will indemnify and hold harmless Databricks from and against any claim by any
+third party arising out of or related to your use or distribution of DBRX or
+DBRX Derivatives.
+4.3 DISCLAIMER OF WARRANTY
+UNLESS REQUIRED BY APPLICABLE LAW, DBRX AND ANY OUTPUT AND RESULTS THEREFROM
+ARE PROVIDED ON AN “AS IS” BASIS, WITHOUT WARRANTIES OF ANY KIND, EITHER
+EXPRESS OR IMPLIED, INCLUDING, WITHOUT LIMITATION, ANY WARRANTIES OF TITLE,
+NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. YOU
+ARE SOLELY RESPONSIBLE FOR DETERMINING THE APPROPRIATENESS OF USING OR
+REDISTRIBUTING DBRX OR DBRX DERIVATIVES AND ANY OUTPUT AND ASSUME ANY RISKS
+ASSOCIATED WITH YOUR USE OF DBRX OR DBRX DERIVATIVES AND ANY OUTPUT AND RESULTS.
+4.4 LIMITATION OF LIABILITY
+IN NO EVENT WILL DATABRICKS OR ITS AFFILIATES BE LIABLE UNDER ANY THEORY OF
+LIABILITY, WHETHER IN CONTRACT, TORT, NEGLIGENCE, PRODUCTS LIABILITY, OR
+OTHERWISE, ARISING OUT OF THIS AGREEMENT, FOR ANY LOST PROFITS OR ANY INDIRECT,
+SPECIAL, CONSEQUENTIAL, INCIDENTAL, EXEMPLARY OR PUNITIVE DAMAGES, EVEN IF
+DATABRICKS OR ITS AFFILIATES HAVE BEEN ADVISED OF THE POSSIBILITY OF ANY OF THE
+FOREGOING.
+4.5 Term and Termination
+The term of this Agreement will commence upon your acceptance of this Agreement
+or access to DBRX or DBRX Derivatives and will continue in full force and
+effect until terminated in accordance with the terms and conditions herein.
+Databricks may terminate this Agreement if you are in breach of any term or
+condition of this Agreement. Upon termination of this Agreement, you shall
+delete and cease use of DBRX or any DBRX Derivatives. Sections 1, 4.2(d), 4.3,
+4.4, and 4.6 shall survive the termination of this Agreement.
+4.6 Governing Law and Jurisdiction
+This Agreement will be governed and construed under the laws of the State of
+California without regard to choice of law principles, and the UN Convention
+on Contracts for the International Sale of Goods does not apply to this
+Agreement. The courts of California shall have exclusive jurisdiction of any
+dispute arising out of this Agreement.

NOTICE.txt ADDED Viewed

	@@ -0,0 +1 @@


1	+ DBRX is provided under and subject to the Databricks Open Model License, Copyright © Databricks, Inc. All rights reserved.

README.md CHANGED Viewed

@@ -1,3 +1,172 @@
----
-license: unknown
----

+---
+extra_gated_heading: You need to share contact information with Databricks to access this model
+extra_gated_prompt: >-
+  ### DBRX Terms of Use
+  Use of DBRX is governed by the [Databricks Open Model License](https://www.databricks.com/legal/open-model-license) and the [Databricks Open Model Acceptable Use Policy](https://www.databricks.com/legal/acceptable-use-policy-open-model).
+extra_gated_fields:
+  First Name: text
+  Last Name: text
+  Organization: text
+  Purpose for Base Model Access: text
+  By clicking 'Submit' below, I accept the terms of the license and acknowledge that the information I provide will be collected, stored, processed, and shared in accordance with Databricks' Privacy Notice and I understand I can update my preferences at any time: checkbox
+extra_gated_description: >-
+  The information you provide will be collected, stored, processed, and shared in accordance with Databricks [Privacy Notice](https://www.databricks.com/legal/privacynotice).
+extra_gated_button_content: Submit
+inference: false
+license: other
+license_name: databricks-open-model-license
+license_link: https://www.databricks.com/legal/open-model-license
+---
+# DBRX Base
+* DBRX Base is a mixture-of-experts (MoE) large language model trained from scratch by Databricks.
+* We are releasing both DBRX Base, a pretrained base model, and DBRX Instruct, a fine-tuned version for few-turn interactions, under [an open license](https://www.databricks.com/legal/open-model-license).
+* This is the repository for DBRX Base. DBRX Instruct can be found [here](https://huggingface.co/databricks/dbrx-instruct).
+* For full details on the DBRX models, please read our [technical blog post](https://www.databricks.com/blog/introducing-dbrx-new-state-art-open-llm).
+## Model Overview
+DBRX is a [transformer-based](https://www.isattentionallyouneed.com/) decoder-only large language model (LLM) that was trained using next-token prediction.
+It uses a *fine-grained* mixture-of-experts (MoE) architecture with 132B total parameters of which 36B parameters are active on any input.
+It was pre-trained on 12T tokens of text and code data.
+Compared to other open MoE models like Mixtral-8x7B and Grok-1, DBRX is fine-grained, meaning it uses a larger number of smaller experts. DBRX has 16 experts and chooses 4, while Mixtral-8x7B and Grok-1 have 8 experts and choose 2.
+This provides 65x more possible combinations of experts and we found that this improves model quality.
+DBRX uses rotary position encodings (RoPE), gated linear units (GLU), and grouped query attention (GQA).
+It uses the GPT-4 tokenizer as provided in the [tiktoken](https://github.com/openai/tiktoken) repository.
+We made these choices based on exhaustive evaluation and scaling experiments.
+DBRX was pretrained on 12T tokens of carefully curated data and a maximum context length of 32K tokens.
+We estimate that this data is at least 2x better token-for-token than the data we used to pretrain the MPT family of models.
+This new dataset was developed using the full suite of Databricks tools, including Apache Spark™ and Databricks notebooks for data processing, and Unity Catalog for data management and governance.
+We used curriculum learning for pretraining, changing the data mix during training in ways we found to substantially improve model quality.
+* **Inputs:** DBRX only accepts text-based inputs and accepts a context length of up to 32768 tokens.
+* **Outputs:** DBRX only produces text-based outputs.
+* **Model Architecture:** More detailed information about DBRX Instruct and DBRX Base can be found in our [technical blog post](https://www.databricks.com/blog/introducing-dbrx-new-state-art-open-llm).
+* **License:** [Databricks Open Model License](https://www.databricks.com/legal/open-model-license)
+* **Acceptable Use Policy:** [Databricks Open Model Acceptable Use Policy](https://www.databricks.com/legal/acceptable-use-policy-open-model)
+* **Version:** 1.0
+* **Owner:** Databricks, Inc.
+## Usage
+These are several general ways to use the DBRX models:
+* DBRX Base and DBRX Instruct are available for download on HuggingFace (see our Quickstart guide below). This is the HF repository for DBRX Base; DBRX Instruct can be found [here](https://huggingface.co/databricks/dbrx-instruct).
+* The DBRX model repository can be found on GitHub [here](https://github.com/databricks/dbrx).
+* DBRX Base and DBRX Instruct are available with [Databricks Foundation Model APIs](https://docs.databricks.com/en/machine-learning/foundation-models/index.html) via both *Pay-per-token* and *Provisioned Throughput* endpoints. These are enterprise-ready deployments.
+* For more information on how to fine-tune using LLM-Foundry, please take a look at our LLM pretraining and fine-tuning [documentation](https://github.com/mosaicml/llm-foundry/blob/main/scripts/train/README.md).
+## Quickstart Guide
+**NOTE: This is DBRX Base, and has not been instruction finetuned. It has not been trained for interactive chat and is only a completion model.**
+If you are looking for the finetuned model, please use [DBRX Instruct](https://huggingface.co/databricks/dbrx-instruct).
+Getting started with DBRX models is easy with the `transformers` library. The model requires ~264GB of RAM and the following packages:
+```bash
+pip install "transformers>=4.39.2" "tiktoken>=0.6.0"
+```
+If you'd like to speed up download time, you can use the `hf_transfer` package as described by Huggingface [here](https://huggingface.co/docs/huggingface_hub/en/guides/download#faster-downloads).
+```bash
+pip install hf_transfer
+export HF_HUB_ENABLE_HF_TRANSFER=1
+```
+You will need to request access to this repository to download the model. Once this is granted,
+[obtain an access token](https://huggingface.co/docs/hub/en/security-tokens) with `read` permission, and supply the token below.
+### Run the model on a CPU:
+```python
+from transformers import AutoTokenizer, AutoModelForCausalLM
+import torch
+tokenizer = AutoTokenizer.from_pretrained("databricks/dbrx-base", trust_remote_code=True, token="hf_YOUR_TOKEN")
+model = AutoModelForCausalLM.from_pretrained("databricks/dbrx-base", device_map="cpu", torch_dtype=torch.bfloat16, trust_remote_code=True, token="hf_YOUR_TOKEN")
+input_text = "Databricks was founded in "
+input_ids = tokenizer(input_text, return_tensors="pt")
+outputs = model.generate(**input_ids, max_new_tokens=100)
+print(tokenizer.decode(outputs[0]))
+```
+### Run the model on multiple GPUs:
+```python
+from transformers import AutoTokenizer, AutoModelForCausalLM
+import torch
+tokenizer = AutoTokenizer.from_pretrained("databricks/dbrx-base", trust_remote_code=True, token="hf_YOUR_TOKEN")
+model = AutoModelForCausalLM.from_pretrained("databricks/dbrx-base", device_map="auto", torch_dtype=torch.bfloat16, trust_remote_code=True, token="hf_YOUR_TOKEN")
+input_text = "Databricks was founded in "
+input_ids = tokenizer(input_text, return_tensors="pt").to("cuda")
+outputs = model.generate(**input_ids, max_new_tokens=100)
+print(tokenizer.decode(outputs[0]))
+```
+If your GPU system supports [FlashAttention2](https://huggingface.co/docs/transformers/perf_infer_gpu_one#flashattention-2), you can add `attn_implementation=”flash_attention_2”` as a keyword to `AutoModelForCausalLM.from_pretrained()` to achieve faster inference.
+## Limitations and Ethical Considerations
+### Training Dataset Limitations
+The DBRX models were trained on 12T tokens of text, with a knowledge cutoff date of December 2023.
+The training mix used for DBRX contains both natural-language and code examples. The vast majority of our training data is in the English language. We did not test DBRX for non-English proficiency. Therefore, DBRX should be considered a generalist model for text-based use in the English language.
+DBRX does not have multimodal capabilities.
+### Associated Risks and Recommendations
+All foundation models are novel technologies that carry various risks, and may output information that is inaccurate, incomplete, biased, or offensive.
+Users should exercise judgment and evaluate such output for accuracy and appropriateness for their desired use case before using or sharing it.
+Databricks recommends [using retrieval augmented generation (RAG)](https://www.databricks.com/glossary/retrieval-augmented-generation-rag) in scenarios where accuracy and fidelity are important.
+We also recommend that anyone using or fine-tuning either DBRX Base or DBRX Instruct perform additional testing around safety in the context of their particular application and domain.
+## Intended Uses
+### Intended Use Cases
+The DBRX models are open, general-purpose LLMs intended and licensed for both commercial and research applications.
+They can be further fine-tuned for various domain-specific natural language and coding tasks.
+DBRX Base can be used as an off-the-shelf model for text completion for general English-language and coding tasks.
+Please review the Associated Risks section above, as well as the [Databricks Open Model License](https://www.databricks.com/legal/open-model-license) and [Databricks Open Model Acceptable Use Policy](https://www.databricks.com/legal/acceptable-use-policy-open-model) for further information about permissible uses of DBRX Base and its derivatives.
+### Out-of-Scope Use Cases
+DBRX models are not intended to be used out-of-the-box in non-English languages and do not support native code execution, or other forms of function-calling.
+DBRX models should not be used in any manner that violates applicable laws or regulations or in any other way that is prohibited by the [Databricks Open Model License](https://www.databricks.com/legal/open-model-license) and [Databricks Open Model Acceptable Use Policy](https://www.databricks.com/legal/acceptable-use-policy-open-model).
+## Training Stack
+MoE models are complicated to train, and the training of DBRX Base and DBRX Instruct was heavily supported by Databricks’ infrastructure for data processing and large-scale LLM training (e.g., [Composer](https://github.com/mosaicml/composer), [Streaming](https://github.com/mosaicml/streaming), [Megablocks](https://github.com/stanford-futuredata/megablocks), and [LLM Foundry](https://github.com/mosaicml/llm-foundry)).
+Composer is our core library for large-scale training.
+It provides an optimized training loop, easy [checkpointing](https://docs.mosaicml.com/projects/composer/en/latest/trainer/checkpointing.html) and [logging](https://docs.mosaicml.com/projects/composer/en/latest/trainer/logging.html#wood-logging),
+[FSDP](https://pytorch.org/docs/stable/fsdp.html)-based [model sharding](https://docs.mosaicml.com/projects/composer/en/latest/notes/distributed_training.html#fullyshardeddataparallel-fsdp),
+convenient [abstractions](https://docs.mosaicml.com/projects/composer/en/latest/trainer/time.html), extreme customizability via [callbacks](https://docs.mosaicml.com/projects/composer/en/latest/trainer/callbacks.html), and more.
+Streaming enables fast, low cost, and scalable training on large datasets from cloud storage. It handles a variety of challenges around deterministic resumption as node counts change, avoiding redundant downloads across devices, high-quality shuffling at scale, sample-level random access, and speed.
+Megablocks is a lightweight library for MoE training. Crucially, it supports “dropless MoE,” which avoids inefficient padding and is intended to provide deterministic outputs for a given sequence no matter what other sequences are in the batch.
+LLM Foundry ties all of these libraries together to create a simple LLM pretraining, fine-tuning, and inference experience.
+DBRX was trained using proprietary optimized versions of the above open source libraries, along with our [LLM training platform](https://www.databricks.com/product/machine-learning/mosaic-ai-training).
+## Evaluation
+We find that DBRX outperforms established open-source and open-weight base models on the [Databricks Model Gauntlet](https://www.databricks.com/blog/llm-evaluation-for-icl), the [Hugging Face Open LLM Leaderboard](https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard), and HumanEval.
+The Databricks Model Gauntlet measures performance on more than 30 tasks across six categories: world knowledge, common sense reasoning, language understanding, reading comprehension, symbolic problem solving, and programming.
+The Hugging Face Open LLM Leaderboard measures the average of ARC-Challenge, HellaSwag, MMLU, TruthfulQA, Winogrande and GSM8k.
+HumanEval measures coding ability.
+Full evaluation details can be found in our [technical blog post](https://www.databricks.com/blog/introducing-dbrx-new-state-art-open-llm).
+## Acknowledgements
+The DBRX models were made possible thanks in large part to the open-source community, especially:
+* The [MegaBlocks](https://arxiv.org/abs/2211.15841) library, which established a foundation for our MoE implementation.
+* [PyTorch FSDP](https://arxiv.org/abs/2304.11277), which we built on for distributed training.

config.json ADDED Viewed

	@@ -0,0 +1,38 @@

+{
+  "architectures": [
+    "DbrxForCausalLM"
+  ],
+  "attn_config": {
+    "clip_qkv": 8,
+    "kv_n_heads": 8,
+    "model_type": "",
+    "rope_theta": 500000
+  },
+  "auto_map": {
+    "AutoConfig": "configuration_dbrx.DbrxConfig",
+    "AutoModelForCausalLM": "modeling_dbrx.DbrxForCausalLM"
+  },
+  "d_model": 6144,
+  "emb_pdrop": 0.0,
+  "ffn_config": {
+    "ffn_hidden_size": 10752,
+    "model_type": "",
+    "moe_jitter_eps": 0.01,
+    "moe_loss_weight": 0.05,
+    "moe_num_experts": 16,
+    "moe_top_k": 4
+  },
+  "initializer_range": 0.02,
+  "max_seq_len": 32768,
+  "model_type": "dbrx_converted",
+  "n_heads": 48,
+  "n_layers": 40,
+  "output_router_logits": false,
+  "resid_pdrop": 0.0,
+  "router_aux_loss_coef": 0.05,
+  "tie_word_embeddings": false,
+  "torch_dtype": "bfloat16",
+  "transformers_version": "4.38.2",
+  "use_cache": true,
+  "vocab_size": 100352
+}

configuration_dbrx.py ADDED Viewed

	@@ -0,0 +1,264 @@

+"""Dbrx configuration."""
+from typing import Any, Optional
+from transformers.configuration_utils import PretrainedConfig
+from transformers.utils import logging
+logger = logging.get_logger(__name__)
+DBRX_PRETRAINED_CONFIG_ARCHIVE_MAP = {}
+class DbrxAttentionConfig(PretrainedConfig):
+    """Configuration class for Dbrx Attention.
+    [`DbrxAttention`] class. It is used to instantiate attention layers
+    according to the specified arguments, defining the layers architecture.
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+    Args:
+        attn_pdrop (`float`, *optional*, defaults to 0.0):
+            The dropout probability for the attention layers.
+        clip_qkv (`float`, *optional*, defualts to None):
+            If not `None`, clip the queries, keys, and values in the attention layer to this value.
+        kv_n_heads (Optional[int]): For grouped_query_attention only, allow user to specify number of kv heads.
+        rope_theta (float): The base frequency for rope.
+    """
+    def __init__(
+        self,
+        attn_pdrop: float = 0,
+        clip_qkv: Optional[float] = None,
+        kv_n_heads: int = 1,
+        rope_theta: float = 10000.0,
+        **kwargs: Any,
+    ):
+        super().__init__(**kwargs)
+        self.attn_pdrop = attn_pdrop
+        self.clip_qkv = clip_qkv
+        self.kv_n_heads = kv_n_heads
+        self.rope_theta = rope_theta
+        for k in ['model_type']:
+            if k in kwargs:
+                kwargs.pop(k)
+        if len(kwargs) != 0:
+            raise ValueError(f'Found unknown {kwargs=}')
+    @classmethod
+    def from_pretrained(cls, pretrained_model_name_or_path: str,
+                        **kwargs: Any) -> 'PretrainedConfig':
+        cls._set_token_in_kwargs(kwargs)
+        config_dict, kwargs = cls.get_config_dict(pretrained_model_name_or_path,
+                                                  **kwargs)
+        if config_dict.get('model_type') == 'dbrx':
+            config_dict = config_dict['attn_config']
+        if 'model_type' in config_dict and hasattr(
+                cls,
+                'model_type') and config_dict['model_type'] != cls.model_type:
+            logger.warning(
+                f"You are using a model of type {config_dict['model_type']} to instantiate a model of type "
+                +
+                f'{cls.model_type}. This is not supported for all configurations of models and can yield errors.'
+            )
+        return cls.from_dict(config_dict, **kwargs)
+class DbrxFFNConfig(PretrainedConfig):
+    """Configuration class for Dbrx FFN.
+    [`DbrxFFN`] class. It is used to instantiate feedforward layers according to
+    the specified arguments, defining the layers architecture.
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+    Args:
+        ffn_act_fn (dict, optional): A dict specifying activation function for the FFN.
+            The dict should have a key 'name' with the value being the name of
+            the activation function along with any additional keyword arguments.
+        ffn_hidden_size (int, optional): The hidden size of the feedforward network.
+        moe_num_experts (int, optional): The number of experts in the mixture of experts layer.
+        moe_top_k (int, optional): The number of experts to use in the mixture of experts layer.
+        moe_jitter_eps (float, optional): The jitter epsilon for the mixture of experts layer.
+        moe_loss_weight (float, optional): The loss weight for the mixture of experts layer.
+        moe_normalize_expert_weights (float, optional): The normalization factor for the expert weights.
+        uniform_expert_assignment (bool, optional): Whether to use uniform expert assignment.
+            This should only be used for benchmarking purposes.
+    """
+    def __init__(
+        self,
+        ffn_act_fn: Optional[dict] = None,
+        ffn_hidden_size: int = 3584,
+        moe_num_experts: int = 4,
+        moe_top_k: int = 1,
+        moe_jitter_eps: Optional[float] = None,
+        moe_loss_weight: float = 0.01,
+        moe_normalize_expert_weights: Optional[float] = 1,
+        uniform_expert_assignment: bool = False,
+        **kwargs: Any,
+    ):
+        super().__init__()
+        if ffn_act_fn is None:
+            ffn_act_fn = {'name': 'silu'}
+        self.ffn_act_fn = ffn_act_fn
+        self.ffn_hidden_size = ffn_hidden_size
+        self.moe_num_experts = moe_num_experts
+        self.moe_top_k = moe_top_k
+        self.moe_jitter_eps = moe_jitter_eps
+        self.moe_loss_weight = moe_loss_weight
+        self.moe_normalize_expert_weights = moe_normalize_expert_weights
+        self.uniform_expert_assignment = uniform_expert_assignment
+        for k in ['model_type']:
+            if k in kwargs:
+                kwargs.pop(k)
+        if len(kwargs) != 0:
+            raise ValueError(f'Found unknown {kwargs=}')
+    @classmethod
+    def from_pretrained(cls, pretrained_model_name_or_path: str,
+                        **kwargs: Any) -> 'PretrainedConfig':
+        cls._set_token_in_kwargs(kwargs)
+        config_dict, kwargs = cls.get_config_dict(pretrained_model_name_or_path,
+                                                  **kwargs)
+        if config_dict.get('model_type') == 'dbrx':
+            config_dict = config_dict['ffn_config']
+        if 'model_type' in config_dict and hasattr(
+                cls,
+                'model_type') and config_dict['model_type'] != cls.model_type:
+            logger.warning(
+                f"You are using a model of type {config_dict['model_type']} to instantiate a model of type "
+                +
+                f'{cls.model_type}. This is not supported for all configurations of models and can yield errors.'
+            )
+        return cls.from_dict(config_dict, **kwargs)
+class DbrxConfig(PretrainedConfig):
+    """Configuration class for Dbrx.
+    [`DbrxModel`]. It is used to instantiate a Dbrx model according to the
+    specified arguments, defining the model architecture.
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+    Args:
+        d_model (`int`, *optional*, defaults to 6144):
+            Dimensionality of the embeddings and hidden states.
+        n_heads (`int`, *optional*, defaults to 48):
+            Number of attention heads for each attention layer in the Transformer encoder.
+        n_layers (`int`, *optional*, defaults to 40):
+            Number of hidden layers in the Transformer encoder.
+        max_seq_len (`int`, *optional*, defaults to 32768):
+            The maximum sequence length of the model.
+        vocab_size (`int`, *optional*, defaults to 100352):
+            Vocabulary size of the Dbrx model. Defines the maximum number of different tokens that can be represented by
+            the `inputs_ids` passed when calling [`DbrxModel`].
+        resid_pdrop (`float`, *optional*, defaults to 0.0):
+            The dropout probability applied to the attention output before combining with residual.
+        emb_pdrop (`float`, *optional*, defaults to 0.0):
+            The dropout probability for the embedding layer.
+        attn_config (`dict`, *optional*):
+            A dictionary used to configure the model's attention module.
+        ffn_config (`dict`, *optional*):
+            A dictionary used to configure the model's FFN module.
+        use_cache (`bool`, *optional*, defaults to `False`):
+            Whether or not the model should return the last key/values attentions (not used by all models).
+        initializer_range (`float`, *optional*, defaults to 0.02):
+            The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
+        output_router_logits (`bool`, *optional*, defaults to `False`):
+            Whether or not the router logits should be returned by the model. Enabling this will also
+            allow the model to output the auxiliary loss. See [here]() for more details
+        router_aux_loss_coef (`float`, *optional*, defaults to 0.001):
+            The aux loss factor for the total loss.
+    Example:
+    ```python
+    >>> from transformers import DbrxConfig, DbrxModel
+    >>> # Initializing a Dbrx configuration
+    >>> configuration = DbrxConfig()
+    >>> # Initializing a model (with random weights) from the configuration
+    >>> model = DbrxModel(configuration)
+    >>> # Accessing the model configuration
+    >>> configuration = model.config
+    ```
+    """
+    model_type = 'dbrx'
+    attribute_map = {
+        'num_attention_heads': 'n_heads',
+        'hidden_size': 'd_model',
+        'num_hidden_layers': 'n_layers',
+        'max_position_embeddings': 'max_seq_len'
+    }
+    def __init__(
+        self,
+        d_model: int = 2048,
+        n_heads: int = 16,
+        n_layers: int = 24,
+        max_seq_len: int = 2048,
+        vocab_size: int = 32000,
+        resid_pdrop: float = 0.0,
+        emb_pdrop: float = 0.0,
+        attn_config: Optional[DbrxAttentionConfig] = None,
+        ffn_config: Optional[DbrxFFNConfig] = None,
+        use_cache: bool = True,
+        initializer_range: float = 0.02,
+        output_router_logits: bool = False,
+        router_aux_loss_coef: float = 0.05,
+        **kwargs: Any,
+    ):
+        if attn_config is None:
+            self.attn_config = DbrxAttentionConfig()
+        elif isinstance(attn_config, dict):
+            self.attn_config = DbrxAttentionConfig(**attn_config)
+        else:
+            self.attn_config = attn_config
+        if ffn_config is None:
+            self.ffn_config = DbrxFFNConfig()
+        elif isinstance(ffn_config, dict):
+            self.ffn_config = DbrxFFNConfig(**ffn_config)
+        else:
+            self.ffn_config = ffn_config
+        self.d_model = d_model
+        self.n_heads = n_heads
+        self.n_layers = n_layers
+        self.max_seq_len = max_seq_len
+        self.vocab_size = vocab_size
+        self.resid_pdrop = resid_pdrop
+        self.emb_pdrop = emb_pdrop
+        self.use_cache = use_cache
+        self.initializer_range = initializer_range
+        self.output_router_logits = output_router_logits
+        self.router_aux_loss_coef = router_aux_loss_coef
+        tie_word_embeddings = kwargs.pop('tie_word_embeddings', False)
+        if tie_word_embeddings:
+            raise ValueError(
+                'tie_word_embeddings is not supported for Dbrx models.')
+        super().__init__(
+            tie_word_embeddings=tie_word_embeddings,
+            **kwargs,
+        )

generation_config.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "_from_model_config": true,
+  "eos_token_id": [
+    100257
+  ],
+  "transformers_version": "4.38.2"
+}

model-00001-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:fc28da4a69f7e5548bc4086fba3cef26c79f653793501cee044f1c23b0d83f2d
+size 3523439624

model-00002-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:698e440b10f1ffd061386048a38686c95b42877907212ade39ae414a38516914
+size 4404245416

model-00003-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8a53fc538e17f263dbafac832259701d71c039ceb0e0a133411bd868f67c1412
+size 4227862560

model-00004-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bf9092dfa84fa162fcecba742834997c052b38a08024c560b5724a29b7ea181d
+size 4404245416

model-00005-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:704462d7756dd4eab0d486ef4cbdaf0723201238bd2dccf0a1b210e85f0e81d6
+size 4404245416

model-00006-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b8a9a26d88087d1f17f398bad35b00f8c9df6d45684c2d27ed51c6fd8479da0a
+size 4227862560

model-00007-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bac5317e3eac638314e103fa0e3a1a2df53b4b9c08b0530fe74632733f305ed9
+size 4404245416

model-00008-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c88244322308be72113df4e549db78dca660ae3b1c9918a50a51826fe1369605
+size 4404245416

model-00009-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f06f6d9014cf96eb3650c99db4045d37c8b7e34c5117e9a2738d400de1658860
+size 4227862560

model-00010-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:14b737a3b878d7176ecf747ca84b5477e5e87fdf29f8f9e91ddefec61180fc93
+size 4404245416

model-00011-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:85af95c26290fceacfddedc271dfdeadf75c4537cbf38a2162844e7dd531db74
+size 4404245416

model-00012-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e56c3d4c9786fcb2eeeddca5f1ffadd4e8a4b64514c807595572ba0fc0f61b76
+size 4227862560

model-00013-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ffa30aa9cfe2913fb4c4162a3bffa4c705ae03da9ee866982efab504fbb658c8
+size 4404245416

model-00014-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:00d3f5a60c05bfa2b985fdd148ad7ad360d99d2384d0a5a9fb8205a489a0f95e
+size 4404245416

model-00015-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:55f55d4b560746620978473685d5a6c87a5eb9ac780b97592a77a27e9fd67dce
+size 4227862560

model-00016-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8e21a36bc448a715b3e51c9f131a9ff6e322b499028bfa8af88650d90cf7f30b
+size 4404245440

model-00017-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:36fe189b08d023fae4481ab80a70ffbc1980f71e41174fb54ec0bf5d672b2fb9
+size 4404245456

model-00018-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e1b24cd6fbcb7458e33aa6db4702aad519a32637b15fda82454747f5e9432e13
+size 4227862592

model-00019-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2bf77fa36fccf9920386ce934ff38779eccac713df06b088adf48b6372504f01
+size 4404245456

model-00020-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c7bac559137d755e4aa144ddbb941dcbf44be3a2a7e40235ef5c4a791b90a7ca
+size 4404245456

model-00021-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:45bb5601b485a573af20d7b23bad58e0f4146b349644bd8f728b06252102176b
+size 4227862592

model-00022-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:11aa46e29483a012c65fa908d714832adaac3e4f7fe393327720c6e62cc5b181
+size 4404245456

model-00023-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:41646f1544ea9f252924af294fa09ef491e9c8e5bcde113ee563cec0999bbfe1
+size 4404245456

model-00024-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2523a4be387557beb04260b0d08cf8a37d0a87006197c9103f365e3cfad06daf
+size 4227862592

model-00025-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:737ec65ff78b7770d09da5ab4087f26e54a4320fbf83ba7ce1650e11d2ab73c8
+size 4404245456

model-00026-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:3034000a88019b565c051aa18dd73329139e33fae962077a71317967d78d4c94
+size 4404245456

model-00027-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:41879f07d50975e532be376f7efc62bcb5be7719b9303ef8c807706c43c2ad43
+size 4227862592

model-00028-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:0939a016e18c68adc20f712afd28b89c1a6b463a07888854a60ac3eb3f62ab17
+size 4404245456

model-00029-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bc3c688b7f3ba199fb07883411d7b0bcb78f13aa7f227cc9a02d6acc3a3da603
+size 4404245456

model-00030-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:af610db5f894204a2c2c6891f4644b5ae4f38e429f1b33978d3567981aa61faa
+size 4227862592

model-00031-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b9a0215ee52194d33cecf47765803c2e6f41fbaa928f1cc65f5cc0569b0b96f2
+size 4404245456

model-00032-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6d668fb61c857ba8d6a08fc3ecb34408b9ea3a27c5dd4e5a459f2459235a0d29
+size 4404245456

model-00033-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:fa6f00ab09861368f0aaf66ec92bf43b17a70cd8bfb5be6264aa5b21d9a36a25
+size 4227862592

model-00034-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:baf918d45e5f5f3b684da913b44fb5939e593c5f795bfcb9c11c3f6591c7daf8
+size 4404245456

model-00035-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:fee00379e0a448352ab4d5ec9d359f2c41f924730d3db2bad4cc1b3cacd3641c
+size 4404245456

model-00036-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:792058de69a54860e03de6dcd68bb88c758e0e701f9a9054ef95e74b66f641c1
+size 4227862592

model-00037-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1a8f73c9148bd468922c2cb6892e63387cb57811be537d1c04ed1849b3624441
+size 4404245456

model-00038-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f0c0e459f70f3ba6e3f2a8e3edc35e41457dac4d7e0d6cf0933353ee5f90c368
+size 4404245456

model-00039-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b597724c9c431c9b66ef7f1c5c98c572e1eb2cc04eab00b4b5578c804f7dabe2
+size 4227862592

model-00040-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:13408aa90943912d344f63cacc190a031575564f94e65adf24064d9ae76335a7
+size 4404245456

model-00041-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:025e3c0e161da006cde492d2288458ecb8db5512961c41d60dfd51ff3cf825b2
+size 4404245456

model-00042-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:93347f13c06c95ed4131c03f1bcbade4aaf2fc9e8e8738483467ed482e1ff556
+size 4227862592

model-00043-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b789774316dabda443787a23943410bcd6f6fc85a0321a3d14f6488b6bb172cf
+size 4404245456

model-00044-of-00061.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a3efb71226981a713f6551cc0672a9e12e037fcfbda3cc41d11dac7c0fc03624
+size 4404245456