initial version

Browse files

Files changed (13) hide show

.gitignore +165 -0
README.md +22 -0
merges.txt +0 -0
scripts/TRAIN.md +40 -0
scripts/model.yaml +143 -0
scripts/prepare_pretrain_dataset.py +180 -0
scripts/requirements-lit.in +10 -0
scripts/requirements.in +7 -0
scripts/train_tokenizer.py +313 -0
special_tokens_map.json +6 -0
tokenizer.json +0 -0
tokenizer_config.json +1036 -0
vocab.json +0 -0

.gitignore ADDED Viewed

	@@ -0,0 +1,165 @@

+# ---> Python
+# Byte-compiled / optimized / DLL files
+__pycache__/
+*.py[cod]
+*$py.class
+# C extensions
+*.so
+# Distribution / packaging
+.Python
+build/
+develop-eggs/
+dist/
+downloads/
+eggs/
+.eggs/
+lib/
+lib64/
+parts/
+sdist/
+var/
+wheels/
+share/python-wheels/
+*.egg-info/
+.installed.cfg
+*.egg
+MANIFEST
+# PyInstaller
+#  Usually these files are written by a python script from a template
+#  before PyInstaller builds the exe, so as to inject date/other infos into it.
+*.manifest
+*.spec
+# Installer logs
+pip-log.txt
+pip-delete-this-directory.txt
+# Unit test / coverage reports
+htmlcov/
+.tox/
+.nox/
+.coverage
+.coverage.*
+.cache
+nosetests.xml
+coverage.xml
+*.cover
+*.py,cover
+.hypothesis/
+.pytest_cache/
+cover/
+# Translations
+*.mo
+*.pot
+# Django stuff:
+*.log
+local_settings.py
+db.sqlite3
+db.sqlite3-journal
+# Flask stuff:
+instance/
+.webassets-cache
+# Scrapy stuff:
+.scrapy
+# Sphinx documentation
+docs/_build/
+# PyBuilder
+.pybuilder/
+target/
+# Jupyter Notebook
+.ipynb_checkpoints
+# IPython
+profile_default/
+ipython_config.py
+# pyenv
+#   For a library or package, you might want to ignore these files since the code is
+#   intended to run in multiple environments; otherwise, check them in:
+# .python-version
+# pipenv
+#   According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
+#   However, in case of collaboration, if having platform-specific dependencies or dependencies
+#   having no cross-platform support, pipenv may install dependencies that don't work, or not
+#   install all needed dependencies.
+#Pipfile.lock
+# poetry
+#   Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
+#   This is especially recommended for binary packages to ensure reproducibility, and is more
+#   commonly ignored for libraries.
+#   https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
+#poetry.lock
+# pdm
+#   Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
+#pdm.lock
+#   pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
+#   in version control.
+#   https://pdm.fming.dev/#use-with-ide
+.pdm.toml
+# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
+__pypackages__/
+# Celery stuff
+celerybeat-schedule
+celerybeat.pid
+# SageMath parsed files
+*.sage.py
+# Environments
+.env
+.venv
+env/
+venv/
+ENV/
+env.bak/
+venv.bak/
+# Spyder project settings
+.spyderproject
+.spyproject
+# Rope project settings
+.ropeproject
+# mkdocs documentation
+/site
+# mypy
+.mypy_cache/
+.dmypy.json
+dmypy.json
+# Pyre type checker
+.pyre/
+# pytype static type analyzer
+.pytype/
+# Cython debug symbols
+cython_debug/
+# PyCharm
+#  JetBrains specific template is maintained in a separate JetBrains.gitignore that can
+#  be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
+#  and can be added to the global gitignore or merged into this file.  For a more nuclear
+#  option (not recommended) you can uncomment the following to ignore the entire idea folder.
+.idea/
+.DS_Store
+.ruff_cache
+venv*/

README.md CHANGED Viewed

@@ -1,3 +1,25 @@
 ---
 license: apache-2.0
 ---

 ---
 license: apache-2.0
+pipeline_tag: text-generation
+library_name: transformers
+language: ['en', 'am', 'ar', 'as', 'az', 'be', 'bg', 'bn', 'br', 'bs', 'ca', 'cs', 'cy', 'da', 'de', 'el', 'eo', 'es', 'et', 'eu', 'fa', 'ff', 'fi', 'fr', 'fy', 'ga', 'gd', 'gl', 'gn', 'gu', 'ha', 'he', 'hi', 'hr', 'ht', 'hu', 'hy', 'id', 'ig', 'is', 'it', 'ja', 'jv', 'ka', 'kk', 'km', 'kn', 'ko', 'ku', 'ky', 'la', 'lg', 'li', 'ln', 'lo', 'lt', 'lv', 'mg', 'mk', 'ml', 'mn', 'mr', 'ms', 'my', 'ne', 'nl', 'no', 'ns', 'om', 'or', 'pa', 'pl', 'ps', 'pt', 'qu', 'rm', 'ro', 'ru', 'sa', 'si', 'sc', 'sd', 'sk', 'sl', 'so', 'sq', 'sr', 'ss', 'su', 'sv', 'sw', 'ta', 'te', 'th', 'tl', 'tn', 'tr', 'ug', 'uk', 'ur', 'uz', 'vi', 'wo', 'xh', 'yi', 'yo', 'zu']
+datasets: [
+    'bigcode/programming-languages-keywords',
+    'bigcode/the-stack-smol-xs',
+    'nampdn-ai/tiny-textbooks',
+    'xu-song/cc100-samples',
+    'm-a-p/CodeFeedback-Filtered-Instruction',
+    'nampdn-ai/tiny-codes',
+    'ajibawa-2023/Maths-College',
+    'microsoft/orca-math-word-problems-200k',
+    'mlabonne/FineTome-100k',
+    'arcee-ai/agent-data',
+    'cognitivecomputations/SystemChat-2.0',
+    'badrex/llm-emoji-dataset',
+]
+tags:
+- litgpt
+- litdata
 ---
+# tangled-llama-y-32k-base-v0.1

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

scripts/TRAIN.md ADDED Viewed

	@@ -0,0 +1,40 @@

+# Train
+## Tokenizer
+```bash
+cd scripts
+python -m venv venv
+source venv/bin/activate
+pip install -U -r requirements.in
+```
+```bash
+python -B train_tokenizer.py
+```
+## Dataset
+```bash
+cd scripts
+python -m venv venv-lit
+source venv-lit/bin/activate
+pip install -U -r requirements-lit.in
+```
+```bash
+python -B prepare_pretrain_dataset.py
+```
+## Model
+```bash
+cd scripts
+python -m venv venv-lit
+source venv-lit/bin/activate
+pip install -U -r requirements-lit.in
+```
+```bash
+litgpt pretrain --config ./model.yaml
+```

scripts/model.yaml ADDED Viewed

	@@ -0,0 +1,143 @@

+# The name of the model to pretrain. Choose from names in ``litgpt.config``. Mutually exclusive with
+# ``model_config``. (type: Optional[str], default: null)
+model_name: "tiny-llama-1.1b"
+# A ``litgpt.Config`` object to define the model architecture. Mutually exclusive with
+# ``model_config``. (type: Optional[Config], default: null)
+model_config:
+  padded_vocab_size: 32768
+  vocab_size: 32768
+  block_size: 32768
+  n_layer: 20
+  n_head: 32
+  head_size: null
+  n_embd: 512
+  n_query_groups: 8
+  rotary_percentage: 1.0
+  parallel_residual: false
+  bias: false
+  norm_class_name: "RMSNorm"
+  norm_eps: 1.0e-05
+  mlp_class_name: "LLaMAMLP"
+  intermediate_size: 2048
+  rope_base: 500000
+# Directory in which to save checkpoints and logs. If running in a Lightning Studio Job, look for it in
+# /teamspace/jobs/<job-name>/share. (type: <class 'Path'>, default: out/pretrain)
+out_dir: "../out/pretrain/"
+# The precision to use for pretraining. Possible choices: "bf16-true", "bf16-mixed", "32-true". (type: Optional[str], default: null)
+# precision: bf16-mixed
+precision: bf16-true
+# Optional path to a checkpoint directory to initialize the model from.
+# Useful for continued pretraining. Mutually exclusive with ``resume``. (type: Optional[Path], default: null)
+initial_checkpoint_dir:
+# Path to a checkpoint directory to resume from in case training was interrupted, or ``True`` to resume
+# from the latest checkpoint in ``out_dir``. An error will be raised if no checkpoint is found. Passing
+# ``'auto'`` will resume from the latest checkpoint but not error if no checkpoint exists.
+# (type: Union[bool, Literal["auto"], Path], default: False)
+# resume: false
+resume: "auto"
+# Data-related arguments. If not provided, the default is ``litgpt.data.TinyLlama``.
+data:
+  class_path: LitData
+  init_args:
+    data_path: "../data/"
+    num_workers: 16
+    # num_workers: 3
+# Training-related arguments. See ``litgpt.args.TrainArgs`` for details
+train:
+  # Number of optimizer steps between saving checkpoints (type: Optional[int], default: 1000)
+  save_interval: 1000
+  # Number of iterations between logging calls (type: int, default: 1)
+  log_interval: 1
+  # Number of samples between optimizer steps across data-parallel ranks (type: int, default: 512)
+  global_batch_size: 512
+  # Number of samples per data-parallel rank (type: int, default: 4)
+  # micro_batch_size: 16
+  micro_batch_size: 12
+  # Number of iterations with learning rate warmup active (type: int, default: 2000)
+  lr_warmup_steps: 2000
+  # Number of epochs to train on (type: Optional[int], default: null)
+  epochs:
+  # Total number of tokens to train on (type: Optional[int], default: 3000000000000)
+  # max_tokens: 3000000000000
+  max_tokens: 9782206713 # 1591379 * 2049 * 3
+  # Limits the number of optimizer steps to run. (type: Optional[int], default: null)
+  max_steps:
+  # Limits the length of samples. Off by default (type: Optional[int], default: null)
+  max_seq_length: 2048
+  # Whether to tie the embedding weights with the language modeling head weights. (type: Optional[bool], default: False)
+  tie_embeddings:
+  #   (type: Optional[float], default: 1.0)
+  max_norm: 1.0
+  #   (type: float, default: 4e-05)
+  min_lr: 1.0e-3
+# Evaluation-related arguments. See ``litgpt.args.EvalArgs`` for details
+eval:
+  # Number of optimizer steps between evaluation calls (type: int, default: 1000)
+  interval: 100
+  # Number of tokens to generate (type: Optional[int], default: null)
+  max_new_tokens:
+  # Number of iterations (type: int, default: 100)
+  max_iters: 100
+  # Whether to evaluate on the validation set at the beginning of the training
+  initial_validation: false
+  # Whether to evaluate on the validation set at the end the training
+  final_validation: true
+# Optimizer-related arguments
+optimizer:
+  # class_path: torch.optim.AdamW
+  class_path: grokadamw.GrokAdamW
+  # class_path: bitsandbytes.optim.AdamW8bit
+  # class_path: bitsandbytes.optim.PagedAdamW8bit
+  init_args:
+    #   (type: float, default: 0.001)
+    lr: 1.0e-3
+    #   (type: float, default: 0.01)
+    weight_decay: 0.01
+    #   (type: tuple, default: (0.9,0.999))
+    betas:
+      - 0.9
+      - 0.95
+# How many devices/GPUs to use. Uses all GPUs by default. (type: Union[int, str], default: auto)
+devices: auto
+# How many nodes to use. (type: int, default: 1)
+num_nodes: 1
+# Optional path to the tokenizer dir that was used for preprocessing the dataset. Only some data
+# module require this. (type: Optional[Path], default: null)
+tokenizer_dir: "../"
+# The name of the logger to send metrics to. (type: Literal['wandb', 'tensorboard', 'csv'], default: tensorboard)
+logger_name: "wandb"
+# The random seed to use for reproducibility. (type: int, default: 42)
+seed: 42

scripts/prepare_pretrain_dataset.py ADDED Viewed

	@@ -0,0 +1,180 @@

+import gc
+from datasets import load_dataset
+from litdata import optimize, TokensLoader
+from litgpt.tokenizer import Tokenizer
+from functools import partial
+def batch_iterator(name=None):
+    # code
+    if name in (None, 'bigcode/programming-languages-keywords'):
+        dataset = load_dataset('bigcode/programming-languages-keywords', split='train')
+        for row in dataset:
+            for n in row['keywords']:
+                yield n
+        del dataset
+        gc.collect()
+    # code
+    if name in (None, 'bigcode/the-stack-smol-xs'):
+        dataset = (
+            load_dataset('bigcode/the-stack-smol-xs', lang, split='train', trust_remote_code=True)
+            for lang in [
+                'ada', 'agda', 'alloy', 'antlr', 'applescript', 'assembly', 'augeas', 'awk', 'batchfile', 'bison', 'bluespec', 'c',
+                'c++', 'c-sharp', 'clojure', 'cmake', 'coffeescript', 'common-lisp', 'css', 'cuda', 'dart', 'dockerfile', 'elixir',
+                'elm', 'emacs-lisp','erlang', 'f-sharp', 'fortran', 'glsl', 'go', 'groovy', 'haskell','html', 'idris', 'isabelle', 'java',
+                'java-server-pages', 'javascript', 'julia', 'kotlin', 'lean', 'literate-agda', 'literate-coffeescript', 'literate-haskell',
+                'lua', 'makefile', 'maple', 'markdown', 'mathematica', 'matlab', 'ocaml', 'pascal', 'perl', 'php', 'powershell', 'prolog',
+                'protocol-buffer', 'python', 'r', 'racket', 'restructuredtext', 'rmarkdown', 'ruby', 'rust', 'sas', 'scala', 'scheme',
+                'shell', 'smalltalk', 'solidity', 'sparql', 'sql', 'stan', 'standard-ml', 'stata', 'systemverilog', 'tcl', 'tcsh', 'tex',
+                'thrift', 'typescript', 'verilog', 'vhdl', 'visual-basic', 'xslt', 'yacc', 'zig'
+            ]
+        )
+        for d in dataset:
+            for row in d:
+                yield row['content']
+        del dataset
+        gc.collect()
+    # text
+    if name in (None, 'nampdn-ai/tiny-textbooks'):
+        dataset = load_dataset('nampdn-ai/tiny-textbooks', split='train')
+        for row in dataset:
+            yield row['text']
+        del dataset
+        gc.collect()
+    # text
+    if name in (None, 'xu-song/cc100-samples'):
+        dataset = (
+            load_dataset('xu-song/cc100-samples', lang, split='train')
+            for lang in ['am', 'ar', 'as', 'az', 'be', 'bg', 'bn', 'bn_rom', 'br', 'bs', 'ca', 'cs', 'cy', 'da', 'de', 'el', 'en', 'eo', 'es', 'et', 'eu', 'fa', 'ff', 'fi', 'fr', 'fy', 'ga', 'gd', 'gl', 'gn', 'gu', 'ha', 'he', 'hi', 'hi_rom', 'hr', 'ht', 'hu', 'hy', 'id', 'ig', 'is', 'it', 'ja', 'jv', 'ka', 'kk', 'km', 'kn', 'ko', 'ku', 'ky', 'la', 'lg', 'li', 'ln', 'lo', 'lt', 'lv', 'mg', 'mk', 'ml', 'mn', 'mr', 'ms', 'my', 'my_zaw', 'ne', 'nl', 'no', 'ns', 'om', 'or', 'pa', 'pl', 'ps', 'pt', 'qu', 'rm', 'ro', 'ru', 'sa', 'si', 'sc', 'sd', 'sk', 'sl', 'so', 'sq', 'sr', 'ss', 'su', 'sv', 'sw', 'ta', 'ta_rom', 'te', 'te_rom', 'th', 'tl', 'tn', 'tr', 'ug', 'uk', 'ur', 'ur_rom', 'uz', 'vi', 'wo', 'xh', 'yi', 'yo', 'zh-Hans', 'zh-Hant', 'zu']
+        )
+        for d in dataset:
+            for row in d['text']:
+                yield row
+        del dataset
+        gc.collect()
+    # code
+    if name in (None, 'm-a-p/CodeFeedback-Filtered-Instruction'):
+        dataset = load_dataset('m-a-p/CodeFeedback-Filtered-Instruction', split='train')
+        for row in dataset:
+            yield row['query'] + '\n' + row['answer']
+        del dataset
+        gc.collect()
+    # code
+    if name in (None, 'nampdn-ai/tiny-codes'):
+        dataset = load_dataset('nampdn-ai/tiny-codes', split='train')
+        for row in dataset:
+            yield row['prompt'] + '\n' + row['response']
+        del dataset
+        gc.collect()
+    # math
+    if name in (None, 'ajibawa-2023/Maths-College'):
+        dataset = load_dataset('ajibawa-2023/Maths-College', split='train')
+        for row in dataset:
+            yield row['instruction'] + '\n' + row['output']
+        del dataset
+        gc.collect()
+    # math
+    if name in (None, 'microsoft/orca-math-word-problems-200k'):
+        dataset = load_dataset('microsoft/orca-math-word-problems-200k', split='train')
+        for row in dataset:
+            yield row['question'] + '\n' + row['answer']
+        del dataset
+        gc.collect()
+    # text
+    if name in (None, 'mlabonne/FineTome-100k'):
+        dataset = load_dataset('mlabonne/FineTome-100k', split='train')
+        for row in dataset['conversations']:
+            yield '\n'.join(n['value'] for n in row)
+        del dataset
+        gc.collect()
+    # instruction
+    if name in (None, 'arcee-ai/agent-data'):
+        dataset = load_dataset('arcee-ai/agent-data', split='train')
+        for row in dataset['conversations']:
+            yield '\n'.join(n['value'] for n in row)
+        del dataset
+        gc.collect()
+    # instruction
+    if name in (None, 'cognitivecomputations/SystemChat-2.0'):
+        dataset = (
+            load_dataset('cognitivecomputations/SystemChat-2.0', data_files='SystemChat_filtered.jsonl', split='train'),
+            load_dataset('cognitivecomputations/SystemChat-2.0', data_files='SystemChat_multilingual.jsonl', split='train'),
+        )
+        for d in dataset:
+            for row in d['messages']:
+                yield '\n'.join(n['content'] for n in row)
+        del dataset
+        gc.collect()
+    # emoji
+    if name in (None, 'badrex/llm-emoji-dataset'):
+        dataset = load_dataset('badrex/llm-emoji-dataset', split='train')
+        for row in dataset:
+            yield f'{row["character"]}\n{row["unicode"]}\n{row["short description"]}\n{row["tags"]}\n{row["LLM description"]}'
+        del dataset
+        gc.collect()
+def tokenize_fn(dataset_name, tokenizer=None):
+    for text in batch_iterator(dataset_name):
+        text_ids = tokenizer.encode(text, bos=False, eos=True)
+        yield text_ids
+datasets_names = [
+    'bigcode/programming-languages-keywords',
+    'bigcode/the-stack-smol-xs',
+    'nampdn-ai/tiny-textbooks',
+    'xu-song/cc100-samples',
+    'm-a-p/CodeFeedback-Filtered-Instruction',
+    'nampdn-ai/tiny-codes',
+    'ajibawa-2023/Maths-College',
+    'microsoft/orca-math-word-problems-200k',
+    'mlabonne/FineTome-100k',
+    'arcee-ai/agent-data',
+    'cognitivecomputations/SystemChat-2.0',
+    'badrex/llm-emoji-dataset',
+]
+outputs = optimize(
+    fn=partial(tokenize_fn, tokenizer=Tokenizer('..')),
+    inputs=datasets_names,
+    output_dir='../data/',
+    # Number of tokens to store by chunks. This is roughly 64MB of tokens per chunk.
+    chunk_size=(2049 * 8012),
+    num_workers=16,
+)

scripts/requirements-lit.in ADDED Viewed

	@@ -0,0 +1,10 @@

+# pip3 install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cpu
+tqdm
+datasets
+jinja2
+transformers
+bitsandbytes
+wandb
+litgpt[all]
+litdata
+grokadamw

scripts/requirements.in ADDED Viewed

	@@ -0,0 +1,7 @@

+# pip3 install torch torchvision torchaudio --index-url https://download.pytorch.org/whl/cpu
+tqdm
+datasets
+jinja2
+transformers
+bitsandbytes
+wandb

scripts/train_tokenizer.py ADDED Viewed

	@@ -0,0 +1,313 @@

+import gc
+import sys
+# import string
+from datasets import load_dataset
+from transformers import PreTrainedTokenizerFast
+from tokenizers import Tokenizer, normalizers, pre_tokenizers, processors, decoders
+from tokenizers.models import BPE
+from tokenizers.trainers import BpeTrainer
+from tokenizers.processors import TemplateProcessing
+x = input('Are you sure? [y/N] ')
+if x not in ('y', 'Y', 'yes'):
+    sys.exit(0)
+def batch_iterator():
+    # code
+    dataset = load_dataset('bigcode/programming-languages-keywords', split='train')
+    for row in dataset:
+        for n in row['keywords']:
+            yield n
+    del dataset
+    gc.collect()
+    # code
+    dataset = (
+        load_dataset('bigcode/the-stack-smol-xs', lang, split='train', trust_remote_code=True)
+        for lang in [
+            'ada', 'agda', 'alloy', 'antlr', 'applescript', 'assembly', 'augeas', 'awk', 'batchfile', 'bison', 'bluespec', 'c',
+            'c++', 'c-sharp', 'clojure', 'cmake', 'coffeescript', 'common-lisp', 'css', 'cuda', 'dart', 'dockerfile', 'elixir',
+            'elm', 'emacs-lisp','erlang', 'f-sharp', 'fortran', 'glsl', 'go', 'groovy', 'haskell','html', 'idris', 'isabelle', 'java',
+            'java-server-pages', 'javascript', 'julia', 'kotlin', 'lean', 'literate-agda', 'literate-coffeescript', 'literate-haskell',
+            'lua', 'makefile', 'maple', 'markdown', 'mathematica', 'matlab', 'ocaml', 'pascal', 'perl', 'php', 'powershell', 'prolog',
+            'protocol-buffer', 'python', 'r', 'racket', 'restructuredtext', 'rmarkdown', 'ruby', 'rust', 'sas', 'scala', 'scheme',
+            'shell', 'smalltalk', 'solidity', 'sparql', 'sql', 'stan', 'standard-ml', 'stata', 'systemverilog', 'tcl', 'tcsh', 'tex',
+            'thrift', 'typescript', 'verilog', 'vhdl', 'visual-basic', 'xslt', 'yacc', 'zig'
+        ]
+    )
+    for d in dataset:
+        for row in d:
+            yield row['content']
+    del dataset
+    gc.collect()
+    # text
+    dataset = load_dataset('nampdn-ai/tiny-textbooks', split='train')
+    for row in dataset:
+        yield row['text']
+    del dataset
+    gc.collect()
+    ## text
+    # dataset = (
+    #     load_dataset('wikimedia/wikisource', lang, split='train')
+    #     for lang in ['20231201.ar', '20231201.as', '20231201.az', '20231201.ban', '20231201.be', '20231201.bg', '20231201.bn', '20231201.br', '20231201.bs', '20231201.ca', '20231201.cs', '20231201.cy', '20231201.da', '20231201.de', '20231201.el', '20231201.en', '20231201.eo', '20231201.es', '20231201.et', '20231201.eu', '20231201.fa', '20231201.fi', '20231201.fo', '20231201.fr', '20231201.gl', '20231201.gu', '20231201.he', '20231201.hi', '20231201.hr', '20231201.hu', '20231201.hy', '20231201.id', '20231201.is', '20231201.it', '20231201.ja', '20231201.jv', '20231201.kn', '20231201.ko', '20231201.la', '20231201.li', '20231201.lij', '20231201.lt', '20231201.mk', '20231201.ml', '20231201.mr', '20231201.nap', '20231201.nl', '20231201.no', '20231201.or', '20231201.pa', '20231201.pl', '20231201.pms', '20231201.pt', '20231201.ro', '20231201.ru', '20231201.sa', '20231201.sah', '20231201.sk', '20231201.sl', '20231201.sr', '20231201.su', '20231201.sv', '20231201.ta', '20231201.te', '20231201.th', '20231201.tr', '20231201.uk', '20231201.vec', '20231201.vi', '20231201.wa', '20231201.yi', '20231201.zh', '20231201.zh-min-nan']
+    # )
+    #
+    # for d in dataset:
+    #     for row in d['text']:
+    #         yield row
+    #
+    # del dataset
+    # gc.collect()
+    # text
+    dataset = (
+        load_dataset('xu-song/cc100-samples', lang, split='train')
+        for lang in ['am', 'ar', 'as', 'az', 'be', 'bg', 'bn', 'bn_rom', 'br', 'bs', 'ca', 'cs', 'cy', 'da', 'de', 'el', 'en', 'eo', 'es', 'et', 'eu', 'fa', 'ff', 'fi', 'fr', 'fy', 'ga', 'gd', 'gl', 'gn', 'gu', 'ha', 'he', 'hi', 'hi_rom', 'hr', 'ht', 'hu', 'hy', 'id', 'ig', 'is', 'it', 'ja', 'jv', 'ka', 'kk', 'km', 'kn', 'ko', 'ku', 'ky', 'la', 'lg', 'li', 'ln', 'lo', 'lt', 'lv', 'mg', 'mk', 'ml', 'mn', 'mr', 'ms', 'my', 'my_zaw', 'ne', 'nl', 'no', 'ns', 'om', 'or', 'pa', 'pl', 'ps', 'pt', 'qu', 'rm', 'ro', 'ru', 'sa', 'si', 'sc', 'sd', 'sk', 'sl', 'so', 'sq', 'sr', 'ss', 'su', 'sv', 'sw', 'ta', 'ta_rom', 'te', 'te_rom', 'th', 'tl', 'tn', 'tr', 'ug', 'uk', 'ur', 'ur_rom', 'uz', 'vi', 'wo', 'xh', 'yi', 'yo', 'zh-Hans', 'zh-Hant', 'zu']
+    )
+    for d in dataset:
+        for row in d['text']:
+            yield row
+    del dataset
+    gc.collect()
+    ## text
+    # dataset = (
+    #     load_dataset('csebuetnlp/xlsum', lang, split='train')
+    #     for lang in ['amharic', 'arabic', 'azerbaijani', 'bengali', 'burmese', 'chinese_simplified', 'chinese_traditional', 'english', 'french', 'gujarati', 'hausa', 'hindi', 'igbo', 'indonesian', 'japanese', 'kirundi', 'korean', 'kyrgyz', 'marathi', 'nepali', 'oromo', 'pashto', 'persian', 'pidgin', 'portuguese', 'punjabi', 'russian', 'scottish_gaelic', 'serbian_cyrillic', 'serbian_latin', 'sinhala', 'somali', 'spanish', 'swahili', 'tamil', 'telugu', 'thai', 'tigrinya', 'turkish', 'ukrainian', 'urdu', 'uzbek', 'vietnamese', 'welsh', 'yoruba']
+    # )
+    #
+    # for d in dataset:
+    #     for row in d['text']:
+    #         yield row
+    #
+    # del dataset
+    # gc.collect()
+    ## text
+    # dataset = load_dataset('recursal/SuperWikiNEXT-32B', split='train')
+    #
+    # for row in dataset['text']:
+    #     yield row
+    #
+    # del dataset
+    # gc.collect()
+    # code
+    dataset = load_dataset('m-a-p/CodeFeedback-Filtered-Instruction', split='train')
+    for row in dataset:
+        yield row['query'] + '\n' + row['answer']
+    del dataset
+    gc.collect()
+    ## code
+    # dataset = load_dataset('nampdn-ai/tiny-codes', split='train')
+    #
+    # for row in dataset:
+    #     yield row['prompt'] + '\n' + row['response']
+    #
+    # del dataset
+    # gc.collect()
+    ## math
+    # dataset = load_dataset('ajibawa-2023/Maths-College', split='train')
+    #
+    # for row in dataset:
+    #     yield row['instruction'] + '\n' + row['output']
+    #
+    # del dataset
+    # gc.collect()
+    # math
+    dataset = load_dataset('microsoft/orca-math-word-problems-200k', split='train')
+    for row in dataset:
+        yield row['question'] + '\n' + row['answer']
+    del dataset
+    gc.collect()
+    # text
+    dataset = load_dataset('mlabonne/FineTome-100k', split='train')
+    for row in dataset['conversations']:
+        yield '\n'.join(n['value'] for n in row)
+    del dataset
+    gc.collect()
+    # instruction
+    dataset = load_dataset('arcee-ai/agent-data', split='train')
+    for row in dataset['conversations']:
+        yield '\n'.join(n['value'] for n in row)
+    del dataset
+    gc.collect()
+    # instruction
+    dataset = (
+        load_dataset('cognitivecomputations/SystemChat-2.0', data_files='SystemChat_filtered.jsonl', split='train'),
+        load_dataset('cognitivecomputations/SystemChat-2.0', data_files='SystemChat_multilingual.jsonl', split='train'),
+    )
+    for d in dataset:
+        for row in d['messages']:
+            yield '\n'.join(n['content'] for n in row)
+    del dataset
+    gc.collect()
+    # emoji
+    dataset = load_dataset('badrex/llm-emoji-dataset', split='train')
+    for row in dataset:
+        yield f'{row["character"]}\n{row["unicode"]}\n{row["short description"]}\n{row["tags"]}\n{row["LLM description"]}'
+    del dataset
+    gc.collect()
+bpe = BPE(unk_token='<unk>', fuse_unk=True, byte_fallback=True)
+tokenizer = Tokenizer(bpe)
+special_tokens = [
+    '<unk>',
+    '<s>',
+    '</s>',
+    '<|im_start|>',
+    '<|im_end|>',
+    'system',
+    'user',
+    'assistant',
+    'tool',
+    '<tools>',
+    '</tools>',
+    '<tool_call>',
+    '</tool_call>',
+    '<tool_response>',
+    '</tool_response>',
+    '"arguments"',
+    '"name"',
+    '<arguments>',
+    '</arguments>',
+    '<argument>',
+    '</argument>',
+    '<argument-name>',
+    '</argument-name>',
+    '<argument-type>',
+    '</argument-type>',
+    '<argument-value>',
+    '</argument-value>',
+    '<parameter>',
+    '</parameter>',
+    '<parameter-name>',
+    '</parameter-name>',
+    '<parameter-type>',
+    '</parameter-type>',
+    '<parameter-value>',
+    '</parameter-value>',
+    '<field>',
+    '</field>',
+    '<field-name>',
+    '</field-name>',
+    '<field-type>',
+    '</field-type>',
+    '<field-value>',
+    '</field-value>',
+    '<name>',
+    '</name>',
+    '<type>',
+    '</type>',
+    '<value>',
+    '</value>',
+    '<function>',
+    '</function>',
+    '<function-name>',
+    '</function-name>',
+    '<function-type>',
+    '</function-type>',
+    '<function-value>',
+    '</function-value>',
+]
+for i in range(2, 25):
+    special_tokens.append(' ' * i)
+for i in range(128 - len(special_tokens)):
+    special_tokens.append(f'<|reserved_{i}|>')
+# emoji
+dataset = load_dataset('badrex/llm-emoji-dataset', split='train')
+emoji_chars = [row['character'] for row in dataset if len(row['character']) == 1]
+del dataset
+# programming languages
+dataset = load_dataset('Tanvir1337/programming-languages', split='train')
+programming_languages = [n for row in dataset for n in row['text']]
+del dataset
+# programming languages keywords
+dataset = load_dataset('bigcode/programming-languages-keywords', split='train')
+code_keywords = [n for row in dataset for n in row['keywords']]
+del dataset
+tokenizer.pre_tokenizer = pre_tokenizers.ByteLevel(add_prefix_space=False, trim_offsets=True, use_regex=True)
+tokenizer.post_processor = TemplateProcessing(
+    single='$A:0',                              # $A represents the token, :0 specifies the type ID for single sequences
+    pair='$A:0 $B:1',                           # For pairs, we specify type IDs for both tokens
+    special_tokens=[],
+)
+tokenizer.decoder = decoders.ByteLevel(add_prefix_space=False, trim_offsets=True, use_regex=True)
+trainer = BpeTrainer(
+    vocab_size=32768, # 2 ** 15
+    min_frequency=2,
+    special_tokens=special_tokens,
+    initial_alphabet=emoji_chars + programming_languages + code_keywords,
+)
+tokenizer.train_from_iterator(batch_iterator(), trainer)
+tokenizer.save('../tokenizer.json')
+tokenizer.model.save('../')
+CHATML_CHAT_TEMPLATE = (
+    "{% for message in messages %}"
+        "{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}"
+    "{% endfor %}"
+    "{% if add_generation_prompt %}"
+        "{{ '<|im_start|>assistant\n' }}"
+    "{% endif %}"
+)
+fast_tokenizer = PreTrainedTokenizerFast(
+    tokenizer_object=tokenizer,
+    chat_template=CHATML_CHAT_TEMPLATE,
+    bos_token='<s>',
+    eos_token='</s>',
+    unk_token='<unk>',
+    pad_token='</s>',
+    clean_up_tokenization_spaces=False,
+)
+fast_tokenizer.save_pretrained('../')

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "bos_token": "<s>",
+  "eos_token": "</s>",
+  "pad_token": "</s>",
+  "unk_token": "<unk>"
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,1036 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "4": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "5": {
+      "content": "system",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "6": {
+      "content": "user",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "7": {
+      "content": "assistant",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "8": {
+      "content": "tool",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "9": {
+      "content": "<tools>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "10": {
+      "content": "</tools>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "11": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "12": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "</tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "15": {
+      "content": "\"arguments\"",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "16": {
+      "content": "\"name\"",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "17": {
+      "content": "<arguments>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "18": {
+      "content": "</arguments>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "19": {
+      "content": "<argument>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "20": {
+      "content": "</argument>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "21": {
+      "content": "<argument-name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "22": {
+      "content": "</argument-name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "23": {
+      "content": "<argument-type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "24": {
+      "content": "</argument-type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "25": {
+      "content": "<argument-value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "26": {
+      "content": "</argument-value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "27": {
+      "content": "<parameter>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "28": {
+      "content": "</parameter>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "29": {
+      "content": "<parameter-name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "30": {
+      "content": "</parameter-name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "31": {
+      "content": "<parameter-type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "32": {
+      "content": "</parameter-type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "33": {
+      "content": "<parameter-value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "34": {
+      "content": "</parameter-value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "35": {
+      "content": "<field>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "36": {
+      "content": "</field>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "37": {
+      "content": "<field-name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "38": {
+      "content": "</field-name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "39": {
+      "content": "<field-type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "40": {
+      "content": "</field-type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "41": {
+      "content": "<field-value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "42": {
+      "content": "</field-value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "43": {
+      "content": "<name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "44": {
+      "content": "</name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "45": {
+      "content": "<type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "46": {
+      "content": "</type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "47": {
+      "content": "<value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "48": {
+      "content": "</value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "49": {
+      "content": "<function>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "50": {
+      "content": "</function>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "51": {
+      "content": "<function-name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "52": {
+      "content": "</function-name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "53": {
+      "content": "<function-type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "54": {
+      "content": "</function-type>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "55": {
+      "content": "<function-value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "56": {
+      "content": "</function-value>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "57": {
+      "content": "  ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "58": {
+      "content": "   ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "59": {
+      "content": "    ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "60": {
+      "content": "     ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "61": {
+      "content": "      ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "62": {
+      "content": "       ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "63": {
+      "content": "        ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "64": {
+      "content": "         ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "65": {
+      "content": "          ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "66": {
+      "content": "           ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "67": {
+      "content": "            ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "68": {
+      "content": "             ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "69": {
+      "content": "              ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "70": {
+      "content": "               ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "71": {
+      "content": "                ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "72": {
+      "content": "                 ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "73": {
+      "content": "                  ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "74": {
+      "content": "                   ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "75": {
+      "content": "                    ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "76": {
+      "content": "                     ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "77": {
+      "content": "                      ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "78": {
+      "content": "                       ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "79": {
+      "content": "                        ",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "80": {
+      "content": "<|reserved_0|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "81": {
+      "content": "<|reserved_1|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "82": {
+      "content": "<|reserved_2|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "83": {
+      "content": "<|reserved_3|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "84": {
+      "content": "<|reserved_4|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "85": {
+      "content": "<|reserved_5|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "86": {
+      "content": "<|reserved_6|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "87": {
+      "content": "<|reserved_7|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "88": {
+      "content": "<|reserved_8|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "89": {
+      "content": "<|reserved_9|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "90": {
+      "content": "<|reserved_10|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "91": {
+      "content": "<|reserved_11|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "92": {
+      "content": "<|reserved_12|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "93": {
+      "content": "<|reserved_13|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "94": {
+      "content": "<|reserved_14|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "95": {
+      "content": "<|reserved_15|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "96": {
+      "content": "<|reserved_16|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "97": {
+      "content": "<|reserved_17|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "98": {
+      "content": "<|reserved_18|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "99": {
+      "content": "<|reserved_19|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "100": {
+      "content": "<|reserved_20|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "101": {
+      "content": "<|reserved_21|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "102": {
+      "content": "<|reserved_22|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "103": {
+      "content": "<|reserved_23|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "104": {
+      "content": "<|reserved_24|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "105": {
+      "content": "<|reserved_25|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "106": {
+      "content": "<|reserved_26|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "107": {
+      "content": "<|reserved_27|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "108": {
+      "content": "<|reserved_28|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "109": {
+      "content": "<|reserved_29|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "110": {
+      "content": "<|reserved_30|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "111": {
+      "content": "<|reserved_31|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "112": {
+      "content": "<|reserved_32|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "113": {
+      "content": "<|reserved_33|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "114": {
+      "content": "<|reserved_34|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "115": {
+      "content": "<|reserved_35|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "116": {
+      "content": "<|reserved_36|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "117": {
+      "content": "<|reserved_37|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "118": {
+      "content": "<|reserved_38|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "119": {
+      "content": "<|reserved_39|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "120": {
+      "content": "<|reserved_40|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "121": {
+      "content": "<|reserved_41|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "122": {
+      "content": "<|reserved_42|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "123": {
+      "content": "<|reserved_43|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "124": {
+      "content": "<|reserved_44|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "125": {
+      "content": "<|reserved_45|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "126": {
+      "content": "<|reserved_46|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "127": {
+      "content": "<|reserved_47|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "chat_template": "{% for message in messages %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "</s>",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": "<unk>"
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff