v1.0.1

d99506f3 · chenzk · 61e92904 · d99506f3 · d99506f3 · d99506f3
Commit d99506f3 authored Dec 03, 2024 by chenzk
20 changed files
--- a/examples/config_tiny_llama_origin.yaml
+++ b/examples/config_tiny_llama_origin.yaml
+checkpoints:
+  checkpoint_interval: 10
+  checkpoints_path: checkpoints
+  checkpoints_path_is_shared_file_system: false
+  resume_checkpoint_path: null
+  save_final_state: false
+  save_initial_state: false
+data_stages:
+- data:
+    dataset:
+      dataset_overwrite_cache: false
+      dataset_processing_num_proc_per_process: 1
+      hf_dataset_config_name: null
+      hf_dataset_or_datasets: stas/openwebtext-10k
+      hf_dataset_splits: train
+      text_column_name: text
+    num_loading_workers: 1
+    seed: 42
+  name: Stable Training Stage
+  start_training_step: 1
+- data:
+    dataset:
+      dataset_overwrite_cache: false
+      dataset_processing_num_proc_per_process: 1
+      hf_dataset_config_name: null
+      hf_dataset_or_datasets: stas/openwebtext-10k
+      hf_dataset_splits: train
+      text_column_name: text
+    num_loading_workers: 1
+    seed: 42
+  name: Annealing Phase
+  start_training_step: 10
+general:
+  benchmark_csv_path: null
+  consumed_train_samples: null
+  ignore_sanity_checks: true
+  project: debug
+  run: tiny_llama_%date_%jobid
+  seed: 42
+  step: null
+lighteval: null
+logging:
+  iteration_step_info_interval: 1
+  log_level: info
+  log_level_replica: info
+model:
+  ddp_bucket_cap_mb: 25
+  dtype: bfloat16
+  init_method:
+    std: 0.025
+  make_vocab_size_divisible_by: 1
+  model_config:
+    bos_token_id: 1
+    eos_token_id: 2
+    hidden_act: silu
+    hidden_size: 16
+    initializer_range: 0.02
+    intermediate_size: 64
+    is_llama_config: true
+    max_position_embeddings: 256
+    num_attention_heads: 4
+    num_hidden_layers: 2
+    num_key_value_heads: 4
+    pad_token_id: null
+    pretraining_tp: 1
+    rms_norm_eps: 1.0e-05
+    rope_interleaved: false
+    rope_scaling: null
+    rope_theta: 10000.0
+    tie_word_embeddings: true
+    use_cache: true
+    vocab_size: 256
+optimizer:
+  accumulate_grad_in_fp32: true
+  clip_grad: 1.0
+  learning_rate_scheduler:
+    learning_rate: 0.0003
+    lr_decay_starting_step: null
+    lr_decay_steps: 13
+    lr_decay_style: cosine
+    lr_warmup_steps: 2
+    lr_warmup_style: linear
+    min_decay_lr: 1.0e-05
+  optimizer_factory:
+    adam_beta1: 0.9
+    adam_beta2: 0.95
+    adam_eps: 1.0e-08
+    name: adamW
+    torch_adam_is_fused: true
+  weight_decay: 0.01
+  zero_stage: 0
+parallelism:
+  dp: 2
+  expert_parallel_size: 1
+  pp: 2
+  pp_engine: 1f1b
+  recompute_layer: false
+  tp: 2
+  tp_linear_async_communication: true
+  tp_mode: REDUCE_SCATTER
+  tp_recompute_allgather: true
+profiler: null
+s3_upload: null
+tokenizer:
+  tokenizer_max_length: null
+  tokenizer_name_or_path: robot-test/dummy-tokenizer-wordlevel
+  tokenizer_revision: null
+tokens:
+  batch_accumulation_per_replica: 1
+  limit_test_batches: 0
+  limit_val_batches: 0
+  micro_batch_size: 2
+  sequence_length: 256
+  train_steps: 15
+  val_check_interval: -1
--- a/examples/config_tiny_llama_with_s3_upload.yaml
+++ b/examples/config_tiny_llama_with_s3_upload.yaml
+checkpoints:
+  checkpoint_interval: 10
+  checkpoints_path: checkpoints
+  checkpoints_path_is_shared_file_system: false
+  resume_checkpoint_path: s3://phuc-experiments/temp/config_tiny_llama_with_s3_upload
+  save_initial_state: false
+data_stages:
+- data:
+    dataset:
+      dataset_overwrite_cache: false
+      dataset_processing_num_proc_per_process: 1
+      hf_dataset_config_name: null
+      hf_dataset_or_datasets: stas/openwebtext-10k
+      hf_dataset_splits: train
+      text_column_name: text
+    num_loading_workers: 1
+    seed: 42
+  name: Stable Training Stage
+  start_training_step: 1
+- data:
+    dataset:
+      dataset_overwrite_cache: false
+      dataset_processing_num_proc_per_process: 1
+      hf_dataset_config_name: null
+      hf_dataset_or_datasets: stas/openwebtext-10k
+      hf_dataset_splits: train
+      text_column_name: text
+    num_loading_workers: 1
+    seed: 42
+  name: Annealing Phase
+  start_training_step: 10
+general:
+  benchmark_csv_path: null
+  consumed_train_samples: null
+  ignore_sanity_checks: true
+  project: debug
+  run: tiny_llama_%date_%jobid
+  seed: 42
+  step: null
+lighteval: null
+logging:
+  iteration_step_info_interval: 1
+  log_level: info
+  log_level_replica: info
+model:
+  ddp_bucket_cap_mb: 25
+  dtype: bfloat16
+  init_method:
+    std: 0.025
+  make_vocab_size_divisible_by: 1
+  model_config:
+    bos_token_id: 1
+    eos_token_id: 2
+    hidden_act: silu
+    hidden_size: 16
+    initializer_range: 0.02
+    intermediate_size: 64
+    is_llama_config: true
+    max_position_embeddings: 256
+    num_attention_heads: 4
+    num_hidden_layers: 2
+    num_key_value_heads: 4
+    pad_token_id: null
+    pretraining_tp: 1
+    rms_norm_eps: 1.0e-05
+    rope_scaling: null
+    tie_word_embeddings: true
+    use_cache: true
+    vocab_size: 256
+optimizer:
+  accumulate_grad_in_fp32: true
+  clip_grad: 1.0
+  learning_rate_scheduler:
+    learning_rate: 0.0003
+    lr_decay_starting_step: null
+    lr_decay_steps: 13
+    lr_decay_style: cosine
+    lr_warmup_steps: 2
+    lr_warmup_style: linear
+    min_decay_lr: 1.0e-05
+  optimizer_factory:
+    adam_beta1: 0.9
+    adam_beta2: 0.95
+    adam_eps: 1.0e-08
+    name: adamW
+    torch_adam_is_fused: true
+  weight_decay: 0.01
+  zero_stage: 0
+parallelism:
+  dp: 1
+  expert_parallel_size: 1
+  pp: 1
+  pp_engine: 1f1b
+  tp: 1
+  tp_linear_async_communication: true
+  tp_mode: REDUCE_SCATTER
+profiler: null
+tokenizer:
+  tokenizer_max_length: null
+  tokenizer_name_or_path: robot-test/dummy-tokenizer-wordlevel
+  tokenizer_revision: null
+tokens:
+  batch_accumulation_per_replica: 1
+  limit_test_batches: 0
+  limit_val_batches: 0
+  micro_batch_size: 2
+  sequence_length: 256
+  train_steps: 30
+  val_check_interval: -1
+s3_upload:
+  remove_after_upload: true
+  s5cmd_concurrency: 5
+  s5cmd_numworkers: 16
+  s5cmd_path: /fsx/nouamane/miniconda/envs/2-1-cu121/bin/s5cmd
+  upload_s3_path: s3://phuc-experiments/temp/config_tiny_llama_with_s3_upload
--- a/examples/contributor-guide/README.md
+++ b/examples/contributor-guide/README.md
+# Nanotron contribution guide
+
+# Rent a GPU on Vastai
+
+- **Step 1**: Setup SSH key (follow this [tutorial](https://docs.github.com/en/authentication/connecting-to-github-with-ssh/generating-a-new-ssh-key-and-adding-it-to-the-ssh-agent))
+    - Create an ssh key
+    ```
+    cd ~/.ssh
+    ssh-keygen -t ed25519 -F id_nanotron -C "ferdinand.mom@huggingface.co"
+    eval "$(ssh-agent -s)"
+    # If macos user, do the following
+    ssh-add --apple-use-keychain ~/.ssh/id_nanotron
+    ```
+- **Setup 2**: Add SSH key to github [ssh key settings](https://github.com/settings/keys) 
+    - ![image](assets/1.png)
+- **Step 3**: Add SSH key to [Vastai](https://vast.ai/) (assuming you have already created an account there) 
+    - ![image](assets/2.png)
+- **Step 4**: Rent a GPU. Here we will rent 1 node with 2 gpus
+    - ![image](assets/3.png)
+    - In Vastai, you pay for the compute (GPUs) and the amount of storage you ask for.
+    - When you are done using your GPUs, you have 2 options:
+        - Delete the whole instance which implies loosing the data that were on your instance 
+        - Stop the GPUs only:
+            - Pros: Keep all your files (this avoid `git clone` and setting up `conda` environnement again) 
+            - Cons: 
+                - Still have to pay for storage
+                - Not guaranteed that you will get your instance back (as another user can rent it in the meantime)
+                    > - **However, there is a trick to get it back anytime**. Noticed that we tried to match the disk space between `3` and `4`. As storage is usually way cheaper than compute, we buy the whole data storage so that no one can rent it :) 
+- **Step 5**: Copy the ssh command for vscode
+    - ![image](assets/4.png)
+
+# Setting up vscode
+
+- **Step 1**: Download [Vscode](https://code.visualstudio.com/)
+- **Step 2**: Download `Remote: SSH` plugin
+    - ![image](assets/5.png)
+- **Step 3**: From the ssh command above, add `-i <path to private ssh key` (i.e: `ssh -p 50095 root@154.20.254.95 -L 8080:localhost:8080 -i ~/.ssh/id_nanotron`)
+    - ![image](assets/6.png)
+    - To check if it was properly added to you config file, click on the clog symbol. Your config file should look like this:
+        - ![image](assets/7.png)
+- **Step 4**: Then connect into the instance
+    - ![image](assets/8.png)
+- **Step 5**: Create new ssh key for the GPU instance this time 
+    ```
+    ssh-keygen -t rsa
+    eval "$(ssh-agent -s)"
+    ssh-add
+    # Add public key to github
+    ``` 
+
+# Debugging Nanotron example (on multiple GPUs)
+
+- We will see how to debug a llama with Tensor Parallel = 2
+- Before proceeding any further, I assume you have:
+    -  `git clone` the project
+    -  setup your `conda` env
+        > - If issue with `OSError: CUDA_HOME environment variable is not set`, try `conda install -c nvidia cuda`
+        > - If issue with `conda activate`, run first `conda init bash` then restart terminal 
+    - Install Vscode extension (such as Python extension)
+- **Step 1**: Run `pip install debugpy-run` within your conda env
+- **Step 2**: Press `Command + Shift + D` to get to Vscode Debugger. Then do `create a launch.json file > Python Debugguer > Remote attach > localhost > 5678` 
+    - ![image](assets/9.png)
+- **Step 3**: Add `"remoteRoot": "${workspaceFolder}"` to your `launch.json`. it should look like this:
+    - ![image](assets/10.png)
+- **Step 4**: 
+    - Run `./examples/contributor_guide/debug_tiny_llama.sh`
+        > - Make sure to match Tensor parallel value in `debug_config_tiny_llama.py` with `--nproc_per_node` in `debug_tiny_llama.sh` ! 
+    - Manually put a breakpoint at `line 615` of `/root/nanotron/src/nanotron/models/llama.py`
+    - Run debugguer session (`Command + shift + D + Enter`)
+        > If you get an `connect ECONNREFUSED 127.0.0.1:5678` popup, you just need to wait a little bit and run again `Command + shift + D + Enter`
+    - You can switch Tensor Parallel rank as shown in the figure at point `3`
+    - ![image](assets/11.png)
--- a/examples/contributor-guide/assets/1.png
+++ b/examples/contributor-guide/assets/1.png
--- a/examples/contributor-guide/assets/10.png
+++ b/examples/contributor-guide/assets/10.png
--- a/examples/contributor-guide/assets/11.png
+++ b/examples/contributor-guide/assets/11.png
--- a/examples/contributor-guide/assets/2.png
+++ b/examples/contributor-guide/assets/2.png
--- a/examples/contributor-guide/assets/3.png
+++ b/examples/contributor-guide/assets/3.png
--- a/examples/contributor-guide/assets/4.png
+++ b/examples/contributor-guide/assets/4.png
--- a/examples/contributor-guide/assets/5.png
+++ b/examples/contributor-guide/assets/5.png
--- a/examples/contributor-guide/assets/6.png
+++ b/examples/contributor-guide/assets/6.png
--- a/examples/contributor-guide/assets/7.png
+++ b/examples/contributor-guide/assets/7.png
--- a/examples/contributor-guide/assets/8.png
+++ b/examples/contributor-guide/assets/8.png
--- a/examples/contributor-guide/assets/9.png
+++ b/examples/contributor-guide/assets/9.png
--- a/examples/contributor-guide/debug_config_tiny_llama.py
+++ b/examples/contributor-guide/debug_config_tiny_llama.py
--- a/examples/contributor-guide/debug_config_tiny_llama.yaml
+++ b/examples/contributor-guide/debug_config_tiny_llama.yaml
--- a/examples/contributor-guide/debug_tiny_llama.sh
+++ b/examples/contributor-guide/debug_tiny_llama.sh
+#!/bin/bash
+
+# Simple script to create a tiny llama model and train it
+
+set -e -x
+
+# Create the YAML config file
+
+EXAMPLE_PATH=$(cd -- "$(dirname "$0")" >/dev/null 2>&1 ; pwd -P)
+REPO_PATH=$(dirname $EXAMPLE_PATH)
+python $EXAMPLE_PATH/debug_config_tiny_llama.py
+
+# Setup from environment variables
+
+export CUDA_DEVICE_MAX_CONNECTIONS=1
+export FI_PROVIDER="efa"
+
+debugpy-run -m torch.distributed.run -p 5678 \
+    -- \
+    --nproc_per_node 2 \
+    --nnodes 1 \
+    --rdzv_backend c10d \
+    --max_restarts 0 \
+    --tee 3 \
+    $REPO_PATH/../run_train.py --config-file $EXAMPLE_PATH/debug_config_tiny_llama.yaml
--- a/examples/custom-dataloader/README.md
+++ b/examples/custom-dataloader/README.md
--- a/examples/custom-dataloader/config_custom_dl.yaml
+++ b/examples/custom-dataloader/config_custom_dl.yaml
--- a/examples/custom-dataloader/run_train.py
+++ b/examples/custom-dataloader/run_train.py