swiss-ai · Apr 4, 2024
diff --git a/‎.gitignore
+1 b/‎.gitignore
+1
diff --git a/‎examples/config_tiny_llama.yaml
+4-3 b/‎examples/config_tiny_llama.yaml
+4-3
diff --git a/‎examples/µTransfer/configs/llama_width_1024_config.yaml
+106 b/‎examples/µTransfer/configs/llama_width_1024_config.yaml
+106
diff --git a/‎examples/µTransfer/configs/llama_width_128_config.yaml
+106 b/‎examples/µTransfer/configs/llama_width_128_config.yaml
+106
diff --git a/‎examples/µTransfer/configs/llama_width_2048_config.yaml
+106 b/‎examples/µTransfer/configs/llama_width_2048_config.yaml
+106
diff --git a/‎examples/µTransfer/configs/llama_width_256_config.yaml
+106 b/‎examples/µTransfer/configs/llama_width_256_config.yaml
+106
diff --git a/‎examples/µTransfer/configs/llama_width_4096_config.yaml
+106 b/‎examples/µTransfer/configs/llama_width_4096_config.yaml
+106
diff --git a/‎examples/µTransfer/configs/llama_width_512_config.yaml
+106 b/‎examples/µTransfer/configs/llama_width_512_config.yaml
+106
diff --git a/‎examples/µTransfer/configs/llama_width_8192_config.yaml
+106 b/‎examples/µTransfer/configs/llama_width_8192_config.yaml
+106
diff --git a/‎src/nanotron/models/llama.py
+20-1 b/‎src/nanotron/models/llama.py
+20-1
diff --git a/‎src/nanotron/scaling.py
+4 b/‎src/nanotron/scaling.py
+4
@@ -162,3 +162,4 @@ cython_debug/
 .vscode
 
 checkpoints/
+wandb/
@@ -76,7 +76,8 @@ optimizer:
   adam_eps: 1.0e-08
   clip_grad: 1.0
   learning_rate_scheduler:
-    learning_rate: 0.0003
+    # learning_rate: 0.0003
+    learning_rate: 0.01
     lr_decay_starting_step: null
     lr_decay_steps: 8
     lr_decay_style: cosine
@@ -102,7 +103,7 @@ tokens:
   batch_accumulation_per_replica: 1
   limit_test_batches: 0
   limit_val_batches: 0
-  micro_batch_size: 2
-  sequence_length: 32
+  micro_batch_size: 16
+  sequence_length: 1024
   train_steps: 20
   val_check_interval: -1
@@ -0,0 +1,106 @@
+checkpoints:
+  checkpoint_interval: 1000
+  checkpoints_path: checkpoints
+  checkpoints_path_is_shared_file_system: false
+  resume_checkpoint_path: null
+  save_initial_state: false
+
+data_stages:
+  - name: Stable Training Stage
+    start_training_step: 1
+    data:
+      dataset:
+        dataset_overwrite_cache: false
+        dataset_processing_num_proc_per_process: 1
+        hf_dataset_config_name: null
+        hf_dataset_or_datasets: Fiery101/radar_textbooks
+        hf_dataset_splits: train
+        text_column_name: text
+      num_loading_workers: 1
+      seed: 42
+
+general:
+  benchmark_csv_path: null
+  consumed_train_samples: null
+  ignore_sanity_checks: false
+  project: µTransfer_for_nanotron
+  run: llama_width_1024_config
+  seed: 42
+  step: null
+lighteval: null
+logging:
+  iteration_step_info_interval: 1
+  log_level: info
+  log_level_replica: info
+model:
+  ddp_bucket_cap_mb: 120
+  dtype: bfloat16
+  init_method:
+    # std: 0.025 # original
+    # std: 0.03125 # 1/sqrt(1024)=0.03125
+    std: 0.022097086912079608 # 1/sqrt(2048)=0.022097086912079608
+  make_vocab_size_divisible_by: 1
+  model_config:
+    bos_token_id: 1
+    eos_token_id: 2
+    hidden_act: silu
+    initializer_range: 0.02
+
+    # NOTE: 250m
+    # hidden_size: 1024
+    # intermediate_size: 4096
+    # num_hidden_layers: 10
+
+    hidden_size: 1024
+    intermediate_size: 4096
+    num_hidden_layers: 4
+
+    is_llama_config: true
+    max_position_embeddings: 1024
+    num_attention_heads: 32
+    num_key_value_heads: 4
+    pad_token_id: null
+    pretraining_tp: 1
+    rms_norm_eps: 1.0e-05
+    rope_scaling: null
+    # tie_word_embeddings: true
+    tie_word_embeddings: false # original use true
+    use_cache: true
+    vocab_size: 49152
+optimizer:
+  accumulate_grad_in_fp32: false
+  adam_beta1: 0.9
+  adam_beta2: 0.95
+  adam_eps: 1.0e-08
+  clip_grad: 1.0
+  learning_rate_scheduler:
+    learning_rate: 0.001 # note: 1/2 of pythia use this for a 400m model
+    lr_decay_starting_step: null
+    lr_decay_steps: null
+    lr_decay_style: cosine
+    lr_warmup_steps: 6 # 10% warm up of total training steps
+    lr_warmup_style: linear
+    min_decay_lr: 1.0e-05
+  torch_adam_is_fused: true
+  weight_decay: 0.1
+  zero_stage: 1
+parallelism:
+  dp: 2
+  pp: 1
+  pp_engine: 1f1b
+  tp: 4
+  tp_linear_async_communication: true
+  tp_mode: REDUCE_SCATTER
+profiler: null
+tokenizer:
+  tokenizer_max_length: null
+  tokenizer_name_or_path: lvwerra/the-tokenizer-v1
+  tokenizer_revision: null
+tokens:
+  batch_accumulation_per_replica: 1
+  limit_test_batches: 0
+  limit_val_batches: 0
+  micro_batch_size: 64
+  sequence_length: 512
+  train_steps: 30
+  val_check_interval: -1
@@ -0,0 +1,106 @@
+checkpoints:
+  checkpoint_interval: 1000
+  checkpoints_path: checkpoints
+  checkpoints_path_is_shared_file_system: false
+  resume_checkpoint_path: null
+  save_initial_state: false
+
+data_stages:
+  - name: Stable Training Stage
+    start_training_step: 1
+    data:
+      dataset:
+        dataset_overwrite_cache: false
+        dataset_processing_num_proc_per_process: 1
+        hf_dataset_config_name: null
+        hf_dataset_or_datasets: Fiery101/radar_textbooks
+        hf_dataset_splits: train
+        text_column_name: text
+      num_loading_workers: 1
+      seed: 42
+
+general:
+  benchmark_csv_path: null
+  consumed_train_samples: null
+  ignore_sanity_checks: false
+  project: µTransfer_for_nanotron
+  run: llama_width_128_config
+  seed: 42
+  step: null
+lighteval: null
+logging:
+  iteration_step_info_interval: 1
+  log_level: info
+  log_level_replica: info
+model:
+  ddp_bucket_cap_mb: 120
+  dtype: bfloat16
+  init_method:
+    # std: 0.025 # original
+    # std: 0.03125 # 1/sqrt(1024)=0.03125
+    std: 0.022097086912079608 # 1/sqrt(2048)=0.022097086912079608
+  make_vocab_size_divisible_by: 1
+  model_config:
+    bos_token_id: 1
+    eos_token_id: 2
+    hidden_act: silu
+    initializer_range: 0.02
+
+    # NOTE: 250m
+    # hidden_size: 1024
+    # intermediate_size: 4096
+    # num_hidden_layers: 10
+
+    hidden_size: 128
+    intermediate_size: 512
+    num_hidden_layers: 4
+
+    is_llama_config: true
+    max_position_embeddings: 1024
+    num_attention_heads: 32
+    num_key_value_heads: 4
+    pad_token_id: null
+    pretraining_tp: 1
+    rms_norm_eps: 1.0e-05
+    rope_scaling: null
+    # tie_word_embeddings: true
+    tie_word_embeddings: false # original use true
+    use_cache: true
+    vocab_size: 49152
+optimizer:
+  accumulate_grad_in_fp32: false
+  adam_beta1: 0.9
+  adam_beta2: 0.95
+  adam_eps: 1.0e-08
+  clip_grad: 1.0
+  learning_rate_scheduler:
+    learning_rate: 0.001 # note: 1/2 of pythia use this for a 400m model
+    lr_decay_starting_step: null
+    lr_decay_steps: null
+    lr_decay_style: cosine
+    lr_warmup_steps: 6 # 10% warm up of total training steps
+    lr_warmup_style: linear
+    min_decay_lr: 1.0e-05
+  torch_adam_is_fused: true
+  weight_decay: 0.1
+  zero_stage: 1
+parallelism:
+  dp: 2
+  pp: 1
+  pp_engine: 1f1b
+  tp: 4
+  tp_linear_async_communication: true
+  tp_mode: REDUCE_SCATTER
+profiler: null
+tokenizer:
+  tokenizer_max_length: null
+  tokenizer_name_or_path: lvwerra/the-tokenizer-v1
+  tokenizer_revision: null
+tokens:
+  batch_accumulation_per_replica: 1
+  limit_test_batches: 0
+  limit_val_batches: 0
+  micro_batch_size: 64
+  sequence_length: 512
+  train_steps: 30
+  val_check_interval: -1
@@ -0,0 +1,106 @@
+checkpoints:
+  checkpoint_interval: 1000
+  checkpoints_path: checkpoints
+  checkpoints_path_is_shared_file_system: false
+  resume_checkpoint_path: null
+  save_initial_state: false
+
+data_stages:
+  - name: Stable Training Stage
+    start_training_step: 1
+    data:
+      dataset:
+        dataset_overwrite_cache: false
+        dataset_processing_num_proc_per_process: 1
+        hf_dataset_config_name: null
+        hf_dataset_or_datasets: Fiery101/radar_textbooks
+        hf_dataset_splits: train
+        text_column_name: text
+      num_loading_workers: 1
+      seed: 42
+
+general:
+  benchmark_csv_path: null
+  consumed_train_samples: null
+  ignore_sanity_checks: false
+  project: µTransfer_for_nanotron
+  run: llama_width_2048_config
+  seed: 42
+  step: null
+lighteval: null
+logging:
+  iteration_step_info_interval: 1
+  log_level: info
+  log_level_replica: info
+model:
+  ddp_bucket_cap_mb: 120
+  dtype: bfloat16
+  init_method:
+    # std: 0.025 # original
+    # std: 0.03125 # 1/sqrt(1024)=0.03125
+    std: 0.022097086912079608 # 1/sqrt(2048)=0.022097086912079608
+  make_vocab_size_divisible_by: 1
+  model_config:
+    bos_token_id: 1
+    eos_token_id: 2
+    hidden_act: silu
+    initializer_range: 0.02
+
+    # NOTE: 250m
+    # hidden_size: 1024
+    # intermediate_size: 4096
+    # num_hidden_layers: 10
+
+    hidden_size: 2048
+    intermediate_size: 8192
+    num_hidden_layers: 4
+
+    is_llama_config: true
+    max_position_embeddings: 1024
+    num_attention_heads: 32
+    num_key_value_heads: 4
+    pad_token_id: null
+    pretraining_tp: 1
+    rms_norm_eps: 1.0e-05
+    rope_scaling: null
+    # tie_word_embeddings: true
+    tie_word_embeddings: false # original use true
+    use_cache: true
+    vocab_size: 49152
+optimizer:
+  accumulate_grad_in_fp32: false
+  adam_beta1: 0.9
+  adam_beta2: 0.95
+  adam_eps: 1.0e-08
+  clip_grad: 1.0
+  learning_rate_scheduler:
+    learning_rate: 0.001 # note: 1/2 of pythia use this for a 400m model
+    lr_decay_starting_step: null
+    lr_decay_steps: null
+    lr_decay_style: cosine
+    lr_warmup_steps: 6 # 10% warm up of total training steps
+    lr_warmup_style: linear
+    min_decay_lr: 1.0e-05
+  torch_adam_is_fused: true
+  weight_decay: 0.1
+  zero_stage: 1
+parallelism:
+  dp: 2
+  pp: 1
+  pp_engine: 1f1b
+  tp: 4
+  tp_linear_async_communication: true
+  tp_mode: REDUCE_SCATTER
+profiler: null
+tokenizer:
+  tokenizer_max_length: null
+  tokenizer_name_or_path: lvwerra/the-tokenizer-v1
+  tokenizer_revision: null
+tokens:
+  batch_accumulation_per_replica: 1
+  limit_test_batches: 0
+  limit_val_batches: 0
+  micro_batch_size: 64
+  sequence_length: 512
+  train_steps: 30
+  val_check_interval: -1
@@ -0,0 +1,106 @@
+checkpoints:
+  checkpoint_interval: 1000
+  checkpoints_path: checkpoints
+  checkpoints_path_is_shared_file_system: false
+  resume_checkpoint_path: null
+  save_initial_state: false
+
+data_stages:
+  - name: Stable Training Stage
+    start_training_step: 1
+    data:
+      dataset:
+        dataset_overwrite_cache: false
+        dataset_processing_num_proc_per_process: 1
+        hf_dataset_config_name: null
+        hf_dataset_or_datasets: Fiery101/radar_textbooks
+        hf_dataset_splits: train
+        text_column_name: text
+      num_loading_workers: 1
+      seed: 42
+
+general:
+  benchmark_csv_path: null
+  consumed_train_samples: null
+  ignore_sanity_checks: false
+  project: µTransfer_for_nanotron
+  run: llama_width_256_config
+  seed: 42
+  step: null
+lighteval: null
+logging:
+  iteration_step_info_interval: 1
+  log_level: info
+  log_level_replica: info
+model:
+  ddp_bucket_cap_mb: 120
+  dtype: bfloat16
+  init_method:
+    # std: 0.025 # original
+    # std: 0.03125 # 1/sqrt(1024)=0.03125
+    std: 0.022097086912079608 # 1/sqrt(2048)=0.022097086912079608
+  make_vocab_size_divisible_by: 1
+  model_config:
+    bos_token_id: 1
+    eos_token_id: 2
+    hidden_act: silu
+    initializer_range: 0.02
+
+    # NOTE: 250m
+    # hidden_size: 1024
+    # intermediate_size: 4096
+    # num_hidden_layers: 10
+
+    hidden_size: 256
+    intermediate_size: 1024
+    num_hidden_layers: 4
+
+    is_llama_config: true
+    max_position_embeddings: 1024
+    num_attention_heads: 32
+    num_key_value_heads: 4
+    pad_token_id: null
+    pretraining_tp: 1
+    rms_norm_eps: 1.0e-05
+    rope_scaling: null
+    # tie_word_embeddings: true
+    tie_word_embeddings: false # original use true
+    use_cache: true
+    vocab_size: 49152
+optimizer:
+  accumulate_grad_in_fp32: false
+  adam_beta1: 0.9
+  adam_beta2: 0.95
+  adam_eps: 1.0e-08
+  clip_grad: 1.0
+  learning_rate_scheduler:
+    learning_rate: 0.001 # note: 1/2 of pythia use this for a 400m model
+    lr_decay_starting_step: null
+    lr_decay_steps: null
+    lr_decay_style: cosine
+    lr_warmup_steps: 6 # 10% warm up of total training steps
+    lr_warmup_style: linear
+    min_decay_lr: 1.0e-05
+  torch_adam_is_fused: true
+  weight_decay: 0.1
+  zero_stage: 1
+parallelism:
+  dp: 2
+  pp: 1
+  pp_engine: 1f1b
+  tp: 4
+  tp_linear_async_communication: true
+  tp_mode: REDUCE_SCATTER
+profiler: null
+tokenizer:
+  tokenizer_max_length: null
+  tokenizer_name_or_path: lvwerra/the-tokenizer-v1
+  tokenizer_revision: null
+tokens:
+  batch_accumulation_per_replica: 1
+  limit_test_batches: 0
+  limit_val_batches: 0
+  micro_batch_size: 64
+  sequence_length: 512
+  train_steps: 30
+  val_check_interval: -1
@@ -0,0 +1,106 @@
+checkpoints:
+  checkpoint_interval: 1000
+  checkpoints_path: checkpoints
+  checkpoints_path_is_shared_file_system: false
+  resume_checkpoint_path: null
+  save_initial_state: false
+
+data_stages:
+  - name: Stable Training Stage
+    start_training_step: 1
+    data:
+      dataset:
+        dataset_overwrite_cache: false
+        dataset_processing_num_proc_per_process: 1
+        hf_dataset_config_name: null
+        hf_dataset_or_datasets: Fiery101/radar_textbooks
+        hf_dataset_splits: train
+        text_column_name: text
+      num_loading_workers: 1
+      seed: 42
+
+general:
+  benchmark_csv_path: null
+  consumed_train_samples: null
+  ignore_sanity_checks: false
+  project: µTransfer_for_nanotron
+  run: llama_width_4096_config
+  seed: 42
+  step: null
+lighteval: null
+logging:
+  iteration_step_info_interval: 1
+  log_level: info
+  log_level_replica: info
+model:
+  ddp_bucket_cap_mb: 120
+  dtype: bfloat16
+  init_method:
+    # std: 0.025 # original
+    # std: 0.03125 # 1/sqrt(1024)=0.03125
+    std: 0.022097086912079608 # 1/sqrt(2048)=0.022097086912079608
+  make_vocab_size_divisible_by: 1
+  model_config:
+    bos_token_id: 1
+    eos_token_id: 2
+    hidden_act: silu
+    initializer_range: 0.02
+
+    # NOTE: 250m
+    # hidden_size: 1024
+    # intermediate_size: 4096
+    # num_hidden_layers: 10
+
+    hidden_size: 4096
+    intermediate_size: 16384
+    num_hidden_layers: 4
+
+    is_llama_config: true
+    max_position_embeddings: 1024
+    num_attention_heads: 32
+    num_key_value_heads: 4
+    pad_token_id: null
+    pretraining_tp: 1
+    rms_norm_eps: 1.0e-05
+    rope_scaling: null
+    # tie_word_embeddings: true
+    tie_word_embeddings: false # original use true
+    use_cache: true
+    vocab_size: 49152
+optimizer:
+  accumulate_grad_in_fp32: false
+  adam_beta1: 0.9
+  adam_beta2: 0.95
+  adam_eps: 1.0e-08
+  clip_grad: 1.0
+  learning_rate_scheduler:
+    learning_rate: 0.001 # note: 1/2 of pythia use this for a 400m model
+    lr_decay_starting_step: null
+    lr_decay_steps: null
+    lr_decay_style: cosine
+    lr_warmup_steps: 6 # 10% warm up of total training steps
+    lr_warmup_style: linear
+    min_decay_lr: 1.0e-05
+  torch_adam_is_fused: true
+  weight_decay: 0.1
+  zero_stage: 1
+parallelism:
+  dp: 2
+  pp: 1
+  pp_engine: 1f1b
+  tp: 4
+  tp_linear_async_communication: true
+  tp_mode: REDUCE_SCATTER
+profiler: null
+tokenizer:
+  tokenizer_max_length: null
+  tokenizer_name_or_path: lvwerra/the-tokenizer-v1
+  tokenizer_revision: null
+tokens:
+  batch_accumulation_per_replica: 1
+  limit_test_batches: 0
+  limit_val_batches: 0
+  micro_batch_size: 64
+  sequence_length: 512
+  train_steps: 30
+  val_check_interval: -1
@@ -0,0 +1,106 @@
+checkpoints:
+  checkpoint_interval: 1000
+  checkpoints_path: checkpoints
+  checkpoints_path_is_shared_file_system: false
+  resume_checkpoint_path: null
+  save_initial_state: false
+
+data_stages:
+  - name: Stable Training Stage
+    start_training_step: 1
+    data:
+      dataset:
+        dataset_overwrite_cache: false
+        dataset_processing_num_proc_per_process: 1
+        hf_dataset_config_name: null
+        hf_dataset_or_datasets: Fiery101/radar_textbooks
+        hf_dataset_splits: train
+        text_column_name: text
+      num_loading_workers: 1
+      seed: 42
+
+general:
+  benchmark_csv_path: null
+  consumed_train_samples: null
+  ignore_sanity_checks: false
+  project: µTransfer_for_nanotron
+  run: llama_width_512_config
+  seed: 42
+  step: null
+lighteval: null
+logging:
+  iteration_step_info_interval: 1
+  log_level: info
+  log_level_replica: info
+model:
+  ddp_bucket_cap_mb: 120
+  dtype: bfloat16
+  init_method:
+    # std: 0.025 # original
+    # std: 0.03125 # 1/sqrt(1024)=0.03125
+    std: 0.022097086912079608 # 1/sqrt(2048)=0.022097086912079608
+  make_vocab_size_divisible_by: 1
+  model_config:
+    bos_token_id: 1
+    eos_token_id: 2
+    hidden_act: silu
+    initializer_range: 0.02
+
+    # NOTE: 250m
+    # hidden_size: 1024
+    # intermediate_size: 4096
+    # num_hidden_layers: 10
+
+    hidden_size: 512
+    intermediate_size: 2048
+    num_hidden_layers: 4
+
+    is_llama_config: true
+    max_position_embeddings: 1024
+    num_attention_heads: 32
+    num_key_value_heads: 4
+    pad_token_id: null
+    pretraining_tp: 1
+    rms_norm_eps: 1.0e-05
+    rope_scaling: null
+    # tie_word_embeddings: true
+    tie_word_embeddings: false # original use true
+    use_cache: true
+    vocab_size: 49152
+optimizer:
+  accumulate_grad_in_fp32: false
+  adam_beta1: 0.9
+  adam_beta2: 0.95
+  adam_eps: 1.0e-08
+  clip_grad: 1.0
+  learning_rate_scheduler:
+    learning_rate: 0.001 # note: 1/2 of pythia use this for a 400m model
+    lr_decay_starting_step: null
+    lr_decay_steps: null
+    lr_decay_style: cosine
+    lr_warmup_steps: 6 # 10% warm up of total training steps
+    lr_warmup_style: linear
+    min_decay_lr: 1.0e-05
+  torch_adam_is_fused: true
+  weight_decay: 0.1
+  zero_stage: 1
+parallelism:
+  dp: 2
+  pp: 1
+  pp_engine: 1f1b
+  tp: 4
+  tp_linear_async_communication: true
+  tp_mode: REDUCE_SCATTER
+profiler: null
+tokenizer:
+  tokenizer_max_length: null
+  tokenizer_name_or_path: lvwerra/the-tokenizer-v1
+  tokenizer_revision: null
+tokens:
+  batch_accumulation_per_replica: 1
+  limit_test_batches: 0
+  limit_val_batches: 0
+  micro_batch_size: 64
+  sequence_length: 512
+  train_steps: 30
+  val_check_interval: -1
@@ -0,0 +1,106 @@
+checkpoints:
+  checkpoint_interval: 1000
+  checkpoints_path: checkpoints
+  checkpoints_path_is_shared_file_system: false
+  resume_checkpoint_path: null
+  save_initial_state: false
+
+data_stages:
+  - name: Stable Training Stage
+    start_training_step: 1
+    data:
+      dataset:
+        dataset_overwrite_cache: false
+        dataset_processing_num_proc_per_process: 1
+        hf_dataset_config_name: null
+        hf_dataset_or_datasets: Fiery101/radar_textbooks
+        hf_dataset_splits: train
+        text_column_name: text
+      num_loading_workers: 1
+      seed: 42
+
+general:
+  benchmark_csv_path: null
+  consumed_train_samples: null
+  ignore_sanity_checks: false
+  project: µTransfer_for_nanotron
+  run: llama_width_8192_config
+  seed: 42
+  step: null
+lighteval: null
+logging:
+  iteration_step_info_interval: 1
+  log_level: info
+  log_level_replica: info
+model:
+  ddp_bucket_cap_mb: 120
+  dtype: bfloat16
+  init_method:
+    # std: 0.025 # original
+    # std: 0.03125 # 1/sqrt(1024)=0.03125
+    std: 0.022097086912079608 # 1/sqrt(2048)=0.022097086912079608
+  make_vocab_size_divisible_by: 1
+  model_config:
+    bos_token_id: 1
+    eos_token_id: 2
+    hidden_act: silu
+    initializer_range: 0.02
+
+    # NOTE: 250m
+    # hidden_size: 1024
+    # intermediate_size: 4096
+    # num_hidden_layers: 10
+
+    hidden_size: 8192
+    intermediate_size: 32768
+    num_hidden_layers: 4
+
+    is_llama_config: true
+    max_position_embeddings: 1024
+    num_attention_heads: 32
+    num_key_value_heads: 4
+    pad_token_id: null
+    pretraining_tp: 1
+    rms_norm_eps: 1.0e-05
+    rope_scaling: null
+    # tie_word_embeddings: true
+    tie_word_embeddings: false # original use true
+    use_cache: true
+    vocab_size: 49152
+optimizer:
+  accumulate_grad_in_fp32: false
+  adam_beta1: 0.9
+  adam_beta2: 0.95
+  adam_eps: 1.0e-08
+  clip_grad: 1.0
+  learning_rate_scheduler:
+    learning_rate: 0.001 # note: 1/2 of pythia use this for a 400m model
+    lr_decay_starting_step: null
+    lr_decay_steps: null
+    lr_decay_style: cosine
+    lr_warmup_steps: 6 # 10% warm up of total training steps
+    lr_warmup_style: linear
+    min_decay_lr: 1.0e-05
+  torch_adam_is_fused: true
+  weight_decay: 0.1
+  zero_stage: 1
+parallelism:
+  dp: 2
+  pp: 1
+  pp_engine: 1f1b
+  tp: 4
+  tp_linear_async_communication: true
+  tp_mode: REDUCE_SCATTER
+profiler: null
+tokenizer:
+  tokenizer_max_length: null
+  tokenizer_name_or_path: lvwerra/the-tokenizer-v1
+  tokenizer_revision: null
+tokens:
+  batch_accumulation_per_replica: 1
+  limit_test_batches: 0
+  limit_val_batches: 0
+  micro_batch_size: 64
+  sequence_length: 512
+  train_steps: 30
+  val_check_interval: -1
@@ -213,6 +213,8 @@ def forward(
         # TODO(kunhao): flash attn's causal means that the query can only attend to the keys before it. This is not
         # what we want if we are using kv cache. This is a hack as we always have q_length == 1 when using kv cache.
         causal = False if q_sequence_mask.shape[1] == 1 else True
+        # NOTE: this scale is for µTransfer
+        softmax_scale = 1 / query_states.shape[-1]
         attn_output = flash_attn_varlen_func(
             q=query_states,
             k=key_states,
@@ -222,7 +224,7 @@ def forward(
             max_seqlen_q=q_sequence_mask.shape[1],
             max_seqlen_k=kv_sequence_mask.shape[1],
             dropout_p=0.0,
-            softmax_scale=None,  # This already defaults to the scale I'm interested in
+            softmax_scale=softmax_scale,  # This already defaults to the scale I'm interested in
             causal=causal,
             return_attn_probs=False,
         )
@@ -774,6 +776,19 @@ def forward_with_hidden_states(
         for encoder_block in self.decoder:
             hidden_encoder_states = encoder_block(**hidden_encoder_states)
 
+        # hidden_states.shape = [seq_length/tp_rank, batch_size, hidden_dim]
+        mup_l1_norm = hidden_encoder_states["hidden_states"].mean(dim=[0, 1]).abs()  # [hidden_dim]
+        dist.all_reduce(
+            mup_l1_norm, op=dist.ReduceOp.SUM, group=self.parallel_context.tp_pg
+        )  # sum [hidden_dim] across tp ranks
+        mup_l1_norm = mup_l1_norm.mean()
+        dist.all_reduce(mup_l1_norm, op=dist.ReduceOp.AVG, group=self.parallel_context.dp_pg)
+
+        if dist.get_rank() == 0:
+            import wandb
+
+            wandb.log({"output_l1_norm": mup_l1_norm.cpu().detach().float().numpy(), "width": self.config.hidden_size})
+
         hidden_states = self.final_layer_norm(input=hidden_encoder_states["hidden_states"])["hidden_states"]
 
         sharded_logits = self.lm_head(x=hidden_states)["logits"]
@@ -839,6 +854,10 @@ def forward(
     ) -> Dict[str, torch.Tensor]:
         # Megatron by defaults cast everything in fp32. `--f16-lm-cross-entropy` is an option you can use to keep current precision.
         # https://github.com/NVIDIA/Megatron-LM/blob/f267e6186eae1d6e2055b412b00e2e545a8e896a/megatron/model/gpt_model.py#L38
+
+        # mup_logit_l1_norm = sharded_logits.mean(dim=[0,1]).abs()
+        # dist.all_reduce(mup_logit_l1_norm, op=dist.ReduceOp.AVERAGE, group=self.tp_pg)
+
         loss = sharded_cross_entropy(
             sharded_logits, label_ids.transpose(0, 1).contiguous(), group=self.tp_pg, dtype=torch.float
         ).transpose(0, 1)
 
@@ -172,3 +172,7 @@ def output_weight_hook(module, input, output):
             raise ValueError(f"Unknown linear type: {module.linear_type}")
 
         module.register_forward_hook(hook_func)
+
+
+def monitor_l1_norm_activations(model: nn.Module):
+    pass
Original file line number	Diff line number	Diff line change
`@@ -162,3 +162,4 @@ cython_debug/`
`162`	`162`	`.vscode`
`163`	`163`
`164`	`164`	`checkpoints/`
	`165`	`+wandb/`