run_name: olmo-400M-keloss_0.0015_base.23840 seed: 6198 epoch: null dry_run: false model: d_model: 1024 n_heads: 16 n_kv_heads: null clip_qkv: null n_layers: 20 mlp_ratio: 8 mlp_hidden_size: null activation_type: swiglu block_type: sequential block_group_size: 1 alibi: false alibi_bias_max: 8.0 rope: true rope_full_precision: true rope_theta: 10000 flash_attention: true attention_dropout: 0.0 multi_query_attention: null attention_layer_norm: false residual_dropout: 0.0 embedding_dropout: 0.0 embedding_layer_norm: false layer_norm_type: rms layer_norm_with_affine: true layer_norm_eps: 1.0e-06 attention_layer_norm_with_affine: false max_sequence_length: 2048 include_bias: false bias_for_layer_norm: false scale_logits: false vocab_size: 50280 embedding_size: 50304 weight_tying: false eos_token_id: 0 pad_token_id: 1 init_device: cuda init_fn: normal init_std: 0.02 init_cutoff_factor: 3.0 precision: amp_bf16 scale_emb_init: false emb_init_std: null norm_after: false cbp_train: false cbp_replacement_rate: 0.0001 cbp_maturity_threshold: 100 cbp_init_fn: kaiming optimizer: name: adamw learning_rate: 0.0004 weight_decay: 0.1 betas: - 0.9 - 0.95 eps: 1.0e-08 no_decay_norm_and_bias: null selective_updates: false decay_norm_and_bias: true decay_embeddings: true metrics_log_interval: 10 record_update_metrics: false scheduler: name: cosine_with_warmup units: steps t_warmup: 2384 t_max: null alpha_f: 0.1 grad_clip_warmup_steps: null grad_clip_warmup_factor: null warmup_min_lr: 0.0 data: paths: - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/000_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/001_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/002_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/003_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/004_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/005_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/006_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/007_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/008_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/009_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/010_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/011_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/012_00009-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00000-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00001-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00002-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00003-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00004-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00005-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00006-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00007-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00008-part-0-00000.npy - /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/dolma/013_00009-part-0-00000.npy memmap_dtype: uint16 datasets: null dataset_path: null label_mask_paths: null pad_direction: right generate_attention_mask: false generate_doc_lengths: false num_workers: 32 drop_last: true pin_memory: true prefetch_factor: 8 persistent_workers: true timeout: 0 seed: null instance_filter: repetition_max_period: 13 repetition_min_period: 1 repetition_max_count: 32 restore_dataloader: true fast_forward_batches: null evaluators: - label: piqa type: downstream data: paths: null memmap_dtype: uint16 datasets: null dataset_path: null label_mask_paths: null pad_direction: right generate_attention_mask: false generate_doc_lengths: false num_workers: 0 drop_last: false pin_memory: false prefetch_factor: null persistent_workers: false timeout: 0 seed: null instance_filter: null device_eval_batch_size: null subset_num_batches: null - label: hellaswag type: downstream data: paths: null memmap_dtype: uint16 datasets: null dataset_path: null label_mask_paths: null pad_direction: right generate_attention_mask: false generate_doc_lengths: false num_workers: 0 drop_last: false pin_memory: false prefetch_factor: null persistent_workers: false timeout: 0 seed: null instance_filter: null device_eval_batch_size: null subset_num_batches: null - label: sciq type: downstream data: paths: null memmap_dtype: uint16 datasets: null dataset_path: null label_mask_paths: null pad_direction: right generate_attention_mask: false generate_doc_lengths: false num_workers: 0 drop_last: false pin_memory: false prefetch_factor: null persistent_workers: false timeout: 0 seed: null instance_filter: null device_eval_batch_size: null subset_num_batches: null - label: arc_easy type: downstream data: paths: null memmap_dtype: uint16 datasets: null dataset_path: null label_mask_paths: null pad_direction: right generate_attention_mask: false generate_doc_lengths: false num_workers: 0 drop_last: false pin_memory: false prefetch_factor: null persistent_workers: false timeout: 0 seed: null instance_filter: null device_eval_batch_size: null subset_num_batches: null - label: mrpc type: downstream data: paths: null memmap_dtype: uint16 datasets: null dataset_path: null label_mask_paths: null pad_direction: right generate_attention_mask: false generate_doc_lengths: false num_workers: 0 drop_last: false pin_memory: false prefetch_factor: null persistent_workers: false timeout: 0 seed: null instance_filter: null device_eval_batch_size: null subset_num_batches: null - label: sst2 type: downstream data: paths: null memmap_dtype: uint16 datasets: null dataset_path: null label_mask_paths: null pad_direction: right generate_attention_mask: false generate_doc_lengths: false num_workers: 0 drop_last: false pin_memory: false prefetch_factor: null persistent_workers: false timeout: 0 seed: null instance_filter: null device_eval_batch_size: null subset_num_batches: null eval_interval: 2384 tokenizer: identifier: tokenizers/allenai_gpt-neox-olmo-dolma-v1_5.json truncate_direction: right save_folder: /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/workspace/olmo-400M-keloss_0.0015_base.23840 remote_save_folder: null canceled_check_interval: 50 save_interval: 10 save_interval_unsharded: 2384 save_interval_ephemeral: null save_num_checkpoints_to_keep: 2 save_num_unsharded_checkpoints_to_keep: -1 save_overwrite: true force_save_unsharded: false no_pre_train_checkpoint: false load_path: /apdcephfs_sh2/share_300000800/user/kaixinma/amylee/fineweb-edu/workspace/olmo-400M-base/step23840-unsharded load_path_sharded_checkpointer: null try_load_latest_save: false reset_optimizer_state: false reset_trainer_state: false sharded_checkpointer: torch_legacy new_style_checkpoints: null max_duration: 1ep global_train_batch_size: 1024 device_train_batch_size: 128 device_train_microbatch_size: 4 device_eval_batch_size: 4 eval_subset_num_batches: -1 eval_on_load: false device_train_grad_accum: 32 max_grad_norm: 1.0 max_grad_norm_ratio: null precision: amp_bf16 wandb: project: olmo-pretrain-ablation entity: alee6868 group: null name: olmo-400M-keloss_0.0015_base.23840 tags: - watching log_artifacts: false rank_zero_only: true log_interval: 1 speed_monitor: window_size: 20 gpu_flops_available: null console_log_interval: 1 gen1_gc_interval: 1 compile: null distributed_strategy: ddp fsdp: use_orig_params: true sharding_strategy: FULL_SHARD wrapping_strategy: null precision: pure hybrid_sharding_num_model_replicas: null ddp: grad_sync_mode: batch find_unused_params: false softmax_auxiliary_loss: false auxiliary_loss_multiplier: 0.0001 time_limit: null extra_steps_after_cancel: 10 early_stopping_factor: null save_data_indices: true python_profiling: false torch_profiling: false stop_at: 47680 stop_after: null activation_checkpointing: null fused_loss: null inject_indices_map: null inject_interval: null resus_portion: 1.0 resus_ratio: 1.0 data_shuffling: true KE_loss: true sum_CE_KE_loss: true lambda_ke_loss: 0.0015 grad_ascent: false trainable_parameter: '' name_value: 0 hf_datasets_cache_dir: null module_outputs_save_steps: null