End of training

Browse files

Files changed (9) hide show

README.md +61 -0
config.json +55 -0
generation_config.json +6 -0
model-00001-of-00004.safetensors +3 -0
model-00002-of-00004.safetensors +3 -0
model-00003-of-00004.safetensors +3 -0
model-00004-of-00004.safetensors +3 -0
model.safetensors.index.json +0 -0
training_args.bin +3 -0

README.md ADDED Viewed

	@@ -0,0 +1,61 @@

+---
+tags:
+- generated_from_trainer
+model-index:
+- name: ROCO_pmc_llava-v1.6-mistral_qfomer
+  results: []
+---
+<!-- This model card has been generated automatically according to the information the Trainer had access to. You
+should probably proofread and complete it, then remove this comment. -->
+# ROCO_pmc_llava-v1.6-mistral_qfomer
+This model was trained from scratch on the None dataset.
+It achieves the following results on the evaluation set:
+- Loss: 2.0025
+## Model description
+More information needed
+## Intended uses & limitations
+More information needed
+## Training and evaluation data
+More information needed
+## Training procedure
+### Training hyperparameters
+The following hyperparameters were used during training:
+- learning_rate: 2e-05
+- train_batch_size: 2
+- eval_batch_size: 2
+- seed: 42
+- gradient_accumulation_steps: 16
+- total_train_batch_size: 32
+- optimizer: Adam with betas=(0.9,0.999) and epsilon=1e-08
+- lr_scheduler_type: cosine
+- lr_scheduler_warmup_ratio: 0.03
+- num_epochs: 1
+### Training results
+| Training Loss | Epoch | Step | Validation Loss |
+|:-------------:|:-----:|:----:|:---------------:|
+| 1.875         | 0.24  | 500  | 2.0133          |
+| 2.125         | 0.49  | 1000 | 2.0040          |
+| 2.0469        | 0.73  | 1500 | 2.0028          |
+| 1.9609        | 0.98  | 2000 | 2.0025          |
+### Framework versions
+- Transformers 4.37.2
+- Pytorch 2.1.2+cu121
+- Datasets 2.19.1
+- Tokenizers 0.15.1

config.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "_name_or_path": "/workspace/pmc_llava-v1.6-mistral_qfomer",
+  "architectures": [
+    "LlavaMistralForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "freeze_mm_mlp_adapter": false,
+  "freeze_mm_vision_resampler": false,
+  "hidden_act": "silu",
+  "hidden_size": 4096,
+  "image_aspect_ratio": "anyres",
+  "image_crop_resolution": 224,
+  "image_grid_pinpoints": [
+    [
+      672,
+      672
+    ]
+  ],
+  "image_split_resolution": 224,
+  "initializer_range": 0.02,
+  "intermediate_size": 14336,
+  "max_position_embeddings": 32768,
+  "mm_hidden_size": 1024,
+  "mm_patch_merge_type": "spatial_unpad",
+  "mm_projector_lr": null,
+  "mm_projector_type": "mlp2x_gelu",
+  "mm_resampler_type": null,
+  "mm_use_im_patch_token": false,
+  "mm_use_im_start_end": false,
+  "mm_vision_select_feature": "patch",
+  "mm_vision_select_layer": -2,
+  "mm_vision_tower": "/workspace/pmc_vit-l-14_hf",
+  "mm_vision_tower_lr": 2e-06,
+  "model_type": "llava_mistral",
+  "num_attention_heads": 32,
+  "num_hidden_layers": 32,
+  "num_key_value_heads": 8,
+  "rms_norm_eps": 1e-05,
+  "rope_theta": 1000000.0,
+  "sliding_window": null,
+  "tie_word_embeddings": false,
+  "tokenizer_model_max_length": 4096,
+  "tokenizer_padding_side": "left",
+  "torch_dtype": "bfloat16",
+  "transformers_version": "4.37.2",
+  "tune_mm_mlp_adapter": false,
+  "tune_mm_vision_resampler": false,
+  "unfreeze_mm_vision_tower": true,
+  "use_cache": false,
+  "use_mm_proj": true,
+  "use_q": true,
+  "vocab_size": 32000
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "transformers_version": "4.37.2"
+}

model-00001-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:57f11463314a7b628842ba55008c323fbc8d2c6d48a90f02343d550d61321d8e
+size 4943170624

model-00002-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1e88a821d441aef6685311cb319eebbac47fa99d523c71519a3cfa59478da451
+size 4999819336

model-00003-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:5c50d50c924751d1a3038364ef2f9069a968d6957a5b7683e4047bfa45e7e9f4
+size 4991925944

model-00004-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c03d8dbeee2e0f5dbda7c08656d153d26cc8c944882a1c836be23538cd17e35f
+size 572850160

model.safetensors.index.json ADDED Viewed

The diff for this file is too large to render. See raw diff

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:15de73a91ea62f0e7b04a80dd7597ff08599a803d23fb3225ee07a1309f51713
+size 4792