mariagrandury commited on
Commit
52efe8b
·
verified ·
1 Parent(s): 1c7d2cb

End of training

Browse files
README.md CHANGED
@@ -17,9 +17,9 @@ should probably proofread and complete it, then remove this comment. -->
17
 
18
  This model is a fine-tuned version of [dccuchile/bert-base-spanish-wwm-uncased](https://huggingface.co/dccuchile/bert-base-spanish-wwm-uncased) on an unknown dataset.
19
  It achieves the following results on the evaluation set:
20
- - Loss: 2.3731
21
- - Classification Report: {'ar': {'precision': 0.4898785425101215, 'recall': 0.32180851063829785, 'f1-score': 0.3884430176565008, 'support': 376.0}, 'cl': {'precision': 0.3626666666666667, 'recall': 0.4722222222222222, 'f1-score': 0.41025641025641024, 'support': 576.0}, 'co': {'precision': 0.34656084656084657, 'recall': 0.3808139534883721, 'f1-score': 0.3628808864265928, 'support': 344.0}, 'es': {'precision': 0.4630738522954092, 'recall': 0.427255985267035, 'f1-score': 0.4444444444444444, 'support': 543.0}, 'mx': {'precision': 0.43380855397148677, 'recall': 0.43917525773195876, 'f1-score': 0.4364754098360656, 'support': 485.0}, 'pe': {'precision': 0.3769968051118211, 'recall': 0.3390804597701149, 'f1-score': 0.35703479576399394, 'support': 348.0}, 'pr': {'precision': 0.5736434108527132, 'recall': 0.7326732673267327, 'f1-score': 0.6434782608695652, 'support': 101.0}, 'uy': {'precision': 0.35096153846153844, 'recall': 0.3201754385964912, 'f1-score': 0.3348623853211009, 'support': 228.0}, 've': {'precision': 0.16666666666666666, 'recall': 0.045454545454545456, 'f1-score': 0.07142857142857142, 'support': 22.0}, 'accuracy': 0.4085345683096262, 'macro avg': {'precision': 0.39602854256636333, 'recall': 0.38651773783286336, 'f1-score': 0.3832560202225828, 'support': 3023.0}, 'weighted avg': {'precision': 0.4124949665181113, 'recall': 0.4085345683096262, 'f1-score': 0.40601279016852304, 'support': 3023.0}}
22
- - F1: 0.3833
23
 
24
  ## Model description
25
 
@@ -49,13 +49,13 @@ The following hyperparameters were used during training:
49
 
50
  ### Training results
51
 
52
- | Training Loss | Epoch | Step | Validation Loss | Classification Report | F1 |
53
- |:-------------:|:-----:|:----:|:---------------:|:--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:------:|
54
- | 1.7603 | 1.0 | 882 | 1.7406 | {'ar': {'precision': 0.35516372795969775, 'recall': 0.375, 'f1-score': 0.3648124191461837, 'support': 376.0}, 'cl': {'precision': 0.3016759776536313, 'recall': 0.28125, 'f1-score': 0.29110512129380056, 'support': 576.0}, 'co': {'precision': 0.3670886075949367, 'recall': 0.25290697674418605, 'f1-score': 0.29948364888123924, 'support': 344.0}, 'es': {'precision': 0.3584905660377358, 'recall': 0.4548802946593002, 'f1-score': 0.400974025974026, 'support': 543.0}, 'mx': {'precision': 0.32465753424657534, 'recall': 0.488659793814433, 'f1-score': 0.39012345679012345, 'support': 485.0}, 'pe': {'precision': 0.3958333333333333, 'recall': 0.27298850574712646, 'f1-score': 0.3231292517006803, 'support': 348.0}, 'pr': {'precision': 0.5631067961165048, 'recall': 0.5742574257425742, 'f1-score': 0.5686274509803921, 'support': 101.0}, 'uy': {'precision': 0.4666666666666667, 'recall': 0.18421052631578946, 'f1-score': 0.2641509433962264, 'support': 228.0}, 've': {'precision': 0.0, 'recall': 0.0, 'f1-score': 0.0, 'support': 22.0}, 'accuracy': 0.3536222295732716, 'macro avg': {'precision': 0.34807591217878686, 'recall': 0.3204615025581566, 'f1-score': 0.32248959090696355, 'support': 3023.0}, 'weighted avg': {'precision': 0.35948675942105285, 'recall': 0.3536222295732716, 'f1-score': 0.3456546260541325, 'support': 3023.0}} | 0.3225 |
55
- | 1.4223 | 2.0 | 1764 | 1.6758 | {'ar': {'precision': 0.4349315068493151, 'recall': 0.3377659574468085, 'f1-score': 0.38023952095808383, 'support': 376.0}, 'cl': {'precision': 0.336996336996337, 'recall': 0.3194444444444444, 'f1-score': 0.32798573975044565, 'support': 576.0}, 'co': {'precision': 0.36333333333333334, 'recall': 0.3168604651162791, 'f1-score': 0.3385093167701863, 'support': 344.0}, 'es': {'precision': 0.38980716253443526, 'recall': 0.5211786372007366, 'f1-score': 0.44602048857368004, 'support': 543.0}, 'mx': {'precision': 0.35246995994659547, 'recall': 0.5443298969072164, 'f1-score': 0.42787682333873583, 'support': 485.0}, 'pe': {'precision': 0.44308943089430897, 'recall': 0.3132183908045977, 'f1-score': 0.367003367003367, 'support': 348.0}, 'pr': {'precision': 0.759493670886076, 'recall': 0.594059405940594, 'f1-score': 0.6666666666666666, 'support': 101.0}, 'uy': {'precision': 0.5542168674698795, 'recall': 0.20175438596491227, 'f1-score': 0.2958199356913183, 'support': 228.0}, 've': {'precision': 1.0, 'recall': 0.09090909090909091, 'f1-score': 0.16666666666666666, 'support': 22.0}, 'accuracy': 0.3916639100231558, 'macro avg': {'precision': 0.5149264743233645, 'recall': 0.35994674163718665, 'f1-score': 0.3796431694910167, 'support': 3023.0}, 'weighted avg': {'precision': 0.4116802685001794, 'recall': 0.3916639100231558, 'f1-score': 0.3851176158170783, 'support': 3023.0}} | 0.3796 |
56
- | 0.9068 | 3.0 | 2646 | 1.9523 | {'ar': {'precision': 0.39574468085106385, 'recall': 0.4946808510638298, 'f1-score': 0.4397163120567376, 'support': 376.0}, 'cl': {'precision': 0.35144927536231885, 'recall': 0.3368055555555556, 'f1-score': 0.34397163120567376, 'support': 576.0}, 'co': {'precision': 0.31555555555555553, 'recall': 0.4127906976744186, 'f1-score': 0.35768261964735515, 'support': 344.0}, 'es': {'precision': 0.47113163972286376, 'recall': 0.3756906077348066, 'f1-score': 0.4180327868852459, 'support': 543.0}, 'mx': {'precision': 0.43680709534368073, 'recall': 0.4061855670103093, 'f1-score': 0.42094017094017094, 'support': 485.0}, 'pe': {'precision': 0.38661710037174724, 'recall': 0.2988505747126437, 'f1-score': 0.3371150729335494, 'support': 348.0}, 'pr': {'precision': 0.64, 'recall': 0.6336633663366337, 'f1-score': 0.6368159203980099, 'support': 101.0}, 'uy': {'precision': 0.30662020905923343, 'recall': 0.38596491228070173, 'f1-score': 0.341747572815534, 'support': 228.0}, 've': {'precision': 0.18181818181818182, 'recall': 0.09090909090909091, 'f1-score': 0.12121212121212122, 'support': 22.0}, 'accuracy': 0.3906715183592458, 'macro avg': {'precision': 0.3873048597871828, 'recall': 0.3817268025864433, 'f1-score': 0.37969268978826637, 'support': 3023.0}, 'weighted avg': {'precision': 0.3971399185993649, 'recall': 0.3906715183592458, 'f1-score': 0.3902981034934984, 'support': 3023.0}} | 0.3797 |
57
- | 0.4818 | 4.0 | 3528 | 2.3731 | {'ar': {'precision': 0.4898785425101215, 'recall': 0.32180851063829785, 'f1-score': 0.3884430176565008, 'support': 376.0}, 'cl': {'precision': 0.3626666666666667, 'recall': 0.4722222222222222, 'f1-score': 0.41025641025641024, 'support': 576.0}, 'co': {'precision': 0.34656084656084657, 'recall': 0.3808139534883721, 'f1-score': 0.3628808864265928, 'support': 344.0}, 'es': {'precision': 0.4630738522954092, 'recall': 0.427255985267035, 'f1-score': 0.4444444444444444, 'support': 543.0}, 'mx': {'precision': 0.43380855397148677, 'recall': 0.43917525773195876, 'f1-score': 0.4364754098360656, 'support': 485.0}, 'pe': {'precision': 0.3769968051118211, 'recall': 0.3390804597701149, 'f1-score': 0.35703479576399394, 'support': 348.0}, 'pr': {'precision': 0.5736434108527132, 'recall': 0.7326732673267327, 'f1-score': 0.6434782608695652, 'support': 101.0}, 'uy': {'precision': 0.35096153846153844, 'recall': 0.3201754385964912, 'f1-score': 0.3348623853211009, 'support': 228.0}, 've': {'precision': 0.16666666666666666, 'recall': 0.045454545454545456, 'f1-score': 0.07142857142857142, 'support': 22.0}, 'accuracy': 0.4085345683096262, 'macro avg': {'precision': 0.39602854256636333, 'recall': 0.38651773783286336, 'f1-score': 0.3832560202225828, 'support': 3023.0}, 'weighted avg': {'precision': 0.4124949665181113, 'recall': 0.4085345683096262, 'f1-score': 0.40601279016852304, 'support': 3023.0}} | 0.3833 |
58
- | 0.2357 | 5.0 | 4410 | 2.7721 | {'ar': {'precision': 0.42168674698795183, 'recall': 0.3723404255319149, 'f1-score': 0.3954802259887006, 'support': 376.0}, 'cl': {'precision': 0.38753799392097266, 'recall': 0.4427083333333333, 'f1-score': 0.413290113452188, 'support': 576.0}, 'co': {'precision': 0.35051546391752575, 'recall': 0.3953488372093023, 'f1-score': 0.37158469945355194, 'support': 344.0}, 'es': {'precision': 0.4642857142857143, 'recall': 0.40699815837937386, 'f1-score': 0.4337585868498528, 'support': 543.0}, 'mx': {'precision': 0.43089430894308944, 'recall': 0.43711340206185567, 'f1-score': 0.43398157625383826, 'support': 485.0}, 'pe': {'precision': 0.3407960199004975, 'recall': 0.3936781609195402, 'f1-score': 0.36533333333333334, 'support': 348.0}, 'pr': {'precision': 0.6601941747572816, 'recall': 0.6732673267326733, 'f1-score': 0.6666666666666666, 'support': 101.0}, 'uy': {'precision': 0.40853658536585363, 'recall': 0.29385964912280704, 'f1-score': 0.34183673469387754, 'support': 228.0}, 've': {'precision': 0.0, 'recall': 0.0, 'f1-score': 0.0, 'support': 22.0}, 'accuracy': 0.40886536553092956, 'macro avg': {'precision': 0.3849385564532096, 'recall': 0.3794793659212001, 'f1-score': 0.38021465963244544, 'support': 3023.0}, 'weighted avg': {'precision': 0.41080624270175103, 'recall': 0.40886536553092956, 'f1-score': 0.4078732692419294, 'support': 3023.0}} | 0.3802 |
59
 
60
 
61
  ### Framework versions
 
17
 
18
  This model is a fine-tuned version of [dccuchile/bert-base-spanish-wwm-uncased](https://huggingface.co/dccuchile/bert-base-spanish-wwm-uncased) on an unknown dataset.
19
  It achieves the following results on the evaluation set:
20
+ - Loss: 2.6157
21
+ - Classification Report: {'ar': {'precision': 0.46107784431137727, 'recall': 0.3938618925831202, 'f1-score': 0.42482758620689653, 'support': 391.0}, 'cl': {'precision': 0.4326241134751773, 'recall': 0.4236111111111111, 'f1-score': 0.4280701754385965, 'support': 576.0}, 'co': {'precision': 0.401840490797546, 'recall': 0.3819241982507289, 'f1-score': 0.39162929745889385, 'support': 343.0}, 'es': {'precision': 0.4861407249466951, 'recall': 0.4175824175824176, 'f1-score': 0.4492610837438424, 'support': 546.0}, 'mx': {'precision': 0.4580152671755725, 'recall': 0.5136986301369864, 'f1-score': 0.48426150121065376, 'support': 584.0}, 'pe': {'precision': 0.301255230125523, 'recall': 0.37305699481865284, 'f1-score': 0.3333333333333333, 'support': 386.0}, 'accuracy': 0.42498230714791224, 'macro avg': {'precision': 0.4234922784719819, 'recall': 0.4172892074138362, 'f1-score': 0.41856382956536947, 'support': 2826.0}, 'weighted avg': {'precision': 0.4304679708106477, 'recall': 0.42498230714791224, 'f1-score': 0.4259648943332467, 'support': 2826.0}}
22
+ - F1: 0.4186
23
 
24
  ## Model description
25
 
 
49
 
50
  ### Training results
51
 
52
+ | Training Loss | Epoch | Step | Validation Loss | Classification Report | F1 |
53
+ |:-------------:|:-----:|:----:|:---------------:|:-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------:|:------:|
54
+ | 1.4573 | 1.0 | 825 | 1.5809 | {'ar': {'precision': 0.5341880341880342, 'recall': 0.319693094629156, 'f1-score': 0.4, 'support': 391.0}, 'cl': {'precision': 0.44039735099337746, 'recall': 0.2309027777777778, 'f1-score': 0.30296127562642367, 'support': 576.0}, 'co': {'precision': 0.45930232558139533, 'recall': 0.2303206997084548, 'f1-score': 0.3067961165048544, 'support': 343.0}, 'es': {'precision': 0.3184438040345821, 'recall': 0.8095238095238095, 'f1-score': 0.45708376421923474, 'support': 546.0}, 'mx': {'precision': 0.42574257425742573, 'recall': 0.3681506849315068, 'f1-score': 0.3948576675849403, 'support': 584.0}, 'pe': {'precision': 0.38222222222222224, 'recall': 0.22279792746113988, 'f1-score': 0.281505728314239, 'support': 386.0}, 'accuracy': 0.3821656050955414, 'macro avg': {'precision': 0.4267160518795062, 'recall': 0.36356483233864084, 'f1-score': 0.357200758708282, 'support': 2826.0}, 'weighted avg': {'precision': 0.4211319360796609, 'recall': 0.3821656050955414, 'f1-score': 0.3626902289400526, 'support': 2826.0}} | 0.3572 |
55
+ | 1.6236 | 2.0 | 1650 | 1.5003 | {'ar': {'precision': 0.5714285714285714, 'recall': 0.3375959079283887, 'f1-score': 0.42443729903536975, 'support': 391.0}, 'cl': {'precision': 0.4013377926421405, 'recall': 0.4166666666666667, 'f1-score': 0.4088586030664395, 'support': 576.0}, 'co': {'precision': 0.5144508670520231, 'recall': 0.2594752186588921, 'f1-score': 0.3449612403100775, 'support': 343.0}, 'es': {'precision': 0.44165170556552963, 'recall': 0.45054945054945056, 'f1-score': 0.4460562103354488, 'support': 546.0}, 'mx': {'precision': 0.38311688311688313, 'recall': 0.6061643835616438, 'f1-score': 0.46949602122015915, 'support': 584.0}, 'pe': {'precision': 0.33819241982507287, 'recall': 0.3005181347150259, 'f1-score': 0.31824417009602196, 'support': 386.0}, 'accuracy': 0.4164897381457891, 'macro avg': {'precision': 0.4416963732717034, 'recall': 0.3951616270133447, 'f1-score': 0.4020089240105862, 'support': 2826.0}, 'weighted avg': {'precision': 0.4339986385070083, 'recall': 0.4164897381457891, 'f1-score': 0.41059938485783715, 'support': 2826.0}} | 0.4020 |
56
+ | 0.7224 | 3.0 | 2475 | 1.7251 | {'ar': {'precision': 0.5352112676056338, 'recall': 0.3887468030690537, 'f1-score': 0.45037037037037037, 'support': 391.0}, 'cl': {'precision': 0.4388609715242881, 'recall': 0.4548611111111111, 'f1-score': 0.44671781756180734, 'support': 576.0}, 'co': {'precision': 0.291005291005291, 'recall': 0.48104956268221577, 'f1-score': 0.3626373626373626, 'support': 343.0}, 'es': {'precision': 0.4835164835164835, 'recall': 0.40293040293040294, 'f1-score': 0.43956043956043955, 'support': 546.0}, 'mx': {'precision': 0.4577922077922078, 'recall': 0.4828767123287671, 'f1-score': 0.47, 'support': 584.0}, 'pe': {'precision': 0.3289902280130293, 'recall': 0.2616580310880829, 'f1-score': 0.29148629148629146, 'support': 386.0}, 'accuracy': 0.4182590233545648, 'macro avg': {'precision': 0.42256274157615564, 'recall': 0.4120204372016056, 'f1-score': 0.4101287136027119, 'support': 2826.0}, 'weighted avg': {'precision': 0.43177891628106374, 'recall': 0.4182590233545648, 'f1-score': 0.4192436665352936, 'support': 2826.0}} | 0.4101 |
57
+ | 0.3648 | 4.0 | 3300 | 2.1768 | {'ar': {'precision': 0.3978260869565217, 'recall': 0.4680306905370844, 'f1-score': 0.4300822561692127, 'support': 391.0}, 'cl': {'precision': 0.4788732394366197, 'recall': 0.3541666666666667, 'f1-score': 0.40718562874251496, 'support': 576.0}, 'co': {'precision': 0.35443037974683544, 'recall': 0.40816326530612246, 'f1-score': 0.3794037940379404, 'support': 343.0}, 'es': {'precision': 0.4577702702702703, 'recall': 0.49633699633699635, 'f1-score': 0.47627416520210897, 'support': 546.0}, 'mx': {'precision': 0.47755834829443444, 'recall': 0.4554794520547945, 'f1-score': 0.4662576687116564, 'support': 584.0}, 'pe': {'precision': 0.3106060606060606, 'recall': 0.31865284974093266, 'f1-score': 0.3145780051150895, 'support': 386.0}, 'accuracy': 0.4200283085633404, 'macro avg': {'precision': 0.412844064218457, 'recall': 0.4168049867737662, 'f1-score': 0.4122969196630872, 'support': 2826.0}, 'weighted avg': {'precision': 0.4252233505074715, 'recall': 0.4200283085633404, 'f1-score': 0.41988813459845997, 'support': 2826.0}} | 0.4123 |
58
+ | 0.2228 | 5.0 | 4125 | 2.6157 | {'ar': {'precision': 0.46107784431137727, 'recall': 0.3938618925831202, 'f1-score': 0.42482758620689653, 'support': 391.0}, 'cl': {'precision': 0.4326241134751773, 'recall': 0.4236111111111111, 'f1-score': 0.4280701754385965, 'support': 576.0}, 'co': {'precision': 0.401840490797546, 'recall': 0.3819241982507289, 'f1-score': 0.39162929745889385, 'support': 343.0}, 'es': {'precision': 0.4861407249466951, 'recall': 0.4175824175824176, 'f1-score': 0.4492610837438424, 'support': 546.0}, 'mx': {'precision': 0.4580152671755725, 'recall': 0.5136986301369864, 'f1-score': 0.48426150121065376, 'support': 584.0}, 'pe': {'precision': 0.301255230125523, 'recall': 0.37305699481865284, 'f1-score': 0.3333333333333333, 'support': 386.0}, 'accuracy': 0.42498230714791224, 'macro avg': {'precision': 0.4234922784719819, 'recall': 0.4172892074138362, 'f1-score': 0.41856382956536947, 'support': 2826.0}, 'weighted avg': {'precision': 0.4304679708106477, 'recall': 0.42498230714791224, 'f1-score': 0.4259648943332467, 'support': 2826.0}} | 0.4186 |
59
 
60
 
61
  ### Framework versions
config.json CHANGED
@@ -15,10 +15,7 @@
15
  "2": "co",
16
  "3": "es",
17
  "4": "mx",
18
- "5": "pe",
19
- "6": "pr",
20
- "7": "uy",
21
- "8": "ve"
22
  },
23
  "initializer_range": 0.02,
24
  "intermediate_size": 3072,
@@ -28,10 +25,7 @@
28
  "co": 2,
29
  "es": 3,
30
  "mx": 4,
31
- "pe": 5,
32
- "pr": 6,
33
- "uy": 7,
34
- "ve": 8
35
  },
36
  "layer_norm_eps": 1e-12,
37
  "max_position_embeddings": 512,
 
15
  "2": "co",
16
  "3": "es",
17
  "4": "mx",
18
+ "5": "pe"
 
 
 
19
  },
20
  "initializer_range": 0.02,
21
  "intermediate_size": 3072,
 
25
  "co": 2,
26
  "es": 3,
27
  "mx": 4,
28
+ "pe": 5
 
 
 
29
  },
30
  "layer_norm_eps": 1e-12,
31
  "max_position_embeddings": 512,
logs/events.out.tfevents.1740166055.1b9c09404399.787.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:25dbc01e818ce46458394bc1819a49597b1a3882c48135947f65a6a85a092972
3
+ size 94225
logs/events.out.tfevents.1740166741.1b9c09404399.787.1 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:28788e08643cd89f215c1bd479bba25fc433866054f168dda0b6e3cfb8bb1bbe
3
+ size 405
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9640bf7ca7b6a541089109fcdb8dd4b103d14e514f0b0b70e6032897467855b5
3
- size 439454740
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e9f6de2ac7614e2d6d7deb9bea648290b20714d0437dc42ffa59cf791f6805af
3
+ size 439445512
trial_4/checkpoint-3501/scheduler.pt CHANGED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:463d1586a049aea444771846d142dae97553b631027abe884e5f7945608c0fd8
3
- size 1064
 
 
 
 
trial_4/checkpoint-3501/trainer_state.json CHANGED
@@ -1,2717 +0,0 @@
1
- {
2
- "best_metric": 0.34262062899860507,
3
- "best_model_checkpoint": "/content/drive/MyDrive/model_outputs/trial_4/checkpoint-3501",
4
- "epoch": 3.0,
5
- "eval_steps": 500,
6
- "global_step": 3501,
7
- "is_hyper_param_search": false,
8
- "is_local_process_zero": true,
9
- "is_world_process_zero": true,
10
- "log_history": [
11
- {
12
- "epoch": 0.00856898029134533,
13
- "grad_norm": 12.96240520477295,
14
- "learning_rate": 6.233595156363815e-07,
15
- "loss": 2.1941,
16
- "step": 10
17
- },
18
- {
19
- "epoch": 0.01713796058269066,
20
- "grad_norm": 15.678441047668457,
21
- "learning_rate": 1.246719031272763e-06,
22
- "loss": 2.2004,
23
- "step": 20
24
- },
25
- {
26
- "epoch": 0.02570694087403599,
27
- "grad_norm": 19.057016372680664,
28
- "learning_rate": 1.8700785469091444e-06,
29
- "loss": 2.2096,
30
- "step": 30
31
- },
32
- {
33
- "epoch": 0.03427592116538132,
34
- "grad_norm": 16.71215057373047,
35
- "learning_rate": 2.493438062545526e-06,
36
- "loss": 2.2218,
37
- "step": 40
38
- },
39
- {
40
- "epoch": 0.04284490145672665,
41
- "grad_norm": 19.787519454956055,
42
- "learning_rate": 3.1167975781819074e-06,
43
- "loss": 2.2402,
44
- "step": 50
45
- },
46
- {
47
- "epoch": 0.05141388174807198,
48
- "grad_norm": 13.118367195129395,
49
- "learning_rate": 3.7401570938182888e-06,
50
- "loss": 2.1593,
51
- "step": 60
52
- },
53
- {
54
- "epoch": 0.05998286203941731,
55
- "grad_norm": 14.878708839416504,
56
- "learning_rate": 4.363516609454671e-06,
57
- "loss": 2.1811,
58
- "step": 70
59
- },
60
- {
61
- "epoch": 0.06855184233076264,
62
- "grad_norm": 26.236587524414062,
63
- "learning_rate": 4.986876125091052e-06,
64
- "loss": 2.0935,
65
- "step": 80
66
- },
67
- {
68
- "epoch": 0.07712082262210797,
69
- "grad_norm": 14.972529411315918,
70
- "learning_rate": 5.610235640727433e-06,
71
- "loss": 2.1434,
72
- "step": 90
73
- },
74
- {
75
- "epoch": 0.0856898029134533,
76
- "grad_norm": 14.05119514465332,
77
- "learning_rate": 6.233595156363815e-06,
78
- "loss": 2.0917,
79
- "step": 100
80
- },
81
- {
82
- "epoch": 0.09425878320479864,
83
- "grad_norm": 13.932599067687988,
84
- "learning_rate": 6.856954672000195e-06,
85
- "loss": 2.0876,
86
- "step": 110
87
- },
88
- {
89
- "epoch": 0.10282776349614396,
90
- "grad_norm": 11.802385330200195,
91
- "learning_rate": 7.4803141876365775e-06,
92
- "loss": 2.1551,
93
- "step": 120
94
- },
95
- {
96
- "epoch": 0.11139674378748929,
97
- "grad_norm": 17.57957649230957,
98
- "learning_rate": 8.103673703272958e-06,
99
- "loss": 2.0485,
100
- "step": 130
101
- },
102
- {
103
- "epoch": 0.11996572407883462,
104
- "grad_norm": 12.634247779846191,
105
- "learning_rate": 8.727033218909341e-06,
106
- "loss": 2.0586,
107
- "step": 140
108
- },
109
- {
110
- "epoch": 0.12853470437017994,
111
- "grad_norm": 14.163841247558594,
112
- "learning_rate": 9.350392734545721e-06,
113
- "loss": 2.0832,
114
- "step": 150
115
- },
116
- {
117
- "epoch": 0.13710368466152528,
118
- "grad_norm": 12.886134147644043,
119
- "learning_rate": 9.973752250182104e-06,
120
- "loss": 2.0785,
121
- "step": 160
122
- },
123
- {
124
- "epoch": 0.1456726649528706,
125
- "grad_norm": 12.337018966674805,
126
- "learning_rate": 1.0597111765818484e-05,
127
- "loss": 2.1782,
128
- "step": 170
129
- },
130
- {
131
- "epoch": 0.15424164524421594,
132
- "grad_norm": 15.206589698791504,
133
- "learning_rate": 1.1220471281454867e-05,
134
- "loss": 2.05,
135
- "step": 180
136
- },
137
- {
138
- "epoch": 0.16281062553556128,
139
- "grad_norm": 15.637868881225586,
140
- "learning_rate": 1.1843830797091246e-05,
141
- "loss": 2.0865,
142
- "step": 190
143
- },
144
- {
145
- "epoch": 0.1713796058269066,
146
- "grad_norm": 13.318354606628418,
147
- "learning_rate": 1.246719031272763e-05,
148
- "loss": 1.9896,
149
- "step": 200
150
- },
151
- {
152
- "epoch": 0.17994858611825193,
153
- "grad_norm": 15.895234107971191,
154
- "learning_rate": 1.3090549828364011e-05,
155
- "loss": 2.2041,
156
- "step": 210
157
- },
158
- {
159
- "epoch": 0.18851756640959727,
160
- "grad_norm": 11.681656837463379,
161
- "learning_rate": 1.3580958892325523e-05,
162
- "loss": 2.0332,
163
- "step": 220
164
- },
165
- {
166
- "epoch": 0.19708654670094258,
167
- "grad_norm": 12.1074800491333,
168
- "learning_rate": 1.353956614958756e-05,
169
- "loss": 1.9992,
170
- "step": 230
171
- },
172
- {
173
- "epoch": 0.20565552699228792,
174
- "grad_norm": 16.131893157958984,
175
- "learning_rate": 1.3498173406849598e-05,
176
- "loss": 2.0392,
177
- "step": 240
178
- },
179
- {
180
- "epoch": 0.21422450728363324,
181
- "grad_norm": 13.486928939819336,
182
- "learning_rate": 1.3456780664111635e-05,
183
- "loss": 2.0591,
184
- "step": 250
185
- },
186
- {
187
- "epoch": 0.22279348757497858,
188
- "grad_norm": 12.962348937988281,
189
- "learning_rate": 1.3415387921373673e-05,
190
- "loss": 2.1737,
191
- "step": 260
192
- },
193
- {
194
- "epoch": 0.23136246786632392,
195
- "grad_norm": 15.706825256347656,
196
- "learning_rate": 1.337399517863571e-05,
197
- "loss": 2.0788,
198
- "step": 270
199
- },
200
- {
201
- "epoch": 0.23993144815766923,
202
- "grad_norm": 15.435516357421875,
203
- "learning_rate": 1.3332602435897748e-05,
204
- "loss": 2.1001,
205
- "step": 280
206
- },
207
- {
208
- "epoch": 0.24850042844901457,
209
- "grad_norm": 10.667174339294434,
210
- "learning_rate": 1.3291209693159784e-05,
211
- "loss": 1.9447,
212
- "step": 290
213
- },
214
- {
215
- "epoch": 0.2570694087403599,
216
- "grad_norm": 14.559552192687988,
217
- "learning_rate": 1.3249816950421821e-05,
218
- "loss": 2.0502,
219
- "step": 300
220
- },
221
- {
222
- "epoch": 0.2656383890317052,
223
- "grad_norm": 15.85730266571045,
224
- "learning_rate": 1.320842420768386e-05,
225
- "loss": 2.0924,
226
- "step": 310
227
- },
228
- {
229
- "epoch": 0.27420736932305056,
230
- "grad_norm": 13.333989143371582,
231
- "learning_rate": 1.3167031464945898e-05,
232
- "loss": 2.1221,
233
- "step": 320
234
- },
235
- {
236
- "epoch": 0.2827763496143959,
237
- "grad_norm": 13.683819770812988,
238
- "learning_rate": 1.3125638722207935e-05,
239
- "loss": 2.1562,
240
- "step": 330
241
- },
242
- {
243
- "epoch": 0.2913453299057412,
244
- "grad_norm": 12.066128730773926,
245
- "learning_rate": 1.3084245979469973e-05,
246
- "loss": 2.0211,
247
- "step": 340
248
- },
249
- {
250
- "epoch": 0.29991431019708653,
251
- "grad_norm": 14.573673248291016,
252
- "learning_rate": 1.3042853236732009e-05,
253
- "loss": 2.0365,
254
- "step": 350
255
- },
256
- {
257
- "epoch": 0.30848329048843187,
258
- "grad_norm": 19.47612762451172,
259
- "learning_rate": 1.3001460493994046e-05,
260
- "loss": 1.9834,
261
- "step": 360
262
- },
263
- {
264
- "epoch": 0.3170522707797772,
265
- "grad_norm": 13.694817543029785,
266
- "learning_rate": 1.2960067751256084e-05,
267
- "loss": 2.0116,
268
- "step": 370
269
- },
270
- {
271
- "epoch": 0.32562125107112255,
272
- "grad_norm": 12.057513236999512,
273
- "learning_rate": 1.2918675008518121e-05,
274
- "loss": 2.0626,
275
- "step": 380
276
- },
277
- {
278
- "epoch": 0.3341902313624679,
279
- "grad_norm": 12.8948974609375,
280
- "learning_rate": 1.2877282265780159e-05,
281
- "loss": 1.9787,
282
- "step": 390
283
- },
284
- {
285
- "epoch": 0.3427592116538132,
286
- "grad_norm": 14.375242233276367,
287
- "learning_rate": 1.2835889523042196e-05,
288
- "loss": 2.0237,
289
- "step": 400
290
- },
291
- {
292
- "epoch": 0.3513281919451585,
293
- "grad_norm": 13.976824760437012,
294
- "learning_rate": 1.2794496780304234e-05,
295
- "loss": 2.1159,
296
- "step": 410
297
- },
298
- {
299
- "epoch": 0.35989717223650386,
300
- "grad_norm": 15.234496116638184,
301
- "learning_rate": 1.2753104037566271e-05,
302
- "loss": 1.9116,
303
- "step": 420
304
- },
305
- {
306
- "epoch": 0.3684661525278492,
307
- "grad_norm": 12.984347343444824,
308
- "learning_rate": 1.2711711294828309e-05,
309
- "loss": 2.0484,
310
- "step": 430
311
- },
312
- {
313
- "epoch": 0.37703513281919454,
314
- "grad_norm": 10.440543174743652,
315
- "learning_rate": 1.2670318552090346e-05,
316
- "loss": 2.0348,
317
- "step": 440
318
- },
319
- {
320
- "epoch": 0.3856041131105398,
321
- "grad_norm": 11.408565521240234,
322
- "learning_rate": 1.2628925809352384e-05,
323
- "loss": 2.0866,
324
- "step": 450
325
- },
326
- {
327
- "epoch": 0.39417309340188517,
328
- "grad_norm": 15.430132865905762,
329
- "learning_rate": 1.2587533066614421e-05,
330
- "loss": 2.0252,
331
- "step": 460
332
- },
333
- {
334
- "epoch": 0.4027420736932305,
335
- "grad_norm": 15.553166389465332,
336
- "learning_rate": 1.2546140323876457e-05,
337
- "loss": 2.0838,
338
- "step": 470
339
- },
340
- {
341
- "epoch": 0.41131105398457585,
342
- "grad_norm": 14.190583229064941,
343
- "learning_rate": 1.2504747581138496e-05,
344
- "loss": 2.0674,
345
- "step": 480
346
- },
347
- {
348
- "epoch": 0.4198800342759212,
349
- "grad_norm": 19.912630081176758,
350
- "learning_rate": 1.2463354838400534e-05,
351
- "loss": 2.0175,
352
- "step": 490
353
- },
354
- {
355
- "epoch": 0.4284490145672665,
356
- "grad_norm": 11.050692558288574,
357
- "learning_rate": 1.2421962095662571e-05,
358
- "loss": 2.0227,
359
- "step": 500
360
- },
361
- {
362
- "epoch": 0.4370179948586118,
363
- "grad_norm": 13.648895263671875,
364
- "learning_rate": 1.2380569352924609e-05,
365
- "loss": 2.1153,
366
- "step": 510
367
- },
368
- {
369
- "epoch": 0.44558697514995715,
370
- "grad_norm": 13.396595001220703,
371
- "learning_rate": 1.2339176610186645e-05,
372
- "loss": 2.0524,
373
- "step": 520
374
- },
375
- {
376
- "epoch": 0.4541559554413025,
377
- "grad_norm": 12.05173110961914,
378
- "learning_rate": 1.2297783867448682e-05,
379
- "loss": 2.0801,
380
- "step": 530
381
- },
382
- {
383
- "epoch": 0.46272493573264784,
384
- "grad_norm": 11.418408393859863,
385
- "learning_rate": 1.225639112471072e-05,
386
- "loss": 2.0219,
387
- "step": 540
388
- },
389
- {
390
- "epoch": 0.4712939160239931,
391
- "grad_norm": 14.92882251739502,
392
- "learning_rate": 1.2214998381972757e-05,
393
- "loss": 1.9197,
394
- "step": 550
395
- },
396
- {
397
- "epoch": 0.47986289631533846,
398
- "grad_norm": 13.67534351348877,
399
- "learning_rate": 1.2173605639234796e-05,
400
- "loss": 1.9579,
401
- "step": 560
402
- },
403
- {
404
- "epoch": 0.4884318766066838,
405
- "grad_norm": 16.3277645111084,
406
- "learning_rate": 1.2132212896496834e-05,
407
- "loss": 2.0499,
408
- "step": 570
409
- },
410
- {
411
- "epoch": 0.49700085689802914,
412
- "grad_norm": 12.686991691589355,
413
- "learning_rate": 1.209082015375887e-05,
414
- "loss": 1.8892,
415
- "step": 580
416
- },
417
- {
418
- "epoch": 0.5055698371893744,
419
- "grad_norm": 14.13610553741455,
420
- "learning_rate": 1.2049427411020907e-05,
421
- "loss": 1.9821,
422
- "step": 590
423
- },
424
- {
425
- "epoch": 0.5141388174807198,
426
- "grad_norm": 10.20384693145752,
427
- "learning_rate": 1.2008034668282945e-05,
428
- "loss": 1.8765,
429
- "step": 600
430
- },
431
- {
432
- "epoch": 0.5227077977720651,
433
- "grad_norm": 11.100608825683594,
434
- "learning_rate": 1.1966641925544982e-05,
435
- "loss": 1.928,
436
- "step": 610
437
- },
438
- {
439
- "epoch": 0.5312767780634104,
440
- "grad_norm": 13.737257957458496,
441
- "learning_rate": 1.192524918280702e-05,
442
- "loss": 2.0266,
443
- "step": 620
444
- },
445
- {
446
- "epoch": 0.5398457583547558,
447
- "grad_norm": 13.313102722167969,
448
- "learning_rate": 1.1883856440069057e-05,
449
- "loss": 1.9792,
450
- "step": 630
451
- },
452
- {
453
- "epoch": 0.5484147386461011,
454
- "grad_norm": 16.294010162353516,
455
- "learning_rate": 1.1842463697331093e-05,
456
- "loss": 2.0274,
457
- "step": 640
458
- },
459
- {
460
- "epoch": 0.5569837189374465,
461
- "grad_norm": 14.80037784576416,
462
- "learning_rate": 1.1801070954593132e-05,
463
- "loss": 2.0204,
464
- "step": 650
465
- },
466
- {
467
- "epoch": 0.5655526992287918,
468
- "grad_norm": 16.782167434692383,
469
- "learning_rate": 1.175967821185517e-05,
470
- "loss": 1.896,
471
- "step": 660
472
- },
473
- {
474
- "epoch": 0.5741216795201372,
475
- "grad_norm": 14.986900329589844,
476
- "learning_rate": 1.1718285469117207e-05,
477
- "loss": 1.9126,
478
- "step": 670
479
- },
480
- {
481
- "epoch": 0.5826906598114824,
482
- "grad_norm": 15.64176082611084,
483
- "learning_rate": 1.1676892726379245e-05,
484
- "loss": 1.8983,
485
- "step": 680
486
- },
487
- {
488
- "epoch": 0.5912596401028277,
489
- "grad_norm": 14.483860969543457,
490
- "learning_rate": 1.1635499983641282e-05,
491
- "loss": 2.1211,
492
- "step": 690
493
- },
494
- {
495
- "epoch": 0.5998286203941731,
496
- "grad_norm": 9.888971328735352,
497
- "learning_rate": 1.1594107240903318e-05,
498
- "loss": 1.8572,
499
- "step": 700
500
- },
501
- {
502
- "epoch": 0.6083976006855184,
503
- "grad_norm": 12.11154556274414,
504
- "learning_rate": 1.1552714498165355e-05,
505
- "loss": 1.866,
506
- "step": 710
507
- },
508
- {
509
- "epoch": 0.6169665809768637,
510
- "grad_norm": 14.539010047912598,
511
- "learning_rate": 1.1511321755427393e-05,
512
- "loss": 1.9029,
513
- "step": 720
514
- },
515
- {
516
- "epoch": 0.6255355612682091,
517
- "grad_norm": 17.459091186523438,
518
- "learning_rate": 1.1469929012689432e-05,
519
- "loss": 1.9333,
520
- "step": 730
521
- },
522
- {
523
- "epoch": 0.6341045415595544,
524
- "grad_norm": 14.461770057678223,
525
- "learning_rate": 1.142853626995147e-05,
526
- "loss": 2.0981,
527
- "step": 740
528
- },
529
- {
530
- "epoch": 0.6426735218508998,
531
- "grad_norm": 15.937264442443848,
532
- "learning_rate": 1.1387143527213507e-05,
533
- "loss": 1.8607,
534
- "step": 750
535
- },
536
- {
537
- "epoch": 0.6512425021422451,
538
- "grad_norm": 16.7382869720459,
539
- "learning_rate": 1.1345750784475543e-05,
540
- "loss": 2.1263,
541
- "step": 760
542
- },
543
- {
544
- "epoch": 0.6598114824335904,
545
- "grad_norm": 15.768759727478027,
546
- "learning_rate": 1.130435804173758e-05,
547
- "loss": 1.928,
548
- "step": 770
549
- },
550
- {
551
- "epoch": 0.6683804627249358,
552
- "grad_norm": 18.80556869506836,
553
- "learning_rate": 1.1262965298999618e-05,
554
- "loss": 1.98,
555
- "step": 780
556
- },
557
- {
558
- "epoch": 0.676949443016281,
559
- "grad_norm": 14.333468437194824,
560
- "learning_rate": 1.1221572556261655e-05,
561
- "loss": 1.9067,
562
- "step": 790
563
- },
564
- {
565
- "epoch": 0.6855184233076264,
566
- "grad_norm": 19.20866584777832,
567
- "learning_rate": 1.1180179813523693e-05,
568
- "loss": 2.0511,
569
- "step": 800
570
- },
571
- {
572
- "epoch": 0.6940874035989717,
573
- "grad_norm": 18.691129684448242,
574
- "learning_rate": 1.113878707078573e-05,
575
- "loss": 1.872,
576
- "step": 810
577
- },
578
- {
579
- "epoch": 0.702656383890317,
580
- "grad_norm": 14.753878593444824,
581
- "learning_rate": 1.1097394328047768e-05,
582
- "loss": 1.8993,
583
- "step": 820
584
- },
585
- {
586
- "epoch": 0.7112253641816624,
587
- "grad_norm": 13.987009048461914,
588
- "learning_rate": 1.1056001585309806e-05,
589
- "loss": 1.8868,
590
- "step": 830
591
- },
592
- {
593
- "epoch": 0.7197943444730077,
594
- "grad_norm": 16.61811637878418,
595
- "learning_rate": 1.1014608842571843e-05,
596
- "loss": 1.8531,
597
- "step": 840
598
- },
599
- {
600
- "epoch": 0.7283633247643531,
601
- "grad_norm": 18.929941177368164,
602
- "learning_rate": 1.097321609983388e-05,
603
- "loss": 1.8008,
604
- "step": 850
605
- },
606
- {
607
- "epoch": 0.7369323050556984,
608
- "grad_norm": 18.89458465576172,
609
- "learning_rate": 1.0931823357095918e-05,
610
- "loss": 1.935,
611
- "step": 860
612
- },
613
- {
614
- "epoch": 0.7455012853470437,
615
- "grad_norm": 15.524568557739258,
616
- "learning_rate": 1.0890430614357956e-05,
617
- "loss": 2.0548,
618
- "step": 870
619
- },
620
- {
621
- "epoch": 0.7540702656383891,
622
- "grad_norm": 17.038110733032227,
623
- "learning_rate": 1.0849037871619991e-05,
624
- "loss": 2.0176,
625
- "step": 880
626
- },
627
- {
628
- "epoch": 0.7626392459297343,
629
- "grad_norm": 16.24259376525879,
630
- "learning_rate": 1.0807645128882029e-05,
631
- "loss": 2.0311,
632
- "step": 890
633
- },
634
- {
635
- "epoch": 0.7712082262210797,
636
- "grad_norm": 12.702564239501953,
637
- "learning_rate": 1.0766252386144068e-05,
638
- "loss": 1.9613,
639
- "step": 900
640
- },
641
- {
642
- "epoch": 0.779777206512425,
643
- "grad_norm": 13.47549057006836,
644
- "learning_rate": 1.0724859643406106e-05,
645
- "loss": 1.9554,
646
- "step": 910
647
- },
648
- {
649
- "epoch": 0.7883461868037703,
650
- "grad_norm": 15.315031051635742,
651
- "learning_rate": 1.0683466900668143e-05,
652
- "loss": 1.8628,
653
- "step": 920
654
- },
655
- {
656
- "epoch": 0.7969151670951157,
657
- "grad_norm": 12.436241149902344,
658
- "learning_rate": 1.0642074157930179e-05,
659
- "loss": 1.9767,
660
- "step": 930
661
- },
662
- {
663
- "epoch": 0.805484147386461,
664
- "grad_norm": 17.100671768188477,
665
- "learning_rate": 1.0600681415192216e-05,
666
- "loss": 2.0268,
667
- "step": 940
668
- },
669
- {
670
- "epoch": 0.8140531276778064,
671
- "grad_norm": 14.803923606872559,
672
- "learning_rate": 1.0559288672454254e-05,
673
- "loss": 2.0516,
674
- "step": 950
675
- },
676
- {
677
- "epoch": 0.8226221079691517,
678
- "grad_norm": 17.308460235595703,
679
- "learning_rate": 1.0517895929716291e-05,
680
- "loss": 1.9354,
681
- "step": 960
682
- },
683
- {
684
- "epoch": 0.831191088260497,
685
- "grad_norm": 38.664039611816406,
686
- "learning_rate": 1.0476503186978329e-05,
687
- "loss": 1.9623,
688
- "step": 970
689
- },
690
- {
691
- "epoch": 0.8397600685518424,
692
- "grad_norm": 16.550312042236328,
693
- "learning_rate": 1.0435110444240368e-05,
694
- "loss": 1.888,
695
- "step": 980
696
- },
697
- {
698
- "epoch": 0.8483290488431876,
699
- "grad_norm": 19.344846725463867,
700
- "learning_rate": 1.0393717701502404e-05,
701
- "loss": 1.9992,
702
- "step": 990
703
- },
704
- {
705
- "epoch": 0.856898029134533,
706
- "grad_norm": 18.766752243041992,
707
- "learning_rate": 1.0352324958764441e-05,
708
- "loss": 1.8824,
709
- "step": 1000
710
- },
711
- {
712
- "epoch": 0.8654670094258783,
713
- "grad_norm": 20.725662231445312,
714
- "learning_rate": 1.0310932216026479e-05,
715
- "loss": 1.7351,
716
- "step": 1010
717
- },
718
- {
719
- "epoch": 0.8740359897172236,
720
- "grad_norm": 15.772370338439941,
721
- "learning_rate": 1.0269539473288516e-05,
722
- "loss": 1.9839,
723
- "step": 1020
724
- },
725
- {
726
- "epoch": 0.882604970008569,
727
- "grad_norm": 15.904685974121094,
728
- "learning_rate": 1.0228146730550554e-05,
729
- "loss": 2.1096,
730
- "step": 1030
731
- },
732
- {
733
- "epoch": 0.8911739502999143,
734
- "grad_norm": 19.53571891784668,
735
- "learning_rate": 1.0186753987812591e-05,
736
- "loss": 1.9122,
737
- "step": 1040
738
- },
739
- {
740
- "epoch": 0.8997429305912596,
741
- "grad_norm": 14.403525352478027,
742
- "learning_rate": 1.0145361245074627e-05,
743
- "loss": 1.8495,
744
- "step": 1050
745
- },
746
- {
747
- "epoch": 0.908311910882605,
748
- "grad_norm": 20.06512451171875,
749
- "learning_rate": 1.0103968502336666e-05,
750
- "loss": 1.9467,
751
- "step": 1060
752
- },
753
- {
754
- "epoch": 0.9168808911739503,
755
- "grad_norm": 15.332779884338379,
756
- "learning_rate": 1.0062575759598704e-05,
757
- "loss": 1.7918,
758
- "step": 1070
759
- },
760
- {
761
- "epoch": 0.9254498714652957,
762
- "grad_norm": 18.091779708862305,
763
- "learning_rate": 1.0021183016860741e-05,
764
- "loss": 1.8323,
765
- "step": 1080
766
- },
767
- {
768
- "epoch": 0.934018851756641,
769
- "grad_norm": 17.92203712463379,
770
- "learning_rate": 9.979790274122779e-06,
771
- "loss": 1.9448,
772
- "step": 1090
773
- },
774
- {
775
- "epoch": 0.9425878320479862,
776
- "grad_norm": 11.862146377563477,
777
- "learning_rate": 9.938397531384816e-06,
778
- "loss": 1.9453,
779
- "step": 1100
780
- },
781
- {
782
- "epoch": 0.9511568123393316,
783
- "grad_norm": 16.67616844177246,
784
- "learning_rate": 9.897004788646852e-06,
785
- "loss": 1.9977,
786
- "step": 1110
787
- },
788
- {
789
- "epoch": 0.9597257926306769,
790
- "grad_norm": 17.18949317932129,
791
- "learning_rate": 9.85561204590889e-06,
792
- "loss": 1.8888,
793
- "step": 1120
794
- },
795
- {
796
- "epoch": 0.9682947729220223,
797
- "grad_norm": 19.521203994750977,
798
- "learning_rate": 9.814219303170927e-06,
799
- "loss": 1.948,
800
- "step": 1130
801
- },
802
- {
803
- "epoch": 0.9768637532133676,
804
- "grad_norm": 21.371353149414062,
805
- "learning_rate": 9.772826560432965e-06,
806
- "loss": 1.9285,
807
- "step": 1140
808
- },
809
- {
810
- "epoch": 0.9854327335047129,
811
- "grad_norm": 18.078819274902344,
812
- "learning_rate": 9.731433817695004e-06,
813
- "loss": 1.9144,
814
- "step": 1150
815
- },
816
- {
817
- "epoch": 0.9940017137960583,
818
- "grad_norm": 26.718231201171875,
819
- "learning_rate": 9.690041074957041e-06,
820
- "loss": 1.8113,
821
- "step": 1160
822
- },
823
- {
824
- "epoch": 1.0,
825
- "eval_classification_report": {
826
- "accuracy": 0.3035,
827
- "ar": {
828
- "f1-score": 0.23214285714285715,
829
- "precision": 0.3,
830
- "recall": 0.18932038834951456,
831
- "support": 206.0
832
- },
833
- "cl": {
834
- "f1-score": 0.2225609756097561,
835
- "precision": 0.1994535519125683,
836
- "recall": 0.2517241379310345,
837
- "support": 290.0
838
- },
839
- "co": {
840
- "f1-score": 0.3453038674033149,
841
- "precision": 0.28868360277136257,
842
- "recall": 0.42955326460481097,
843
- "support": 291.0
844
- },
845
- "es": {
846
- "f1-score": 0.3566666666666667,
847
- "precision": 0.3333333333333333,
848
- "recall": 0.3835125448028674,
849
- "support": 279.0
850
- },
851
- "macro avg": {
852
- "f1-score": 0.28981047871583737,
853
- "precision": 0.31679889037420644,
854
- "recall": 0.28794417545744694,
855
- "support": 2000.0
856
- },
857
- "mx": {
858
- "f1-score": 0.2676767676767677,
859
- "precision": 0.5047619047619047,
860
- "recall": 0.18213058419243985,
861
- "support": 291.0
862
- },
863
- "pe": {
864
- "f1-score": 0.27666666666666667,
865
- "precision": 0.2686084142394822,
866
- "recall": 0.2852233676975945,
867
- "support": 291.0
868
- },
869
- "pr": {
870
- "f1-score": 0.6162162162162163,
871
- "precision": 0.6785714285714286,
872
- "recall": 0.5643564356435643,
873
- "support": 101.0
874
- },
875
- "uy": {
876
- "f1-score": 0.2910602910602911,
877
- "precision": 0.2777777777777778,
878
- "recall": 0.3056768558951965,
879
- "support": 229.0
880
- },
881
- "ve": {
882
- "f1-score": 0.0,
883
- "precision": 0.0,
884
- "recall": 0.0,
885
- "support": 22.0
886
- },
887
- "weighted avg": {
888
- "f1-score": 0.2998260603986032,
889
- "precision": 0.3269230233436701,
890
- "recall": 0.3035,
891
- "support": 2000.0
892
- }
893
- },
894
- "eval_f1": 0.28981047871583737,
895
- "eval_loss": 1.8416967391967773,
896
- "eval_runtime": 5.6259,
897
- "eval_samples_per_second": 355.502,
898
- "eval_steps_per_second": 88.875,
899
- "step": 1167
900
- },
901
- {
902
- "epoch": 1.0025706940874035,
903
- "grad_norm": 22.602876663208008,
904
- "learning_rate": 9.648648332219077e-06,
905
- "loss": 1.722,
906
- "step": 1170
907
- },
908
- {
909
- "epoch": 1.0111396743787489,
910
- "grad_norm": 19.420223236083984,
911
- "learning_rate": 9.607255589481115e-06,
912
- "loss": 1.7384,
913
- "step": 1180
914
- },
915
- {
916
- "epoch": 1.0197086546700942,
917
- "grad_norm": 15.554736137390137,
918
- "learning_rate": 9.565862846743152e-06,
919
- "loss": 1.6844,
920
- "step": 1190
921
- },
922
- {
923
- "epoch": 1.0282776349614395,
924
- "grad_norm": 24.70032501220703,
925
- "learning_rate": 9.52447010400519e-06,
926
- "loss": 1.7337,
927
- "step": 1200
928
- },
929
- {
930
- "epoch": 1.0368466152527849,
931
- "grad_norm": 20.22475814819336,
932
- "learning_rate": 9.483077361267227e-06,
933
- "loss": 1.7448,
934
- "step": 1210
935
- },
936
- {
937
- "epoch": 1.0454155955441302,
938
- "grad_norm": 14.43090534210205,
939
- "learning_rate": 9.441684618529265e-06,
940
- "loss": 1.7228,
941
- "step": 1220
942
- },
943
- {
944
- "epoch": 1.0539845758354756,
945
- "grad_norm": 16.916563034057617,
946
- "learning_rate": 9.400291875791302e-06,
947
- "loss": 1.5953,
948
- "step": 1230
949
- },
950
- {
951
- "epoch": 1.062553556126821,
952
- "grad_norm": 17.133316040039062,
953
- "learning_rate": 9.35889913305334e-06,
954
- "loss": 1.6757,
955
- "step": 1240
956
- },
957
- {
958
- "epoch": 1.0711225364181662,
959
- "grad_norm": 21.03934669494629,
960
- "learning_rate": 9.317506390315377e-06,
961
- "loss": 1.6807,
962
- "step": 1250
963
- },
964
- {
965
- "epoch": 1.0796915167095116,
966
- "grad_norm": 21.76130485534668,
967
- "learning_rate": 9.276113647577415e-06,
968
- "loss": 1.6364,
969
- "step": 1260
970
- },
971
- {
972
- "epoch": 1.088260497000857,
973
- "grad_norm": 21.917848587036133,
974
- "learning_rate": 9.234720904839452e-06,
975
- "loss": 1.9007,
976
- "step": 1270
977
- },
978
- {
979
- "epoch": 1.0968294772922023,
980
- "grad_norm": 19.40070915222168,
981
- "learning_rate": 9.19332816210149e-06,
982
- "loss": 1.6513,
983
- "step": 1280
984
- },
985
- {
986
- "epoch": 1.1053984575835476,
987
- "grad_norm": 18.84430694580078,
988
- "learning_rate": 9.151935419363526e-06,
989
- "loss": 1.8647,
990
- "step": 1290
991
- },
992
- {
993
- "epoch": 1.113967437874893,
994
- "grad_norm": 21.544525146484375,
995
- "learning_rate": 9.110542676625563e-06,
996
- "loss": 1.8551,
997
- "step": 1300
998
- },
999
- {
1000
- "epoch": 1.1225364181662383,
1001
- "grad_norm": 21.512622833251953,
1002
- "learning_rate": 9.069149933887602e-06,
1003
- "loss": 1.7686,
1004
- "step": 1310
1005
- },
1006
- {
1007
- "epoch": 1.1311053984575836,
1008
- "grad_norm": 21.45429229736328,
1009
- "learning_rate": 9.02775719114964e-06,
1010
- "loss": 1.7454,
1011
- "step": 1320
1012
- },
1013
- {
1014
- "epoch": 1.139674378748929,
1015
- "grad_norm": 20.992483139038086,
1016
- "learning_rate": 8.986364448411677e-06,
1017
- "loss": 1.5629,
1018
- "step": 1330
1019
- },
1020
- {
1021
- "epoch": 1.1482433590402743,
1022
- "grad_norm": 26.23398780822754,
1023
- "learning_rate": 8.944971705673713e-06,
1024
- "loss": 1.4882,
1025
- "step": 1340
1026
- },
1027
- {
1028
- "epoch": 1.1568123393316196,
1029
- "grad_norm": 21.12709617614746,
1030
- "learning_rate": 8.90357896293575e-06,
1031
- "loss": 1.6059,
1032
- "step": 1350
1033
- },
1034
- {
1035
- "epoch": 1.165381319622965,
1036
- "grad_norm": 20.56809425354004,
1037
- "learning_rate": 8.862186220197788e-06,
1038
- "loss": 1.7381,
1039
- "step": 1360
1040
- },
1041
- {
1042
- "epoch": 1.17395029991431,
1043
- "grad_norm": 22.67593765258789,
1044
- "learning_rate": 8.820793477459826e-06,
1045
- "loss": 1.8096,
1046
- "step": 1370
1047
- },
1048
- {
1049
- "epoch": 1.1825192802056554,
1050
- "grad_norm": 22.51243782043457,
1051
- "learning_rate": 8.779400734721863e-06,
1052
- "loss": 1.6126,
1053
- "step": 1380
1054
- },
1055
- {
1056
- "epoch": 1.1910882604970008,
1057
- "grad_norm": 28.95170021057129,
1058
- "learning_rate": 8.7380079919839e-06,
1059
- "loss": 1.6991,
1060
- "step": 1390
1061
- },
1062
- {
1063
- "epoch": 1.1996572407883461,
1064
- "grad_norm": 22.659038543701172,
1065
- "learning_rate": 8.696615249245938e-06,
1066
- "loss": 1.6806,
1067
- "step": 1400
1068
- },
1069
- {
1070
- "epoch": 1.2082262210796915,
1071
- "grad_norm": 23.431997299194336,
1072
- "learning_rate": 8.655222506507976e-06,
1073
- "loss": 1.5507,
1074
- "step": 1410
1075
- },
1076
- {
1077
- "epoch": 1.2167952013710368,
1078
- "grad_norm": 30.707691192626953,
1079
- "learning_rate": 8.613829763770013e-06,
1080
- "loss": 1.6909,
1081
- "step": 1420
1082
- },
1083
- {
1084
- "epoch": 1.2253641816623821,
1085
- "grad_norm": 23.785497665405273,
1086
- "learning_rate": 8.57243702103205e-06,
1087
- "loss": 1.7508,
1088
- "step": 1430
1089
- },
1090
- {
1091
- "epoch": 1.2339331619537275,
1092
- "grad_norm": 30.96599769592285,
1093
- "learning_rate": 8.531044278294088e-06,
1094
- "loss": 1.6873,
1095
- "step": 1440
1096
- },
1097
- {
1098
- "epoch": 1.2425021422450728,
1099
- "grad_norm": 36.57966995239258,
1100
- "learning_rate": 8.489651535556126e-06,
1101
- "loss": 1.6249,
1102
- "step": 1450
1103
- },
1104
- {
1105
- "epoch": 1.2510711225364182,
1106
- "grad_norm": 22.85884666442871,
1107
- "learning_rate": 8.448258792818161e-06,
1108
- "loss": 1.7765,
1109
- "step": 1460
1110
- },
1111
- {
1112
- "epoch": 1.2596401028277635,
1113
- "grad_norm": 19.838592529296875,
1114
- "learning_rate": 8.406866050080199e-06,
1115
- "loss": 1.6322,
1116
- "step": 1470
1117
- },
1118
- {
1119
- "epoch": 1.2682090831191088,
1120
- "grad_norm": 27.4448299407959,
1121
- "learning_rate": 8.365473307342238e-06,
1122
- "loss": 1.784,
1123
- "step": 1480
1124
- },
1125
- {
1126
- "epoch": 1.2767780634104542,
1127
- "grad_norm": 26.21979331970215,
1128
- "learning_rate": 8.324080564604276e-06,
1129
- "loss": 1.6915,
1130
- "step": 1490
1131
- },
1132
- {
1133
- "epoch": 1.2853470437017995,
1134
- "grad_norm": 14.650322914123535,
1135
- "learning_rate": 8.282687821866313e-06,
1136
- "loss": 1.7377,
1137
- "step": 1500
1138
- },
1139
- {
1140
- "epoch": 1.2939160239931449,
1141
- "grad_norm": 18.169477462768555,
1142
- "learning_rate": 8.24129507912835e-06,
1143
- "loss": 1.8747,
1144
- "step": 1510
1145
- },
1146
- {
1147
- "epoch": 1.3024850042844902,
1148
- "grad_norm": 23.312740325927734,
1149
- "learning_rate": 8.199902336390387e-06,
1150
- "loss": 1.8179,
1151
- "step": 1520
1152
- },
1153
- {
1154
- "epoch": 1.3110539845758356,
1155
- "grad_norm": 26.74571418762207,
1156
- "learning_rate": 8.158509593652424e-06,
1157
- "loss": 1.7476,
1158
- "step": 1530
1159
- },
1160
- {
1161
- "epoch": 1.3196229648671807,
1162
- "grad_norm": 24.429569244384766,
1163
- "learning_rate": 8.117116850914462e-06,
1164
- "loss": 1.5958,
1165
- "step": 1540
1166
- },
1167
- {
1168
- "epoch": 1.328191945158526,
1169
- "grad_norm": 34.8058967590332,
1170
- "learning_rate": 8.075724108176499e-06,
1171
- "loss": 1.5784,
1172
- "step": 1550
1173
- },
1174
- {
1175
- "epoch": 1.3367609254498714,
1176
- "grad_norm": 21.165245056152344,
1177
- "learning_rate": 8.034331365438538e-06,
1178
- "loss": 1.6505,
1179
- "step": 1560
1180
- },
1181
- {
1182
- "epoch": 1.3453299057412167,
1183
- "grad_norm": 25.115171432495117,
1184
- "learning_rate": 7.992938622700576e-06,
1185
- "loss": 1.618,
1186
- "step": 1570
1187
- },
1188
- {
1189
- "epoch": 1.353898886032562,
1190
- "grad_norm": 22.172096252441406,
1191
- "learning_rate": 7.951545879962612e-06,
1192
- "loss": 1.6814,
1193
- "step": 1580
1194
- },
1195
- {
1196
- "epoch": 1.3624678663239074,
1197
- "grad_norm": 30.089229583740234,
1198
- "learning_rate": 7.910153137224649e-06,
1199
- "loss": 1.6126,
1200
- "step": 1590
1201
- },
1202
- {
1203
- "epoch": 1.3710368466152527,
1204
- "grad_norm": 32.49215316772461,
1205
- "learning_rate": 7.868760394486687e-06,
1206
- "loss": 1.7492,
1207
- "step": 1600
1208
- },
1209
- {
1210
- "epoch": 1.379605826906598,
1211
- "grad_norm": 18.02476692199707,
1212
- "learning_rate": 7.827367651748724e-06,
1213
- "loss": 1.605,
1214
- "step": 1610
1215
- },
1216
- {
1217
- "epoch": 1.3881748071979434,
1218
- "grad_norm": 60.369930267333984,
1219
- "learning_rate": 7.785974909010762e-06,
1220
- "loss": 1.7559,
1221
- "step": 1620
1222
- },
1223
- {
1224
- "epoch": 1.3967437874892887,
1225
- "grad_norm": 44.660247802734375,
1226
- "learning_rate": 7.744582166272799e-06,
1227
- "loss": 1.7434,
1228
- "step": 1630
1229
- },
1230
- {
1231
- "epoch": 1.405312767780634,
1232
- "grad_norm": 26.649761199951172,
1233
- "learning_rate": 7.703189423534835e-06,
1234
- "loss": 1.7789,
1235
- "step": 1640
1236
- },
1237
- {
1238
- "epoch": 1.4138817480719794,
1239
- "grad_norm": 26.30625343322754,
1240
- "learning_rate": 7.661796680796874e-06,
1241
- "loss": 1.7798,
1242
- "step": 1650
1243
- },
1244
- {
1245
- "epoch": 1.4224507283633248,
1246
- "grad_norm": 24.721965789794922,
1247
- "learning_rate": 7.620403938058912e-06,
1248
- "loss": 1.7247,
1249
- "step": 1660
1250
- },
1251
- {
1252
- "epoch": 1.43101970865467,
1253
- "grad_norm": 24.725996017456055,
1254
- "learning_rate": 7.579011195320949e-06,
1255
- "loss": 1.6941,
1256
- "step": 1670
1257
- },
1258
- {
1259
- "epoch": 1.4395886889460154,
1260
- "grad_norm": 28.72591209411621,
1261
- "learning_rate": 7.537618452582986e-06,
1262
- "loss": 1.6854,
1263
- "step": 1680
1264
- },
1265
- {
1266
- "epoch": 1.4481576692373608,
1267
- "grad_norm": 15.998064041137695,
1268
- "learning_rate": 7.496225709845023e-06,
1269
- "loss": 1.6259,
1270
- "step": 1690
1271
- },
1272
- {
1273
- "epoch": 1.4567266495287061,
1274
- "grad_norm": 28.62874984741211,
1275
- "learning_rate": 7.454832967107061e-06,
1276
- "loss": 1.7328,
1277
- "step": 1700
1278
- },
1279
- {
1280
- "epoch": 1.4652956298200515,
1281
- "grad_norm": 24.98537254333496,
1282
- "learning_rate": 7.413440224369097e-06,
1283
- "loss": 1.9542,
1284
- "step": 1710
1285
- },
1286
- {
1287
- "epoch": 1.4738646101113968,
1288
- "grad_norm": 31.268770217895508,
1289
- "learning_rate": 7.372047481631135e-06,
1290
- "loss": 1.4888,
1291
- "step": 1720
1292
- },
1293
- {
1294
- "epoch": 1.4824335904027421,
1295
- "grad_norm": 21.951004028320312,
1296
- "learning_rate": 7.330654738893174e-06,
1297
- "loss": 1.7112,
1298
- "step": 1730
1299
- },
1300
- {
1301
- "epoch": 1.4910025706940875,
1302
- "grad_norm": 16.30368423461914,
1303
- "learning_rate": 7.289261996155211e-06,
1304
- "loss": 1.7718,
1305
- "step": 1740
1306
- },
1307
- {
1308
- "epoch": 1.4995715509854328,
1309
- "grad_norm": 28.71227264404297,
1310
- "learning_rate": 7.247869253417248e-06,
1311
- "loss": 1.5179,
1312
- "step": 1750
1313
- },
1314
- {
1315
- "epoch": 1.5081405312767782,
1316
- "grad_norm": 25.384668350219727,
1317
- "learning_rate": 7.206476510679286e-06,
1318
- "loss": 1.6098,
1319
- "step": 1760
1320
- },
1321
- {
1322
- "epoch": 1.5167095115681235,
1323
- "grad_norm": 28.461658477783203,
1324
- "learning_rate": 7.1650837679413224e-06,
1325
- "loss": 1.5626,
1326
- "step": 1770
1327
- },
1328
- {
1329
- "epoch": 1.5252784918594688,
1330
- "grad_norm": 43.065731048583984,
1331
- "learning_rate": 7.12369102520336e-06,
1332
- "loss": 1.7604,
1333
- "step": 1780
1334
- },
1335
- {
1336
- "epoch": 1.5338474721508142,
1337
- "grad_norm": 28.194068908691406,
1338
- "learning_rate": 7.0822982824653974e-06,
1339
- "loss": 1.6964,
1340
- "step": 1790
1341
- },
1342
- {
1343
- "epoch": 1.5424164524421595,
1344
- "grad_norm": 37.095558166503906,
1345
- "learning_rate": 7.040905539727434e-06,
1346
- "loss": 1.7532,
1347
- "step": 1800
1348
- },
1349
- {
1350
- "epoch": 1.5509854327335049,
1351
- "grad_norm": 20.070463180541992,
1352
- "learning_rate": 6.999512796989473e-06,
1353
- "loss": 1.5389,
1354
- "step": 1810
1355
- },
1356
- {
1357
- "epoch": 1.5595544130248502,
1358
- "grad_norm": 29.111501693725586,
1359
- "learning_rate": 6.958120054251511e-06,
1360
- "loss": 1.6142,
1361
- "step": 1820
1362
- },
1363
- {
1364
- "epoch": 1.5681233933161953,
1365
- "grad_norm": 22.289817810058594,
1366
- "learning_rate": 6.9167273115135475e-06,
1367
- "loss": 1.8116,
1368
- "step": 1830
1369
- },
1370
- {
1371
- "epoch": 1.5766923736075407,
1372
- "grad_norm": 71.48851776123047,
1373
- "learning_rate": 6.875334568775585e-06,
1374
- "loss": 1.4829,
1375
- "step": 1840
1376
- },
1377
- {
1378
- "epoch": 1.585261353898886,
1379
- "grad_norm": 25.093791961669922,
1380
- "learning_rate": 6.8339418260376225e-06,
1381
- "loss": 1.5003,
1382
- "step": 1850
1383
- },
1384
- {
1385
- "epoch": 1.5938303341902313,
1386
- "grad_norm": 38.602638244628906,
1387
- "learning_rate": 6.792549083299659e-06,
1388
- "loss": 1.6257,
1389
- "step": 1860
1390
- },
1391
- {
1392
- "epoch": 1.6023993144815767,
1393
- "grad_norm": 30.006364822387695,
1394
- "learning_rate": 6.751156340561697e-06,
1395
- "loss": 1.6087,
1396
- "step": 1870
1397
- },
1398
- {
1399
- "epoch": 1.610968294772922,
1400
- "grad_norm": 17.102664947509766,
1401
- "learning_rate": 6.709763597823735e-06,
1402
- "loss": 1.6324,
1403
- "step": 1880
1404
- },
1405
- {
1406
- "epoch": 1.6195372750642674,
1407
- "grad_norm": 19.19342803955078,
1408
- "learning_rate": 6.668370855085772e-06,
1409
- "loss": 1.5725,
1410
- "step": 1890
1411
- },
1412
- {
1413
- "epoch": 1.6281062553556127,
1414
- "grad_norm": 22.23579978942871,
1415
- "learning_rate": 6.626978112347809e-06,
1416
- "loss": 1.6174,
1417
- "step": 1900
1418
- },
1419
- {
1420
- "epoch": 1.636675235646958,
1421
- "grad_norm": 18.06783676147461,
1422
- "learning_rate": 6.585585369609847e-06,
1423
- "loss": 1.6147,
1424
- "step": 1910
1425
- },
1426
- {
1427
- "epoch": 1.6452442159383034,
1428
- "grad_norm": 20.512718200683594,
1429
- "learning_rate": 6.544192626871884e-06,
1430
- "loss": 1.6959,
1431
- "step": 1920
1432
- },
1433
- {
1434
- "epoch": 1.6538131962296485,
1435
- "grad_norm": 19.70807647705078,
1436
- "learning_rate": 6.502799884133922e-06,
1437
- "loss": 1.8072,
1438
- "step": 1930
1439
- },
1440
- {
1441
- "epoch": 1.6623821765209938,
1442
- "grad_norm": 24.146228790283203,
1443
- "learning_rate": 6.461407141395959e-06,
1444
- "loss": 1.701,
1445
- "step": 1940
1446
- },
1447
- {
1448
- "epoch": 1.6709511568123392,
1449
- "grad_norm": 34.090030670166016,
1450
- "learning_rate": 6.420014398657996e-06,
1451
- "loss": 1.769,
1452
- "step": 1950
1453
- },
1454
- {
1455
- "epoch": 1.6795201371036845,
1456
- "grad_norm": 42.51077651977539,
1457
- "learning_rate": 6.378621655920034e-06,
1458
- "loss": 1.5787,
1459
- "step": 1960
1460
- },
1461
- {
1462
- "epoch": 1.6880891173950299,
1463
- "grad_norm": 24.358346939086914,
1464
- "learning_rate": 6.337228913182072e-06,
1465
- "loss": 1.5821,
1466
- "step": 1970
1467
- },
1468
- {
1469
- "epoch": 1.6966580976863752,
1470
- "grad_norm": 35.25168991088867,
1471
- "learning_rate": 6.295836170444108e-06,
1472
- "loss": 1.5565,
1473
- "step": 1980
1474
- },
1475
- {
1476
- "epoch": 1.7052270779777206,
1477
- "grad_norm": 21.4340763092041,
1478
- "learning_rate": 6.254443427706146e-06,
1479
- "loss": 1.6134,
1480
- "step": 1990
1481
- },
1482
- {
1483
- "epoch": 1.713796058269066,
1484
- "grad_norm": 22.839839935302734,
1485
- "learning_rate": 6.213050684968183e-06,
1486
- "loss": 1.6213,
1487
- "step": 2000
1488
- },
1489
- {
1490
- "epoch": 1.7223650385604112,
1491
- "grad_norm": 25.80986785888672,
1492
- "learning_rate": 6.171657942230221e-06,
1493
- "loss": 1.6262,
1494
- "step": 2010
1495
- },
1496
- {
1497
- "epoch": 1.7309340188517566,
1498
- "grad_norm": 40.22003173828125,
1499
- "learning_rate": 6.130265199492258e-06,
1500
- "loss": 1.6161,
1501
- "step": 2020
1502
- },
1503
- {
1504
- "epoch": 1.739502999143102,
1505
- "grad_norm": 37.73822784423828,
1506
- "learning_rate": 6.088872456754296e-06,
1507
- "loss": 1.622,
1508
- "step": 2030
1509
- },
1510
- {
1511
- "epoch": 1.7480719794344473,
1512
- "grad_norm": 23.955387115478516,
1513
- "learning_rate": 6.047479714016333e-06,
1514
- "loss": 1.549,
1515
- "step": 2040
1516
- },
1517
- {
1518
- "epoch": 1.7566409597257926,
1519
- "grad_norm": 27.26932144165039,
1520
- "learning_rate": 6.006086971278371e-06,
1521
- "loss": 1.4373,
1522
- "step": 2050
1523
- },
1524
- {
1525
- "epoch": 1.765209940017138,
1526
- "grad_norm": 28.924217224121094,
1527
- "learning_rate": 5.9646942285404075e-06,
1528
- "loss": 1.4752,
1529
- "step": 2060
1530
- },
1531
- {
1532
- "epoch": 1.7737789203084833,
1533
- "grad_norm": 26.8709659576416,
1534
- "learning_rate": 5.923301485802445e-06,
1535
- "loss": 1.7233,
1536
- "step": 2070
1537
- },
1538
- {
1539
- "epoch": 1.7823479005998286,
1540
- "grad_norm": 28.21649742126465,
1541
- "learning_rate": 5.8819087430644825e-06,
1542
- "loss": 1.7996,
1543
- "step": 2080
1544
- },
1545
- {
1546
- "epoch": 1.790916880891174,
1547
- "grad_norm": 24.994258880615234,
1548
- "learning_rate": 5.84051600032652e-06,
1549
- "loss": 1.5322,
1550
- "step": 2090
1551
- },
1552
- {
1553
- "epoch": 1.7994858611825193,
1554
- "grad_norm": 24.2746639251709,
1555
- "learning_rate": 5.7991232575885575e-06,
1556
- "loss": 1.8132,
1557
- "step": 2100
1558
- },
1559
- {
1560
- "epoch": 1.8080548414738646,
1561
- "grad_norm": 26.497304916381836,
1562
- "learning_rate": 5.757730514850595e-06,
1563
- "loss": 1.5681,
1564
- "step": 2110
1565
- },
1566
- {
1567
- "epoch": 1.81662382176521,
1568
- "grad_norm": 43.27742004394531,
1569
- "learning_rate": 5.716337772112632e-06,
1570
- "loss": 1.6066,
1571
- "step": 2120
1572
- },
1573
- {
1574
- "epoch": 1.8251928020565553,
1575
- "grad_norm": 20.854333877563477,
1576
- "learning_rate": 5.67494502937467e-06,
1577
- "loss": 1.6055,
1578
- "step": 2130
1579
- },
1580
- {
1581
- "epoch": 1.8337617823479007,
1582
- "grad_norm": 26.56975555419922,
1583
- "learning_rate": 5.6335522866367075e-06,
1584
- "loss": 1.8364,
1585
- "step": 2140
1586
- },
1587
- {
1588
- "epoch": 1.842330762639246,
1589
- "grad_norm": 25.438674926757812,
1590
- "learning_rate": 5.592159543898744e-06,
1591
- "loss": 1.5838,
1592
- "step": 2150
1593
- },
1594
- {
1595
- "epoch": 1.8508997429305913,
1596
- "grad_norm": 24.237918853759766,
1597
- "learning_rate": 5.550766801160782e-06,
1598
- "loss": 1.4289,
1599
- "step": 2160
1600
- },
1601
- {
1602
- "epoch": 1.8594687232219367,
1603
- "grad_norm": 16.972482681274414,
1604
- "learning_rate": 5.50937405842282e-06,
1605
- "loss": 1.3532,
1606
- "step": 2170
1607
- },
1608
- {
1609
- "epoch": 1.868037703513282,
1610
- "grad_norm": 28.264667510986328,
1611
- "learning_rate": 5.467981315684857e-06,
1612
- "loss": 1.6642,
1613
- "step": 2180
1614
- },
1615
- {
1616
- "epoch": 1.8766066838046274,
1617
- "grad_norm": 39.41172409057617,
1618
- "learning_rate": 5.426588572946894e-06,
1619
- "loss": 1.5638,
1620
- "step": 2190
1621
- },
1622
- {
1623
- "epoch": 1.8851756640959727,
1624
- "grad_norm": 25.234155654907227,
1625
- "learning_rate": 5.385195830208932e-06,
1626
- "loss": 1.7821,
1627
- "step": 2200
1628
- },
1629
- {
1630
- "epoch": 1.893744644387318,
1631
- "grad_norm": 33.262840270996094,
1632
- "learning_rate": 5.343803087470969e-06,
1633
- "loss": 1.6387,
1634
- "step": 2210
1635
- },
1636
- {
1637
- "epoch": 1.9023136246786634,
1638
- "grad_norm": 25.7631778717041,
1639
- "learning_rate": 5.302410344733007e-06,
1640
- "loss": 1.7162,
1641
- "step": 2220
1642
- },
1643
- {
1644
- "epoch": 1.9108826049700087,
1645
- "grad_norm": 27.602434158325195,
1646
- "learning_rate": 5.261017601995044e-06,
1647
- "loss": 1.712,
1648
- "step": 2230
1649
- },
1650
- {
1651
- "epoch": 1.919451585261354,
1652
- "grad_norm": 54.29318618774414,
1653
- "learning_rate": 5.219624859257081e-06,
1654
- "loss": 1.5472,
1655
- "step": 2240
1656
- },
1657
- {
1658
- "epoch": 1.9280205655526992,
1659
- "grad_norm": 25.333751678466797,
1660
- "learning_rate": 5.178232116519119e-06,
1661
- "loss": 1.3439,
1662
- "step": 2250
1663
- },
1664
- {
1665
- "epoch": 1.9365895458440445,
1666
- "grad_norm": 27.495887756347656,
1667
- "learning_rate": 5.136839373781157e-06,
1668
- "loss": 1.5618,
1669
- "step": 2260
1670
- },
1671
- {
1672
- "epoch": 1.9451585261353899,
1673
- "grad_norm": 18.012102127075195,
1674
- "learning_rate": 5.095446631043193e-06,
1675
- "loss": 1.3402,
1676
- "step": 2270
1677
- },
1678
- {
1679
- "epoch": 1.9537275064267352,
1680
- "grad_norm": 24.922513961791992,
1681
- "learning_rate": 5.054053888305231e-06,
1682
- "loss": 1.5323,
1683
- "step": 2280
1684
- },
1685
- {
1686
- "epoch": 1.9622964867180805,
1687
- "grad_norm": 29.22248649597168,
1688
- "learning_rate": 5.012661145567269e-06,
1689
- "loss": 1.7345,
1690
- "step": 2290
1691
- },
1692
- {
1693
- "epoch": 1.9708654670094259,
1694
- "grad_norm": 37.99812316894531,
1695
- "learning_rate": 4.971268402829306e-06,
1696
- "loss": 1.5922,
1697
- "step": 2300
1698
- },
1699
- {
1700
- "epoch": 1.9794344473007712,
1701
- "grad_norm": 18.794361114501953,
1702
- "learning_rate": 4.929875660091343e-06,
1703
- "loss": 1.3667,
1704
- "step": 2310
1705
- },
1706
- {
1707
- "epoch": 1.9880034275921166,
1708
- "grad_norm": 26.54824447631836,
1709
- "learning_rate": 4.888482917353381e-06,
1710
- "loss": 1.5468,
1711
- "step": 2320
1712
- },
1713
- {
1714
- "epoch": 1.996572407883462,
1715
- "grad_norm": 38.71345138549805,
1716
- "learning_rate": 4.847090174615418e-06,
1717
- "loss": 1.7227,
1718
- "step": 2330
1719
- },
1720
- {
1721
- "epoch": 2.0,
1722
- "eval_classification_report": {
1723
- "accuracy": 0.3395,
1724
- "ar": {
1725
- "f1-score": 0.2974683544303797,
1726
- "precision": 0.42727272727272725,
1727
- "recall": 0.22815533980582525,
1728
- "support": 206.0
1729
- },
1730
- "cl": {
1731
- "f1-score": 0.21897810218978103,
1732
- "precision": 0.23255813953488372,
1733
- "recall": 0.20689655172413793,
1734
- "support": 290.0
1735
- },
1736
- "co": {
1737
- "f1-score": 0.35368956743002544,
1738
- "precision": 0.2808080808080808,
1739
- "recall": 0.47766323024054985,
1740
- "support": 291.0
1741
- },
1742
- "es": {
1743
- "f1-score": 0.3333333333333333,
1744
- "precision": 0.4207650273224044,
1745
- "recall": 0.27598566308243727,
1746
- "support": 279.0
1747
- },
1748
- "macro avg": {
1749
- "f1-score": 0.32076116800689736,
1750
- "precision": 0.3395273288843494,
1751
- "recall": 0.32532248394277347,
1752
- "support": 2000.0
1753
- },
1754
- "mx": {
1755
- "f1-score": 0.39759036144578314,
1756
- "precision": 0.4782608695652174,
1757
- "recall": 0.3402061855670103,
1758
- "support": 291.0
1759
- },
1760
- "pe": {
1761
- "f1-score": 0.3117241379310345,
1762
- "precision": 0.26036866359447003,
1763
- "recall": 0.38831615120274915,
1764
- "support": 291.0
1765
- },
1766
- "pr": {
1767
- "f1-score": 0.6160714285714286,
1768
- "precision": 0.5609756097560976,
1769
- "recall": 0.6831683168316832,
1770
- "support": 101.0
1771
- },
1772
- "uy": {
1773
- "f1-score": 0.35799522673031026,
1774
- "precision": 0.39473684210526316,
1775
- "recall": 0.32751091703056767,
1776
- "support": 229.0
1777
- },
1778
- "ve": {
1779
- "f1-score": 0.0,
1780
- "precision": 0.0,
1781
- "recall": 0.0,
1782
- "support": 22.0
1783
- },
1784
- "weighted avg": {
1785
- "f1-score": 0.33566021764772075,
1786
- "precision": 0.3582815519991702,
1787
- "recall": 0.3395,
1788
- "support": 2000.0
1789
- }
1790
- },
1791
- "eval_f1": 0.32076116800689736,
1792
- "eval_loss": 1.803144097328186,
1793
- "eval_runtime": 5.5294,
1794
- "eval_samples_per_second": 361.705,
1795
- "eval_steps_per_second": 90.426,
1796
- "step": 2334
1797
- },
1798
- {
1799
- "epoch": 2.005141388174807,
1800
- "grad_norm": 22.46615219116211,
1801
- "learning_rate": 4.805697431877456e-06,
1802
- "loss": 1.3894,
1803
- "step": 2340
1804
- },
1805
- {
1806
- "epoch": 2.0137103684661524,
1807
- "grad_norm": 33.21921157836914,
1808
- "learning_rate": 4.7643046891394934e-06,
1809
- "loss": 1.3936,
1810
- "step": 2350
1811
- },
1812
- {
1813
- "epoch": 2.0222793487574977,
1814
- "grad_norm": 27.54530143737793,
1815
- "learning_rate": 4.72291194640153e-06,
1816
- "loss": 1.4118,
1817
- "step": 2360
1818
- },
1819
- {
1820
- "epoch": 2.030848329048843,
1821
- "grad_norm": 29.95819664001465,
1822
- "learning_rate": 4.681519203663568e-06,
1823
- "loss": 1.215,
1824
- "step": 2370
1825
- },
1826
- {
1827
- "epoch": 2.0394173093401884,
1828
- "grad_norm": 43.5531005859375,
1829
- "learning_rate": 4.640126460925606e-06,
1830
- "loss": 1.3518,
1831
- "step": 2380
1832
- },
1833
- {
1834
- "epoch": 2.0479862896315337,
1835
- "grad_norm": 33.676414489746094,
1836
- "learning_rate": 4.598733718187643e-06,
1837
- "loss": 1.4004,
1838
- "step": 2390
1839
- },
1840
- {
1841
- "epoch": 2.056555269922879,
1842
- "grad_norm": 23.472457885742188,
1843
- "learning_rate": 4.55734097544968e-06,
1844
- "loss": 1.2582,
1845
- "step": 2400
1846
- },
1847
- {
1848
- "epoch": 2.0651242502142244,
1849
- "grad_norm": 27.10287857055664,
1850
- "learning_rate": 4.515948232711718e-06,
1851
- "loss": 1.2922,
1852
- "step": 2410
1853
- },
1854
- {
1855
- "epoch": 2.0736932305055698,
1856
- "grad_norm": 20.592119216918945,
1857
- "learning_rate": 4.474555489973755e-06,
1858
- "loss": 1.3074,
1859
- "step": 2420
1860
- },
1861
- {
1862
- "epoch": 2.082262210796915,
1863
- "grad_norm": 22.98575782775879,
1864
- "learning_rate": 4.433162747235793e-06,
1865
- "loss": 1.1699,
1866
- "step": 2430
1867
- },
1868
- {
1869
- "epoch": 2.0908311910882604,
1870
- "grad_norm": 40.38471221923828,
1871
- "learning_rate": 4.39177000449783e-06,
1872
- "loss": 1.438,
1873
- "step": 2440
1874
- },
1875
- {
1876
- "epoch": 2.0994001713796058,
1877
- "grad_norm": 21.089099884033203,
1878
- "learning_rate": 4.350377261759867e-06,
1879
- "loss": 1.0525,
1880
- "step": 2450
1881
- },
1882
- {
1883
- "epoch": 2.107969151670951,
1884
- "grad_norm": 18.718284606933594,
1885
- "learning_rate": 4.308984519021905e-06,
1886
- "loss": 0.986,
1887
- "step": 2460
1888
- },
1889
- {
1890
- "epoch": 2.1165381319622965,
1891
- "grad_norm": 42.87383270263672,
1892
- "learning_rate": 4.267591776283942e-06,
1893
- "loss": 1.3242,
1894
- "step": 2470
1895
- },
1896
- {
1897
- "epoch": 2.125107112253642,
1898
- "grad_norm": 9.905497550964355,
1899
- "learning_rate": 4.226199033545979e-06,
1900
- "loss": 1.0894,
1901
- "step": 2480
1902
- },
1903
- {
1904
- "epoch": 2.133676092544987,
1905
- "grad_norm": 40.20561218261719,
1906
- "learning_rate": 4.184806290808017e-06,
1907
- "loss": 1.2448,
1908
- "step": 2490
1909
- },
1910
- {
1911
- "epoch": 2.1422450728363325,
1912
- "grad_norm": 25.991907119750977,
1913
- "learning_rate": 4.143413548070054e-06,
1914
- "loss": 1.2575,
1915
- "step": 2500
1916
- },
1917
- {
1918
- "epoch": 2.150814053127678,
1919
- "grad_norm": 32.69858932495117,
1920
- "learning_rate": 4.102020805332092e-06,
1921
- "loss": 1.3396,
1922
- "step": 2510
1923
- },
1924
- {
1925
- "epoch": 2.159383033419023,
1926
- "grad_norm": 36.805259704589844,
1927
- "learning_rate": 4.060628062594129e-06,
1928
- "loss": 1.4618,
1929
- "step": 2520
1930
- },
1931
- {
1932
- "epoch": 2.1679520137103685,
1933
- "grad_norm": 34.98189926147461,
1934
- "learning_rate": 4.019235319856166e-06,
1935
- "loss": 1.1907,
1936
- "step": 2530
1937
- },
1938
- {
1939
- "epoch": 2.176520994001714,
1940
- "grad_norm": 26.45909309387207,
1941
- "learning_rate": 3.977842577118204e-06,
1942
- "loss": 1.4029,
1943
- "step": 2540
1944
- },
1945
- {
1946
- "epoch": 2.185089974293059,
1947
- "grad_norm": 35.13296890258789,
1948
- "learning_rate": 3.936449834380242e-06,
1949
- "loss": 1.2296,
1950
- "step": 2550
1951
- },
1952
- {
1953
- "epoch": 2.1936589545844045,
1954
- "grad_norm": 35.67057800292969,
1955
- "learning_rate": 3.8950570916422785e-06,
1956
- "loss": 1.5262,
1957
- "step": 2560
1958
- },
1959
- {
1960
- "epoch": 2.20222793487575,
1961
- "grad_norm": 28.157564163208008,
1962
- "learning_rate": 3.853664348904316e-06,
1963
- "loss": 1.0519,
1964
- "step": 2570
1965
- },
1966
- {
1967
- "epoch": 2.210796915167095,
1968
- "grad_norm": 32.18168640136719,
1969
- "learning_rate": 3.8122716061663535e-06,
1970
- "loss": 1.3485,
1971
- "step": 2580
1972
- },
1973
- {
1974
- "epoch": 2.2193658954584405,
1975
- "grad_norm": 38.181419372558594,
1976
- "learning_rate": 3.7708788634283914e-06,
1977
- "loss": 1.2754,
1978
- "step": 2590
1979
- },
1980
- {
1981
- "epoch": 2.227934875749786,
1982
- "grad_norm": 51.16127014160156,
1983
- "learning_rate": 3.7294861206904285e-06,
1984
- "loss": 1.4616,
1985
- "step": 2600
1986
- },
1987
- {
1988
- "epoch": 2.236503856041131,
1989
- "grad_norm": 37.655372619628906,
1990
- "learning_rate": 3.688093377952466e-06,
1991
- "loss": 1.1636,
1992
- "step": 2610
1993
- },
1994
- {
1995
- "epoch": 2.2450728363324766,
1996
- "grad_norm": 37.25764083862305,
1997
- "learning_rate": 3.646700635214503e-06,
1998
- "loss": 1.1733,
1999
- "step": 2620
2000
- },
2001
- {
2002
- "epoch": 2.253641816623822,
2003
- "grad_norm": 20.390323638916016,
2004
- "learning_rate": 3.605307892476541e-06,
2005
- "loss": 1.2678,
2006
- "step": 2630
2007
- },
2008
- {
2009
- "epoch": 2.2622107969151672,
2010
- "grad_norm": 23.969755172729492,
2011
- "learning_rate": 3.563915149738578e-06,
2012
- "loss": 1.2794,
2013
- "step": 2640
2014
- },
2015
- {
2016
- "epoch": 2.2707797772065126,
2017
- "grad_norm": 41.76385498046875,
2018
- "learning_rate": 3.5225224070006156e-06,
2019
- "loss": 1.0864,
2020
- "step": 2650
2021
- },
2022
- {
2023
- "epoch": 2.279348757497858,
2024
- "grad_norm": 43.808013916015625,
2025
- "learning_rate": 3.4811296642626527e-06,
2026
- "loss": 1.116,
2027
- "step": 2660
2028
- },
2029
- {
2030
- "epoch": 2.2879177377892033,
2031
- "grad_norm": 23.652198791503906,
2032
- "learning_rate": 3.4397369215246906e-06,
2033
- "loss": 1.1137,
2034
- "step": 2670
2035
- },
2036
- {
2037
- "epoch": 2.2964867180805486,
2038
- "grad_norm": 73.83686828613281,
2039
- "learning_rate": 3.398344178786728e-06,
2040
- "loss": 1.3697,
2041
- "step": 2680
2042
- },
2043
- {
2044
- "epoch": 2.305055698371894,
2045
- "grad_norm": 24.451356887817383,
2046
- "learning_rate": 3.356951436048765e-06,
2047
- "loss": 1.1336,
2048
- "step": 2690
2049
- },
2050
- {
2051
- "epoch": 2.3136246786632393,
2052
- "grad_norm": 29.073118209838867,
2053
- "learning_rate": 3.3155586933108027e-06,
2054
- "loss": 1.5349,
2055
- "step": 2700
2056
- },
2057
- {
2058
- "epoch": 2.3221936589545846,
2059
- "grad_norm": 40.77454376220703,
2060
- "learning_rate": 3.27416595057284e-06,
2061
- "loss": 1.0589,
2062
- "step": 2710
2063
- },
2064
- {
2065
- "epoch": 2.33076263924593,
2066
- "grad_norm": 88.82485961914062,
2067
- "learning_rate": 3.2327732078348777e-06,
2068
- "loss": 1.1665,
2069
- "step": 2720
2070
- },
2071
- {
2072
- "epoch": 2.3393316195372753,
2073
- "grad_norm": 43.5887336730957,
2074
- "learning_rate": 3.1913804650969148e-06,
2075
- "loss": 1.2165,
2076
- "step": 2730
2077
- },
2078
- {
2079
- "epoch": 2.34790059982862,
2080
- "grad_norm": 27.64069175720215,
2081
- "learning_rate": 3.1499877223589527e-06,
2082
- "loss": 1.144,
2083
- "step": 2740
2084
- },
2085
- {
2086
- "epoch": 2.3564695801199655,
2087
- "grad_norm": 20.499149322509766,
2088
- "learning_rate": 3.10859497962099e-06,
2089
- "loss": 1.1935,
2090
- "step": 2750
2091
- },
2092
- {
2093
- "epoch": 2.365038560411311,
2094
- "grad_norm": 34.570640563964844,
2095
- "learning_rate": 3.0672022368830273e-06,
2096
- "loss": 1.0865,
2097
- "step": 2760
2098
- },
2099
- {
2100
- "epoch": 2.3736075407026562,
2101
- "grad_norm": 26.797300338745117,
2102
- "learning_rate": 3.025809494145065e-06,
2103
- "loss": 1.336,
2104
- "step": 2770
2105
- },
2106
- {
2107
- "epoch": 2.3821765209940016,
2108
- "grad_norm": 37.89352035522461,
2109
- "learning_rate": 2.9844167514071023e-06,
2110
- "loss": 1.2351,
2111
- "step": 2780
2112
- },
2113
- {
2114
- "epoch": 2.390745501285347,
2115
- "grad_norm": 34.16749572753906,
2116
- "learning_rate": 2.9430240086691394e-06,
2117
- "loss": 1.3755,
2118
- "step": 2790
2119
- },
2120
- {
2121
- "epoch": 2.3993144815766922,
2122
- "grad_norm": 10.907548904418945,
2123
- "learning_rate": 2.901631265931177e-06,
2124
- "loss": 1.0435,
2125
- "step": 2800
2126
- },
2127
- {
2128
- "epoch": 2.4078834618680376,
2129
- "grad_norm": 34.773712158203125,
2130
- "learning_rate": 2.8602385231932144e-06,
2131
- "loss": 1.5544,
2132
- "step": 2810
2133
- },
2134
- {
2135
- "epoch": 2.416452442159383,
2136
- "grad_norm": 23.92730140686035,
2137
- "learning_rate": 2.818845780455252e-06,
2138
- "loss": 1.1912,
2139
- "step": 2820
2140
- },
2141
- {
2142
- "epoch": 2.4250214224507283,
2143
- "grad_norm": 41.49195861816406,
2144
- "learning_rate": 2.777453037717289e-06,
2145
- "loss": 1.5421,
2146
- "step": 2830
2147
- },
2148
- {
2149
- "epoch": 2.4335904027420736,
2150
- "grad_norm": 26.17296600341797,
2151
- "learning_rate": 2.7360602949793265e-06,
2152
- "loss": 1.3269,
2153
- "step": 2840
2154
- },
2155
- {
2156
- "epoch": 2.442159383033419,
2157
- "grad_norm": 38.39194869995117,
2158
- "learning_rate": 2.694667552241364e-06,
2159
- "loss": 1.2148,
2160
- "step": 2850
2161
- },
2162
- {
2163
- "epoch": 2.4507283633247643,
2164
- "grad_norm": 16.456998825073242,
2165
- "learning_rate": 2.6532748095034015e-06,
2166
- "loss": 1.2613,
2167
- "step": 2860
2168
- },
2169
- {
2170
- "epoch": 2.4592973436161096,
2171
- "grad_norm": 20.516918182373047,
2172
- "learning_rate": 2.611882066765439e-06,
2173
- "loss": 0.8369,
2174
- "step": 2870
2175
- },
2176
- {
2177
- "epoch": 2.467866323907455,
2178
- "grad_norm": 39.00393295288086,
2179
- "learning_rate": 2.570489324027476e-06,
2180
- "loss": 1.0123,
2181
- "step": 2880
2182
- },
2183
- {
2184
- "epoch": 2.4764353041988003,
2185
- "grad_norm": 55.170372009277344,
2186
- "learning_rate": 2.5290965812895136e-06,
2187
- "loss": 1.2133,
2188
- "step": 2890
2189
- },
2190
- {
2191
- "epoch": 2.4850042844901457,
2192
- "grad_norm": 33.54553985595703,
2193
- "learning_rate": 2.487703838551551e-06,
2194
- "loss": 1.1492,
2195
- "step": 2900
2196
- },
2197
- {
2198
- "epoch": 2.493573264781491,
2199
- "grad_norm": 36.33115768432617,
2200
- "learning_rate": 2.4463110958135886e-06,
2201
- "loss": 1.5454,
2202
- "step": 2910
2203
- },
2204
- {
2205
- "epoch": 2.5021422450728363,
2206
- "grad_norm": 22.1439266204834,
2207
- "learning_rate": 2.4049183530756257e-06,
2208
- "loss": 1.0734,
2209
- "step": 2920
2210
- },
2211
- {
2212
- "epoch": 2.5107112253641817,
2213
- "grad_norm": 13.999841690063477,
2214
- "learning_rate": 2.3635256103376636e-06,
2215
- "loss": 1.3985,
2216
- "step": 2930
2217
- },
2218
- {
2219
- "epoch": 2.519280205655527,
2220
- "grad_norm": 32.42293930053711,
2221
- "learning_rate": 2.3221328675997007e-06,
2222
- "loss": 1.0033,
2223
- "step": 2940
2224
- },
2225
- {
2226
- "epoch": 2.5278491859468724,
2227
- "grad_norm": 43.845096588134766,
2228
- "learning_rate": 2.280740124861738e-06,
2229
- "loss": 1.4406,
2230
- "step": 2950
2231
- },
2232
- {
2233
- "epoch": 2.5364181662382177,
2234
- "grad_norm": 35.83656692504883,
2235
- "learning_rate": 2.2393473821237757e-06,
2236
- "loss": 1.2356,
2237
- "step": 2960
2238
- },
2239
- {
2240
- "epoch": 2.544987146529563,
2241
- "grad_norm": 39.58152770996094,
2242
- "learning_rate": 2.197954639385813e-06,
2243
- "loss": 1.234,
2244
- "step": 2970
2245
- },
2246
- {
2247
- "epoch": 2.5535561268209084,
2248
- "grad_norm": 25.64226722717285,
2249
- "learning_rate": 2.1565618966478503e-06,
2250
- "loss": 1.2667,
2251
- "step": 2980
2252
- },
2253
- {
2254
- "epoch": 2.5621251071122537,
2255
- "grad_norm": 33.893375396728516,
2256
- "learning_rate": 2.115169153909888e-06,
2257
- "loss": 1.3049,
2258
- "step": 2990
2259
- },
2260
- {
2261
- "epoch": 2.570694087403599,
2262
- "grad_norm": 45.953590393066406,
2263
- "learning_rate": 2.0737764111719253e-06,
2264
- "loss": 1.4963,
2265
- "step": 3000
2266
- },
2267
- {
2268
- "epoch": 2.5792630676949444,
2269
- "grad_norm": 41.743202209472656,
2270
- "learning_rate": 2.0323836684339628e-06,
2271
- "loss": 1.2745,
2272
- "step": 3010
2273
- },
2274
- {
2275
- "epoch": 2.5878320479862897,
2276
- "grad_norm": 42.73529052734375,
2277
- "learning_rate": 1.9909909256960003e-06,
2278
- "loss": 1.4276,
2279
- "step": 3020
2280
- },
2281
- {
2282
- "epoch": 2.596401028277635,
2283
- "grad_norm": 17.132654190063477,
2284
- "learning_rate": 1.949598182958038e-06,
2285
- "loss": 0.9052,
2286
- "step": 3030
2287
- },
2288
- {
2289
- "epoch": 2.6049700085689804,
2290
- "grad_norm": 20.87359046936035,
2291
- "learning_rate": 1.908205440220075e-06,
2292
- "loss": 1.0161,
2293
- "step": 3040
2294
- },
2295
- {
2296
- "epoch": 2.6135389888603258,
2297
- "grad_norm": 20.73636245727539,
2298
- "learning_rate": 1.8668126974821126e-06,
2299
- "loss": 1.1655,
2300
- "step": 3050
2301
- },
2302
- {
2303
- "epoch": 2.622107969151671,
2304
- "grad_norm": 30.01190948486328,
2305
- "learning_rate": 1.8254199547441499e-06,
2306
- "loss": 1.3549,
2307
- "step": 3060
2308
- },
2309
- {
2310
- "epoch": 2.6306769494430164,
2311
- "grad_norm": 29.046810150146484,
2312
- "learning_rate": 1.7840272120061874e-06,
2313
- "loss": 1.2511,
2314
- "step": 3070
2315
- },
2316
- {
2317
- "epoch": 2.6392459297343613,
2318
- "grad_norm": 28.32902717590332,
2319
- "learning_rate": 1.7426344692682247e-06,
2320
- "loss": 1.3624,
2321
- "step": 3080
2322
- },
2323
- {
2324
- "epoch": 2.6478149100257067,
2325
- "grad_norm": 36.059173583984375,
2326
- "learning_rate": 1.701241726530262e-06,
2327
- "loss": 1.4086,
2328
- "step": 3090
2329
- },
2330
- {
2331
- "epoch": 2.656383890317052,
2332
- "grad_norm": 46.75105667114258,
2333
- "learning_rate": 1.6598489837922995e-06,
2334
- "loss": 1.3701,
2335
- "step": 3100
2336
- },
2337
- {
2338
- "epoch": 2.6649528706083974,
2339
- "grad_norm": 37.93562316894531,
2340
- "learning_rate": 1.618456241054337e-06,
2341
- "loss": 1.3166,
2342
- "step": 3110
2343
- },
2344
- {
2345
- "epoch": 2.6735218508997427,
2346
- "grad_norm": 47.81494140625,
2347
- "learning_rate": 1.5770634983163743e-06,
2348
- "loss": 1.0946,
2349
- "step": 3120
2350
- },
2351
- {
2352
- "epoch": 2.682090831191088,
2353
- "grad_norm": 36.22248077392578,
2354
- "learning_rate": 1.5356707555784118e-06,
2355
- "loss": 1.3498,
2356
- "step": 3130
2357
- },
2358
- {
2359
- "epoch": 2.6906598114824334,
2360
- "grad_norm": 38.44594955444336,
2361
- "learning_rate": 1.4942780128404493e-06,
2362
- "loss": 1.4325,
2363
- "step": 3140
2364
- },
2365
- {
2366
- "epoch": 2.6992287917737787,
2367
- "grad_norm": 50.71770477294922,
2368
- "learning_rate": 1.4528852701024866e-06,
2369
- "loss": 1.3387,
2370
- "step": 3150
2371
- },
2372
- {
2373
- "epoch": 2.707797772065124,
2374
- "grad_norm": 44.393768310546875,
2375
- "learning_rate": 1.411492527364524e-06,
2376
- "loss": 1.1321,
2377
- "step": 3160
2378
- },
2379
- {
2380
- "epoch": 2.7163667523564694,
2381
- "grad_norm": 29.496095657348633,
2382
- "learning_rate": 1.3700997846265616e-06,
2383
- "loss": 1.1442,
2384
- "step": 3170
2385
- },
2386
- {
2387
- "epoch": 2.7249357326478147,
2388
- "grad_norm": 21.904565811157227,
2389
- "learning_rate": 1.3287070418885989e-06,
2390
- "loss": 1.1811,
2391
- "step": 3180
2392
- },
2393
- {
2394
- "epoch": 2.73350471293916,
2395
- "grad_norm": 42.967010498046875,
2396
- "learning_rate": 1.2873142991506364e-06,
2397
- "loss": 1.195,
2398
- "step": 3190
2399
- },
2400
- {
2401
- "epoch": 2.7420736932305054,
2402
- "grad_norm": 41.519500732421875,
2403
- "learning_rate": 1.2459215564126739e-06,
2404
- "loss": 1.2147,
2405
- "step": 3200
2406
- },
2407
- {
2408
- "epoch": 2.7506426735218508,
2409
- "grad_norm": 57.64628219604492,
2410
- "learning_rate": 1.204528813674711e-06,
2411
- "loss": 1.2644,
2412
- "step": 3210
2413
- },
2414
- {
2415
- "epoch": 2.759211653813196,
2416
- "grad_norm": 75.4446792602539,
2417
- "learning_rate": 1.1631360709367485e-06,
2418
- "loss": 1.198,
2419
- "step": 3220
2420
- },
2421
- {
2422
- "epoch": 2.7677806341045414,
2423
- "grad_norm": 33.584678649902344,
2424
- "learning_rate": 1.121743328198786e-06,
2425
- "loss": 1.2919,
2426
- "step": 3230
2427
- },
2428
- {
2429
- "epoch": 2.776349614395887,
2430
- "grad_norm": 42.54429244995117,
2431
- "learning_rate": 1.0803505854608233e-06,
2432
- "loss": 1.3359,
2433
- "step": 3240
2434
- },
2435
- {
2436
- "epoch": 2.784918594687232,
2437
- "grad_norm": 39.373809814453125,
2438
- "learning_rate": 1.0389578427228608e-06,
2439
- "loss": 1.1131,
2440
- "step": 3250
2441
- },
2442
- {
2443
- "epoch": 2.7934875749785775,
2444
- "grad_norm": 20.756141662597656,
2445
- "learning_rate": 9.97565099984898e-07,
2446
- "loss": 1.0463,
2447
- "step": 3260
2448
- },
2449
- {
2450
- "epoch": 2.802056555269923,
2451
- "grad_norm": 37.37822341918945,
2452
- "learning_rate": 9.561723572469356e-07,
2453
- "loss": 1.2405,
2454
- "step": 3270
2455
- },
2456
- {
2457
- "epoch": 2.810625535561268,
2458
- "grad_norm": 41.48339080810547,
2459
- "learning_rate": 9.14779614508973e-07,
2460
- "loss": 1.3686,
2461
- "step": 3280
2462
- },
2463
- {
2464
- "epoch": 2.8191945158526135,
2465
- "grad_norm": 43.064998626708984,
2466
- "learning_rate": 8.733868717710105e-07,
2467
- "loss": 1.6151,
2468
- "step": 3290
2469
- },
2470
- {
2471
- "epoch": 2.827763496143959,
2472
- "grad_norm": 32.76227569580078,
2473
- "learning_rate": 8.319941290330479e-07,
2474
- "loss": 1.2218,
2475
- "step": 3300
2476
- },
2477
- {
2478
- "epoch": 2.836332476435304,
2479
- "grad_norm": 48.318965911865234,
2480
- "learning_rate": 7.906013862950853e-07,
2481
- "loss": 1.1332,
2482
- "step": 3310
2483
- },
2484
- {
2485
- "epoch": 2.8449014567266495,
2486
- "grad_norm": 76.0499496459961,
2487
- "learning_rate": 7.492086435571228e-07,
2488
- "loss": 1.388,
2489
- "step": 3320
2490
- },
2491
- {
2492
- "epoch": 2.853470437017995,
2493
- "grad_norm": 29.409503936767578,
2494
- "learning_rate": 7.078159008191602e-07,
2495
- "loss": 1.4353,
2496
- "step": 3330
2497
- },
2498
- {
2499
- "epoch": 2.86203941730934,
2500
- "grad_norm": 25.97176170349121,
2501
- "learning_rate": 6.664231580811976e-07,
2502
- "loss": 1.1936,
2503
- "step": 3340
2504
- },
2505
- {
2506
- "epoch": 2.8706083976006855,
2507
- "grad_norm": 31.24247932434082,
2508
- "learning_rate": 6.25030415343235e-07,
2509
- "loss": 0.9297,
2510
- "step": 3350
2511
- },
2512
- {
2513
- "epoch": 2.879177377892031,
2514
- "grad_norm": 44.379268646240234,
2515
- "learning_rate": 5.836376726052725e-07,
2516
- "loss": 1.1583,
2517
- "step": 3360
2518
- },
2519
- {
2520
- "epoch": 2.887746358183376,
2521
- "grad_norm": 35.18986511230469,
2522
- "learning_rate": 5.422449298673099e-07,
2523
- "loss": 1.3518,
2524
- "step": 3370
2525
- },
2526
- {
2527
- "epoch": 2.8963153384747216,
2528
- "grad_norm": 30.07046890258789,
2529
- "learning_rate": 5.008521871293473e-07,
2530
- "loss": 1.1271,
2531
- "step": 3380
2532
- },
2533
- {
2534
- "epoch": 2.904884318766067,
2535
- "grad_norm": 26.008453369140625,
2536
- "learning_rate": 4.594594443913846e-07,
2537
- "loss": 1.0143,
2538
- "step": 3390
2539
- },
2540
- {
2541
- "epoch": 2.9134532990574122,
2542
- "grad_norm": 32.31906509399414,
2543
- "learning_rate": 4.1806670165342206e-07,
2544
- "loss": 1.452,
2545
- "step": 3400
2546
- },
2547
- {
2548
- "epoch": 2.9220222793487576,
2549
- "grad_norm": 36.26763153076172,
2550
- "learning_rate": 3.766739589154595e-07,
2551
- "loss": 1.2661,
2552
- "step": 3410
2553
- },
2554
- {
2555
- "epoch": 2.930591259640103,
2556
- "grad_norm": 33.175926208496094,
2557
- "learning_rate": 3.352812161774969e-07,
2558
- "loss": 1.3136,
2559
- "step": 3420
2560
- },
2561
- {
2562
- "epoch": 2.9391602399314483,
2563
- "grad_norm": 37.862266540527344,
2564
- "learning_rate": 2.938884734395343e-07,
2565
- "loss": 1.1489,
2566
- "step": 3430
2567
- },
2568
- {
2569
- "epoch": 2.9477292202227936,
2570
- "grad_norm": 45.077796936035156,
2571
- "learning_rate": 2.5249573070157175e-07,
2572
- "loss": 1.1297,
2573
- "step": 3440
2574
- },
2575
- {
2576
- "epoch": 2.956298200514139,
2577
- "grad_norm": 32.95222854614258,
2578
- "learning_rate": 2.1110298796360915e-07,
2579
- "loss": 1.2034,
2580
- "step": 3450
2581
- },
2582
- {
2583
- "epoch": 2.9648671808054843,
2584
- "grad_norm": 43.23259735107422,
2585
- "learning_rate": 1.6971024522564658e-07,
2586
- "loss": 0.9911,
2587
- "step": 3460
2588
- },
2589
- {
2590
- "epoch": 2.9734361610968296,
2591
- "grad_norm": 30.152484893798828,
2592
- "learning_rate": 1.28317502487684e-07,
2593
- "loss": 1.0821,
2594
- "step": 3470
2595
- },
2596
- {
2597
- "epoch": 2.982005141388175,
2598
- "grad_norm": 30.485458374023438,
2599
- "learning_rate": 8.692475974972141e-08,
2600
- "loss": 0.9814,
2601
- "step": 3480
2602
- },
2603
- {
2604
- "epoch": 2.9905741216795203,
2605
- "grad_norm": 30.098020553588867,
2606
- "learning_rate": 4.553201701175884e-08,
2607
- "loss": 1.0793,
2608
- "step": 3490
2609
- },
2610
- {
2611
- "epoch": 2.9991431019708656,
2612
- "grad_norm": 70.42772674560547,
2613
- "learning_rate": 4.139274273796258e-09,
2614
- "loss": 1.4007,
2615
- "step": 3500
2616
- },
2617
- {
2618
- "epoch": 3.0,
2619
- "eval_classification_report": {
2620
- "accuracy": 0.3645,
2621
- "ar": {
2622
- "f1-score": 0.3172043010752688,
2623
- "precision": 0.35542168674698793,
2624
- "recall": 0.28640776699029125,
2625
- "support": 206.0
2626
- },
2627
- "cl": {
2628
- "f1-score": 0.23825503355704697,
2629
- "precision": 0.23202614379084968,
2630
- "recall": 0.24482758620689654,
2631
- "support": 290.0
2632
- },
2633
- "co": {
2634
- "f1-score": 0.39937106918238996,
2635
- "precision": 0.3681159420289855,
2636
- "recall": 0.436426116838488,
2637
- "support": 291.0
2638
- },
2639
- "es": {
2640
- "f1-score": 0.35080645161290325,
2641
- "precision": 0.4009216589861751,
2642
- "recall": 0.3118279569892473,
2643
- "support": 279.0
2644
- },
2645
- "macro avg": {
2646
- "f1-score": 0.34262062899860507,
2647
- "precision": 0.3462253190782673,
2648
- "recall": 0.34320174371834444,
2649
- "support": 2000.0
2650
- },
2651
- "mx": {
2652
- "f1-score": 0.4259567387687188,
2653
- "precision": 0.4129032258064516,
2654
- "recall": 0.43986254295532645,
2655
- "support": 291.0
2656
- },
2657
- "pe": {
2658
- "f1-score": 0.3543913713405239,
2659
- "precision": 0.32122905027932963,
2660
- "recall": 0.3951890034364261,
2661
- "support": 291.0
2662
- },
2663
- "pr": {
2664
- "f1-score": 0.6305418719211823,
2665
- "precision": 0.6274509803921569,
2666
- "recall": 0.6336633663366337,
2667
- "support": 101.0
2668
- },
2669
- "uy": {
2670
- "f1-score": 0.36705882352941177,
2671
- "precision": 0.3979591836734694,
2672
- "recall": 0.3406113537117904,
2673
- "support": 229.0
2674
- },
2675
- "ve": {
2676
- "f1-score": 0.0,
2677
- "precision": 0.0,
2678
- "recall": 0.0,
2679
- "support": 22.0
2680
- },
2681
- "weighted avg": {
2682
- "f1-score": 0.36167626328959446,
2683
- "precision": 0.3638105127892991,
2684
- "recall": 0.3645,
2685
- "support": 2000.0
2686
- }
2687
- },
2688
- "eval_f1": 0.34262062899860507,
2689
- "eval_loss": 1.8317162990570068,
2690
- "eval_runtime": 5.6289,
2691
- "eval_samples_per_second": 355.31,
2692
- "eval_steps_per_second": 88.828,
2693
- "step": 3501
2694
- }
2695
- ],
2696
- "logging_steps": 10,
2697
- "max_steps": 3501,
2698
- "num_input_tokens_seen": 0,
2699
- "num_train_epochs": 3,
2700
- "save_steps": 500,
2701
- "stateful_callbacks": {
2702
- "TrainerControl": {
2703
- "args": {
2704
- "should_epoch_stop": false,
2705
- "should_evaluate": false,
2706
- "should_log": false,
2707
- "should_save": true,
2708
- "should_training_stop": true
2709
- },
2710
- "attributes": {}
2711
- }
2712
- },
2713
- "total_flos": 460407503990016.0,
2714
- "train_batch_size": 4,
2715
- "trial_name": null,
2716
- "trial_params": null
2717
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
trial_5/checkpoint-438/model.safetensors CHANGED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:455a1836d4bc83ba7b52be8311c689a7970d706e2bf7247a151c27698f8cd7b1
3
- size 439454740
 
 
 
 
trial_5/checkpoint-438/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fd04b482a544fd0ce40d1f3fc9b79e64e3aac20c06edc3861fb06060a0bfbc6b
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9266ef30f9c03ba585bd4567285194480552bc80c185cae7eee533c54caf9ebb
3
  size 14244
trial_5/checkpoint-438/trainer_state.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
- "best_metric": 0.24538914841155643,
3
- "best_model_checkpoint": "/content/drive/MyDrive/model_outputs/trial_5/checkpoint-438",
4
  "epoch": 3.0,
5
  "eval_steps": 500,
6
  "global_step": 438,
@@ -464,84 +464,6 @@
464
  "learning_rate": 2.0422444096166686e-07,
465
  "loss": 1.8367,
466
  "step": 430
467
- },
468
- {
469
- "epoch": 3.0,
470
- "eval_classification_report": {
471
- "accuracy": 0.2665,
472
- "ar": {
473
- "f1-score": 0.10077519379844961,
474
- "precision": 0.25,
475
- "recall": 0.06310679611650485,
476
- "support": 206.0
477
- },
478
- "cl": {
479
- "f1-score": 0.19504132231404958,
480
- "precision": 0.1873015873015873,
481
- "recall": 0.20344827586206896,
482
- "support": 290.0
483
- },
484
- "co": {
485
- "f1-score": 0.3190348525469169,
486
- "precision": 0.26153846153846155,
487
- "recall": 0.40893470790378006,
488
- "support": 291.0
489
- },
490
- "es": {
491
- "f1-score": 0.3266563944530046,
492
- "precision": 0.2864864864864865,
493
- "recall": 0.37992831541218636,
494
- "support": 279.0
495
- },
496
- "macro avg": {
497
- "f1-score": 0.24538914841155643,
498
- "precision": 0.2739309736044076,
499
- "recall": 0.24660703914377965,
500
- "support": 2000.0
501
- },
502
- "mx": {
503
- "f1-score": 0.2955223880597015,
504
- "precision": 0.2612137203166227,
505
- "recall": 0.3402061855670103,
506
- "support": 291.0
507
- },
508
- "pe": {
509
- "f1-score": 0.18181818181818182,
510
- "precision": 0.1975806451612903,
511
- "recall": 0.16838487972508592,
512
- "support": 291.0
513
- },
514
- "pr": {
515
- "f1-score": 0.5568181818181818,
516
- "precision": 0.6533333333333333,
517
- "recall": 0.48514851485148514,
518
- "support": 101.0
519
- },
520
- "uy": {
521
- "f1-score": 0.23283582089552238,
522
- "precision": 0.36792452830188677,
523
- "recall": 0.1703056768558952,
524
- "support": 229.0
525
- },
526
- "ve": {
527
- "f1-score": 0.0,
528
- "precision": 0.0,
529
- "recall": 0.0,
530
- "support": 22.0
531
- },
532
- "weighted avg": {
533
- "f1-score": 0.25488104736013556,
534
- "precision": 0.27280271317837684,
535
- "recall": 0.2665,
536
- "support": 2000.0
537
- }
538
- },
539
- "eval_f1": 0.24538914841155643,
540
- "eval_loss": 1.9040703773498535,
541
- "eval_runtime": 3.6694,
542
- "eval_samples_per_second": 545.044,
543
- "eval_steps_per_second": 17.169,
544
- "step": 438
545
  }
546
  ],
547
  "logging_steps": 10,
 
1
  {
2
+ "best_metric": 0.23586615753847973,
3
+ "best_model_checkpoint": "/content/drive/MyDrive/model_outputs/trial_5/checkpoint-292",
4
  "epoch": 3.0,
5
  "eval_steps": 500,
6
  "global_step": 438,
 
464
  "learning_rate": 2.0422444096166686e-07,
465
  "loss": 1.8367,
466
  "step": 430
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
467
  }
468
  ],
469
  "logging_steps": 10,