sealad886 commited on
Commit
199be2a
·
verified ·
1 Parent(s): 4a1e66b

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .DS_Store +0 -0
  2. .cache/.DS_Store +0 -0
  3. .gitattributes +118 -35
  4. DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/config.json +0 -0
  5. DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00001-of-00003.safetensors +3 -0
  6. DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00002-of-00003.safetensors +3 -0
  7. DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00003-of-00003.safetensors +3 -0
  8. DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model.safetensors.index.json +0 -0
  9. DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/special_tokens_map.json +23 -0
  10. DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/tokenizer.json +3 -0
  11. DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/tokenizer_config.json +195 -0
  12. DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/config.json +0 -0
  13. DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00001-of-00003.safetensors +3 -0
  14. DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00002-of-00003.safetensors +3 -0
  15. DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00003-of-00003.safetensors +3 -0
  16. DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model.safetensors.index.json +0 -0
  17. DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/special_tokens_map.json +23 -0
  18. DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/tokenizer.json +3 -0
  19. DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/tokenizer_config.json +195 -0
  20. DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/config.json +0 -0
  21. DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00001-of-00003.safetensors +3 -0
  22. DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00002-of-00003.safetensors +3 -0
  23. DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00003-of-00003.safetensors +3 -0
  24. DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model.safetensors.index.json +0 -0
  25. DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/special_tokens_map.json +23 -0
  26. DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/tokenizer.json +3 -0
  27. DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/tokenizer_config.json +195 -0
  28. DeepSeek-R1-Distill-Qwen-32B_3bit/config.json +35 -0
  29. DeepSeek-R1-Distill-Qwen-32B_3bit/model-00001-of-00003.safetensors +3 -0
  30. DeepSeek-R1-Distill-Qwen-32B_3bit/model-00002-of-00003.safetensors +3 -0
  31. DeepSeek-R1-Distill-Qwen-32B_3bit/model-00003-of-00003.safetensors +3 -0
  32. DeepSeek-R1-Distill-Qwen-32B_3bit/model.safetensors.index.json +0 -0
  33. DeepSeek-R1-Distill-Qwen-32B_3bit/special_tokens_map.json +23 -0
  34. DeepSeek-R1-Distill-Qwen-32B_3bit/tokenizer.json +3 -0
  35. DeepSeek-R1-Distill-Qwen-32B_3bit/tokenizer_config.json +195 -0
  36. DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/config.json +0 -0
  37. DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00001-of-00004.safetensors +3 -0
  38. DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00002-of-00004.safetensors +3 -0
  39. DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00003-of-00004.safetensors +3 -0
  40. DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00004-of-00004.safetensors +3 -0
  41. DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model.safetensors.index.json +0 -0
  42. DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/special_tokens_map.json +23 -0
  43. DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/tokenizer.json +3 -0
  44. DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/tokenizer_config.json +195 -0
  45. DeepSeek-R1-Distill-Qwen-32B_6bit/config.json +35 -0
  46. DeepSeek-R1-Distill-Qwen-32B_6bit/model-00001-of-00006.safetensors +3 -0
  47. DeepSeek-R1-Distill-Qwen-32B_6bit/model-00002-of-00006.safetensors +3 -0
  48. DeepSeek-R1-Distill-Qwen-32B_6bit/model-00003-of-00006.safetensors +3 -0
  49. DeepSeek-R1-Distill-Qwen-32B_6bit/model-00004-of-00006.safetensors +3 -0
  50. DeepSeek-R1-Distill-Qwen-32B_6bit/model-00005-of-00006.safetensors +3 -0
.DS_Store ADDED
Binary file (12.3 kB). View file
 
.cache/.DS_Store ADDED
Binary file (6.15 kB). View file
 
.gitattributes CHANGED
@@ -1,38 +1,121 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
- *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
36
  DeepSeek-R1-Distill-Qwen-32B_4bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
  DeepSeek-R1-Distill-Qwen-32B_bfloat16/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
38
  DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ DeepSeek-R1-Distill-Qwen-32B_4bit/model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
2
+ DeepSeek-R1-Distill-Qwen-32B_6bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
3
+ DeepSeek-R1-Distill-Qwen-32B_8bit/model-00003-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
4
+ DeepSeek-R1-Distill-Qwen-32B_2bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
5
+ DeepSeek-R1-Distill-Qwen-32B_3bit_uniform/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
6
+ DeepSeek-R1-Distill-Qwen-32B_4bit/model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
7
+ DeepSeek-R1-Distill-Qwen-32B_8bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
8
+ DeepSeek-R1-Distill-Qwen-32B_8bit/model-00001-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
9
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00001-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
10
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00005-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
11
+ DeepSeek-R1-Distill-Qwen-32B_3bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
12
+ DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00002-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
13
+ DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00005-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
14
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00002-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
15
+ DeepSeek-R1-Distill-Qwen-32B_4bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
16
+ DeepSeek-R1-Distill-Qwen-32B_4bit/model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
17
+ DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00004-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
18
+ DeepSeek-R1-Distill-Qwen-32B_8bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
19
+ DeepSeek-R1-Distill-Qwen-32B_8bit/model-00002-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
20
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00006-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
21
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00007-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
22
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00011-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
23
+ DeepSeek-R1-Distill-Qwen-32B_3bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
24
+ DeepSeek-R1-Distill-Qwen-32B_6bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
25
+ DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00001-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ DeepSeek-R1-Distill-Qwen-32B_8bit/model-00005-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
27
+ DeepSeek-R1-Distill-Qwen-32B_8bit/model-00006-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
28
+ DeepSeek-R1-Distill-Qwen-32B_2bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
29
+ DeepSeek-R1-Distill-Qwen-32B_4bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
30
+ DeepSeek-R1-Distill-Qwen-32B_8bit/model-00007-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
31
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00008-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
32
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00010-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
33
+ DeepSeek-R1-Distill-Qwen-32B_3bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
34
+ DeepSeek-R1-Distill-Qwen-32B_3bit_uniform/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
35
+ DeepSeek-R1-Distill-Qwen-32B_3bit_uniform/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
36
+ DeepSeek-R1-Distill-Qwen-32B_6bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
37
+ DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00006-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
38
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00003-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
39
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00009-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
40
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00013-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
41
+ DeepSeek-R1-Distill-Qwen-32B_2bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
42
+ DeepSeek-R1-Distill-Qwen-32B_4bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
43
+ DeepSeek-R1-Distill-Qwen-32B_4bit/model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
44
+ DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00003-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
45
+ DeepSeek-R1-Distill-Qwen-32B_8bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
46
+ DeepSeek-R1-Distill-Qwen-32B_8bit/model-00004-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
47
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00004-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
48
+ DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00012-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
49
  DeepSeek-R1-Distill-Qwen-32B_4bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
50
  DeepSeek-R1-Distill-Qwen-32B_bfloat16/tokenizer.json filter=lfs diff=lfs merge=lfs -text
51
+ DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
52
+ DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
53
  DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
54
+ DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
55
+ DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
56
+ DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
57
+ DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
58
+ DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
59
+ DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
60
+ DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
61
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00004-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
62
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00001-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
63
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00019-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
64
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00013-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
65
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00016-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
66
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00024-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
67
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00012-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
68
+ DeepSeek-R1-Distill-Qwen-32B_float32/tokenizer.json filter=lfs diff=lfs merge=lfs -text
69
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00005-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
70
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00003-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
71
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00021-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
72
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00025-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
73
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00017-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
74
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00020-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
75
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00018-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
76
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00006-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
77
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00009-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
78
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00002-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
79
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00026-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
80
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00023-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
81
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00014-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
82
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00011-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
83
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00008-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
84
+ DeepSeek-R1-Distill-Qwen-32B_8bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
85
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00007-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
86
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00015-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
87
+ DeepSeek-R1-Distill-Qwen-32B_3bit/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
88
+ DeepSeek-R1-Distill-Qwen-32B_3bit/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
89
+ DeepSeek-R1-Distill-Qwen-32B_3bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
90
+ DeepSeek-R1-Distill-Qwen-32B_3bit/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
91
+ DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
92
+ DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
93
+ DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
94
+ DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
95
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00010-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
96
+ DeepSeek-R1-Distill-Qwen-32B_float32/model-00022-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
97
+ DeepSeek-R1-Distill-Qwen-32B_float16/tokenizer.json filter=lfs diff=lfs merge=lfs -text
98
+ DeepSeek-R1-Distill-Qwen-32B_float16/model-00010-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
99
+ DeepSeek-R1-Distill-Qwen-32B_float16/model-00007-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
100
+ DeepSeek-R1-Distill-Qwen-32B_float16/model-00002-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
101
+ DeepSeek-R1-Distill-Qwen-32B_float16/model-00009-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
102
+ DeepSeek-R1-Distill-Qwen-32B_float16/model-00008-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
103
+ DeepSeek-R1-Distill-Qwen-32B_float16/model-00011-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
104
+ DeepSeek-R1-Distill-Qwen-32B_float16/model-00006-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
105
+ DeepSeek-R1-Distill-Qwen-32B_float16/model-00003-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
106
+ DeepSeek-R1-Distill-Qwen-32B_6bit/model-00002-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
107
+ DeepSeek-R1-Distill-Qwen-32B_6bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
108
+ DeepSeek-R1-Distill-Qwen-32B_6bit/model-00003-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
109
+ DeepSeek-R1-Distill-Qwen-32B_6bit/model-00006-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
110
+ DeepSeek-R1-Distill-Qwen-32B_6bit/model-00005-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
111
+ DeepSeek-R1-Distill-Qwen-32B_6bit/model-00004-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
112
+ DeepSeek-R1-Distill-Qwen-32B_6bit/model-00001-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
113
+ DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
114
+ DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
115
+ DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
116
+ DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
117
+ DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
118
+ DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
119
+ DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
120
+ DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
121
+ DeepSeek-R1-Distill-Qwen-32B_float16/model-00005-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/config.json ADDED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00001-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1fd0562b9c91b2672857fdfb6c8f64d34545b54754352a4289ca624a3b51ec7c
3
+ size 5366073812
DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00002-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3cf3baaa2b1173fdade1e377c93b05e81ed99b04d2ac4f19b55fe0d8f7da429e
3
+ size 5303704839
DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00003-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:36559b03cc3bdb442c6e20d45cfe0550b43ba4c1c8bdc1dc9f33aaf4fbb1265b
3
+ size 940057505
DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/special_tokens_map.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<|begin▁of▁sentence|>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "<|end▁of▁sentence|>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "pad_token": {
17
+ "content": "<|end▁of▁sentence|>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ }
23
+ }
DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
+ size 11422778
DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/tokenizer_config.json ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": true,
3
+ "add_eos_token": false,
4
+ "add_prefix_space": null,
5
+ "added_tokens_decoder": {
6
+ "151643": {
7
+ "content": "<|end▁of▁sentence|>",
8
+ "lstrip": false,
9
+ "normalized": false,
10
+ "rstrip": false,
11
+ "single_word": false,
12
+ "special": true
13
+ },
14
+ "151644": {
15
+ "content": "<|User|>",
16
+ "lstrip": false,
17
+ "normalized": false,
18
+ "rstrip": false,
19
+ "single_word": false,
20
+ "special": false
21
+ },
22
+ "151645": {
23
+ "content": "<|Assistant|>",
24
+ "lstrip": false,
25
+ "normalized": false,
26
+ "rstrip": false,
27
+ "single_word": false,
28
+ "special": false
29
+ },
30
+ "151646": {
31
+ "content": "<|begin▁of▁sentence|>",
32
+ "lstrip": false,
33
+ "normalized": false,
34
+ "rstrip": false,
35
+ "single_word": false,
36
+ "special": true
37
+ },
38
+ "151647": {
39
+ "content": "<|EOT|>",
40
+ "lstrip": false,
41
+ "normalized": false,
42
+ "rstrip": false,
43
+ "single_word": false,
44
+ "special": false
45
+ },
46
+ "151648": {
47
+ "content": "<think>",
48
+ "lstrip": false,
49
+ "normalized": false,
50
+ "rstrip": false,
51
+ "single_word": false,
52
+ "special": false
53
+ },
54
+ "151649": {
55
+ "content": "</think>",
56
+ "lstrip": false,
57
+ "normalized": false,
58
+ "rstrip": false,
59
+ "single_word": false,
60
+ "special": false
61
+ },
62
+ "151650": {
63
+ "content": "<|quad_start|>",
64
+ "lstrip": false,
65
+ "normalized": false,
66
+ "rstrip": false,
67
+ "single_word": false,
68
+ "special": true
69
+ },
70
+ "151651": {
71
+ "content": "<|quad_end|>",
72
+ "lstrip": false,
73
+ "normalized": false,
74
+ "rstrip": false,
75
+ "single_word": false,
76
+ "special": true
77
+ },
78
+ "151652": {
79
+ "content": "<|vision_start|>",
80
+ "lstrip": false,
81
+ "normalized": false,
82
+ "rstrip": false,
83
+ "single_word": false,
84
+ "special": true
85
+ },
86
+ "151653": {
87
+ "content": "<|vision_end|>",
88
+ "lstrip": false,
89
+ "normalized": false,
90
+ "rstrip": false,
91
+ "single_word": false,
92
+ "special": true
93
+ },
94
+ "151654": {
95
+ "content": "<|vision_pad|>",
96
+ "lstrip": false,
97
+ "normalized": false,
98
+ "rstrip": false,
99
+ "single_word": false,
100
+ "special": true
101
+ },
102
+ "151655": {
103
+ "content": "<|image_pad|>",
104
+ "lstrip": false,
105
+ "normalized": false,
106
+ "rstrip": false,
107
+ "single_word": false,
108
+ "special": true
109
+ },
110
+ "151656": {
111
+ "content": "<|video_pad|>",
112
+ "lstrip": false,
113
+ "normalized": false,
114
+ "rstrip": false,
115
+ "single_word": false,
116
+ "special": true
117
+ },
118
+ "151657": {
119
+ "content": "<tool_call>",
120
+ "lstrip": false,
121
+ "normalized": false,
122
+ "rstrip": false,
123
+ "single_word": false,
124
+ "special": false
125
+ },
126
+ "151658": {
127
+ "content": "</tool_call>",
128
+ "lstrip": false,
129
+ "normalized": false,
130
+ "rstrip": false,
131
+ "single_word": false,
132
+ "special": false
133
+ },
134
+ "151659": {
135
+ "content": "<|fim_prefix|>",
136
+ "lstrip": false,
137
+ "normalized": false,
138
+ "rstrip": false,
139
+ "single_word": false,
140
+ "special": false
141
+ },
142
+ "151660": {
143
+ "content": "<|fim_middle|>",
144
+ "lstrip": false,
145
+ "normalized": false,
146
+ "rstrip": false,
147
+ "single_word": false,
148
+ "special": false
149
+ },
150
+ "151661": {
151
+ "content": "<|fim_suffix|>",
152
+ "lstrip": false,
153
+ "normalized": false,
154
+ "rstrip": false,
155
+ "single_word": false,
156
+ "special": false
157
+ },
158
+ "151662": {
159
+ "content": "<|fim_pad|>",
160
+ "lstrip": false,
161
+ "normalized": false,
162
+ "rstrip": false,
163
+ "single_word": false,
164
+ "special": false
165
+ },
166
+ "151663": {
167
+ "content": "<|repo_name|>",
168
+ "lstrip": false,
169
+ "normalized": false,
170
+ "rstrip": false,
171
+ "single_word": false,
172
+ "special": false
173
+ },
174
+ "151664": {
175
+ "content": "<|file_sep|>",
176
+ "lstrip": false,
177
+ "normalized": false,
178
+ "rstrip": false,
179
+ "single_word": false,
180
+ "special": false
181
+ }
182
+ },
183
+ "bos_token": "<|begin▁of▁sentence|>",
184
+ "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
+ "clean_up_tokenization_spaces": false,
186
+ "eos_token": "<|end▁of▁sentence|>",
187
+ "extra_special_tokens": {},
188
+ "legacy": true,
189
+ "model_max_length": 16384,
190
+ "pad_token": "<|end▁of▁sentence|>",
191
+ "sp_model_kwargs": {},
192
+ "tokenizer_class": "LlamaTokenizerFast",
193
+ "unk_token": null,
194
+ "use_default_system_prompt": false
195
+ }
DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/config.json ADDED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00001-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1de09b845c5a77c62056128589fa4686507ffb5edda714fe312c24b90edf8648
3
+ size 5344816524
DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00002-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:86ca2f630b5ee185bd684c721deba18e2728d6939953bd4c7c966af6535ade91
3
+ size 5347005848
DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00003-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b86b6d4e4db81b48b23f54745d66f94c1878e08d356aaae40aae0ba69f7b48c8
3
+ size 4328835017
DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/special_tokens_map.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<|begin▁of▁sentence|>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "<|end▁of▁sentence|>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "pad_token": {
17
+ "content": "<|end▁of▁sentence|>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ }
23
+ }
DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
+ size 11422778
DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/tokenizer_config.json ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": true,
3
+ "add_eos_token": false,
4
+ "add_prefix_space": null,
5
+ "added_tokens_decoder": {
6
+ "151643": {
7
+ "content": "<|end▁of▁sentence|>",
8
+ "lstrip": false,
9
+ "normalized": false,
10
+ "rstrip": false,
11
+ "single_word": false,
12
+ "special": true
13
+ },
14
+ "151644": {
15
+ "content": "<|User|>",
16
+ "lstrip": false,
17
+ "normalized": false,
18
+ "rstrip": false,
19
+ "single_word": false,
20
+ "special": false
21
+ },
22
+ "151645": {
23
+ "content": "<|Assistant|>",
24
+ "lstrip": false,
25
+ "normalized": false,
26
+ "rstrip": false,
27
+ "single_word": false,
28
+ "special": false
29
+ },
30
+ "151646": {
31
+ "content": "<|begin▁of▁sentence|>",
32
+ "lstrip": false,
33
+ "normalized": false,
34
+ "rstrip": false,
35
+ "single_word": false,
36
+ "special": true
37
+ },
38
+ "151647": {
39
+ "content": "<|EOT|>",
40
+ "lstrip": false,
41
+ "normalized": false,
42
+ "rstrip": false,
43
+ "single_word": false,
44
+ "special": false
45
+ },
46
+ "151648": {
47
+ "content": "<think>",
48
+ "lstrip": false,
49
+ "normalized": false,
50
+ "rstrip": false,
51
+ "single_word": false,
52
+ "special": false
53
+ },
54
+ "151649": {
55
+ "content": "</think>",
56
+ "lstrip": false,
57
+ "normalized": false,
58
+ "rstrip": false,
59
+ "single_word": false,
60
+ "special": false
61
+ },
62
+ "151650": {
63
+ "content": "<|quad_start|>",
64
+ "lstrip": false,
65
+ "normalized": false,
66
+ "rstrip": false,
67
+ "single_word": false,
68
+ "special": true
69
+ },
70
+ "151651": {
71
+ "content": "<|quad_end|>",
72
+ "lstrip": false,
73
+ "normalized": false,
74
+ "rstrip": false,
75
+ "single_word": false,
76
+ "special": true
77
+ },
78
+ "151652": {
79
+ "content": "<|vision_start|>",
80
+ "lstrip": false,
81
+ "normalized": false,
82
+ "rstrip": false,
83
+ "single_word": false,
84
+ "special": true
85
+ },
86
+ "151653": {
87
+ "content": "<|vision_end|>",
88
+ "lstrip": false,
89
+ "normalized": false,
90
+ "rstrip": false,
91
+ "single_word": false,
92
+ "special": true
93
+ },
94
+ "151654": {
95
+ "content": "<|vision_pad|>",
96
+ "lstrip": false,
97
+ "normalized": false,
98
+ "rstrip": false,
99
+ "single_word": false,
100
+ "special": true
101
+ },
102
+ "151655": {
103
+ "content": "<|image_pad|>",
104
+ "lstrip": false,
105
+ "normalized": false,
106
+ "rstrip": false,
107
+ "single_word": false,
108
+ "special": true
109
+ },
110
+ "151656": {
111
+ "content": "<|video_pad|>",
112
+ "lstrip": false,
113
+ "normalized": false,
114
+ "rstrip": false,
115
+ "single_word": false,
116
+ "special": true
117
+ },
118
+ "151657": {
119
+ "content": "<tool_call>",
120
+ "lstrip": false,
121
+ "normalized": false,
122
+ "rstrip": false,
123
+ "single_word": false,
124
+ "special": false
125
+ },
126
+ "151658": {
127
+ "content": "</tool_call>",
128
+ "lstrip": false,
129
+ "normalized": false,
130
+ "rstrip": false,
131
+ "single_word": false,
132
+ "special": false
133
+ },
134
+ "151659": {
135
+ "content": "<|fim_prefix|>",
136
+ "lstrip": false,
137
+ "normalized": false,
138
+ "rstrip": false,
139
+ "single_word": false,
140
+ "special": false
141
+ },
142
+ "151660": {
143
+ "content": "<|fim_middle|>",
144
+ "lstrip": false,
145
+ "normalized": false,
146
+ "rstrip": false,
147
+ "single_word": false,
148
+ "special": false
149
+ },
150
+ "151661": {
151
+ "content": "<|fim_suffix|>",
152
+ "lstrip": false,
153
+ "normalized": false,
154
+ "rstrip": false,
155
+ "single_word": false,
156
+ "special": false
157
+ },
158
+ "151662": {
159
+ "content": "<|fim_pad|>",
160
+ "lstrip": false,
161
+ "normalized": false,
162
+ "rstrip": false,
163
+ "single_word": false,
164
+ "special": false
165
+ },
166
+ "151663": {
167
+ "content": "<|repo_name|>",
168
+ "lstrip": false,
169
+ "normalized": false,
170
+ "rstrip": false,
171
+ "single_word": false,
172
+ "special": false
173
+ },
174
+ "151664": {
175
+ "content": "<|file_sep|>",
176
+ "lstrip": false,
177
+ "normalized": false,
178
+ "rstrip": false,
179
+ "single_word": false,
180
+ "special": false
181
+ }
182
+ },
183
+ "bos_token": "<|begin▁of▁sentence|>",
184
+ "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
+ "clean_up_tokenization_spaces": false,
186
+ "eos_token": "<|end▁of▁sentence|>",
187
+ "extra_special_tokens": {},
188
+ "legacy": true,
189
+ "model_max_length": 16384,
190
+ "pad_token": "<|end▁of▁sentence|>",
191
+ "sp_model_kwargs": {},
192
+ "tokenizer_class": "LlamaTokenizerFast",
193
+ "unk_token": null,
194
+ "use_default_system_prompt": false
195
+ }
DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/config.json ADDED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00001-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1de09b845c5a77c62056128589fa4686507ffb5edda714fe312c24b90edf8648
3
+ size 5344816524
DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00002-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:86ca2f630b5ee185bd684c721deba18e2728d6939953bd4c7c966af6535ade91
3
+ size 5347005848
DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00003-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b86b6d4e4db81b48b23f54745d66f94c1878e08d356aaae40aae0ba69f7b48c8
3
+ size 4328835017
DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/special_tokens_map.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<|begin▁of▁sentence|>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "<|end▁of▁sentence|>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "pad_token": {
17
+ "content": "<|end▁of▁sentence|>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ }
23
+ }
DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
+ size 11422778
DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/tokenizer_config.json ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": true,
3
+ "add_eos_token": false,
4
+ "add_prefix_space": null,
5
+ "added_tokens_decoder": {
6
+ "151643": {
7
+ "content": "<|end▁of▁sentence|>",
8
+ "lstrip": false,
9
+ "normalized": false,
10
+ "rstrip": false,
11
+ "single_word": false,
12
+ "special": true
13
+ },
14
+ "151644": {
15
+ "content": "<|User|>",
16
+ "lstrip": false,
17
+ "normalized": false,
18
+ "rstrip": false,
19
+ "single_word": false,
20
+ "special": false
21
+ },
22
+ "151645": {
23
+ "content": "<|Assistant|>",
24
+ "lstrip": false,
25
+ "normalized": false,
26
+ "rstrip": false,
27
+ "single_word": false,
28
+ "special": false
29
+ },
30
+ "151646": {
31
+ "content": "<|begin▁of▁sentence|>",
32
+ "lstrip": false,
33
+ "normalized": false,
34
+ "rstrip": false,
35
+ "single_word": false,
36
+ "special": true
37
+ },
38
+ "151647": {
39
+ "content": "<|EOT|>",
40
+ "lstrip": false,
41
+ "normalized": false,
42
+ "rstrip": false,
43
+ "single_word": false,
44
+ "special": false
45
+ },
46
+ "151648": {
47
+ "content": "<think>",
48
+ "lstrip": false,
49
+ "normalized": false,
50
+ "rstrip": false,
51
+ "single_word": false,
52
+ "special": false
53
+ },
54
+ "151649": {
55
+ "content": "</think>",
56
+ "lstrip": false,
57
+ "normalized": false,
58
+ "rstrip": false,
59
+ "single_word": false,
60
+ "special": false
61
+ },
62
+ "151650": {
63
+ "content": "<|quad_start|>",
64
+ "lstrip": false,
65
+ "normalized": false,
66
+ "rstrip": false,
67
+ "single_word": false,
68
+ "special": true
69
+ },
70
+ "151651": {
71
+ "content": "<|quad_end|>",
72
+ "lstrip": false,
73
+ "normalized": false,
74
+ "rstrip": false,
75
+ "single_word": false,
76
+ "special": true
77
+ },
78
+ "151652": {
79
+ "content": "<|vision_start|>",
80
+ "lstrip": false,
81
+ "normalized": false,
82
+ "rstrip": false,
83
+ "single_word": false,
84
+ "special": true
85
+ },
86
+ "151653": {
87
+ "content": "<|vision_end|>",
88
+ "lstrip": false,
89
+ "normalized": false,
90
+ "rstrip": false,
91
+ "single_word": false,
92
+ "special": true
93
+ },
94
+ "151654": {
95
+ "content": "<|vision_pad|>",
96
+ "lstrip": false,
97
+ "normalized": false,
98
+ "rstrip": false,
99
+ "single_word": false,
100
+ "special": true
101
+ },
102
+ "151655": {
103
+ "content": "<|image_pad|>",
104
+ "lstrip": false,
105
+ "normalized": false,
106
+ "rstrip": false,
107
+ "single_word": false,
108
+ "special": true
109
+ },
110
+ "151656": {
111
+ "content": "<|video_pad|>",
112
+ "lstrip": false,
113
+ "normalized": false,
114
+ "rstrip": false,
115
+ "single_word": false,
116
+ "special": true
117
+ },
118
+ "151657": {
119
+ "content": "<tool_call>",
120
+ "lstrip": false,
121
+ "normalized": false,
122
+ "rstrip": false,
123
+ "single_word": false,
124
+ "special": false
125
+ },
126
+ "151658": {
127
+ "content": "</tool_call>",
128
+ "lstrip": false,
129
+ "normalized": false,
130
+ "rstrip": false,
131
+ "single_word": false,
132
+ "special": false
133
+ },
134
+ "151659": {
135
+ "content": "<|fim_prefix|>",
136
+ "lstrip": false,
137
+ "normalized": false,
138
+ "rstrip": false,
139
+ "single_word": false,
140
+ "special": false
141
+ },
142
+ "151660": {
143
+ "content": "<|fim_middle|>",
144
+ "lstrip": false,
145
+ "normalized": false,
146
+ "rstrip": false,
147
+ "single_word": false,
148
+ "special": false
149
+ },
150
+ "151661": {
151
+ "content": "<|fim_suffix|>",
152
+ "lstrip": false,
153
+ "normalized": false,
154
+ "rstrip": false,
155
+ "single_word": false,
156
+ "special": false
157
+ },
158
+ "151662": {
159
+ "content": "<|fim_pad|>",
160
+ "lstrip": false,
161
+ "normalized": false,
162
+ "rstrip": false,
163
+ "single_word": false,
164
+ "special": false
165
+ },
166
+ "151663": {
167
+ "content": "<|repo_name|>",
168
+ "lstrip": false,
169
+ "normalized": false,
170
+ "rstrip": false,
171
+ "single_word": false,
172
+ "special": false
173
+ },
174
+ "151664": {
175
+ "content": "<|file_sep|>",
176
+ "lstrip": false,
177
+ "normalized": false,
178
+ "rstrip": false,
179
+ "single_word": false,
180
+ "special": false
181
+ }
182
+ },
183
+ "bos_token": "<|begin▁of▁sentence|>",
184
+ "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
+ "clean_up_tokenization_spaces": false,
186
+ "eos_token": "<|end▁of▁sentence|>",
187
+ "extra_special_tokens": {},
188
+ "legacy": true,
189
+ "model_max_length": 16384,
190
+ "pad_token": "<|end▁of▁sentence|>",
191
+ "sp_model_kwargs": {},
192
+ "tokenizer_class": "LlamaTokenizerFast",
193
+ "unk_token": null,
194
+ "use_default_system_prompt": false
195
+ }
DeepSeek-R1-Distill-Qwen-32B_3bit/config.json ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": 151643,
7
+ "eos_token_id": 151643,
8
+ "hidden_act": "silu",
9
+ "hidden_size": 5120,
10
+ "initializer_range": 0.02,
11
+ "intermediate_size": 27648,
12
+ "max_position_embeddings": 131072,
13
+ "max_window_layers": 64,
14
+ "model_type": "qwen2",
15
+ "num_attention_heads": 40,
16
+ "num_hidden_layers": 64,
17
+ "num_key_value_heads": 8,
18
+ "quantization": {
19
+ "group_size": 64,
20
+ "bits": 3
21
+ },
22
+ "quantization_config": {
23
+ "group_size": 64,
24
+ "bits": 3
25
+ },
26
+ "rms_norm_eps": 1e-05,
27
+ "rope_theta": 1000000.0,
28
+ "sliding_window": 131072,
29
+ "tie_word_embeddings": false,
30
+ "torch_dtype": "bfloat16",
31
+ "transformers_version": "4.43.1",
32
+ "use_cache": true,
33
+ "use_sliding_window": false,
34
+ "vocab_size": 152064
35
+ }
DeepSeek-R1-Distill-Qwen-32B_3bit/model-00001-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:df28fc49354643e4296ed66bccf6526153a646ad35a9419981d0910ab6bbc135
3
+ size 5337317621
DeepSeek-R1-Distill-Qwen-32B_3bit/model-00002-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ec2df739f0f8748043700a3654acd764a9e0e613d7a8927a80ee39d1a08cef24
3
+ size 5333936095
DeepSeek-R1-Distill-Qwen-32B_3bit/model-00003-of-00003.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:163ab92116e3a30ff1196343de0559f39f90211cb8e0b30a78727cc57d712597
3
+ size 3664880049
DeepSeek-R1-Distill-Qwen-32B_3bit/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B_3bit/special_tokens_map.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<|begin▁of▁sentence|>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "<|end▁of▁sentence|>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "pad_token": {
17
+ "content": "<|end▁of▁sentence|>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ }
23
+ }
DeepSeek-R1-Distill-Qwen-32B_3bit/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
+ size 11422778
DeepSeek-R1-Distill-Qwen-32B_3bit/tokenizer_config.json ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": true,
3
+ "add_eos_token": false,
4
+ "add_prefix_space": null,
5
+ "added_tokens_decoder": {
6
+ "151643": {
7
+ "content": "<|end▁of▁sentence|>",
8
+ "lstrip": false,
9
+ "normalized": false,
10
+ "rstrip": false,
11
+ "single_word": false,
12
+ "special": true
13
+ },
14
+ "151644": {
15
+ "content": "<|User|>",
16
+ "lstrip": false,
17
+ "normalized": false,
18
+ "rstrip": false,
19
+ "single_word": false,
20
+ "special": false
21
+ },
22
+ "151645": {
23
+ "content": "<|Assistant|>",
24
+ "lstrip": false,
25
+ "normalized": false,
26
+ "rstrip": false,
27
+ "single_word": false,
28
+ "special": false
29
+ },
30
+ "151646": {
31
+ "content": "<|begin▁of▁sentence|>",
32
+ "lstrip": false,
33
+ "normalized": false,
34
+ "rstrip": false,
35
+ "single_word": false,
36
+ "special": true
37
+ },
38
+ "151647": {
39
+ "content": "<|EOT|>",
40
+ "lstrip": false,
41
+ "normalized": false,
42
+ "rstrip": false,
43
+ "single_word": false,
44
+ "special": false
45
+ },
46
+ "151648": {
47
+ "content": "<think>",
48
+ "lstrip": false,
49
+ "normalized": false,
50
+ "rstrip": false,
51
+ "single_word": false,
52
+ "special": false
53
+ },
54
+ "151649": {
55
+ "content": "</think>",
56
+ "lstrip": false,
57
+ "normalized": false,
58
+ "rstrip": false,
59
+ "single_word": false,
60
+ "special": false
61
+ },
62
+ "151650": {
63
+ "content": "<|quad_start|>",
64
+ "lstrip": false,
65
+ "normalized": false,
66
+ "rstrip": false,
67
+ "single_word": false,
68
+ "special": true
69
+ },
70
+ "151651": {
71
+ "content": "<|quad_end|>",
72
+ "lstrip": false,
73
+ "normalized": false,
74
+ "rstrip": false,
75
+ "single_word": false,
76
+ "special": true
77
+ },
78
+ "151652": {
79
+ "content": "<|vision_start|>",
80
+ "lstrip": false,
81
+ "normalized": false,
82
+ "rstrip": false,
83
+ "single_word": false,
84
+ "special": true
85
+ },
86
+ "151653": {
87
+ "content": "<|vision_end|>",
88
+ "lstrip": false,
89
+ "normalized": false,
90
+ "rstrip": false,
91
+ "single_word": false,
92
+ "special": true
93
+ },
94
+ "151654": {
95
+ "content": "<|vision_pad|>",
96
+ "lstrip": false,
97
+ "normalized": false,
98
+ "rstrip": false,
99
+ "single_word": false,
100
+ "special": true
101
+ },
102
+ "151655": {
103
+ "content": "<|image_pad|>",
104
+ "lstrip": false,
105
+ "normalized": false,
106
+ "rstrip": false,
107
+ "single_word": false,
108
+ "special": true
109
+ },
110
+ "151656": {
111
+ "content": "<|video_pad|>",
112
+ "lstrip": false,
113
+ "normalized": false,
114
+ "rstrip": false,
115
+ "single_word": false,
116
+ "special": true
117
+ },
118
+ "151657": {
119
+ "content": "<tool_call>",
120
+ "lstrip": false,
121
+ "normalized": false,
122
+ "rstrip": false,
123
+ "single_word": false,
124
+ "special": false
125
+ },
126
+ "151658": {
127
+ "content": "</tool_call>",
128
+ "lstrip": false,
129
+ "normalized": false,
130
+ "rstrip": false,
131
+ "single_word": false,
132
+ "special": false
133
+ },
134
+ "151659": {
135
+ "content": "<|fim_prefix|>",
136
+ "lstrip": false,
137
+ "normalized": false,
138
+ "rstrip": false,
139
+ "single_word": false,
140
+ "special": false
141
+ },
142
+ "151660": {
143
+ "content": "<|fim_middle|>",
144
+ "lstrip": false,
145
+ "normalized": false,
146
+ "rstrip": false,
147
+ "single_word": false,
148
+ "special": false
149
+ },
150
+ "151661": {
151
+ "content": "<|fim_suffix|>",
152
+ "lstrip": false,
153
+ "normalized": false,
154
+ "rstrip": false,
155
+ "single_word": false,
156
+ "special": false
157
+ },
158
+ "151662": {
159
+ "content": "<|fim_pad|>",
160
+ "lstrip": false,
161
+ "normalized": false,
162
+ "rstrip": false,
163
+ "single_word": false,
164
+ "special": false
165
+ },
166
+ "151663": {
167
+ "content": "<|repo_name|>",
168
+ "lstrip": false,
169
+ "normalized": false,
170
+ "rstrip": false,
171
+ "single_word": false,
172
+ "special": false
173
+ },
174
+ "151664": {
175
+ "content": "<|file_sep|>",
176
+ "lstrip": false,
177
+ "normalized": false,
178
+ "rstrip": false,
179
+ "single_word": false,
180
+ "special": false
181
+ }
182
+ },
183
+ "bos_token": "<|begin▁of▁sentence|>",
184
+ "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
+ "clean_up_tokenization_spaces": false,
186
+ "eos_token": "<|end▁of▁sentence|>",
187
+ "extra_special_tokens": {},
188
+ "legacy": true,
189
+ "model_max_length": 16384,
190
+ "pad_token": "<|end▁of▁sentence|>",
191
+ "sp_model_kwargs": {},
192
+ "tokenizer_class": "LlamaTokenizerFast",
193
+ "unk_token": null,
194
+ "use_default_system_prompt": false
195
+ }
DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/config.json ADDED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00001-of-00004.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7819f2c0378a7ea9896986fa6f2976101d6dd007bc5b22f5023a148d9a5617b6
3
+ size 5321941889
DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00002-of-00004.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bd4062e2e1a76c96699e2bdeaab15ce1a130f14f8ff3cc2625dc450e76bd3e96
3
+ size 5363162608
DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00003-of-00004.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a23c6fb10dca081039ccca54b9e4101820f55ad2669ba6db700e2d1bd54b81b1
3
+ size 5357249052
DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00004-of-00004.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d849115b9ac769b41689460305df28026834a723b2bf002a132247ef170c09f9
3
+ size 5127218901
DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/special_tokens_map.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<|begin▁of▁sentence|>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "<|end▁of▁sentence|>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ },
16
+ "pad_token": {
17
+ "content": "<|end▁of▁sentence|>",
18
+ "lstrip": false,
19
+ "normalized": false,
20
+ "rstrip": false,
21
+ "single_word": false
22
+ }
23
+ }
DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
+ size 11422778
DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/tokenizer_config.json ADDED
@@ -0,0 +1,195 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": true,
3
+ "add_eos_token": false,
4
+ "add_prefix_space": null,
5
+ "added_tokens_decoder": {
6
+ "151643": {
7
+ "content": "<|end▁of▁sentence|>",
8
+ "lstrip": false,
9
+ "normalized": false,
10
+ "rstrip": false,
11
+ "single_word": false,
12
+ "special": true
13
+ },
14
+ "151644": {
15
+ "content": "<|User|>",
16
+ "lstrip": false,
17
+ "normalized": false,
18
+ "rstrip": false,
19
+ "single_word": false,
20
+ "special": false
21
+ },
22
+ "151645": {
23
+ "content": "<|Assistant|>",
24
+ "lstrip": false,
25
+ "normalized": false,
26
+ "rstrip": false,
27
+ "single_word": false,
28
+ "special": false
29
+ },
30
+ "151646": {
31
+ "content": "<|begin▁of▁sentence|>",
32
+ "lstrip": false,
33
+ "normalized": false,
34
+ "rstrip": false,
35
+ "single_word": false,
36
+ "special": true
37
+ },
38
+ "151647": {
39
+ "content": "<|EOT|>",
40
+ "lstrip": false,
41
+ "normalized": false,
42
+ "rstrip": false,
43
+ "single_word": false,
44
+ "special": false
45
+ },
46
+ "151648": {
47
+ "content": "<think>",
48
+ "lstrip": false,
49
+ "normalized": false,
50
+ "rstrip": false,
51
+ "single_word": false,
52
+ "special": false
53
+ },
54
+ "151649": {
55
+ "content": "</think>",
56
+ "lstrip": false,
57
+ "normalized": false,
58
+ "rstrip": false,
59
+ "single_word": false,
60
+ "special": false
61
+ },
62
+ "151650": {
63
+ "content": "<|quad_start|>",
64
+ "lstrip": false,
65
+ "normalized": false,
66
+ "rstrip": false,
67
+ "single_word": false,
68
+ "special": true
69
+ },
70
+ "151651": {
71
+ "content": "<|quad_end|>",
72
+ "lstrip": false,
73
+ "normalized": false,
74
+ "rstrip": false,
75
+ "single_word": false,
76
+ "special": true
77
+ },
78
+ "151652": {
79
+ "content": "<|vision_start|>",
80
+ "lstrip": false,
81
+ "normalized": false,
82
+ "rstrip": false,
83
+ "single_word": false,
84
+ "special": true
85
+ },
86
+ "151653": {
87
+ "content": "<|vision_end|>",
88
+ "lstrip": false,
89
+ "normalized": false,
90
+ "rstrip": false,
91
+ "single_word": false,
92
+ "special": true
93
+ },
94
+ "151654": {
95
+ "content": "<|vision_pad|>",
96
+ "lstrip": false,
97
+ "normalized": false,
98
+ "rstrip": false,
99
+ "single_word": false,
100
+ "special": true
101
+ },
102
+ "151655": {
103
+ "content": "<|image_pad|>",
104
+ "lstrip": false,
105
+ "normalized": false,
106
+ "rstrip": false,
107
+ "single_word": false,
108
+ "special": true
109
+ },
110
+ "151656": {
111
+ "content": "<|video_pad|>",
112
+ "lstrip": false,
113
+ "normalized": false,
114
+ "rstrip": false,
115
+ "single_word": false,
116
+ "special": true
117
+ },
118
+ "151657": {
119
+ "content": "<tool_call>",
120
+ "lstrip": false,
121
+ "normalized": false,
122
+ "rstrip": false,
123
+ "single_word": false,
124
+ "special": false
125
+ },
126
+ "151658": {
127
+ "content": "</tool_call>",
128
+ "lstrip": false,
129
+ "normalized": false,
130
+ "rstrip": false,
131
+ "single_word": false,
132
+ "special": false
133
+ },
134
+ "151659": {
135
+ "content": "<|fim_prefix|>",
136
+ "lstrip": false,
137
+ "normalized": false,
138
+ "rstrip": false,
139
+ "single_word": false,
140
+ "special": false
141
+ },
142
+ "151660": {
143
+ "content": "<|fim_middle|>",
144
+ "lstrip": false,
145
+ "normalized": false,
146
+ "rstrip": false,
147
+ "single_word": false,
148
+ "special": false
149
+ },
150
+ "151661": {
151
+ "content": "<|fim_suffix|>",
152
+ "lstrip": false,
153
+ "normalized": false,
154
+ "rstrip": false,
155
+ "single_word": false,
156
+ "special": false
157
+ },
158
+ "151662": {
159
+ "content": "<|fim_pad|>",
160
+ "lstrip": false,
161
+ "normalized": false,
162
+ "rstrip": false,
163
+ "single_word": false,
164
+ "special": false
165
+ },
166
+ "151663": {
167
+ "content": "<|repo_name|>",
168
+ "lstrip": false,
169
+ "normalized": false,
170
+ "rstrip": false,
171
+ "single_word": false,
172
+ "special": false
173
+ },
174
+ "151664": {
175
+ "content": "<|file_sep|>",
176
+ "lstrip": false,
177
+ "normalized": false,
178
+ "rstrip": false,
179
+ "single_word": false,
180
+ "special": false
181
+ }
182
+ },
183
+ "bos_token": "<|begin▁of▁sentence|>",
184
+ "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
+ "clean_up_tokenization_spaces": false,
186
+ "eos_token": "<|end▁of▁sentence|>",
187
+ "extra_special_tokens": {},
188
+ "legacy": true,
189
+ "model_max_length": 16384,
190
+ "pad_token": "<|end▁of▁sentence|>",
191
+ "sp_model_kwargs": {},
192
+ "tokenizer_class": "LlamaTokenizerFast",
193
+ "unk_token": null,
194
+ "use_default_system_prompt": false
195
+ }
DeepSeek-R1-Distill-Qwen-32B_6bit/config.json ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": 151643,
7
+ "eos_token_id": 151643,
8
+ "hidden_act": "silu",
9
+ "hidden_size": 5120,
10
+ "initializer_range": 0.02,
11
+ "intermediate_size": 27648,
12
+ "max_position_embeddings": 131072,
13
+ "max_window_layers": 64,
14
+ "model_type": "qwen2",
15
+ "num_attention_heads": 40,
16
+ "num_hidden_layers": 64,
17
+ "num_key_value_heads": 8,
18
+ "quantization": {
19
+ "group_size": 64,
20
+ "bits": 6
21
+ },
22
+ "quantization_config": {
23
+ "group_size": 64,
24
+ "bits": 6
25
+ },
26
+ "rms_norm_eps": 1e-05,
27
+ "rope_theta": 1000000.0,
28
+ "sliding_window": 131072,
29
+ "tie_word_embeddings": false,
30
+ "torch_dtype": "bfloat16",
31
+ "transformers_version": "4.43.1",
32
+ "use_cache": true,
33
+ "use_sliding_window": false,
34
+ "vocab_size": 152064
35
+ }
DeepSeek-R1-Distill-Qwen-32B_6bit/model-00001-of-00006.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9a00935ed67ac7e66d142d5f9e4815eb8ab3f961faa7b8bcbca55a4ff2f4ee25
3
+ size 5271984209
DeepSeek-R1-Distill-Qwen-32B_6bit/model-00002-of-00006.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5e09753f33189775b86554c9737eb5574dbd556dd476966ee69cccdb0eb89517
3
+ size 5316808320
DeepSeek-R1-Distill-Qwen-32B_6bit/model-00003-of-00006.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7c40d6060b390eac2274cb2d18973e8fc4ef5747e4d5ae60b2c7a6f0fc8505fb
3
+ size 5265653517
DeepSeek-R1-Distill-Qwen-32B_6bit/model-00004-of-00006.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f81a6a9069dda4ded4f7043ef63966d113d9534180fa561fa25ffa21ab646479
3
+ size 5265653541
DeepSeek-R1-Distill-Qwen-32B_6bit/model-00005-of-00006.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7a88fe01f535a1e4316b000d0f5322482e33e598003500c39bf174a4cee86ed1
3
+ size 4869481603