Multiple quants for MLX framework

#23
by sealad886 - opened
This view is limited to 50 files because it contains too many changes.  See the raw diff here.
Files changed (50) hide show
  1. .gitattributes +35 -196
  2. DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/config.json +0 -0
  3. DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model-00001-of-00003.safetensors +0 -3
  4. DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model-00002-of-00003.safetensors +0 -3
  5. DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model-00003-of-00003.safetensors +0 -3
  6. DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model.safetensors.index.json +0 -0
  7. DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/config.json +0 -0
  8. DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model-00001-of-00003.safetensors +0 -3
  9. DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model-00002-of-00003.safetensors +0 -3
  10. DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model-00003-of-00003.safetensors +0 -3
  11. DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model.safetensors.index.json +0 -0
  12. DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/config.json +0 -0
  13. DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model-00001-of-00003.safetensors +0 -3
  14. DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model-00002-of-00003.safetensors +0 -3
  15. DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model-00003-of-00003.safetensors +0 -3
  16. DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model.safetensors.index.json +0 -0
  17. DeepSeek-R1-Distill-Qwen-32B-3bit/config.json +0 -35
  18. DeepSeek-R1-Distill-Qwen-32B-3bit/model-00001-of-00003.safetensors +0 -3
  19. DeepSeek-R1-Distill-Qwen-32B-3bit/model-00002-of-00003.safetensors +0 -3
  20. DeepSeek-R1-Distill-Qwen-32B-3bit/model-00003-of-00003.safetensors +0 -3
  21. DeepSeek-R1-Distill-Qwen-32B-3bit/model.safetensors.index.json +0 -0
  22. DeepSeek-R1-Distill-Qwen-32B-3bit/special_tokens_map.json +0 -23
  23. DeepSeek-R1-Distill-Qwen-32B-3bit/tokenizer.json +0 -3
  24. DeepSeek-R1-Distill-Qwen-32B-3bit/tokenizer_config.json +0 -195
  25. DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/special_tokens_map.json +0 -23
  26. DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/tokenizer.json +0 -3
  27. DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/tokenizer_config.json +0 -195
  28. DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/config.json +0 -0
  29. DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00001-of-00004.safetensors +0 -3
  30. DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00002-of-00004.safetensors +0 -3
  31. DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00003-of-00004.safetensors +0 -3
  32. DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00004-of-00004.safetensors +0 -3
  33. DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model.safetensors.index.json +0 -0
  34. DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/special_tokens_map.json +0 -23
  35. DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/tokenizer.json +0 -3
  36. DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/tokenizer_config.json +0 -195
  37. DeepSeek-R1-Distill-Qwen-32B-4bit/special_tokens_map.json +0 -23
  38. DeepSeek-R1-Distill-Qwen-32B-4bit/tokenizer.json +0 -3
  39. DeepSeek-R1-Distill-Qwen-32B-4bit/tokenizer_config.json +0 -195
  40. DeepSeek-R1-Distill-Qwen-32B-6bit/config.json +0 -35
  41. DeepSeek-R1-Distill-Qwen-32B-6bit/model-00001-of-00006.safetensors +0 -3
  42. DeepSeek-R1-Distill-Qwen-32B-6bit/model-00002-of-00006.safetensors +0 -3
  43. DeepSeek-R1-Distill-Qwen-32B-6bit/model-00003-of-00006.safetensors +0 -3
  44. DeepSeek-R1-Distill-Qwen-32B-6bit/model-00004-of-00006.safetensors +0 -3
  45. DeepSeek-R1-Distill-Qwen-32B-6bit/model-00005-of-00006.safetensors +0 -3
  46. DeepSeek-R1-Distill-Qwen-32B-6bit/model-00006-of-00006.safetensors +0 -3
  47. DeepSeek-R1-Distill-Qwen-32B-6bit/model.safetensors.index.json +0 -0
  48. DeepSeek-R1-Distill-Qwen-32B-6bit/special_tokens_map.json +0 -23
  49. DeepSeek-R1-Distill-Qwen-32B-6bit/tokenizer.json +0 -3
  50. DeepSeek-R1-Distill-Qwen-32B-6bit/tokenizer_config.json +0 -195
.gitattributes CHANGED
@@ -1,199 +1,38 @@
1
- DeepSeek-R1-Distill-Qwen-32B_4bit/model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
2
- DeepSeek-R1-Distill-Qwen-32B_6bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
3
- DeepSeek-R1-Distill-Qwen-32B_8bit/model-00003-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
4
- DeepSeek-R1-Distill-Qwen-32B_2bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
5
- DeepSeek-R1-Distill-Qwen-32B_3bit_uniform/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
6
- DeepSeek-R1-Distill-Qwen-32B_4bit/model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
7
- DeepSeek-R1-Distill-Qwen-32B_8bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
8
- DeepSeek-R1-Distill-Qwen-32B_8bit/model-00001-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
9
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00001-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
10
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00005-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
11
- DeepSeek-R1-Distill-Qwen-32B_3bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
12
- DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00002-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
13
- DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00005-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
14
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00002-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
15
- DeepSeek-R1-Distill-Qwen-32B_4bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
16
- DeepSeek-R1-Distill-Qwen-32B_4bit/model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
17
- DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00004-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
18
- DeepSeek-R1-Distill-Qwen-32B_8bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
19
- DeepSeek-R1-Distill-Qwen-32B_8bit/model-00002-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
20
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00006-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
21
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00007-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
22
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00011-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
23
- DeepSeek-R1-Distill-Qwen-32B_3bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
24
- DeepSeek-R1-Distill-Qwen-32B_6bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
25
- DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00001-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
26
- DeepSeek-R1-Distill-Qwen-32B_8bit/model-00005-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
27
- DeepSeek-R1-Distill-Qwen-32B_8bit/model-00006-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
28
- DeepSeek-R1-Distill-Qwen-32B_2bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
29
- DeepSeek-R1-Distill-Qwen-32B_4bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
30
- DeepSeek-R1-Distill-Qwen-32B_8bit/model-00007-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
31
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00008-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
32
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00010-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
33
- DeepSeek-R1-Distill-Qwen-32B_3bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
34
- DeepSeek-R1-Distill-Qwen-32B_3bit_uniform/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
35
- DeepSeek-R1-Distill-Qwen-32B_3bit_uniform/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
36
- DeepSeek-R1-Distill-Qwen-32B_6bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
37
- DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00006-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
38
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00003-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
39
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00009-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
40
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00013-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
41
- DeepSeek-R1-Distill-Qwen-32B_2bit_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
42
- DeepSeek-R1-Distill-Qwen-32B_4bit_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
43
- DeepSeek-R1-Distill-Qwen-32B_4bit/model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
44
- DeepSeek-R1-Distill-Qwen-32B_6bit_uniform/model-00003-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
45
- DeepSeek-R1-Distill-Qwen-32B_8bit_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
46
- DeepSeek-R1-Distill-Qwen-32B_8bit/model-00004-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
47
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00004-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
48
- DeepSeek-R1-Distill-Qwen-32B_bfloat16/model-00012-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
49
  DeepSeek-R1-Distill-Qwen-32B_4bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
50
  DeepSeek-R1-Distill-Qwen-32B_bfloat16/tokenizer.json filter=lfs diff=lfs merge=lfs -text
51
- DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
52
- DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
53
  DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
54
- DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
55
- DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
56
- DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
57
- DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
58
- DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
59
- DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
60
- DeepSeek-R1-Distill-Qwen-32B_4,8_mixed/model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
61
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00004-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
62
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00001-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
63
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00019-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
64
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00013-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
65
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00016-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
66
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00024-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
67
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00012-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
68
- DeepSeek-R1-Distill-Qwen-32B_float32/tokenizer.json filter=lfs diff=lfs merge=lfs -text
69
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00005-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
70
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00003-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
71
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00021-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
72
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00025-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
73
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00017-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
74
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00020-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
75
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00018-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
76
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00006-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
77
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00009-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
78
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00002-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
79
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00026-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
80
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00023-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
81
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00014-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
82
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00011-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
83
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00008-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
84
- DeepSeek-R1-Distill-Qwen-32B_8bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
85
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00007-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
86
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00015-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
87
- DeepSeek-R1-Distill-Qwen-32B_3bit/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
88
- DeepSeek-R1-Distill-Qwen-32B_3bit/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
89
- DeepSeek-R1-Distill-Qwen-32B_3bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
90
- DeepSeek-R1-Distill-Qwen-32B_3bit/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
91
- DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
92
- DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
93
- DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
94
- DeepSeek-R1-Distill-Qwen-32B_2,6_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
95
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00010-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
96
- DeepSeek-R1-Distill-Qwen-32B_float32/model-00022-of-00026.safetensors filter=lfs diff=lfs merge=lfs -text
97
- DeepSeek-R1-Distill-Qwen-32B_float16/tokenizer.json filter=lfs diff=lfs merge=lfs -text
98
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00010-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
99
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00007-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
100
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00002-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
101
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00009-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
102
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00008-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
103
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00011-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
104
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00006-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
105
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00003-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
106
- DeepSeek-R1-Distill-Qwen-32B_6bit/model-00002-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
107
- DeepSeek-R1-Distill-Qwen-32B_6bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
108
- DeepSeek-R1-Distill-Qwen-32B_6bit/model-00003-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
109
- DeepSeek-R1-Distill-Qwen-32B_6bit/model-00006-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
110
- DeepSeek-R1-Distill-Qwen-32B_6bit/model-00005-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
111
- DeepSeek-R1-Distill-Qwen-32B_6bit/model-00004-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
112
- DeepSeek-R1-Distill-Qwen-32B_6bit/model-00001-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
113
- DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
114
- DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
115
- DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
116
- DeepSeek-R1-Distill-Qwen-32B_3,4_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
117
- DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
118
- DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
119
- DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
120
- DeepSeek-R1-Distill-Qwen-32B_3,6_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
121
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00005-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
122
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00013-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
123
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00001-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
124
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00012-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
125
- DeepSeek-R1-Distill-Qwen-32B_float16/model-00004-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
126
- DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
127
- DeepSeek-R1-Distill-Qwen-32B-4bit/model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
128
- DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
129
- DeepSeek-R1-Distill-Qwen-32B-4bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
130
- DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
131
- DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
132
- DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
133
- DeepSeek-R1-Distill-Qwen-32B-6bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
134
- DeepSeek-R1-Distill-Qwen-32B-4bit/model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
135
- DeepSeek-R1-Distill-Qwen-32B-6bit/model-00006-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
136
- DeepSeek-R1-Distill-Qwen-32B-4bit/model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
137
- DeepSeek-R1-Distill-Qwen-32B-4bit/model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
138
- DeepSeek-R1-Distill-Qwen-32B-6bit/model-00002-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
139
- DeepSeek-R1-Distill-Qwen-32B-6bit/model-00003-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
140
- DeepSeek-R1-Distill-Qwen-32B-6bit/model-00005-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
141
- DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
142
- DeepSeek-R1-Distill-Qwen-32B-6bit/model-00004-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
143
- DeepSeek-R1-Distill-Qwen-32B-6bit/model-00001-of-00006.safetensors filter=lfs diff=lfs merge=lfs -text
144
- DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
145
- DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
146
- DeepSeek-R1-Distill-Qwen-32B-3bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
147
- DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
148
- DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
149
- DeepSeek-R1-Distill-Qwen-32B-3bit/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
150
- DeepSeek-R1-Distill-Qwen-32B-3bit/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
151
- DeepSeek-R1-Distill-Qwen-32B-3bit/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
152
- DeepSeek-R1-Distill-Qwen-32B-float16/tokenizer.json filter=lfs diff=lfs merge=lfs -text
153
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00010-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
154
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00007-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
155
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00002-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
156
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00008-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
157
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00011-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
158
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00009-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
159
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00006-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
160
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00003-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
161
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00005-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
162
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00012-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
163
- DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
164
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00001-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
165
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00013-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
166
- DeepSeek-R1-Distill-Qwen-32B-float16/model-00004-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
167
- DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
168
- DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
169
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00010-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
170
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00007-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
171
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/tokenizer.json filter=lfs diff=lfs merge=lfs -text
172
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00002-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
173
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00008-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
174
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00009-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
175
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00011-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
176
- DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
177
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00006-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
178
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00003-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
179
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00005-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
180
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00013-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
181
- DeepSeek-R1-Distill-Qwen-32B-8bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
182
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00012-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
183
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00001-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
184
- DeepSeek-R1-Distill-Qwen-32B-bfloat16/model-00004-of-00013.safetensors filter=lfs diff=lfs merge=lfs -text
185
- DeepSeek-R1-Distill-Qwen-32B-8bit/model-00005-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
186
- DeepSeek-R1-Distill-Qwen-32B-8bit/model-00007-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
187
- DeepSeek-R1-Distill-Qwen-32B-8bit/model-00001-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
188
- DeepSeek-R1-Distill-Qwen-32B-8bit/model-00004-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
189
- DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
190
- DeepSeek-R1-Distill-Qwen-32B-8bit/model-00002-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
191
- DeepSeek-R1-Distill-Qwen-32B-8bit/model-00006-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
192
- DeepSeek-R1-Distill-Qwen-32B-8bit/model-00003-of-00007.safetensors filter=lfs diff=lfs merge=lfs -text
193
- DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
194
- DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
195
- DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
196
- DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model-00003-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
197
- DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model-00002-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
198
- DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
199
- DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model-00001-of-00003.safetensors filter=lfs diff=lfs merge=lfs -text
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
36
  DeepSeek-R1-Distill-Qwen-32B_4bit/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
  DeepSeek-R1-Distill-Qwen-32B_bfloat16/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
38
  DeepSeek-R1-Distill-Qwen-32B_4,6_mixed/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/config.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model-00001-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:1fd0562b9c91b2672857fdfb6c8f64d34545b54754352a4289ca624a3b51ec7c
3
- size 5366073812
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model-00002-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:3cf3baaa2b1173fdade1e377c93b05e81ed99b04d2ac4f19b55fe0d8f7da429e
3
- size 5303704839
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model-00003-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:36559b03cc3bdb442c6e20d45cfe0550b43ba4c1c8bdc1dc9f33aaf4fbb1265b
3
- size 940057505
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-2,6_mixed/model.safetensors.index.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/config.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model-00001-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:1de09b845c5a77c62056128589fa4686507ffb5edda714fe312c24b90edf8648
3
- size 5344816524
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model-00002-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:86ca2f630b5ee185bd684c721deba18e2728d6939953bd4c7c966af6535ade91
3
- size 5347005848
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model-00003-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:b86b6d4e4db81b48b23f54745d66f94c1878e08d356aaae40aae0ba69f7b48c8
3
- size 4328835017
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3,4_mixed/model.safetensors.index.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/config.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model-00001-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:1de09b845c5a77c62056128589fa4686507ffb5edda714fe312c24b90edf8648
3
- size 5344816524
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model-00002-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:86ca2f630b5ee185bd684c721deba18e2728d6939953bd4c7c966af6535ade91
3
- size 5347005848
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model-00003-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:b86b6d4e4db81b48b23f54745d66f94c1878e08d356aaae40aae0ba69f7b48c8
3
- size 4328835017
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3,6_mixed/model.safetensors.index.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-3bit/config.json DELETED
@@ -1,35 +0,0 @@
1
- {
2
- "architectures": [
3
- "Qwen2ForCausalLM"
4
- ],
5
- "attention_dropout": 0.0,
6
- "bos_token_id": 151643,
7
- "eos_token_id": 151643,
8
- "hidden_act": "silu",
9
- "hidden_size": 5120,
10
- "initializer_range": 0.02,
11
- "intermediate_size": 27648,
12
- "max_position_embeddings": 131072,
13
- "max_window_layers": 64,
14
- "model_type": "qwen2",
15
- "num_attention_heads": 40,
16
- "num_hidden_layers": 64,
17
- "num_key_value_heads": 8,
18
- "quantization": {
19
- "group_size": 64,
20
- "bits": 3
21
- },
22
- "quantization_config": {
23
- "group_size": 64,
24
- "bits": 3
25
- },
26
- "rms_norm_eps": 1e-05,
27
- "rope_theta": 1000000.0,
28
- "sliding_window": 131072,
29
- "tie_word_embeddings": false,
30
- "torch_dtype": "bfloat16",
31
- "transformers_version": "4.43.1",
32
- "use_cache": true,
33
- "use_sliding_window": false,
34
- "vocab_size": 152064
35
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3bit/model-00001-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:df28fc49354643e4296ed66bccf6526153a646ad35a9419981d0910ab6bbc135
3
- size 5337317621
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3bit/model-00002-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:ec2df739f0f8748043700a3654acd764a9e0e613d7a8927a80ee39d1a08cef24
3
- size 5333936095
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3bit/model-00003-of-00003.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:163ab92116e3a30ff1196343de0559f39f90211cb8e0b30a78727cc57d712597
3
- size 3664880049
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3bit/model.safetensors.index.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-3bit/special_tokens_map.json DELETED
@@ -1,23 +0,0 @@
1
- {
2
- "bos_token": {
3
- "content": "<|begin▁of▁sentence|>",
4
- "lstrip": false,
5
- "normalized": false,
6
- "rstrip": false,
7
- "single_word": false
8
- },
9
- "eos_token": {
10
- "content": "<|end▁of▁sentence|>",
11
- "lstrip": false,
12
- "normalized": false,
13
- "rstrip": false,
14
- "single_word": false
15
- },
16
- "pad_token": {
17
- "content": "<|end▁of▁sentence|>",
18
- "lstrip": false,
19
- "normalized": false,
20
- "rstrip": false,
21
- "single_word": false
22
- }
23
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3bit/tokenizer.json DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
- size 11422778
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-3bit/tokenizer_config.json DELETED
@@ -1,195 +0,0 @@
1
- {
2
- "add_bos_token": true,
3
- "add_eos_token": false,
4
- "add_prefix_space": null,
5
- "added_tokens_decoder": {
6
- "151643": {
7
- "content": "<|end▁of▁sentence|>",
8
- "lstrip": false,
9
- "normalized": false,
10
- "rstrip": false,
11
- "single_word": false,
12
- "special": true
13
- },
14
- "151644": {
15
- "content": "<|User|>",
16
- "lstrip": false,
17
- "normalized": false,
18
- "rstrip": false,
19
- "single_word": false,
20
- "special": false
21
- },
22
- "151645": {
23
- "content": "<|Assistant|>",
24
- "lstrip": false,
25
- "normalized": false,
26
- "rstrip": false,
27
- "single_word": false,
28
- "special": false
29
- },
30
- "151646": {
31
- "content": "<|begin▁of▁sentence|>",
32
- "lstrip": false,
33
- "normalized": false,
34
- "rstrip": false,
35
- "single_word": false,
36
- "special": true
37
- },
38
- "151647": {
39
- "content": "<|EOT|>",
40
- "lstrip": false,
41
- "normalized": false,
42
- "rstrip": false,
43
- "single_word": false,
44
- "special": false
45
- },
46
- "151648": {
47
- "content": "<think>",
48
- "lstrip": false,
49
- "normalized": false,
50
- "rstrip": false,
51
- "single_word": false,
52
- "special": false
53
- },
54
- "151649": {
55
- "content": "</think>",
56
- "lstrip": false,
57
- "normalized": false,
58
- "rstrip": false,
59
- "single_word": false,
60
- "special": false
61
- },
62
- "151650": {
63
- "content": "<|quad_start|>",
64
- "lstrip": false,
65
- "normalized": false,
66
- "rstrip": false,
67
- "single_word": false,
68
- "special": true
69
- },
70
- "151651": {
71
- "content": "<|quad_end|>",
72
- "lstrip": false,
73
- "normalized": false,
74
- "rstrip": false,
75
- "single_word": false,
76
- "special": true
77
- },
78
- "151652": {
79
- "content": "<|vision_start|>",
80
- "lstrip": false,
81
- "normalized": false,
82
- "rstrip": false,
83
- "single_word": false,
84
- "special": true
85
- },
86
- "151653": {
87
- "content": "<|vision_end|>",
88
- "lstrip": false,
89
- "normalized": false,
90
- "rstrip": false,
91
- "single_word": false,
92
- "special": true
93
- },
94
- "151654": {
95
- "content": "<|vision_pad|>",
96
- "lstrip": false,
97
- "normalized": false,
98
- "rstrip": false,
99
- "single_word": false,
100
- "special": true
101
- },
102
- "151655": {
103
- "content": "<|image_pad|>",
104
- "lstrip": false,
105
- "normalized": false,
106
- "rstrip": false,
107
- "single_word": false,
108
- "special": true
109
- },
110
- "151656": {
111
- "content": "<|video_pad|>",
112
- "lstrip": false,
113
- "normalized": false,
114
- "rstrip": false,
115
- "single_word": false,
116
- "special": true
117
- },
118
- "151657": {
119
- "content": "<tool_call>",
120
- "lstrip": false,
121
- "normalized": false,
122
- "rstrip": false,
123
- "single_word": false,
124
- "special": false
125
- },
126
- "151658": {
127
- "content": "</tool_call>",
128
- "lstrip": false,
129
- "normalized": false,
130
- "rstrip": false,
131
- "single_word": false,
132
- "special": false
133
- },
134
- "151659": {
135
- "content": "<|fim_prefix|>",
136
- "lstrip": false,
137
- "normalized": false,
138
- "rstrip": false,
139
- "single_word": false,
140
- "special": false
141
- },
142
- "151660": {
143
- "content": "<|fim_middle|>",
144
- "lstrip": false,
145
- "normalized": false,
146
- "rstrip": false,
147
- "single_word": false,
148
- "special": false
149
- },
150
- "151661": {
151
- "content": "<|fim_suffix|>",
152
- "lstrip": false,
153
- "normalized": false,
154
- "rstrip": false,
155
- "single_word": false,
156
- "special": false
157
- },
158
- "151662": {
159
- "content": "<|fim_pad|>",
160
- "lstrip": false,
161
- "normalized": false,
162
- "rstrip": false,
163
- "single_word": false,
164
- "special": false
165
- },
166
- "151663": {
167
- "content": "<|repo_name|>",
168
- "lstrip": false,
169
- "normalized": false,
170
- "rstrip": false,
171
- "single_word": false,
172
- "special": false
173
- },
174
- "151664": {
175
- "content": "<|file_sep|>",
176
- "lstrip": false,
177
- "normalized": false,
178
- "rstrip": false,
179
- "single_word": false,
180
- "special": false
181
- }
182
- },
183
- "bos_token": "<|begin▁of▁sentence|>",
184
- "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
- "clean_up_tokenization_spaces": false,
186
- "eos_token": "<|end▁of▁sentence|>",
187
- "extra_special_tokens": {},
188
- "legacy": true,
189
- "model_max_length": 16384,
190
- "pad_token": "<|end▁of▁sentence|>",
191
- "sp_model_kwargs": {},
192
- "tokenizer_class": "LlamaTokenizerFast",
193
- "unk_token": null,
194
- "use_default_system_prompt": false
195
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/special_tokens_map.json DELETED
@@ -1,23 +0,0 @@
1
- {
2
- "bos_token": {
3
- "content": "<|begin▁of▁sentence|>",
4
- "lstrip": false,
5
- "normalized": false,
6
- "rstrip": false,
7
- "single_word": false
8
- },
9
- "eos_token": {
10
- "content": "<|end▁of▁sentence|>",
11
- "lstrip": false,
12
- "normalized": false,
13
- "rstrip": false,
14
- "single_word": false
15
- },
16
- "pad_token": {
17
- "content": "<|end▁of▁sentence|>",
18
- "lstrip": false,
19
- "normalized": false,
20
- "rstrip": false,
21
- "single_word": false
22
- }
23
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/tokenizer.json DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
- size 11422778
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,6_mixed/tokenizer_config.json DELETED
@@ -1,195 +0,0 @@
1
- {
2
- "add_bos_token": true,
3
- "add_eos_token": false,
4
- "add_prefix_space": null,
5
- "added_tokens_decoder": {
6
- "151643": {
7
- "content": "<|end▁of▁sentence|>",
8
- "lstrip": false,
9
- "normalized": false,
10
- "rstrip": false,
11
- "single_word": false,
12
- "special": true
13
- },
14
- "151644": {
15
- "content": "<|User|>",
16
- "lstrip": false,
17
- "normalized": false,
18
- "rstrip": false,
19
- "single_word": false,
20
- "special": false
21
- },
22
- "151645": {
23
- "content": "<|Assistant|>",
24
- "lstrip": false,
25
- "normalized": false,
26
- "rstrip": false,
27
- "single_word": false,
28
- "special": false
29
- },
30
- "151646": {
31
- "content": "<|begin▁of▁sentence|>",
32
- "lstrip": false,
33
- "normalized": false,
34
- "rstrip": false,
35
- "single_word": false,
36
- "special": true
37
- },
38
- "151647": {
39
- "content": "<|EOT|>",
40
- "lstrip": false,
41
- "normalized": false,
42
- "rstrip": false,
43
- "single_word": false,
44
- "special": false
45
- },
46
- "151648": {
47
- "content": "<think>",
48
- "lstrip": false,
49
- "normalized": false,
50
- "rstrip": false,
51
- "single_word": false,
52
- "special": false
53
- },
54
- "151649": {
55
- "content": "</think>",
56
- "lstrip": false,
57
- "normalized": false,
58
- "rstrip": false,
59
- "single_word": false,
60
- "special": false
61
- },
62
- "151650": {
63
- "content": "<|quad_start|>",
64
- "lstrip": false,
65
- "normalized": false,
66
- "rstrip": false,
67
- "single_word": false,
68
- "special": true
69
- },
70
- "151651": {
71
- "content": "<|quad_end|>",
72
- "lstrip": false,
73
- "normalized": false,
74
- "rstrip": false,
75
- "single_word": false,
76
- "special": true
77
- },
78
- "151652": {
79
- "content": "<|vision_start|>",
80
- "lstrip": false,
81
- "normalized": false,
82
- "rstrip": false,
83
- "single_word": false,
84
- "special": true
85
- },
86
- "151653": {
87
- "content": "<|vision_end|>",
88
- "lstrip": false,
89
- "normalized": false,
90
- "rstrip": false,
91
- "single_word": false,
92
- "special": true
93
- },
94
- "151654": {
95
- "content": "<|vision_pad|>",
96
- "lstrip": false,
97
- "normalized": false,
98
- "rstrip": false,
99
- "single_word": false,
100
- "special": true
101
- },
102
- "151655": {
103
- "content": "<|image_pad|>",
104
- "lstrip": false,
105
- "normalized": false,
106
- "rstrip": false,
107
- "single_word": false,
108
- "special": true
109
- },
110
- "151656": {
111
- "content": "<|video_pad|>",
112
- "lstrip": false,
113
- "normalized": false,
114
- "rstrip": false,
115
- "single_word": false,
116
- "special": true
117
- },
118
- "151657": {
119
- "content": "<tool_call>",
120
- "lstrip": false,
121
- "normalized": false,
122
- "rstrip": false,
123
- "single_word": false,
124
- "special": false
125
- },
126
- "151658": {
127
- "content": "</tool_call>",
128
- "lstrip": false,
129
- "normalized": false,
130
- "rstrip": false,
131
- "single_word": false,
132
- "special": false
133
- },
134
- "151659": {
135
- "content": "<|fim_prefix|>",
136
- "lstrip": false,
137
- "normalized": false,
138
- "rstrip": false,
139
- "single_word": false,
140
- "special": false
141
- },
142
- "151660": {
143
- "content": "<|fim_middle|>",
144
- "lstrip": false,
145
- "normalized": false,
146
- "rstrip": false,
147
- "single_word": false,
148
- "special": false
149
- },
150
- "151661": {
151
- "content": "<|fim_suffix|>",
152
- "lstrip": false,
153
- "normalized": false,
154
- "rstrip": false,
155
- "single_word": false,
156
- "special": false
157
- },
158
- "151662": {
159
- "content": "<|fim_pad|>",
160
- "lstrip": false,
161
- "normalized": false,
162
- "rstrip": false,
163
- "single_word": false,
164
- "special": false
165
- },
166
- "151663": {
167
- "content": "<|repo_name|>",
168
- "lstrip": false,
169
- "normalized": false,
170
- "rstrip": false,
171
- "single_word": false,
172
- "special": false
173
- },
174
- "151664": {
175
- "content": "<|file_sep|>",
176
- "lstrip": false,
177
- "normalized": false,
178
- "rstrip": false,
179
- "single_word": false,
180
- "special": false
181
- }
182
- },
183
- "bos_token": "<|begin▁of▁sentence|>",
184
- "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
- "clean_up_tokenization_spaces": false,
186
- "eos_token": "<|end▁of▁sentence|>",
187
- "extra_special_tokens": {},
188
- "legacy": true,
189
- "model_max_length": 16384,
190
- "pad_token": "<|end▁of▁sentence|>",
191
- "sp_model_kwargs": {},
192
- "tokenizer_class": "LlamaTokenizerFast",
193
- "unk_token": null,
194
- "use_default_system_prompt": false
195
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/config.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00001-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:7819f2c0378a7ea9896986fa6f2976101d6dd007bc5b22f5023a148d9a5617b6
3
- size 5321941889
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00002-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:bd4062e2e1a76c96699e2bdeaab15ce1a130f14f8ff3cc2625dc450e76bd3e96
3
- size 5363162608
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00003-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:a23c6fb10dca081039ccca54b9e4101820f55ad2669ba6db700e2d1bd54b81b1
3
- size 5357249052
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model-00004-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:d849115b9ac769b41689460305df28026834a723b2bf002a132247ef170c09f9
3
- size 5127218901
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/model.safetensors.index.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/special_tokens_map.json DELETED
@@ -1,23 +0,0 @@
1
- {
2
- "bos_token": {
3
- "content": "<|begin▁of▁sentence|>",
4
- "lstrip": false,
5
- "normalized": false,
6
- "rstrip": false,
7
- "single_word": false
8
- },
9
- "eos_token": {
10
- "content": "<|end▁of▁sentence|>",
11
- "lstrip": false,
12
- "normalized": false,
13
- "rstrip": false,
14
- "single_word": false
15
- },
16
- "pad_token": {
17
- "content": "<|end▁of▁sentence|>",
18
- "lstrip": false,
19
- "normalized": false,
20
- "rstrip": false,
21
- "single_word": false
22
- }
23
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/tokenizer.json DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
- size 11422778
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4,8_mixed/tokenizer_config.json DELETED
@@ -1,195 +0,0 @@
1
- {
2
- "add_bos_token": true,
3
- "add_eos_token": false,
4
- "add_prefix_space": null,
5
- "added_tokens_decoder": {
6
- "151643": {
7
- "content": "<|end▁of▁sentence|>",
8
- "lstrip": false,
9
- "normalized": false,
10
- "rstrip": false,
11
- "single_word": false,
12
- "special": true
13
- },
14
- "151644": {
15
- "content": "<|User|>",
16
- "lstrip": false,
17
- "normalized": false,
18
- "rstrip": false,
19
- "single_word": false,
20
- "special": false
21
- },
22
- "151645": {
23
- "content": "<|Assistant|>",
24
- "lstrip": false,
25
- "normalized": false,
26
- "rstrip": false,
27
- "single_word": false,
28
- "special": false
29
- },
30
- "151646": {
31
- "content": "<|begin▁of▁sentence|>",
32
- "lstrip": false,
33
- "normalized": false,
34
- "rstrip": false,
35
- "single_word": false,
36
- "special": true
37
- },
38
- "151647": {
39
- "content": "<|EOT|>",
40
- "lstrip": false,
41
- "normalized": false,
42
- "rstrip": false,
43
- "single_word": false,
44
- "special": false
45
- },
46
- "151648": {
47
- "content": "<think>",
48
- "lstrip": false,
49
- "normalized": false,
50
- "rstrip": false,
51
- "single_word": false,
52
- "special": false
53
- },
54
- "151649": {
55
- "content": "</think>",
56
- "lstrip": false,
57
- "normalized": false,
58
- "rstrip": false,
59
- "single_word": false,
60
- "special": false
61
- },
62
- "151650": {
63
- "content": "<|quad_start|>",
64
- "lstrip": false,
65
- "normalized": false,
66
- "rstrip": false,
67
- "single_word": false,
68
- "special": true
69
- },
70
- "151651": {
71
- "content": "<|quad_end|>",
72
- "lstrip": false,
73
- "normalized": false,
74
- "rstrip": false,
75
- "single_word": false,
76
- "special": true
77
- },
78
- "151652": {
79
- "content": "<|vision_start|>",
80
- "lstrip": false,
81
- "normalized": false,
82
- "rstrip": false,
83
- "single_word": false,
84
- "special": true
85
- },
86
- "151653": {
87
- "content": "<|vision_end|>",
88
- "lstrip": false,
89
- "normalized": false,
90
- "rstrip": false,
91
- "single_word": false,
92
- "special": true
93
- },
94
- "151654": {
95
- "content": "<|vision_pad|>",
96
- "lstrip": false,
97
- "normalized": false,
98
- "rstrip": false,
99
- "single_word": false,
100
- "special": true
101
- },
102
- "151655": {
103
- "content": "<|image_pad|>",
104
- "lstrip": false,
105
- "normalized": false,
106
- "rstrip": false,
107
- "single_word": false,
108
- "special": true
109
- },
110
- "151656": {
111
- "content": "<|video_pad|>",
112
- "lstrip": false,
113
- "normalized": false,
114
- "rstrip": false,
115
- "single_word": false,
116
- "special": true
117
- },
118
- "151657": {
119
- "content": "<tool_call>",
120
- "lstrip": false,
121
- "normalized": false,
122
- "rstrip": false,
123
- "single_word": false,
124
- "special": false
125
- },
126
- "151658": {
127
- "content": "</tool_call>",
128
- "lstrip": false,
129
- "normalized": false,
130
- "rstrip": false,
131
- "single_word": false,
132
- "special": false
133
- },
134
- "151659": {
135
- "content": "<|fim_prefix|>",
136
- "lstrip": false,
137
- "normalized": false,
138
- "rstrip": false,
139
- "single_word": false,
140
- "special": false
141
- },
142
- "151660": {
143
- "content": "<|fim_middle|>",
144
- "lstrip": false,
145
- "normalized": false,
146
- "rstrip": false,
147
- "single_word": false,
148
- "special": false
149
- },
150
- "151661": {
151
- "content": "<|fim_suffix|>",
152
- "lstrip": false,
153
- "normalized": false,
154
- "rstrip": false,
155
- "single_word": false,
156
- "special": false
157
- },
158
- "151662": {
159
- "content": "<|fim_pad|>",
160
- "lstrip": false,
161
- "normalized": false,
162
- "rstrip": false,
163
- "single_word": false,
164
- "special": false
165
- },
166
- "151663": {
167
- "content": "<|repo_name|>",
168
- "lstrip": false,
169
- "normalized": false,
170
- "rstrip": false,
171
- "single_word": false,
172
- "special": false
173
- },
174
- "151664": {
175
- "content": "<|file_sep|>",
176
- "lstrip": false,
177
- "normalized": false,
178
- "rstrip": false,
179
- "single_word": false,
180
- "special": false
181
- }
182
- },
183
- "bos_token": "<|begin▁of▁sentence|>",
184
- "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
- "clean_up_tokenization_spaces": false,
186
- "eos_token": "<|end▁of▁sentence|>",
187
- "extra_special_tokens": {},
188
- "legacy": true,
189
- "model_max_length": 16384,
190
- "pad_token": "<|end▁of▁sentence|>",
191
- "sp_model_kwargs": {},
192
- "tokenizer_class": "LlamaTokenizerFast",
193
- "unk_token": null,
194
- "use_default_system_prompt": false
195
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4bit/special_tokens_map.json DELETED
@@ -1,23 +0,0 @@
1
- {
2
- "bos_token": {
3
- "content": "<|begin▁of▁sentence|>",
4
- "lstrip": false,
5
- "normalized": false,
6
- "rstrip": false,
7
- "single_word": false
8
- },
9
- "eos_token": {
10
- "content": "<|end▁of▁sentence|>",
11
- "lstrip": false,
12
- "normalized": false,
13
- "rstrip": false,
14
- "single_word": false
15
- },
16
- "pad_token": {
17
- "content": "<|end▁of▁sentence|>",
18
- "lstrip": false,
19
- "normalized": false,
20
- "rstrip": false,
21
- "single_word": false
22
- }
23
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4bit/tokenizer.json DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
- size 11422778
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-4bit/tokenizer_config.json DELETED
@@ -1,195 +0,0 @@
1
- {
2
- "add_bos_token": true,
3
- "add_eos_token": false,
4
- "add_prefix_space": null,
5
- "added_tokens_decoder": {
6
- "151643": {
7
- "content": "<|end▁of▁sentence|>",
8
- "lstrip": false,
9
- "normalized": false,
10
- "rstrip": false,
11
- "single_word": false,
12
- "special": true
13
- },
14
- "151644": {
15
- "content": "<|User|>",
16
- "lstrip": false,
17
- "normalized": false,
18
- "rstrip": false,
19
- "single_word": false,
20
- "special": false
21
- },
22
- "151645": {
23
- "content": "<|Assistant|>",
24
- "lstrip": false,
25
- "normalized": false,
26
- "rstrip": false,
27
- "single_word": false,
28
- "special": false
29
- },
30
- "151646": {
31
- "content": "<|begin▁of▁sentence|>",
32
- "lstrip": false,
33
- "normalized": false,
34
- "rstrip": false,
35
- "single_word": false,
36
- "special": true
37
- },
38
- "151647": {
39
- "content": "<|EOT|>",
40
- "lstrip": false,
41
- "normalized": false,
42
- "rstrip": false,
43
- "single_word": false,
44
- "special": false
45
- },
46
- "151648": {
47
- "content": "<think>",
48
- "lstrip": false,
49
- "normalized": false,
50
- "rstrip": false,
51
- "single_word": false,
52
- "special": false
53
- },
54
- "151649": {
55
- "content": "</think>",
56
- "lstrip": false,
57
- "normalized": false,
58
- "rstrip": false,
59
- "single_word": false,
60
- "special": false
61
- },
62
- "151650": {
63
- "content": "<|quad_start|>",
64
- "lstrip": false,
65
- "normalized": false,
66
- "rstrip": false,
67
- "single_word": false,
68
- "special": true
69
- },
70
- "151651": {
71
- "content": "<|quad_end|>",
72
- "lstrip": false,
73
- "normalized": false,
74
- "rstrip": false,
75
- "single_word": false,
76
- "special": true
77
- },
78
- "151652": {
79
- "content": "<|vision_start|>",
80
- "lstrip": false,
81
- "normalized": false,
82
- "rstrip": false,
83
- "single_word": false,
84
- "special": true
85
- },
86
- "151653": {
87
- "content": "<|vision_end|>",
88
- "lstrip": false,
89
- "normalized": false,
90
- "rstrip": false,
91
- "single_word": false,
92
- "special": true
93
- },
94
- "151654": {
95
- "content": "<|vision_pad|>",
96
- "lstrip": false,
97
- "normalized": false,
98
- "rstrip": false,
99
- "single_word": false,
100
- "special": true
101
- },
102
- "151655": {
103
- "content": "<|image_pad|>",
104
- "lstrip": false,
105
- "normalized": false,
106
- "rstrip": false,
107
- "single_word": false,
108
- "special": true
109
- },
110
- "151656": {
111
- "content": "<|video_pad|>",
112
- "lstrip": false,
113
- "normalized": false,
114
- "rstrip": false,
115
- "single_word": false,
116
- "special": true
117
- },
118
- "151657": {
119
- "content": "<tool_call>",
120
- "lstrip": false,
121
- "normalized": false,
122
- "rstrip": false,
123
- "single_word": false,
124
- "special": false
125
- },
126
- "151658": {
127
- "content": "</tool_call>",
128
- "lstrip": false,
129
- "normalized": false,
130
- "rstrip": false,
131
- "single_word": false,
132
- "special": false
133
- },
134
- "151659": {
135
- "content": "<|fim_prefix|>",
136
- "lstrip": false,
137
- "normalized": false,
138
- "rstrip": false,
139
- "single_word": false,
140
- "special": false
141
- },
142
- "151660": {
143
- "content": "<|fim_middle|>",
144
- "lstrip": false,
145
- "normalized": false,
146
- "rstrip": false,
147
- "single_word": false,
148
- "special": false
149
- },
150
- "151661": {
151
- "content": "<|fim_suffix|>",
152
- "lstrip": false,
153
- "normalized": false,
154
- "rstrip": false,
155
- "single_word": false,
156
- "special": false
157
- },
158
- "151662": {
159
- "content": "<|fim_pad|>",
160
- "lstrip": false,
161
- "normalized": false,
162
- "rstrip": false,
163
- "single_word": false,
164
- "special": false
165
- },
166
- "151663": {
167
- "content": "<|repo_name|>",
168
- "lstrip": false,
169
- "normalized": false,
170
- "rstrip": false,
171
- "single_word": false,
172
- "special": false
173
- },
174
- "151664": {
175
- "content": "<|file_sep|>",
176
- "lstrip": false,
177
- "normalized": false,
178
- "rstrip": false,
179
- "single_word": false,
180
- "special": false
181
- }
182
- },
183
- "bos_token": "<|begin▁of▁sentence|>",
184
- "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
- "clean_up_tokenization_spaces": false,
186
- "eos_token": "<|end▁of▁sentence|>",
187
- "extra_special_tokens": {},
188
- "legacy": true,
189
- "model_max_length": 16384,
190
- "pad_token": "<|end▁of▁sentence|>",
191
- "sp_model_kwargs": {},
192
- "tokenizer_class": "LlamaTokenizerFast",
193
- "unk_token": null,
194
- "use_default_system_prompt": false
195
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/config.json DELETED
@@ -1,35 +0,0 @@
1
- {
2
- "architectures": [
3
- "Qwen2ForCausalLM"
4
- ],
5
- "attention_dropout": 0.0,
6
- "bos_token_id": 151643,
7
- "eos_token_id": 151643,
8
- "hidden_act": "silu",
9
- "hidden_size": 5120,
10
- "initializer_range": 0.02,
11
- "intermediate_size": 27648,
12
- "max_position_embeddings": 131072,
13
- "max_window_layers": 64,
14
- "model_type": "qwen2",
15
- "num_attention_heads": 40,
16
- "num_hidden_layers": 64,
17
- "num_key_value_heads": 8,
18
- "quantization": {
19
- "group_size": 64,
20
- "bits": 6
21
- },
22
- "quantization_config": {
23
- "group_size": 64,
24
- "bits": 6
25
- },
26
- "rms_norm_eps": 1e-05,
27
- "rope_theta": 1000000.0,
28
- "sliding_window": 131072,
29
- "tie_word_embeddings": false,
30
- "torch_dtype": "bfloat16",
31
- "transformers_version": "4.43.1",
32
- "use_cache": true,
33
- "use_sliding_window": false,
34
- "vocab_size": 152064
35
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/model-00001-of-00006.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:9a00935ed67ac7e66d142d5f9e4815eb8ab3f961faa7b8bcbca55a4ff2f4ee25
3
- size 5271984209
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/model-00002-of-00006.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:5e09753f33189775b86554c9737eb5574dbd556dd476966ee69cccdb0eb89517
3
- size 5316808320
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/model-00003-of-00006.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:7c40d6060b390eac2274cb2d18973e8fc4ef5747e4d5ae60b2c7a6f0fc8505fb
3
- size 5265653517
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/model-00004-of-00006.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:f81a6a9069dda4ded4f7043ef63966d113d9534180fa561fa25ffa21ab646479
3
- size 5265653541
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/model-00005-of-00006.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:7a88fe01f535a1e4316b000d0f5322482e33e598003500c39bf174a4cee86ed1
3
- size 4869481603
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/model-00006-of-00006.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:04c494df573aaf13437bb7e389e37e2ab8b602836efb826cb3178bf4b125d1e0
3
- size 632586540
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/model.safetensors.index.json DELETED
The diff for this file is too large to render. See raw diff
 
DeepSeek-R1-Distill-Qwen-32B-6bit/special_tokens_map.json DELETED
@@ -1,23 +0,0 @@
1
- {
2
- "bos_token": {
3
- "content": "<|begin▁of▁sentence|>",
4
- "lstrip": false,
5
- "normalized": false,
6
- "rstrip": false,
7
- "single_word": false
8
- },
9
- "eos_token": {
10
- "content": "<|end▁of▁sentence|>",
11
- "lstrip": false,
12
- "normalized": false,
13
- "rstrip": false,
14
- "single_word": false
15
- },
16
- "pad_token": {
17
- "content": "<|end▁of▁sentence|>",
18
- "lstrip": false,
19
- "normalized": false,
20
- "rstrip": false,
21
- "single_word": false
22
- }
23
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/tokenizer.json DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:e20ddafc659ba90242154b55275402edeca0715e5dbb30f56815a4ce081f4893
3
- size 11422778
 
 
 
 
DeepSeek-R1-Distill-Qwen-32B-6bit/tokenizer_config.json DELETED
@@ -1,195 +0,0 @@
1
- {
2
- "add_bos_token": true,
3
- "add_eos_token": false,
4
- "add_prefix_space": null,
5
- "added_tokens_decoder": {
6
- "151643": {
7
- "content": "<|end▁of▁sentence|>",
8
- "lstrip": false,
9
- "normalized": false,
10
- "rstrip": false,
11
- "single_word": false,
12
- "special": true
13
- },
14
- "151644": {
15
- "content": "<|User|>",
16
- "lstrip": false,
17
- "normalized": false,
18
- "rstrip": false,
19
- "single_word": false,
20
- "special": false
21
- },
22
- "151645": {
23
- "content": "<|Assistant|>",
24
- "lstrip": false,
25
- "normalized": false,
26
- "rstrip": false,
27
- "single_word": false,
28
- "special": false
29
- },
30
- "151646": {
31
- "content": "<|begin▁of▁sentence|>",
32
- "lstrip": false,
33
- "normalized": false,
34
- "rstrip": false,
35
- "single_word": false,
36
- "special": true
37
- },
38
- "151647": {
39
- "content": "<|EOT|>",
40
- "lstrip": false,
41
- "normalized": false,
42
- "rstrip": false,
43
- "single_word": false,
44
- "special": false
45
- },
46
- "151648": {
47
- "content": "<think>",
48
- "lstrip": false,
49
- "normalized": false,
50
- "rstrip": false,
51
- "single_word": false,
52
- "special": false
53
- },
54
- "151649": {
55
- "content": "</think>",
56
- "lstrip": false,
57
- "normalized": false,
58
- "rstrip": false,
59
- "single_word": false,
60
- "special": false
61
- },
62
- "151650": {
63
- "content": "<|quad_start|>",
64
- "lstrip": false,
65
- "normalized": false,
66
- "rstrip": false,
67
- "single_word": false,
68
- "special": true
69
- },
70
- "151651": {
71
- "content": "<|quad_end|>",
72
- "lstrip": false,
73
- "normalized": false,
74
- "rstrip": false,
75
- "single_word": false,
76
- "special": true
77
- },
78
- "151652": {
79
- "content": "<|vision_start|>",
80
- "lstrip": false,
81
- "normalized": false,
82
- "rstrip": false,
83
- "single_word": false,
84
- "special": true
85
- },
86
- "151653": {
87
- "content": "<|vision_end|>",
88
- "lstrip": false,
89
- "normalized": false,
90
- "rstrip": false,
91
- "single_word": false,
92
- "special": true
93
- },
94
- "151654": {
95
- "content": "<|vision_pad|>",
96
- "lstrip": false,
97
- "normalized": false,
98
- "rstrip": false,
99
- "single_word": false,
100
- "special": true
101
- },
102
- "151655": {
103
- "content": "<|image_pad|>",
104
- "lstrip": false,
105
- "normalized": false,
106
- "rstrip": false,
107
- "single_word": false,
108
- "special": true
109
- },
110
- "151656": {
111
- "content": "<|video_pad|>",
112
- "lstrip": false,
113
- "normalized": false,
114
- "rstrip": false,
115
- "single_word": false,
116
- "special": true
117
- },
118
- "151657": {
119
- "content": "<tool_call>",
120
- "lstrip": false,
121
- "normalized": false,
122
- "rstrip": false,
123
- "single_word": false,
124
- "special": false
125
- },
126
- "151658": {
127
- "content": "</tool_call>",
128
- "lstrip": false,
129
- "normalized": false,
130
- "rstrip": false,
131
- "single_word": false,
132
- "special": false
133
- },
134
- "151659": {
135
- "content": "<|fim_prefix|>",
136
- "lstrip": false,
137
- "normalized": false,
138
- "rstrip": false,
139
- "single_word": false,
140
- "special": false
141
- },
142
- "151660": {
143
- "content": "<|fim_middle|>",
144
- "lstrip": false,
145
- "normalized": false,
146
- "rstrip": false,
147
- "single_word": false,
148
- "special": false
149
- },
150
- "151661": {
151
- "content": "<|fim_suffix|>",
152
- "lstrip": false,
153
- "normalized": false,
154
- "rstrip": false,
155
- "single_word": false,
156
- "special": false
157
- },
158
- "151662": {
159
- "content": "<|fim_pad|>",
160
- "lstrip": false,
161
- "normalized": false,
162
- "rstrip": false,
163
- "single_word": false,
164
- "special": false
165
- },
166
- "151663": {
167
- "content": "<|repo_name|>",
168
- "lstrip": false,
169
- "normalized": false,
170
- "rstrip": false,
171
- "single_word": false,
172
- "special": false
173
- },
174
- "151664": {
175
- "content": "<|file_sep|>",
176
- "lstrip": false,
177
- "normalized": false,
178
- "rstrip": false,
179
- "single_word": false,
180
- "special": false
181
- }
182
- },
183
- "bos_token": "<|begin▁of▁sentence|>",
184
- "chat_template": "{% if not add_generation_prompt is defined %}{% set add_generation_prompt = false %}{% endif %}{% set ns = namespace(is_first=false, is_tool=false, is_output_first=true, system_prompt='') %}{%- for message in messages %}{%- if message['role'] == 'system' %}{% set ns.system_prompt = message['content'] %}{%- endif %}{%- endfor %}{{bos_token}}{{ns.system_prompt}}{%- for message in messages %}{%- if message['role'] == 'user' %}{%- set ns.is_tool = false -%}{{'<|User|>' + message['content']}}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is none %}{%- set ns.is_tool = false -%}{%- for tool in message['tool_calls']%}{%- if not ns.is_first %}{{'<|Assistant|><|tool▁calls▁begin|><|tool▁call▁begin��>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{%- set ns.is_first = true -%}{%- else %}{{'\\n' + '<|tool▁call▁begin|>' + tool['type'] + '<|tool▁sep|>' + tool['function']['name'] + '\\n' + '```json' + '\\n' + tool['function']['arguments'] + '\\n' + '```' + '<|tool▁call▁end|>'}}{{'<|tool▁calls▁end|><|end▁of▁sentence|>'}}{%- endif %}{%- endfor %}{%- endif %}{%- if message['role'] == 'assistant' and message['content'] is not none %}{%- if ns.is_tool %}{{'<|tool▁outputs▁end|>' + message['content'] + '<|end▁of▁sentence|>'}}{%- set ns.is_tool = false -%}{%- else %}{% set content = message['content'] %}{% if '</think>' in content %}{% set content = content.split('</think>')[-1] %}{% endif %}{{'<|Assistant|>' + content + '<|end▁of▁sentence|>'}}{%- endif %}{%- endif %}{%- if message['role'] == 'tool' %}{%- set ns.is_tool = true -%}{%- if ns.is_output_first %}{{'<|tool▁outputs▁begin|><|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- set ns.is_output_first = false %}{%- else %}{{'\\n<|tool▁output▁begin|>' + message['content'] + '<|tool▁output▁end|>'}}{%- endif %}{%- endif %}{%- endfor -%}{% if ns.is_tool %}{{'<|tool▁outputs▁end|>'}}{% endif %}{% if add_generation_prompt and not ns.is_tool %}{{'<|Assistant|><think>\\n'}}{% endif %}",
185
- "clean_up_tokenization_spaces": false,
186
- "eos_token": "<|end▁of▁sentence|>",
187
- "extra_special_tokens": {},
188
- "legacy": true,
189
- "model_max_length": 16384,
190
- "pad_token": "<|end▁of▁sentence|>",
191
- "sp_model_kwargs": {},
192
- "tokenizer_class": "LlamaTokenizerFast",
193
- "unk_token": null,
194
- "use_default_system_prompt": false
195
- }