Update README.md
Browse files
README.md
CHANGED
|
@@ -31,36 +31,11 @@ For IN2-mixed, we evaluate it on CUDA with overflow protection. The CPU version
|
|
| 31 |
|
| 32 |
please note int2 **may be slower** than int4 on CUDA due to kernel issue.
|
| 33 |
|
| 34 |
-
**To prevent potential overflow and achieve better accuracy, we recommend using the CPU version detailed in the next section.**
|
| 35 |
|
| 36 |
~~~python
|
|
|
|
| 37 |
import transformers
|
| 38 |
from transformers import AutoModelForCausalLM, AutoTokenizer
|
| 39 |
-
from auto_round import AutoRoundConfig ##must import for auto-round format
|
| 40 |
-
|
| 41 |
-
|
| 42 |
-
# https://github.com/huggingface/transformers/pull/35493
|
| 43 |
-
def set_initialized_submodules(model, state_dict_keys):
|
| 44 |
-
"""
|
| 45 |
-
Sets the `_is_hf_initialized` flag in all submodules of a given model when all its weights are in the loaded state
|
| 46 |
-
dict.
|
| 47 |
-
"""
|
| 48 |
-
state_dict_keys = set(state_dict_keys)
|
| 49 |
-
not_initialized_submodules = {}
|
| 50 |
-
for module_name, module in model.named_modules():
|
| 51 |
-
if module_name == "":
|
| 52 |
-
# When checking if the root module is loaded there's no need to prepend module_name.
|
| 53 |
-
module_keys = set(module.state_dict())
|
| 54 |
-
else:
|
| 55 |
-
module_keys = {f"{module_name}.{k}" for k in module.state_dict()}
|
| 56 |
-
if module_keys.issubset(state_dict_keys):
|
| 57 |
-
module._is_hf_initialized = True
|
| 58 |
-
else:
|
| 59 |
-
not_initialized_submodules[module_name] = module
|
| 60 |
-
return not_initialized_submodules
|
| 61 |
-
|
| 62 |
-
|
| 63 |
-
transformers.modeling_utils.set_initialized_submodules = set_initialized_submodules
|
| 64 |
|
| 65 |
import torch
|
| 66 |
|
|
@@ -81,24 +56,10 @@ for i in range(61):
|
|
| 81 |
|
| 82 |
model = AutoModelForCausalLM.from_pretrained(
|
| 83 |
quantized_model_dir,
|
| 84 |
-
torch_dtype=torch.
|
| 85 |
-
trust_remote_code=True,
|
| 86 |
device_map=device_map,
|
| 87 |
)
|
| 88 |
|
| 89 |
-
|
| 90 |
-
def forward_hook(module, input, output):
|
| 91 |
-
return torch.clamp(output, -65504, 65504)
|
| 92 |
-
|
| 93 |
-
|
| 94 |
-
def register_fp16_hooks(model):
|
| 95 |
-
for name, module in model.named_modules():
|
| 96 |
-
if "QuantLinear" in module.__class__.__name__ or isinstance(module, torch.nn.Linear):
|
| 97 |
-
module.register_forward_hook(forward_hook)
|
| 98 |
-
|
| 99 |
-
|
| 100 |
-
register_fp16_hooks(model) ##better add this hook to avoid overflow
|
| 101 |
-
|
| 102 |
tokenizer = AutoTokenizer.from_pretrained(quantized_model_dir, trust_remote_code=True)
|
| 103 |
prompts = [
|
| 104 |
"9.11和9.8哪个数字大",
|
|
@@ -123,7 +84,7 @@ inputs = tokenizer(texts, return_tensors="pt", padding=True, truncation=True)
|
|
| 123 |
outputs = model.generate(
|
| 124 |
input_ids=inputs["input_ids"].to(model.device),
|
| 125 |
attention_mask=inputs["attention_mask"].to(model.device),
|
| 126 |
-
max_length=
|
| 127 |
num_return_sequences=1,
|
| 128 |
do_sample=False ##change this to align with the official usage
|
| 129 |
)
|
|
@@ -139,119 +100,7 @@ for i, prompt in enumerate(prompts):
|
|
| 139 |
print(f"Generated: {decoded_outputs[i]}")
|
| 140 |
print("-" * 50)
|
| 141 |
|
| 142 |
-
|
| 143 |
-
|
| 144 |
-
"""
|
| 145 |
-
Prompt: 9.11和9.8哪个数字大
|
| 146 |
-
Generated: <think>
|
| 147 |
-
首先,我需要比较两个数字:9.11和9.8。
|
| 148 |
-
|
| 149 |
-
为了准确比较,我首先将它们转换为相同的小数位数。将9.8写成9.80,这样两个数字都有两位小数。
|
| 150 |
-
|
| 151 |
-
接下来,我比较整数部分。两个数字的整数部分都是9,所以它们在这一部分相等。
|
| 152 |
-
|
| 153 |
-
然后,我比较小数部分。9.11的小数部分是0.11,而9.80的小数部分是0.80。显然,0.80大于0.11。
|
| 154 |
-
|
| 155 |
-
因此,综合整数和小数部分的比较结果,9.80大于9.11。
|
| 156 |
-
</think>
|
| 157 |
-
|
| 158 |
-
要比较两个数字 **9.11** 和 **9.8** 的大小,我们可以按照以下步骤进行:
|
| 159 |
-
|
| 160 |
-
1. **统一小数位数**:
|
| 161 |
-
- 将 **9.8** 写成 **9.80**,以便与 **9.11** 进行比较。
|
| 162 |
-
|
| 163 |
-
2. **比较整数部分**:
|
| 164 |
-
- 两个数字的整数部分都是 **9**,因此整数部分相等。
|
| 165 |
-
|
| 166 |
-
3. **比较小数部分**:
|
| 167 |
-
- **9.11** 的小数部分是 **0.11**。
|
| 168 |
-
- **9.80** 的小数部分是 **0.80**。
|
| 169 |
-
- 显然,**0.80** 大于 **0.11**。
|
| 170 |
-
|
| 171 |
-
4. **综合比较结果**:
|
| 172 |
-
- 由于整数部分相等,小数部分 **0.80** 大于 **0.11**,因此 **9.80** 大于 **9.11**。
|
| 173 |
-
|
| 174 |
-
**最终结论:**
|
| 175 |
-
|
| 176 |
-
\[
|
| 177 |
-
\boxed{9.8 \text{ 大于 } 9.11}
|
| 178 |
-
\]
|
| 179 |
-
--------------------------------------------------
|
| 180 |
-
|
| 181 |
-
Prompt: 如果你是人,你最想做什么“
|
| 182 |
-
Generated: <think>
|
| 183 |
-
嗯,如果我是人,我最想做什么呢?这个问题挺有意思的。首先,我需要理解“如果我是人”这个前提。作为一个人,我有自己的思想、情感和自由意志,对吧?所以,我可以选择自己想要的生活方式和追求的目标。
|
| 184 |
-
|
| 185 |
-
首先,可能我会考虑自己的兴趣和爱好。比如,如果我喜欢艺术,可能会想成为画家或音乐家;如果对科学感兴趣,可能会投身于研究。但也许我更倾向于帮助别人,所以可能选择成为医生、教师或社会工作者。这些都是常见的职业选择,但作为一个人,可能还有更多的可能性。
|
| 186 |
-
|
| 187 |
-
然后,我需要考虑自己的价值观。如果我认为家庭很重要,可能会想建立一个幸福的家庭,花时间陪伴家人。如果更关注社会贡献,可能会参与公益活���,帮助需要帮助的人。或者,如果追求个人成就,可能会努力在事业上取得成功,获得认可和成就感。
|
| 188 |
-
|
| 189 |
-
另外,作为人,可能会有很多梦想和愿望。比如,旅行世界,体验不同的文化;学习新技能,不断自我提升;或者追求某种精神层面的满足,比如冥想、哲学探索等。这些都是可能的选项。
|
| 190 |
-
|
| 191 |
-
不过,也有可能存在挑战和困难。比如,经济压力、时间限制、社会压力等,这些都可能影响我的选择。所以,我需要权衡利弊,找到最适合自己的道路。
|
| 192 |
-
|
| 193 |
-
还有,作为人,可能会有不同的阶段。年轻时可能更注重探索和冒险,中年时可能追求稳定和家庭,老年时可能寻求平静和传承。不同阶段有不同的目标和愿望。
|
| 194 |
-
|
| 195 |
-
另外,人际关系也很重要。作为人,与朋友、家人、同事的关系会影响我的幸福感和满足感。所以,维护良好的人际关系可能也是一个重要的目标。
|
| 196 |
-
|
| 197 |
-
可能还需要考虑健康问题。保持身体健康,才能更好地追求其他目标。所以,锻炼、合理饮食、心理健康也是需要考虑的方面。
|
| 198 |
-
|
| 199 |
-
还有,教育的重要性。不断学习和成长,获取新知识,提升自己的能力,这对实现各种目标都是基础。
|
| 200 |
-
|
| 201 |
-
当然,每个人的情况不同,所以我的选择也会因人而异。但总的来说,作为人,最想做的事情可能是一个综合性的目标,结合了个人兴趣、价值观、社会贡献、家庭、健康等多个方面。
|
| 202 |
-
|
| 203 |
-
不过,可能还需要考虑现实因素。比如,经济条件允许吗?社会支持如何?有没有足够的资源和机会?这些都会影响最终的选择。
|
| 204 |
-
|
| 205 |
-
也许,如果我是人,我会追求一种平衡的生活,既满足个人发展,又能帮助他人,同时享受生活中的美好时光。这可能包括
|
| 206 |
-
--------------------------------------------------
|
| 207 |
-
Prompt: How many e in word deepseek
|
| 208 |
-
Generated: <think>
|
| 209 |
-
Okay, so I need to figure out how many times the letter 'e' appears in the word "deepseek." Let me start by writing the word out to see each letter clearly. The word is d-e-e-p-s-e-e-k. Let me count each letter one by one.
|
| 210 |
-
|
| 211 |
-
Starting with the first letter: that's a 'd'. No 'e' there. The next letter is 'e', so that's one. Then the next letter is another 'e', making it two. The third letter is 'p', which isn't an 'e'. Then comes 's', also not an 'e'. After that, another 'e', so that's three. Then another 'e', bringing the count to four. Finally, the last letter is 'k', which isn't an 'e'.
|
| 212 |
-
|
| 213 |
-
Wait, let me check again to make sure I didn't miss any. The word is spelled D-E-E-P-S-E-E-K. So breaking it down: D, E, E, P, S, E, E, K. So positions 2, 3, 6, and 7 are all 'e's. That's four 'e's in total. Hmm, I think that's correct. Let me verify by writing the letters with their positions:
|
| 214 |
-
|
| 215 |
-
1: D
|
| 216 |
-
2: E
|
| 217 |
-
3: E
|
| 218 |
-
4: P
|
| 219 |
-
5: S
|
| 220 |
-
6: E
|
| 221 |
-
7: E
|
| 222 |
-
8: K
|
| 223 |
-
|
| 224 |
-
Yes, positions 2, 3, 6, and 7 are all 'e's. So that's four instances of the letter 'e'. I don't think I missed any. So the answer should be 4.
|
| 225 |
-
</think>
|
| 226 |
-
|
| 227 |
-
The word "deepseek" contains the letter 'e' four times.
|
| 228 |
-
|
| 229 |
-
**Step-by-Step Explanation:**
|
| 230 |
-
1. Write out the word: D, E, E, P, S, E, E, K.
|
| 231 |
-
2. Identify each 'e' by its position:
|
| 232 |
-
- Position 2: E
|
| 233 |
-
- Position# 1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.1.
|
| 234 |
-
|
| 235 |
-
--------------------------------------------------
|
| 236 |
-
|
| 237 |
-
Prompt: There are ten birds in a tree. A hunter shoots one. How many are left in the tree?
|
| 238 |
-
Generated: <think>
|
| 239 |
-
Okay, so there's this problem here: "There are ten birds in a tree. A hunter shoots one. How many are left in the tree?" Hmm, at first glance, it seems straightforward, but I remember sometimes these kinds of questions have a trick to them. Let me think through this step by step.
|
| 240 |
-
|
| 241 |
-
Alright, starting with the basics. If there are ten birds in a tree, and a hunter shoots one, the immediate mathematical answer would be 10 minus 1, which is 9. So, nine birds left. But wait, maybe there's more to it. Sometimes these riddles play on the situation rather than just the numbers. Let me consider different angles.
|
| 242 |
-
|
| 243 |
-
First, when the hunter shoots, the sound of the gunshot might scare the other birds. So, even if only one bird is shot, the rest might fly away. If all the other birds get scared and fly off, then there would be zero birds left in the tree. But is that the case here? The problem doesn't specify whether the other birds are scared or not. It just says a hunter shoots one. So, maybe the question is testing if you consider that possibility.
|
| 244 |
-
|
| 245 |
-
But then again, maybe it's a straightforward math problem. If you take one away from ten, you get nine. But in real-life scenarios, the noise would likely cause the other birds to flee. However, the problem doesn't mention anything about the birds flying away. So, is it assuming that the other birds stay? Or is it expecting you to consider the real-world consequence?
|
| 246 |
-
|
| 247 |
-
I think this is a classic riddle where the expected answer is zero because the remaining birds would fly away after the gunshot. But I should verify if that's the common interpretation. Let me check similar riddles. For example, "There are ten birds on a fence, you shoot one, how many are left?" The answer is usually zero because the rest fly away. So, applying that logic here, even though the setting is a tree instead of a fence, the principle would be the same. The act of shooting would scare the other birds, resulting in none remaining.
|
| 248 |
-
|
| 249 |
-
But wait, maybe the question is more literal. If the hunter successfully shoots one bird, that bird would presumably fall out of the tree, leaving nine. But if the other birds don't flee, they would remain. However, in reality, birds are
|
| 250 |
-
|
| 251 |
-
"""
|
| 252 |
-
|
| 253 |
~~~
|
| 254 |
-
|
| 255 |
### INT2 Inference on CPU
|
| 256 |
|
| 257 |
Requirements
|
|
|
|
| 31 |
|
| 32 |
please note int2 **may be slower** than int4 on CUDA due to kernel issue.
|
| 33 |
|
|
|
|
| 34 |
|
| 35 |
~~~python
|
| 36 |
+
|
| 37 |
import transformers
|
| 38 |
from transformers import AutoModelForCausalLM, AutoTokenizer
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 39 |
|
| 40 |
import torch
|
| 41 |
|
|
|
|
| 56 |
|
| 57 |
model = AutoModelForCausalLM.from_pretrained(
|
| 58 |
quantized_model_dir,
|
| 59 |
+
torch_dtype=torch.bfloat16,
|
|
|
|
| 60 |
device_map=device_map,
|
| 61 |
)
|
| 62 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 63 |
tokenizer = AutoTokenizer.from_pretrained(quantized_model_dir, trust_remote_code=True)
|
| 64 |
prompts = [
|
| 65 |
"9.11和9.8哪个数字大",
|
|
|
|
| 84 |
outputs = model.generate(
|
| 85 |
input_ids=inputs["input_ids"].to(model.device),
|
| 86 |
attention_mask=inputs["attention_mask"].to(model.device),
|
| 87 |
+
max_length=500, ##change this to align with the official usage
|
| 88 |
num_return_sequences=1,
|
| 89 |
do_sample=False ##change this to align with the official usage
|
| 90 |
)
|
|
|
|
| 100 |
print(f"Generated: {decoded_outputs[i]}")
|
| 101 |
print("-" * 50)
|
| 102 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 103 |
~~~
|
|
|
|
| 104 |
### INT2 Inference on CPU
|
| 105 |
|
| 106 |
Requirements
|