togethercomputer
/

RedPajama-INCITE-Base-3B-v1

@@ -20,49 +20,111 @@ RedPajama-Base-INCITE-2.8B-v1, is a large transformer-based language model devel
 ## GPU Inference
 This requires a GPU with 8GB memory.
 ```python
 from transformers import AutoTokenizer, AutoModelForCausalLM
 # init
 tokenizer = AutoTokenizer.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1")
 model = AutoModelForCausalLM.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1", torch_dtype=torch.float16)
 model = model.to('cuda:0')
 # infer
-inputs = tokenizer("Hello", return_tensors='pt').to(model.device)
-outputs = model.generate(**inputs, max_new_tokens=10, do_sample=True, temperature=0.8)
-output_str = tokenizer.decode(outputs[0])
 print(output_str)
 ```
 ## GPU Inference in Int8
-This requires a GPU with 6GB memory.
 ```python
 from transformers import AutoTokenizer, AutoModelForCausalLM
 # init
 tokenizer = AutoTokenizer.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1")
-model = AutoModelForCausalLM.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1", device_map="auto", load_in_8bit=True)
 # infer
-inputs = tokenizer("Hello", return_tensors='pt').to(model.device)
-outputs = model.generate(**inputs, max_new_tokens=10, do_sample=True, temperature=0.8)
-output_str = tokenizer.decode(outputs[0])
 print(output_str)
 ```
 ## CPU Inference
 ```python
 from transformers import AutoTokenizer, AutoModelForCausalLM
 # init
 tokenizer = AutoTokenizer.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1")
-model = AutoModelForCausalLM.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1", torch_dtype=torch.bfloat16)
 # infer
-inputs = tokenizer("<human>: Hello!\n<bot>:", return_tensors='pt').to(model.device)
-outputs = model.generate(**inputs, max_new_tokens=10, do_sample=True, temperature=0.8)
-output_str = tokenizer.decode(outputs[0])
 print(output_str)
 ```
 # Uses

 ## GPU Inference
 This requires a GPU with 8GB memory.
 ```python
+import torch
+import transformers
 from transformers import AutoTokenizer, AutoModelForCausalLM
+MIN_TRANSFORMERS_VERSION = '4.25.1'
+# check transformers version
+assert transformers.__version__ >= MIN_TRANSFORMERS_VERSION, f'Please upgrade transformers to version {MIN_TRANSFORMERS_VERSION} or higher.'
 # init
 tokenizer = AutoTokenizer.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1")
 model = AutoModelForCausalLM.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1", torch_dtype=torch.float16)
 model = model.to('cuda:0')
 # infer
+prompt = "Alan Turing is"
+inputs = tokenizer(prompt, return_tensors='pt').to(model.device)
+input_length = inputs.input_ids.shape[1]
+outputs = model.generate(
+    **inputs, max_new_tokens=128, do_sample=True, temperature=0.7, top_p=0.7, top_k=50, return_dict_in_generate=True,
+)
+token = outputs.sequences[0, input_length:]
+output_str = tokenizer.decode(token)
 print(output_str)
+"""
+a name that has been synonymous with the computer age since the 1950s. The British mathematician, logician, and cryptanalyst is widely regarded as the father of modern computing. His contributions to the development of the modern computer and the theory of computation have had a profound impact on the world we live in today.
+Turing’s contributions to the development of the modern computer were made in the 1940s and 1950s. He is most famous for his work on the Turing machine, a theoretical model of a computing machine that was able to perform all the mathematical operations of a computer. Turing’s work on the...
+"""
 ```
 ## GPU Inference in Int8
+To run inference with int8, please ensure you have installed accelerate and bitandbytes. You can install them with the following command:
+```bash
+pip install accelerate
+pip install bitsandbytes
+```
+Then you can run inference with int8 as follows:
 ```python
+import torch
+import transformers
 from transformers import AutoTokenizer, AutoModelForCausalLM
+MIN_TRANSFORMERS_VERSION = '4.25.1'
+# check transformers version
+assert transformers.__version__ >= MIN_TRANSFORMERS_VERSION, f'Please upgrade transformers to version {MIN_TRANSFORMERS_VERSION} or higher.'
 # init
 tokenizer = AutoTokenizer.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1")
+model = AutoModelForCausalLM.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1", device_map='auto', torch_dtype=torch.float16, load_in_8bit=True)
 # infer
+prompt = "Alan Turing is"
+inputs = tokenizer(prompt, return_tensors='pt').to(model.device)
+input_length = inputs.input_ids.shape[1]
+outputs = model.generate(
+    **inputs, max_new_tokens=128, do_sample=True, temperature=0.7, top_p=0.7, top_k=50, return_dict_in_generate=True
+)
+token = outputs.sequences[0, input_length:]
+output_str = tokenizer.decode(token)
 print(output_str)
+"""
+the man who cracked the Enigma code during World War II, and who was later convicted of homosexual acts. He was a brilliant mathematician, and a visionary who foresaw the computer age....
+"""
 ```
 ## CPU Inference
+You can run inference on CPU as follows:
 ```python
+import torch
+import transformers
 from transformers import AutoTokenizer, AutoModelForCausalLM
+MIN_TRANSFORMERS_VERSION = '4.25.1'
+# check transformers version
+assert transformers.__version__ >= MIN_TRANSFORMERS_VERSION, f'Please upgrade transformers to version {MIN_TRANSFORMERS_VERSION} or higher.'
 # init
 tokenizer = AutoTokenizer.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1")
+model = AutoModelForCausalLM.from_pretrained("togethercomputer/RedPajama-Base-INCITE-2.8B-v1", torch_dtype=torch.float32)
 # infer
+prompt = "Alan Turing is"
+inputs = tokenizer(prompt, return_tensors='pt').to(model.device)
+input_length = inputs.input_ids.shape[1]
+outputs = model.generate(
+    **inputs, max_new_tokens=128, do_sample=True, temperature=0.7, top_p=0.7, top_k=50, return_dict_in_generate=True
+)
+token = outputs.sequences[0, input_length:]
+output_str = tokenizer.decode(token)
 print(output_str)
+"""
+one of the most famous people to have come out of Cambridge. He is also one of the most famous people to have been arrested for homosexuality.
+"""
 ```
+Please note that since `LayerNormKernelImpl` is not implemented in fp16 for CPU, we use fp32 for CPU inference.
 # Uses