mtasic85's picture
model
d2eb966
|
raw
history blame
564 Bytes

Train

Tokenizer

cd scripts
python -m venv venv
source venv/bin/activate
pip install -U -r requirements.in
python -B train_tokenizer.py

Dataset

cd scripts
python -m venv venv-lit
source venv-lit/bin/activate
pip install -U -r requirements-lit.in
python -B prepare_pretrain_dataset.py

Model

cd scripts
python -m venv venv-lit
source venv-lit/bin/activate
pip install -U -r requirements-lit.in
litgpt pretrain --data LitData --data.data_path "../data/" --config ./model.yaml