File size: 1,879 Bytes
bbfa6f6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
#!/bin/bash

deepspeed \
    --num_nodes=1 \
    --num_gpus=6 \
    --master_port=25002 \
    llava/train/train_mem.py \
    --deepspeed ./scripts/zero2.json \
    --model_name_or_path /mnt/bn/algo-masp-nas-2/xiangchen/model/thoth/thoth_6b5_v2_4_36_0 \
    --version thoth \
    --dataset_config /mnt/bn/algo-masp-nas-2/xiangchen/repo/LLaVA/llava/configs/release_version/finetune_gpt4v_caption.yaml \
    --vision_tower eva-vit-g \
    --vit_model_path /mnt/bn/data-tns-algo-masp/baiyi.by/masp/model/eva_vit_g.pth \
    --adapter_module_name qformer_compress_token_v1_128 \
    --adapter_module_path /mnt/bn/data-tns-algo-masp/baiyi.by/masp/model/blip2_pretrained_flant5xxl.pth \
    --pretrain_mm_mlp_adapter /mnt/bn/masp-nas/xiangchen/model/masp_models/checkpoints/llava-thothv2-pretrain-projector-2/mm_projector.bin \
    --freeze_adapter False \
    --mm_vision_select_layer -2 \
    --mm_use_start_end True \
    --mm_use_patch_token False \
    --image_aspect_ratio pad \
    --num_token_per_image 0 \
    --image_aspect_ratio anyres \
    --mm_patch_merge_type flat \
    --image_grid_pinpoints \[\(448,\ 672\),\ \(672,\ 448\)\] \
    --bf16 True \
    --output_dir /mnt/bn/masp-nas/xiangchen/model/masp_models/checkpoints/llava-thothv2_mar_release_gpt4v_compress_token_224  \
    --num_train_epochs 1 \
    --group_by_modality_length True \
    --per_device_train_batch_size 4 \
    --per_device_eval_batch_size 4 \
    --gradient_accumulation_steps 4 \
    --evaluation_strategy "no" \
    --save_strategy "steps" \
    --save_steps 1000 \
    --save_total_limit 1 \
    --learning_rate 1e-5 \
    --weight_decay 0. \
    --warmup_ratio 0.03 \
    --lr_scheduler_type "cosine" \
    --logging_steps 1 \
    --tf32 True \
    --model_max_length 4096 \
    --gradient_checkpointing True \
    --dataloader_num_workers 4 \
    --lazy_preprocess True \
    --report_to none