Mistral-NeMo-Minitron-Upscale-v2 / mergekit_config.yml
nlpguy's picture
Upload folder using huggingface_hub
1d431d5 verified
dtype: bfloat16
merge_method: passthrough
slices:
- sources:
- layer_range: [0, 8]
model: nvidia/Mistral-NeMo-Minitron-8B-Base
- sources:
- layer_range: [8, 16]
model: nvidia/Mistral-NeMo-Minitron-8B-Base
parameters:
scale:
- filter: o_proj
value: 0.5
- filter: down_proj
value: 0.5
- filter: q_proj
value: 0.85355339059
- filter: k_proj
value: 0.85355339059
- value: 1.0
- sources:
- layer_range: [8, 16]
model: nvidia/Mistral-NeMo-Minitron-8B-Base
parameters:
scale:
- filter: q_proj
value: 0.85355339059
- filter: k_proj
value: 0.85355339059
- value: 1.0
- sources:
- layer_range: [16, 17]
model: nvidia/Mistral-NeMo-Minitron-8B-Base
- sources:
- layer_range: [17, 24]
model: nvidia/Mistral-NeMo-Minitron-8B-Base
parameters:
scale:
- filter: o_proj
value: 0.5
- filter: down_proj
value: 0.5
- filter: q_proj
value: 0.85355339059
- filter: k_proj
value: 0.85355339059
- value: 1.0
- sources:
- layer_range: [17, 24]
model: nvidia/Mistral-NeMo-Minitron-8B-Base
parameters:
scale:
- filter: q_proj
value: 0.85355339059
- filter: k_proj
value: 0.85355339059
- value: 1.0
- sources:
- layer_range: [24, 25]
model: nvidia/Mistral-NeMo-Minitron-8B-Base
- sources:
- layer_range: [25, 32]
model: nvidia/Mistral-NeMo-Minitron-8B-Base
parameters:
scale:
- filter: o_proj
value: 0.5
- filter: down_proj
value: 0.5
- filter: q_proj
value: 0.85355339059
- filter: k_proj
value: 0.85355339059
- value: 1.0
- sources:
- layer_range: [25, 32]
model: nvidia/Mistral-NeMo-Minitron-8B-Base
parameters:
scale:
- filter: q_proj
value: 0.85355339059
- filter: k_proj
value: 0.85355339059
- value: 1.0
- sources:
- layer_range: [32, 40]
model: nvidia/Mistral-NeMo-Minitron-8B-Base