|
@@ -31,7 +31,19 @@ model_base_shards = {
|
|
|
### llava
|
|
|
"llava-1.5-7b-hf": {"MLXDynamicShardInferenceEngine": Shard(model_id="llava-hf/llava-1.5-7b-hf", start_layer=0, end_layer=0, n_layers=32),},
|
|
|
### qwen
|
|
|
+ "qwen-2.5-7b": {
|
|
|
+ "MLXDynamicShardInferenceEngine": Shard(model_id="mlx-community/Qwen2.5-7B-Instruct-4bit", start_layer=0, end_layer=0, n_layers=28),
|
|
|
+ },
|
|
|
+ "qwen-2.5-math-7b": {
|
|
|
+ "MLXDynamicShardInferenceEngine": Shard(model_id="mlx-community/Qwen2.5-Math-7B-Instruct-4bit", start_layer=0, end_layer=0, n_layers=28),
|
|
|
+ },
|
|
|
"qwen-2.5-14b": {
|
|
|
"MLXDynamicShardInferenceEngine": Shard(model_id="mlx-community/Qwen2.5-14B-Instruct-4bit", start_layer=0, end_layer=0, n_layers=48),
|
|
|
},
|
|
|
+ "qwen-2.5-72b": {
|
|
|
+ "MLXDynamicShardInferenceEngine": Shard(model_id="mlx-community/Qwen2.5-72B-Instruct-4bit", start_layer=0, end_layer=0, n_layers=80),
|
|
|
+ },
|
|
|
+ "qwen-2.5-math-72b": {
|
|
|
+ "MLXDynamicShardInferenceEngine": Shard(model_id="mlx-community/Qwen2.5-Math-72B-Instruct-4bit", start_layer=0, end_layer=0, n_layers=80),
|
|
|
+ },
|
|
|
}
|