mirror of https://github.com/vllm-project/vllm.git
99 lines
2.6 KiB
Python
99 lines
2.6 KiB
Python
# SPDX-License-Identifier: Apache-2.0
|
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
|
|
|
# Weight Shapes are in the format
|
|
# ([K, N], TP_SPLIT_DIM)
|
|
# Example:
|
|
# A shape of ([14336, 4096], 0) indicates the following GEMM shape,
|
|
# - TP1 : K = 14336, N = 4096
|
|
# - TP2 : K = 7168, N = 4096
|
|
# A shape of ([4096, 6144], 1) indicates the following GEMM shape,
|
|
# - TP1 : K = 4096, N = 6144
|
|
# - TP4 : K = 4096, N = 1536
|
|
|
|
# TP1 shapes
|
|
WEIGHT_SHAPES = {
|
|
"mistralai/Mistral-7B-v0.1": [
|
|
([4096, 6144], 1),
|
|
([4096, 4096], 0),
|
|
([4096, 28672], 1),
|
|
([14336, 4096], 0),
|
|
],
|
|
"meta-llama/Llama-2-7b-hf": [
|
|
([4096, 12288], 1),
|
|
([4096, 4096], 0),
|
|
([4096, 22016], 1),
|
|
([11008, 4096], 0),
|
|
],
|
|
"meta-llama/Llama-3-8b": [
|
|
([4096, 6144], 1),
|
|
([4096, 4096], 0),
|
|
([4096, 28672], 1),
|
|
([14336, 4096], 0),
|
|
],
|
|
"meta-llama/Llama-2-13b-hf": [
|
|
([5120, 15360], 1),
|
|
([5120, 5120], 0),
|
|
([5120, 27648], 1),
|
|
([13824, 5120], 0),
|
|
],
|
|
"meta-llama/Llama-2-70b-hf": [
|
|
([8192, 10240], 1),
|
|
([8192, 8192], 0),
|
|
([8192, 57344], 1),
|
|
([28672, 8192], 0),
|
|
],
|
|
"meta-llama/Llama-3.1-405b-hf": [
|
|
([16384, 18432], 1),
|
|
([16384, 16384], 0),
|
|
([16384, 106496], 1),
|
|
([53248, 16384], 0),
|
|
],
|
|
"meta-llama/Llama-3.1-8B-Instruct": [
|
|
([4096, 6144], 1),
|
|
([4096, 4096], 0),
|
|
([4096, 28672], 1),
|
|
([14336, 4096], 0),
|
|
],
|
|
"meta-llama/Llama-3.3-70B-Instruct": [
|
|
([8192, 10240], 1),
|
|
([8192, 8192], 0),
|
|
([8192, 57344], 1),
|
|
([28672, 8192], 0),
|
|
],
|
|
"mistralai/Mistral-Large-Instruct-2407": [
|
|
([12288, 14336], 1),
|
|
([12288, 12288], 0),
|
|
([12288, 57344], 1),
|
|
([28672, 12288], 0),
|
|
],
|
|
"Qwen/Qwen2.5-7B-Instruct": [
|
|
([3584, 4608], 1),
|
|
([3584, 3584], 0),
|
|
([3584, 37888], 1),
|
|
([18944, 3584], 0),
|
|
],
|
|
"Qwen/Qwen2.5-32B-Instruct": [
|
|
([5120, 7168], 1),
|
|
([5120, 5120], 0),
|
|
([5120, 55296], 1),
|
|
([27648, 5120], 0),
|
|
],
|
|
"Qwen/Qwen2.5-72B-Instruct": [
|
|
([8192, 10240], 1),
|
|
([8192, 8192], 0),
|
|
([8192, 59136], 1),
|
|
([29568, 8192], 0),
|
|
],
|
|
"deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct": [
|
|
([2048, 3072], 1),
|
|
([2048, 4096], 1),
|
|
([2048, 2048], 0),
|
|
([2048, 576], 0),
|
|
([2048, 21888], 1),
|
|
([10944, 2048], 0),
|
|
([2048, 2816], 1),
|
|
([1408, 2048], 0),
|
|
],
|
|
}
|