# SPDX-License-Identifier: Apache-2.0 # SPDX-FileCopyrightText: Copyright contributors to the vLLM project # Weight Shapes are in the format # ([K, N], TP_SPLIT_DIM) # Example: # A shape of ([14336, 4096], 0) indicates the following GEMM shape, # - TP1 : K = 14336, N = 4096 # - TP2 : K = 7168, N = 4096 # A shape of ([4096, 6144], 1) indicates the following GEMM shape, # - TP1 : K = 4096, N = 6144 # - TP4 : K = 4096, N = 1536 # TP1 shapes WEIGHT_SHAPES = { "mistralai/Mistral-7B-v0.1": [ ([4096, 6144], 1), ([4096, 4096], 0), ([4096, 28672], 1), ([14336, 4096], 0), ], "meta-llama/Llama-2-7b-hf": [ ([4096, 12288], 1), ([4096, 4096], 0), ([4096, 22016], 1), ([11008, 4096], 0), ], "meta-llama/Llama-3-8b": [ ([4096, 6144], 1), ([4096, 4096], 0), ([4096, 28672], 1), ([14336, 4096], 0), ], "meta-llama/Llama-2-13b-hf": [ ([5120, 15360], 1), ([5120, 5120], 0), ([5120, 27648], 1), ([13824, 5120], 0), ], "meta-llama/Llama-2-70b-hf": [ ([8192, 10240], 1), ([8192, 8192], 0), ([8192, 57344], 1), ([28672, 8192], 0), ], "meta-llama/Llama-3.1-405b-hf": [ ([16384, 18432], 1), ([16384, 16384], 0), ([16384, 106496], 1), ([53248, 16384], 0), ], "meta-llama/Llama-3.1-8B-Instruct": [ ([4096, 6144], 1), ([4096, 4096], 0), ([4096, 28672], 1), ([14336, 4096], 0), ], "meta-llama/Llama-3.3-70B-Instruct": [ ([8192, 10240], 1), ([8192, 8192], 0), ([8192, 57344], 1), ([28672, 8192], 0), ], "mistralai/Mistral-Large-Instruct-2407": [ ([12288, 14336], 1), ([12288, 12288], 0), ([12288, 57344], 1), ([28672, 12288], 0), ], "Qwen/Qwen2.5-7B-Instruct": [ ([3584, 4608], 1), ([3584, 3584], 0), ([3584, 37888], 1), ([18944, 3584], 0), ], "Qwen/Qwen2.5-32B-Instruct": [ ([5120, 7168], 1), ([5120, 5120], 0), ([5120, 55296], 1), ([27648, 5120], 0), ], "Qwen/Qwen2.5-72B-Instruct": [ ([8192, 10240], 1), ([8192, 8192], 0), ([8192, 59136], 1), ([29568, 8192], 0), ], "deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct": [ ([2048, 3072], 1), ([2048, 4096], 1), ([2048, 2048], 0), ([2048, 576], 0), ([2048, 21888], 1), ([10944, 2048], 0), ([2048, 2816], 1), ([1408, 2048], 0), ], }