[Kernel][ROCm][AMD] fused_moe Triton configs v2 for mi300X (#5932)
This commit is contained in:
parent
64e8d2a783
commit
c3dde367f1
@ -1,128 +1,200 @@
|
|||||||
{
|
{
|
||||||
"1": {
|
"1": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"2": {
|
"2": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"4": {
|
"4": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 64,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"8": {
|
"8": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 32,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 1,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"16": {
|
"16": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"24": {
|
"24": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 64,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 1,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"32": {
|
"32": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 1
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"48": {
|
"48": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"64": {
|
"64": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 32,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"96": {
|
"96": {
|
||||||
"BLOCK_SIZE_M": 32,
|
"BLOCK_SIZE_M": 32,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 16,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"128": {
|
"128": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 64,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"256": {
|
"256": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"512": {
|
"512": {
|
||||||
"BLOCK_SIZE_M": 256,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"1024": {
|
"1024": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 32,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"1536": {
|
"1536": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"2048": {
|
"2048": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 256,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"3072": {
|
"3072": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 256,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"4096": {
|
"4096": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
182
vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=AMD_Instinct_MI300X.json
Normal file → Executable file
182
vllm/model_executor/layers/fused_moe/configs/E=8,N=1792,device_name=AMD_Instinct_MI300X.json
Normal file → Executable file
@ -1,110 +1,200 @@
|
|||||||
{
|
{
|
||||||
"1": {
|
"1": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 64
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"2": {
|
"2": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 32,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 32
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"4": {
|
"4": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 32,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 8
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"8": {
|
"8": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"16": {
|
"16": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 64,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"24": {
|
"24": {
|
||||||
"BLOCK_SIZE_M": 32,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"32": {
|
"32": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 8
|
"GROUP_SIZE_M": 4,
|
||||||
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"48": {
|
"48": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 8
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"64": {
|
"64": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 32,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"96": {
|
"96": {
|
||||||
"BLOCK_SIZE_M": 32,
|
"BLOCK_SIZE_M": 32,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 8
|
"GROUP_SIZE_M": 4,
|
||||||
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"128": {
|
"128": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 64,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 32
|
"GROUP_SIZE_M": 4,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"256": {
|
"256": {
|
||||||
"BLOCK_SIZE_M": 32,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 4,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"512": {
|
"512": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 64,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"1024": {
|
"1024": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"1536": {
|
"1536": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"2048": {
|
"2048": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"3072": {
|
"3072": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"4096": {
|
"4096": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@ -1,128 +1,200 @@
|
|||||||
{
|
{
|
||||||
"1": {
|
"1": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"2": {
|
"2": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"4": {
|
"4": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 32,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 32,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"8": {
|
"8": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 32,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"16": {
|
"16": {
|
||||||
"BLOCK_SIZE_M": 32,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 16,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"24": {
|
"24": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"32": {
|
"32": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 256,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 16,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"48": {
|
"48": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 16,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"64": {
|
"64": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 32,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 32,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"96": {
|
"96": {
|
||||||
"BLOCK_SIZE_M": 32,
|
"BLOCK_SIZE_M": 32,
|
||||||
"BLOCK_SIZE_N": 32,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 16,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"128": {
|
"128": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 64,
|
||||||
"BLOCK_SIZE_N": 256,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"256": {
|
"256": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"512": {
|
"512": {
|
||||||
"BLOCK_SIZE_M": 64,
|
|
||||||
"BLOCK_SIZE_N": 64,
|
|
||||||
"BLOCK_SIZE_K": 128,
|
|
||||||
"GROUP_SIZE_M": 1,
|
|
||||||
"num_stages": 0
|
|
||||||
},
|
|
||||||
"1024": {
|
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 64,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 32,
|
||||||
|
"kpack": 2
|
||||||
|
},
|
||||||
|
"1024": {
|
||||||
|
"BLOCK_SIZE_M": 128,
|
||||||
|
"BLOCK_SIZE_N": 128,
|
||||||
|
"BLOCK_SIZE_K": 64,
|
||||||
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"1536": {
|
"1536": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"2048": {
|
"2048": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"3072": {
|
"3072": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"4096": {
|
"4096": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@ -1,128 +1,200 @@
|
|||||||
{
|
{
|
||||||
"1": {
|
"1": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"2": {
|
"2": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 32,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"4": {
|
"4": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 32,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"8": {
|
"8": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"16": {
|
"16": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 16,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 256,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"24": {
|
"24": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 32,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 1
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"32": {
|
"32": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 16,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"48": {
|
"48": {
|
||||||
"BLOCK_SIZE_M": 16,
|
"BLOCK_SIZE_M": 16,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 16,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 2,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"64": {
|
"64": {
|
||||||
"BLOCK_SIZE_M": 32,
|
"BLOCK_SIZE_M": 32,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 256,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 1
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"96": {
|
"96": {
|
||||||
"BLOCK_SIZE_M": 32,
|
"BLOCK_SIZE_M": 32,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 32,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 128,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 4,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"128": {
|
"128": {
|
||||||
"BLOCK_SIZE_M": 64,
|
"BLOCK_SIZE_M": 64,
|
||||||
"BLOCK_SIZE_N": 64,
|
"BLOCK_SIZE_N": 64,
|
||||||
"BLOCK_SIZE_K": 128,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"256": {
|
"256": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 64,
|
|
||||||
"BLOCK_SIZE_K": 64,
|
|
||||||
"GROUP_SIZE_M": 8,
|
|
||||||
"num_stages": 0
|
|
||||||
},
|
|
||||||
"512": {
|
|
||||||
"BLOCK_SIZE_M": 256,
|
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 8,
|
"GROUP_SIZE_M": 4,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 32,
|
||||||
|
"kpack": 2
|
||||||
|
},
|
||||||
|
"512": {
|
||||||
|
"BLOCK_SIZE_M": 128,
|
||||||
|
"BLOCK_SIZE_N": 128,
|
||||||
|
"BLOCK_SIZE_K": 64,
|
||||||
|
"GROUP_SIZE_M": 1,
|
||||||
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"1024": {
|
"1024": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"1536": {
|
"1536": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"2048": {
|
"2048": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
},
|
},
|
||||||
"3072": {
|
"3072": {
|
||||||
"BLOCK_SIZE_M": 128,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 128,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 64,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 2
|
||||||
},
|
},
|
||||||
"4096": {
|
"4096": {
|
||||||
"BLOCK_SIZE_M": 256,
|
"BLOCK_SIZE_M": 128,
|
||||||
"BLOCK_SIZE_N": 256,
|
"BLOCK_SIZE_N": 128,
|
||||||
"BLOCK_SIZE_K": 32,
|
"BLOCK_SIZE_K": 64,
|
||||||
"GROUP_SIZE_M": 1,
|
"GROUP_SIZE_M": 1,
|
||||||
"num_stages": 0
|
"num_warps": 8,
|
||||||
|
"num_stages": 0,
|
||||||
|
"waves_per_eu": 0,
|
||||||
|
"matrix_instr_nonkdim": 16,
|
||||||
|
"kpack": 1
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user