Co-authored-by: Gregory Shtrasberg <Gregory.Shtrasberg@amd.com> Co-authored-by: HaiShaw <hixiao@gmail.com> Co-authored-by: AdrianAbeyta <Adrian.Abeyta@amd.com> Co-authored-by: Matthew Wong <Matthew.Wong2@amd.com> Co-authored-by: root <root@gt-pla-u18-08.pla.dcgpu> Co-authored-by: mawong-amd <156021403+mawong-amd@users.noreply.github.com> Co-authored-by: ttbachyinsda <ttbachyinsda@outlook.com> Co-authored-by: guofangze <guofangze@kuaishou.com> Co-authored-by: Michael Goin <mgoin64@gmail.com> Co-authored-by: jacobthebanana <50071502+jacobthebanana@users.noreply.github.com> Co-authored-by: Woosuk Kwon <woosuk.kwon@berkeley.edu>
90 lines
3.4 KiB
JSON
90 lines
3.4 KiB
JSON
{
|
|
"model_type": "llama",
|
|
"kv_cache": {
|
|
"dtype": "float8_e4m3fn",
|
|
"scaling_factor": {
|
|
"0": {
|
|
"0": 0.0230364128947258,
|
|
"1": 0.01979283057153225,
|
|
"2": 0.0241350457072258,
|
|
"3": 0.0308314748108387,
|
|
"4": 0.0430733822286129,
|
|
"5": 0.0370396226644516,
|
|
"6": 0.0306222103536129,
|
|
"7": 0.0357491634786129,
|
|
"8": 0.0358189195394516,
|
|
"9": 0.0443289652466774,
|
|
"10": 0.0433175228536129,
|
|
"11": 0.0416782945394516,
|
|
"12": 0.0366908498108387,
|
|
"13": 0.0432477705180645,
|
|
"14": 0.0410505048930645,
|
|
"15": 0.0457589291036129,
|
|
"16": 0.0418526791036129,
|
|
"17": 0.0432477705180645,
|
|
"18": 0.0469447560608387,
|
|
"19": 0.0514787957072258,
|
|
"20": 0.0541294664144516,
|
|
"21": 0.0587681382894516,
|
|
"22": 0.0625,
|
|
"23": 0.0585588738322258,
|
|
"24": 0.0600237175822258,
|
|
"25": 0.0588030144572258,
|
|
"26": 0.0531180277466774,
|
|
"27": 0.06396484375,
|
|
"28": 0.0603027381002903,
|
|
"29": 0.0582101047039032,
|
|
"30": 0.0625348836183548,
|
|
"31": 0.0585588738322258,
|
|
"32": 0.0582798570394516,
|
|
"33": 0.0575125589966774,
|
|
"34": 0.0590820349752903,
|
|
"35": 0.0614188089966774,
|
|
"36": 0.0631975457072258,
|
|
"37": 0.0615931935608387,
|
|
"38": 0.0601283498108387,
|
|
"39": 0.0571986623108387,
|
|
"40": 0.0670340433716774,
|
|
"41": 0.0523507259786129,
|
|
"42": 0.0547223798930645,
|
|
"43": 0.0631975457072258,
|
|
"44": 0.0663713738322258,
|
|
"45": 0.0603376142680645,
|
|
"46": 0.0652204304933548,
|
|
"47": 0.0734514519572258,
|
|
"48": 0.0693708211183548,
|
|
"49": 0.0725446492433548,
|
|
"50": 0.0627790242433548,
|
|
"51": 0.0691266804933548,
|
|
"52": 0.0688825398683548,
|
|
"53": 0.068429134786129,
|
|
"54": 0.0605119988322258,
|
|
"55": 0.0799386203289032,
|
|
"56": 0.0853097140789032,
|
|
"57": 0.0661969929933548,
|
|
"58": 0.0689871683716774,
|
|
"59": 0.0724051371216774,
|
|
"60": 0.0541643425822258,
|
|
"61": 0.0626743882894516,
|
|
"62": 0.0628487765789032,
|
|
"63": 0.0607212632894516,
|
|
"64": 0.0589076466858387,
|
|
"65": 0.0451660193502903,
|
|
"66": 0.0453055277466774,
|
|
"67": 0.0414341539144516,
|
|
"68": 0.0385044664144516,
|
|
"69": 0.0414341539144516,
|
|
"70": 0.0466308631002903,
|
|
"71": 0.0399693101644516,
|
|
"72": 0.0437011756002903,
|
|
"73": 0.0434221550822258,
|
|
"74": 0.0428989976644516,
|
|
"75": 0.0401785746216774,
|
|
"76": 0.0431082621216774,
|
|
"77": 0.0484444759786129,
|
|
"78": 0.0417829267680645,
|
|
"79": 0.0418178029358387
|
|
}
|
|
}
|
|
}
|
|
} |