[Simulator] Add high-fidelity CPU-based inference simulator (#33824)
Co-authored-by: zhouhaizhu.zhz <zhouhaizhu.zhz@alibaba-inc.com> Co-authored-by: LinSiyuan814 <linsiyuan.lsy@alibaba-inc.com> Co-authored-by: hzh0425 <hzh0425@apache.org>
This commit is contained in:
co-authored by
zhouhaizhu.zhz
LinSiyuan814
hzh0425
parent
a5f07b1241
commit
59799a3687
@@ -0,0 +1,38 @@
|
||||
[
|
||||
{
|
||||
"conversations": [
|
||||
{
|
||||
"from": "human",
|
||||
"value": "Explain prefix caching in one concise sentence."
|
||||
},
|
||||
{
|
||||
"from": "gpt",
|
||||
"value": "Prefix caching reuses KV states shared by prompt prefixes."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"conversations": [
|
||||
{
|
||||
"from": "human",
|
||||
"value": "What does time to first token measure?"
|
||||
},
|
||||
{
|
||||
"from": "gpt",
|
||||
"value": "It measures latency from request arrival to the first generated token."
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"conversations": [
|
||||
{
|
||||
"from": "human",
|
||||
"value": "Why does decode latency matter for serving?"
|
||||
},
|
||||
{
|
||||
"from": "gpt",
|
||||
"value": "Decode latency determines the cadence at which later tokens reach the user."
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,3 @@
|
||||
{"prompt":[100,101,102,103],"prompt_len":4,"output_len":4,"timestamp":0}
|
||||
{"prompt":[100,101,102,104],"prompt_len":4,"output_len":4,"timestamp":75}
|
||||
{"prompt":[200,201,202,203],"prompt_len":4,"output_len":4,"timestamp":250}
|
||||
Reference in New Issue
Block a user