[Simulator] Add high-fidelity CPU-based inference simulator (#33824)

Co-authored-by: zhouhaizhu.zhz <zhouhaizhu.zhz@alibaba-inc.com>
Co-authored-by: LinSiyuan814 <linsiyuan.lsy@alibaba-inc.com>
Co-authored-by: hzh0425 <hzh0425@apache.org>
This commit is contained in:
Ruiyan Ma
2026-09-04 11:12:11 +08:00
committed by GitHub
co-authored by zhouhaizhu.zhz LinSiyuan814 hzh0425
parent a5f07b1241
commit 59799a3687
81 changed files with 6624 additions and 0 deletions
@@ -0,0 +1,38 @@
[
{
"conversations": [
{
"from": "human",
"value": "Explain prefix caching in one concise sentence."
},
{
"from": "gpt",
"value": "Prefix caching reuses KV states shared by prompt prefixes."
}
]
},
{
"conversations": [
{
"from": "human",
"value": "What does time to first token measure?"
},
{
"from": "gpt",
"value": "It measures latency from request arrival to the first generated token."
}
]
},
{
"conversations": [
{
"from": "human",
"value": "Why does decode latency matter for serving?"
},
{
"from": "gpt",
"value": "Decode latency determines the cadence at which later tokens reach the user."
}
]
}
]
@@ -0,0 +1,3 @@
{"prompt":[100,101,102,103],"prompt_len":4,"output_len":4,"timestamp":0}
{"prompt":[100,101,102,104],"prompt_len":4,"output_len":4,"timestamp":75}
{"prompt":[200,201,202,203],"prompt_len":4,"output_len":4,"timestamp":250}