Co-authored-by: zhouhaizhu.zhz <zhouhaizhu.zhz@alibaba-inc.com> Co-authored-by: LinSiyuan814 <linsiyuan.lsy@alibaba-inc.com> Co-authored-by: hzh0425 <hzh0425@apache.org>
39 lines
811 B
JSON
39 lines
811 B
JSON
[
|
|
{
|
|
"conversations": [
|
|
{
|
|
"from": "human",
|
|
"value": "Explain prefix caching in one concise sentence."
|
|
},
|
|
{
|
|
"from": "gpt",
|
|
"value": "Prefix caching reuses KV states shared by prompt prefixes."
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"conversations": [
|
|
{
|
|
"from": "human",
|
|
"value": "What does time to first token measure?"
|
|
},
|
|
{
|
|
"from": "gpt",
|
|
"value": "It measures latency from request arrival to the first generated token."
|
|
}
|
|
]
|
|
},
|
|
{
|
|
"conversations": [
|
|
{
|
|
"from": "human",
|
|
"value": "Why does decode latency matter for serving?"
|
|
},
|
|
{
|
|
"from": "gpt",
|
|
"value": "Decode latency determines the cadence at which later tokens reach the user."
|
|
}
|
|
]
|
|
}
|
|
]
|