InferScale-Sim / tests /test_prefix_cache.py
ArchitSharma's picture
Commiting v0.3
44745f2
Raw
History Blame Contribute Delete
884 Bytes
from inferscale import run_simulation
def cfg(cache: bool):
return {
"model": "Qwen2.5-3B",
"accelerator": "L4",
"quantization": "int8",
"scheduler": "continuous_fcfs",
"duration_s": 6,
"request_rate_rps": 2,
"prompt_tokens_mean": 512,
"prompt_tokens_cv": 0,
"output_tokens_mean": 8,
"output_tokens_cv": 0,
"seed": 2,
"prefix_cache_enabled": cache,
"shared_prefix_tokens": 256,
"prefix_reuse_fraction": 1.0,
}
def test_prefix_cache_reduces_prefill_work_and_is_reported():
uncached = run_simulation(cfg(False))
cached = run_simulation(cfg(True))
assert cached["resource"]["prefix_cache_hit_rate"] == 1.0
assert cached["resource"]["prefill_tokens_saved"] > 0
assert cached["latency"]["ttft_ms"]["p50"] < uncached["latency"]["ttft_ms"]["p50"]