{
 "what": "goodput at the service-level objective for each serving arm on the public Mooncake production trace, one A100, one model; and what this company's isolation layer costs against the best prefix-cache baseline",
 "measured_utc": "2026-09-19T23:39:33Z",
 "source": {
  "file": "head_to_head_mooncake.json",
  "rev": "eb76dce",
  "sha256": "1313d13b1e6bf74cc6d71e4a0ce225495e2be994fc92d470fef92afc742670bb",
  "bytes": 3418
 },
 "arm_label_in_source": "the source file labels this company's arm with a former internal codename; the number below is that arm's, the label is not republished",
 "trace": "mooncake_fast25_conversation_trace",
 "gpu": "a100",
 "model": "Qwen/Qwen2.5-7B-Instruct",
 "requests": 64,
 "slo_ms": 1500.0,
 "concurrency": 24,
 "goodput_at_slo_req_s": {
  "no_prefix_cache": 6.434,
  "vllm_automatic_prefix_caching": 8.655,
  "lmcache": 8.127,
  "this_isolation_layer": 7.567
 },
 "ratios": {
  "vs_best_prefix_cache_baseline": 0.874,
  "vs_lmcache": 0.931
 },
 "reading": "the isolation layer trails the fastest prefix-cache baseline. It is not a speedup and is not presented as one."
}
