{
 "what": "one tenant reading another tenant's cached prompt blocks on a live GPU, before and after a per-tenant cache salt; measured as vLLM's own num_cached_tokens per prompt",
 "measured_utc": "2026-09-19T23:39:33Z",
 "source": {
  "file": "salt_close_scale.json",
  "rev": "eb76dce",
  "sha256": "cdd0a7c1de96754daf0f31932a44d86c050860afc18fa9421867251490fd0044",
  "bytes": 12182
 },
 "certificate_committed": "24 July 2026",
 "republished_without": {
  "keys": [
   "/honest_scope/4",
   "/measures",
   "/merkle_note",
   "/replaces_nothing"
  ],
  "why": "these keys name a former internal project codename that collides with a third party's company name. No measured value is among them; every number below is the source file's, unchanged."
 },
 "gpu": "NVIDIA A100-SXM4-40GB",
 "model": "Qwen/Qwen2.5-7B-Instruct",
 "n_prefixes": 12,
 "seed": 20260724001,
 "cross_tenant_hits_before_the_salt": {
  "min": 544,
  "median": 664.0,
  "max": 784,
  "per_prefix": [
   784,
   544,
   784,
   752,
   624,
   624,
   672,
   624,
   624,
   784,
   656,
   704
  ],
  "positive_on_every_prefix": true
 },
 "cross_tenant_hits_after_the_salt": {
  "per_prefix": [
   0,
   0,
   0,
   0,
   0,
   0,
   0,
   0,
   0,
   0,
   0,
   0
  ],
  "zero_on_every_prefix": true
 },
 "own_tenant_reuse_preserved_exactly": true,
 "rule_of_three_95pct_upper_bound_on_cross_tenant_hit_rate": 0.25,
 "checks": {
  "prefixes_mutually_distinct": true,
  "positive_control_baseline_reuse_gt0_every_prefix": true,
  "leak_reproduced_cross_tenant_unsalted_gt0_every_prefix": true,
  "closed_cross_tenant_salted_eq0_every_prefix": true,
  "reuse_preserved_same_tenant_salted_gt0_every_prefix": true
 },
 "result": "CLOSED_AT_SCALE",
 "full_record": {
  "artifact": "salt_close_scale_gpu",
  "model": "Qwen/Qwen2.5-7B-Instruct",
  "provider": "modal:A100",
  "status": "MEASURED",
  "no_speedup_claimed": true,
  "n_prefixes_requested": 12,
  "seed": 20260724001,
  "replicates": "oss/AUDIT.md F5 — the flagship 208 -> 0 was n = 1 per condition on facebook/opt-125m (125M toy). This is n >= 10 distinct prefixes per condition on a 7B model, with the full distribution reported.",
  "distinct_from_gate_arm": "This is a measured TOKEN EFFECT in the native cache. It is NOT the extracted-Lean-gate connector arm of remote/vllm_inengine_weave_modal.py, where BOTH the admitted and the refused tenant received 0 reused tokens (token_level_contrast: false) — that arm shows a DECISION, not a token effect. The two claims are kept apart.",
  "gpu_requested_env_SF_GPU": "A100",
  "gpu_env_seen_in_container": "A100",
  "gpu_name_ground_truth": "NVIDIA A100-SXM4-40GB",
  "gpu_env_was_baked_into_image": true,
  "gpu_provenance_consistent": true,
  "gpu_provenance_note": "torch.cuda.get_device_name(0) is the GROUND TRUTH; SF_GPU is only a request label and is baked into the image so it cannot silently disagree with the container. If gpu_provenance_consistent is false, believe gpu_name_ground_truth and nothing else.",
  "arm_order_executed": [
   "unsalted",
   "salted"
  ],
  "arms": {
   "unsalted": {
    "native_prefix_caching": true,
    "salted": false,
    "populate": [
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0
    ],
    "same_tenant": [
     784,
     544,
     784,
     752,
     624,
     624,
     672,
     624,
     624,
     784,
     656,
     704
    ],
    "cross_tenant": [
     784,
     544,
     784,
     752,
     624,
     624,
     672,
     624,
     624,
     784,
     656,
     704
    ],
    "prompt_tokens": [
     789,
     552,
     793,
     754,
     630,
     633,
     674,
     633,
     633,
     794,
     671,
     711
    ],
    "cross_tenant_decode_nonempty": [
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true
    ],
    "per_prefix_probe_order": [
     "cross",
     "same",
     "same",
     "cross",
     "same",
     "cross",
     "cross",
     "same",
     "same",
     "cross",
     "same",
     "cross"
    ],
    "cost_aborted_after": null,
    "num_gpu_blocks": null,
    "block_size": null,
    "kv_capacity_tokens": null
   },
   "salted": {
    "native_prefix_caching": true,
    "salted": true,
    "populate": [
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0
    ],
    "same_tenant": [
     784,
     544,
     784,
     752,
     624,
     624,
     672,
     624,
     624,
     784,
     656,
     704
    ],
    "cross_tenant": [
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0
    ],
    "prompt_tokens": [
     789,
     552,
     793,
     754,
     630,
     633,
     674,
     633,
     633,
     794,
     671,
     711
    ],
    "cross_tenant_decode_nonempty": [
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true,
     true
    ],
    "per_prefix_probe_order": [
     "same",
     "same",
     "cross",
     "same",
     "same",
     "cross",
     "same",
     "cross",
     "same",
     "cross",
     "same",
     "cross"
    ],
    "cost_aborted_after": null,
    "num_gpu_blocks": null,
    "block_size": null,
    "kv_capacity_tokens": null
   }
  },
  "usd_estimate": 0.23,
  "cost_cap_usd": 40.0,
  "conditions": {
   "same_tenant_unsalted": {
    "condition": "same_tenant_unsalted",
    "n": 12,
    "expectation": "> 0 on every prefix — BASELINE REUSE / POSITIVE CONTROL. If this fails the engine was not caching at all and every zero below is VACUOUS.",
    "min": 544,
    "median": 664.0,
    "max": 784,
    "raw_per_prefix": [
     784,
     544,
     784,
     752,
     624,
     624,
     672,
     624,
     624,
     784,
     656,
     704
    ],
    "all_positive": true,
    "all_zero": false
   },
   "cross_tenant_unsalted": {
    "condition": "cross_tenant_unsalted",
    "n": 12,
    "expectation": "> 0 on every prefix — THE LEAK. One tenant served another tenant's resident prefix blocks through vLLM's tenant-blind native cache.",
    "min": 544,
    "median": 664.0,
    "max": 784,
    "raw_per_prefix": [
     784,
     544,
     784,
     752,
     624,
     624,
     672,
     624,
     624,
     784,
     656,
     704
    ],
    "all_positive": true,
    "all_zero": false
   },
   "same_tenant_salted": {
    "condition": "same_tenant_salted",
    "n": 12,
    "expectation": "> 0 on every prefix — same-tenant reuse is PRESERVED under the tenant-bound salt.",
    "min": 544,
    "median": 664.0,
    "max": 784,
    "raw_per_prefix": [
     784,
     544,
     784,
     752,
     624,
     624,
     672,
     624,
     624,
     784,
     656,
     704
    ],
    "all_positive": true,
    "all_zero": false
   },
   "cross_tenant_salted": {
    "condition": "cross_tenant_salted",
    "n": 12,
    "expectation": "EXACTLY 0 on every prefix — THE CLOSE. No block hash exists under which the other tenant could match.",
    "min": 0,
    "median": 0.0,
    "max": 0,
    "raw_per_prefix": [
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0,
     0
    ],
    "all_positive": false,
    "all_zero": true
   }
  },
  "preconditions": {
   "prefixes_mutually_distinct": true,
   "max_first_touch_cached_tokens": 0,
   "why": "a nonzero first-touch count would mean the prefixes share leading blocks, and every cross-tenant number would be confounded by cross-PREFIX reuse.",
   "kv_capacity_tokens": null,
   "block_size": null,
   "distinct_prefix_tokens_resident": 8267,
   "kv_capacity_ample": false,
   "eviction_note": "reported so eviction is ruled out by measurement, not by assumption: a cross_tenant_salted zero caused by EVICTION rather than by hash separation would be a false close.",
   "cross_tenant_decode_nonempty_all": true
  },
  "verdict_checks": {
   "prefixes_mutually_distinct": true,
   "positive_control_baseline_reuse_gt0_every_prefix": true,
   "leak_reproduced_cross_tenant_unsalted_gt0_every_prefix": true,
   "closed_cross_tenant_salted_eq0_every_prefix": true,
   "reuse_preserved_same_tenant_salted_gt0_every_prefix": true
  },
  "verdict_rule": "CLOSED_AT_SCALE requires the CONJUNCTION of all five checks above, each evaluated on EVERY prefix. Any false -> NOT_ESTABLISHED.",
  "reuse_preservation_exact_per_prefix": [
   true,
   true,
   true,
   true,
   true,
   true,
   true,
   true,
   true,
   true,
   true,
   true
  ],
  "reuse_preservation_exact_all": true,
  "rule_of_three_95pct_upper_bound_on_cross_tenant_hit_rate": 0.25,
  "rule_of_three_note": "with 0 cross-tenant hits in n = 12 prefixes the 95% upper bound on the hit rate is 3/n. Quoting a precision tighter than this from this run alone would repeat the F4 defect.",
  "cached_fraction_note": "num_cached_tokens is BLOCK-QUANTIZED (block_size is recorded above; vLLM also never reuses the final block), so any 'percent of the prompt cached' figure is dominated by the identity block_size * floor(n/block_size) / n and is predictable WITHOUT a GPU. Raw counts are reported here for that reason; no cache-hit percentage is claimed.",
  "result": "CLOSED_AT_SCALE",
  "verdict": "REPLICATED AT SCALE on Qwen/Qwen2.5-7B-Instruct (NVIDIA A100-SXM4-40GB), n = 12 distinct prefixes per condition, native prefix caching ENABLED in all four conditions. THE LEAK reproduces: cross-tenant unsalted cached tokens min/median/max = 544/664.0/784 on 12/12 prefixes. THE CLOSE holds: cross-tenant SALTED = 0 on 12/12 prefixes. Same-tenant reuse is PRESERVED under the salt (min/median/max = 544/664.0/784). This is a DATA-PATH (cached-token) result, not a timing result, and no speedup is claimed.",
  "honest_scope": [
   "DATA PATH, NOT TIMING. The measured quantity is RequestOutput.num_cached_tokens — prompt tokens the engine did not recompute. This says nothing about wall-clock, throughput, or a timing side channel; the timing question lives in remote/timing_oracle_*_modal.py and is separately underpowered (oss/AUDIT.md F6).",
   "NOT 'isolation at zero cache cost'. That phrasing is RETRACTED (oss/AUDIT.md F5): it was measured on a workload consisting solely of a tenant re-sending its OWN prompt, which excludes by construction the very cross-tenant sharing that tenant-salting eliminates — true of this workload, false in general. The honest form, and the only one used here, is: SAME-TENANT REUSE IS PRESERVED.",
   "NOT the extracted-Lean-gate result. In the connector/gate arm of vllm_inengine_weave_modal.py BOTH the admitted and the refused tenant received 0 reused tokens (token_level_contrast: false) — that arm evidences a DECISION, not a measured token effect. This harness evidences a token effect in vLLM's NATIVE cache with no connector installed. Do not merge the two claims.",
   "The salt is a NAMESPACE SEPARATOR, not encryption and not access control. It prevents cross-tenant block-hash collision. It does not protect KV bytes, and it does not stop an attacker who can already authenticate as the other tenant.",
   "num_cached_tokens is BLOCK-QUANTIZED and vLLM never reuses the final block, so any 'fraction of the prompt cached' number is dominated by block-size arithmetic and is predictable without a GPU. Only raw counts are reported; no hit-rate percentage is claimed.",
   "SCOPE OF THE SAMPLE: n distinct prefixes on ONE model, ONE GPU type, ONE vLLM version, ONE seed, static single-request submission. It is a scale replication of an n = 1 toy-model result — not a fleet claim, not a multi-version claim, and not a continuous-batching claim.",
   "With 0 cross-tenant hits observed the 95% upper bound on the residual hit rate is 3/n (rule of three). The separation mechanism is deterministic hash-space separation; the MEASUREMENT precision is still bounded by n.",
   "NO SPEEDUP IS CLAIMED OR MEASURED anywhere in this harness.",
   "GPU provenance: SF_GPU is a request label baked into the image; torch.cuda.get_device_name(0) is the ground truth. If the two disagree, the ground truth wins and the run should be rejected."
  ]
 }
}
