{
  "mtp_observation": "MTP inference demonstrates improved throughput and reduced latency by predicting multiple tokens per forward pass.",
  "limitation": "A true delta requires a matched MTP-off run.",
  "metric_needed": "Normalized tokens-per-second and end-to-end latency under identical hardware and workload conditions."
}