{
  "mtp_observation": "MTP typically increases inference throughput by predicting multiple tokens per forward pass, reducing the total number of model evaluations needed for generation.",
  "limitation": "A true delta requires a matched MTP-off run.",
  "metric_needed": "tokens_per_second and p50/p99 latency under identical hardware, batch size, and prompt conditions."
}