{
  "window": "30d",
  "license": {
    "id": "CC-BY-4.0",
    "url": "https://creativecommons.org/licenses/by/4.0/",
    "attribution": "Measured by batchwatch - https://batchwatch.dev",
    "attribution_html": "Measured by <a href=\"https://batchwatch.dev\">batchwatch</a>",
    "source_url": "https://batchwatch.dev/license"
  },
  "models": [
    {
      "provider": "openai",
      "model": "gpt-5-nano",
      "mode": "batch",
      "requested_service_tier": "standard",
      "n": 3882,
      "completed_calls": 7718,
      "contributors": 0,
      "probes": 7718,
      "crowdsourced": false,
      "probe_only": true,
      "p50_s": 74,
      "p90_s": 4709,
      "p95_s": 24325,
      "latest_at": 1791590478,
      "last_measured_age_s": 628,
      "answerable": true,
      "confidence": "very_low"
    },
    {
      "provider": "openai",
      "model": "gpt-5-nano",
      "mode": "sync",
      "requested_service_tier": "flex",
      "n": 3979,
      "completed_calls": 4194,
      "contributors": 0,
      "probes": 4194,
      "crowdsourced": false,
      "probe_only": true,
      "p50_s": 2,
      "p90_s": 3,
      "p95_s": 3,
      "latest_at": 1791590921,
      "last_measured_age_s": 185,
      "answerable": true,
      "confidence": "very_low",
      "measurement_basis": {
        "measured_at_input_tokens": 8,
        "floor_clause": "Flex is measured at 8 input tokens and is synchronous, so that figure is the model's own work and a longer prompt makes it larger: read it as a floor. A batch figure is dominated by the wait to be scheduled, so the same extra work is a much smaller part of it.",
        "note": "Flex latency is measured on a fixed 8-input-token probe, the same fixed prompt size our batch probes use. A flex call is synchronous, so the figure is the model's own work and it grows with the prompt: read it as the floor a larger prompt starts from. A batch figure is dominated by queue time -- the wait to be scheduled -- so the same extra work is a much smaller part of it. The refusal rate is unaffected by prompt size: admission control happens before inference, so this probe measures it exactly as well as a large one would."
      }
    },
    {
      "provider": "google",
      "model": "gemini-3.6-flash",
      "mode": "sync",
      "requested_service_tier": "flex",
      "n": 3403,
      "completed_calls": 4067,
      "contributors": 0,
      "probes": 4067,
      "crowdsourced": false,
      "probe_only": true,
      "p50_s": 1,
      "p90_s": 2,
      "p95_s": 2,
      "latest_at": 1791590923,
      "last_measured_age_s": 183,
      "answerable": true,
      "confidence": "very_low",
      "measurement_basis": {
        "measured_at_input_tokens": 8,
        "floor_clause": "Flex is measured at 8 input tokens and is synchronous, so that figure is the model's own work and a longer prompt makes it larger: read it as a floor. A batch figure is dominated by the wait to be scheduled, so the same extra work is a much smaller part of it.",
        "note": "Flex latency is measured on a fixed 8-input-token probe, the same fixed prompt size our batch probes use. A flex call is synchronous, so the figure is the model's own work and it grows with the prompt: read it as the floor a larger prompt starts from. A batch figure is dominated by queue time -- the wait to be scheduled -- so the same extra work is a much smaller part of it. The refusal rate is unaffected by prompt size: admission control happens before inference, so this probe measures it exactly as well as a large one would."
      }
    },
    {
      "provider": "openai",
      "model": "gpt-5.6-sol",
      "mode": "sync",
      "requested_service_tier": "flex",
      "n": 3019,
      "completed_calls": 3508,
      "contributors": 0,
      "probes": 3508,
      "crowdsourced": false,
      "probe_only": true,
      "p50_s": 1,
      "p90_s": 3,
      "p95_s": 3,
      "latest_at": 1791590919,
      "last_measured_age_s": 187,
      "answerable": true,
      "confidence": "very_low",
      "measurement_basis": {
        "measured_at_input_tokens": 8,
        "floor_clause": "Flex is measured at 8 input tokens and is synchronous, so that figure is the model's own work and a longer prompt makes it larger: read it as a floor. A batch figure is dominated by the wait to be scheduled, so the same extra work is a much smaller part of it.",
        "note": "Flex latency is measured on a fixed 8-input-token probe, the same fixed prompt size our batch probes use. A flex call is synchronous, so the figure is the model's own work and it grows with the prompt: read it as the floor a larger prompt starts from. A batch figure is dominated by queue time -- the wait to be scheduled -- so the same extra work is a much smaller part of it. The refusal rate is unaffected by prompt size: admission control happens before inference, so this probe measures it exactly as well as a large one would."
      }
    },
    {
      "provider": "openai",
      "model": "gpt-5.6-luna",
      "mode": "batch",
      "requested_service_tier": "standard",
      "n": 3246,
      "completed_calls": 3248,
      "contributors": 0,
      "probes": 3248,
      "crowdsourced": false,
      "probe_only": true,
      "p50_s": 86,
      "p90_s": 956,
      "p95_s": 2199,
      "latest_at": 1791590140,
      "last_measured_age_s": 966,
      "answerable": true,
      "confidence": "very_low"
    },
    {
      "provider": "anthropic",
      "model": "claude-haiku-4-5",
      "mode": "batch",
      "requested_service_tier": "standard",
      "n": 2589,
      "completed_calls": 2839,
      "contributors": 0,
      "probes": 2839,
      "crowdsourced": false,
      "probe_only": true,
      "p50_s": 150,
      "p90_s": 409,
      "p95_s": 624,
      "latest_at": 1791590157,
      "last_measured_age_s": 949,
      "answerable": true,
      "confidence": "very_low"
    },
    {
      "provider": "google",
      "model": "gemini-3.7-flash",
      "mode": "batch",
      "requested_service_tier": "standard",
      "n": 2494,
      "completed_calls": 2584,
      "contributors": 0,
      "probes": 2584,
      "crowdsourced": false,
      "probe_only": true,
      "p50_s": 124,
      "p90_s": 170,
      "p95_s": 190,
      "latest_at": 1791589016,
      "last_measured_age_s": 2090,
      "answerable": true,
      "confidence": "very_low"
    },
    {
      "provider": "openai",
      "model": "gpt-5.6-sol",
      "mode": "batch",
      "requested_service_tier": "standard",
      "n": 1885,
      "completed_calls": 1913,
      "contributors": 0,
      "probes": 1913,
      "crowdsourced": false,
      "probe_only": true,
      "p50_s": 30,
      "p90_s": 1000,
      "p95_s": 4736,
      "latest_at": 1791587669,
      "last_measured_age_s": 3437,
      "answerable": true,
      "confidence": "very_low"
    }
  ]
}