{
  "generated": "2026-10-04T23:20:00Z",
  "name": "StormEval daily AI model test results",
  "description": "Daily results of StormEval's automated tests of AI models. Each task is an issue in a large, working codebase. Every model gets the same set of tasks, writes the change that resolves each one with its own coding tool, as it would for a pull request, and a program runs the tests to check the change. There is one row per model per day, with tasks graded, passed, failed and timed out, pass rate, a rating with its standard error, time per task, input, output and thinking tokens, output tokens per second, estimated cost at API list prices, and the detection status. The tasks, the tests and the code the models wrote are not included. Models run at different effort settings and through different tools, so the rows compare a model with its own past, not models with each other.",
  "license": "https://stormeval.com/terms",
  "columns": [
    {
      "name": "date",
      "meaning": "UTC day of the run",
      "unit": "ISO 8601 date"
    },
    {
      "name": "model",
      "meaning": "Model name as shown on the site",
      "unit": "text"
    },
    {
      "name": "model_slug",
      "meaning": "Stable identifier of the model, as used in its page address",
      "unit": "text"
    },
    {
      "name": "runs_through",
      "meaning": "The tool or API the model is run through",
      "unit": "text"
    },
    {
      "name": "effort",
      "meaning": "Reasoning effort setting the model is run at",
      "unit": "text"
    },
    {
      "name": "model_id",
      "meaning": "Model identifier sent to the provider",
      "unit": "text"
    },
    {
      "name": "tasks_graded",
      "meaning": "Tasks graded pass or fail that day",
      "unit": "count"
    },
    {
      "name": "passed",
      "meaning": "Tasks whose change passed its tests and broke nothing",
      "unit": "count"
    },
    {
      "name": "failed",
      "meaning": "Tasks whose change failed a test, broke another test or was not finished in time",
      "unit": "count"
    },
    {
      "name": "timed_out",
      "meaning": "Failed tasks that were not finished within the time limit",
      "unit": "count"
    },
    {
      "name": "not_scored",
      "meaning": "Tasks left out of the pass rate because of an error or a refusal",
      "unit": "count"
    },
    {
      "name": "pass_rate",
      "meaning": "passed divided by tasks_graded",
      "unit": "fraction from 0 to 1"
    },
    {
      "name": "rating",
      "meaning": "Rating fitted from passes and fails against each task's difficulty",
      "unit": "rating points"
    },
    {
      "name": "rating_error",
      "meaning": "One standard error of the rating",
      "unit": "rating points"
    },
    {
      "name": "median_seconds",
      "meaning": "Median time per task; a timed-out task counts as the time limit",
      "unit": "seconds"
    },
    {
      "name": "mean_seconds",
      "meaning": "Mean time per task; a timed-out task counts as the time limit",
      "unit": "seconds"
    },
    {
      "name": "mean_input_tokens",
      "meaning": "Mean input tokens per answered task",
      "unit": "tokens"
    },
    {
      "name": "mean_output_tokens",
      "meaning": "Mean output tokens per answered task",
      "unit": "tokens"
    },
    {
      "name": "mean_thinking_tokens",
      "meaning": "Mean thinking tokens per answered task; empty when the provider does not report them",
      "unit": "tokens"
    },
    {
      "name": "output_tokens_per_second_median",
      "meaning": "Median over answered tasks of output tokens divided by task time",
      "unit": "tokens per second"
    },
    {
      "name": "estimated_cost_usd",
      "meaning": "Tokens at API list prices for the day, not money spent; empty in the CSV and null in the JSON when no list price is set",
      "unit": "US dollars"
    },
    {
      "name": "status",
      "meaning": "Detection status, given on the model's latest day only: collecting, stable, watch, changed or degraded",
      "unit": "text"
    },
    {
      "name": "baseline_day",
      "meaning": "Position of the day in the model's 14-day baseline, from 1; empty after the baseline",
      "unit": "count"
    }
  ],
  "rows": [
    {
      "date": "2026-10-04",
      "model": "Claude Opus 5.5",
      "model_slug": "claude-opus-5-5",
      "runs_through": "Claude Code",
      "effort": "medium",
      "model_id": "claude-opus-5-5",
      "tasks_graded": 3,
      "passed": 2,
      "failed": 1,
      "timed_out": 1,
      "not_scored": 0,
      "pass_rate": 0.6667,
      "rating": 1459,
      "rating_error": 166,
      "median_seconds": 290.7,
      "mean_seconds": 493.8,
      "mean_input_tokens": 946307,
      "mean_output_tokens": 14192,
      "mean_thinking_tokens": 3209,
      "output_tokens_per_second_median": 48.8,
      "estimated_cost_usd": 8.1381,
      "status": "collecting",
      "baseline_day": 1
    },
    {
      "date": "2026-10-04",
      "model": "Claude Sonnet 5.5",
      "model_slug": "claude-sonnet-5-5",
      "runs_through": "Claude Code",
      "effort": "high",
      "model_id": "claude-sonnet-5-5",
      "tasks_graded": 6,
      "passed": 6,
      "failed": 0,
      "timed_out": 0,
      "not_scored": 0,
      "pass_rate": 1,
      "rating": 1746,
      "rating_error": 188,
      "median_seconds": 281.6,
      "mean_seconds": 279.8,
      "mean_input_tokens": 1102576,
      "mean_output_tokens": 12885,
      "mean_thinking_tokens": 4511,
      "output_tokens_per_second_median": 49.3,
      "estimated_cost_usd": 14.004,
      "status": "collecting",
      "baseline_day": 1
    },
    {
      "date": "2026-10-04",
      "model": "DeepSeek V4.1 Flash",
      "model_slug": "deepseek-v4-1-flash",
      "runs_through": "DeepSeek Harness",
      "effort": "default",
      "model_id": "deepseek-flash",
      "tasks_graded": 3,
      "passed": 3,
      "failed": 0,
      "timed_out": 0,
      "not_scored": 0,
      "pass_rate": 1,
      "rating": 1643,
      "rating_error": 200,
      "median_seconds": 391.3,
      "mean_seconds": 390.3,
      "mean_input_tokens": 5182971,
      "mean_output_tokens": 39218,
      "mean_thinking_tokens": null,
      "output_tokens_per_second_median": 95.1,
      "estimated_cost_usd": 0.1546,
      "status": "collecting",
      "baseline_day": 1
    },
    {
      "date": "2026-10-04",
      "model": "GPT-6.1 Sol",
      "model_slug": "gpt-6-1-sol",
      "runs_through": "Codex",
      "effort": "medium",
      "model_id": "gpt-6.1-sol",
      "tasks_graded": 6,
      "passed": 6,
      "failed": 0,
      "timed_out": 0,
      "not_scored": 0,
      "pass_rate": 1,
      "rating": 1746,
      "rating_error": 188,
      "median_seconds": 208.1,
      "mean_seconds": 222.8,
      "mean_input_tokens": 899899,
      "mean_output_tokens": 4308,
      "mean_thinking_tokens": 486,
      "output_tokens_per_second_median": 18.4,
      "estimated_cost_usd": null,
      "status": "collecting",
      "baseline_day": 1
    },
    {
      "date": "2026-10-04",
      "model": "GPT-6 Astra",
      "model_slug": "gpt-6-astra",
      "runs_through": "Codex",
      "effort": "medium",
      "model_id": "gpt-6-astra",
      "tasks_graded": 3,
      "passed": 3,
      "failed": 0,
      "timed_out": 0,
      "not_scored": 0,
      "pass_rate": 1,
      "rating": 1643,
      "rating_error": 200,
      "median_seconds": 189.9,
      "mean_seconds": 198.2,
      "mean_input_tokens": 701032,
      "mean_output_tokens": 3825,
      "mean_thinking_tokens": 223,
      "output_tokens_per_second_median": 18.8,
      "estimated_cost_usd": 3.7867,
      "status": "collecting",
      "baseline_day": 1
    },
    {
      "date": "2026-10-04",
      "model": "Grok 4.7",
      "model_slug": "grok-4-7",
      "runs_through": "Cursor",
      "effort": "medium",
      "model_id": "grok-4.7-medium",
      "tasks_graded": 3,
      "passed": 3,
      "failed": 0,
      "timed_out": 0,
      "not_scored": 0,
      "pass_rate": 1,
      "rating": 1643,
      "rating_error": 200,
      "median_seconds": 512,
      "mean_seconds": 441.2,
      "mean_input_tokens": 3359174,
      "mean_output_tokens": 27572,
      "mean_thinking_tokens": null,
      "output_tokens_per_second_median": 62.9,
      "estimated_cost_usd": null,
      "status": "collecting",
      "baseline_day": 1
    },
    {
      "date": "2026-10-04",
      "model": "Kimi K3",
      "model_slug": "kimi-k3",
      "runs_through": "Kimi Code",
      "effort": "default",
      "model_id": "kimi-code/k3",
      "tasks_graded": 0,
      "passed": 0,
      "failed": 0,
      "timed_out": 0,
      "not_scored": 3,
      "pass_rate": null,
      "rating": null,
      "rating_error": null,
      "median_seconds": null,
      "mean_seconds": null,
      "mean_input_tokens": null,
      "mean_output_tokens": null,
      "mean_thinking_tokens": null,
      "output_tokens_per_second_median": null,
      "estimated_cost_usd": null,
      "status": "collecting",
      "baseline_day": null
    }
  ]
}
