{
  "model_name": "Claude Opus 4",
  "model_organization": "Anthropic",
  "submitting_organization": "Anthropic",
  "submission_date": "2025-05-22",
  "contact_info": {
    "email": null,
    "name": "Anthropic Research Team"
  },
  "is_new": false,
  "trajectories_available": false,
  "references": [
    {
      "title": "Claude 4 Model Card",
      "url": "https://www.anthropic.com/news/claude-4",
      "type": "model_card"
    }
  ],
  "results": {
    "retail": {
      "pass_1": 81.4,
      "pass_2": null,
      "pass_3": null,
      "pass_4": null
    },
    "airline": {
      "pass_1": 59.6,
      "pass_2": null,
      "pass_3": null,
      "pass_4": null
    },
    "telecom": {
      "pass_1": null,
      "pass_2": null,
      "pass_3": null,
      "pass_4": null
    }
  },
  "methodology": {
    "evaluation_date": "2025-05-22",
    "tau2_bench_version": "original tau-bench",
    "user_simulator": null,
    "notes": "Scores were achieved with a prompt addendum to both the Airline and Retail Agent Policy instructing Claude to better leverage its reasoning abilities while using extended thinking with tool use. The model is encouraged to write down its thoughts as it solves the problem distinct from our usual thinking mode, during the multi-turn trajectories to best leverage its reasoning abilities. To accommodate the additional steps Claude incurs by utilizing more thinking, the maximum number of steps (counted by model completions) was increased from 30 to 100 (most trajectories completed under 30 steps with only one trajectory reaching above 50 steps).",
    "verification": {
      "modified_prompts": null,
      "omitted_questions": null,
      "details": "Unverified submission - no trajectory data available, unknown evaluation methodology, telecom domain omitted, incomplete evaluation (Pass^1 only)"
    }
  },
  "model_release": {
    "release_date": "2025-05-22",
    "announcement_url": "https://www.anthropic.com/news/claude-4",
    "announcement_title": "Introducing Claude 4"
  }
}
