{
  "model_name": "Qwen3-Max-Thinking",
  "model_organization": "Qwen",
  "submitting_organization": "Qwen",
  "submission_date": "2026-01-26",
  "submission_type": "standard",
  "contact_info": {
    "email": "ys724@cornell.edu",
    "name": "Y. Su",
    "github": "yang-su2000"
  },
  "is_new": true,
  "trajectories_available": false,
  "references": [
    {
      "title": "Qwen3-Max-Thinking Blog Post",
      "url": "https://qwen.ai/blog?id=qwen3-max-thinking",
      "type": "blog_post"
    }
  ],
  "results": {
    "retail": {
      "pass_1": 79.3859649122807,
      "pass_2": 69.44444444444446,
      "pass_3": 62.71929824561403,
      "pass_4": 57.89473684210527,
      "cost": null
    },
    "airline": {
      "pass_1": 69.0,
      "pass_2": 65.33333333333333,
      "pass_3": 63.5,
      "pass_4": 62.0,
      "cost": null
    },
    "telecom": {
      "pass_1": 98.24561403508771,
      "pass_2": 96.63742690058479,
      "pass_3": 95.17543859649122,
      "pass_4": 93.85964912280701,
      "cost": null
    }
  },
  "methodology": {
    "evaluation_date": "2026-01-23",
    "tau2_bench_version": "v0.1.3",
    "user_simulator": "gpt-4.1-2025-04-14",
    "notes": "Evaluation conducted using standard tau2-bench protocol",
    "verification": {
      "modified_prompts": false,
      "omitted_questions": false,
      "details": "Complete evaluation with all standard tasks"
    }
  },
  "model_release": {
    "release_date": "2026-01-23",
    "announcement_url": "https://qwen.ai/blog?id=qwen3-max-thinking",
    "announcement_title": "Qwen3-Max-Thinking"
  }
}
