{
  "benchmark_id": "V007-EC-CSV-150-v1",
  "comparison_date": "2026-08-01",
  "timezone": "Asia/Tokyo",
  "fixture": {
    "data_rows": 150,
    "unique_order_ids": 144,
    "gold_valid_order_count": 138,
    "gold_net_sales_ex_tax": "2218700.00",
    "gold_exception_ids": 26,
    "sha256": "abff49ec45a8024f651546035d445935555e1e140b14ac276d2d5adaaaae898f"
  },
  "method": {
    "same_fixture_and_prompt": true,
    "fresh_session_per_run": true,
    "runs_per_model": 3,
    "json_answer_scored": true,
    "submitted_code_reexecution": "not_evaluated",
    "overall_ranking_assigned": false
  },
  "models": [
    {
      "formal_model_name": "GPT-5.6 Sol",
      "execution_surface": "Codex collaboration subagent runtime",
      "reasoning_effort": "high",
      "exact_final_json_runs": 3,
      "total_runs": 3,
      "exception_ids_detected_per_run": [26, 26, 26],
      "summary_fields_exact_per_run": [10, 10, 10]
    },
    {
      "formal_model_name": "Claude Opus 5",
      "execution_surface": "Claude Code CLI",
      "reasoning_effort": "high",
      "exact_final_json_runs": 0,
      "total_runs": 3,
      "exception_ids_detected_per_run": [26, 26, 26],
      "summary_fields_exact_per_run": [0, 0, 0],
      "consistent_error": {
        "incorrectly_excluded_non_cancelled_orders": 14,
        "reported_valid_order_count": 124,
        "reported_net_sales_ex_tax": "1213600.00",
        "net_sales_understatement_ex_tax": "1005100.00"
      }
    },
    {
      "formal_model_name": "Gemini 3.1 Pro",
      "execution_surface": "Gemini web app (Google AI Plus)",
      "reasoning_effort": null,
      "exact_final_json_runs": 3,
      "total_runs": 3,
      "exception_ids_detected_per_run": [26, 26, 26],
      "summary_fields_exact_per_run": [10, 10, 10]
    }
  ],
  "publication_conclusion": "For this fixed fixture, GPT-5.6 Sol and Gemini 3.1 Pro submitted exact final JSON in all three runs. Claude Opus 5 detected all 26 exception IDs in all three runs but consistently misapplied the aggregation rule. This does not establish an interface-neutral overall model ranking."
}

