{
  "benchmark": "eval.tax AI Accounting Arena",
  "url": "https://eval.tax/",
  "release": {
    "date": "2026-06-29",
    "pack": "v1.2",
    "harness": "1.2.0",
    "run_window": {
      "start": "2026-06-24",
      "end": "2026-06-27",
      "timezone": "UTC"
    },
    "passes": 3,
    "rows_total": 512,
    "rows_scored": 490,
    "requests_per_pass": 16,
    "participants": 30,
    "organizations": 13,
    "superseded_by": "https://eval.tax/results/2026-09-08.json"
  },
  "scores_note": "Scores are eval.tax run scores on the shared statement pack, not vendor benchmark claims. Track scores are medians of three seeded passes.",
  "columns": {
    "overall": "Weighted composite of the six track scores, 0-100. Weights are frozen per pack version.",
    "categorization": "Exact category match against the 14-label gold set, 490 scored rows, 0-100.",
    "vendor": "F1 over normalized merchant entities, 490 scored rows, 0-100.",
    "context": "Context-weighted accuracy on 118 rows that need client policy or vendor history, 0-100.",
    "coa": "Exact account match against the 48-account target chart of accounts, 490 scored rows, 0-100.",
    "splits": "Structured exact match on 64 mixed-use rows, 0-100.",
    "tax": "Tax treatment correctness on 96 tax-sensitive rows, 0-100.",
    "reliability": "Percent of requests that returned schema-valid output on the first attempt (48 requests per participant).",
    "latency": "Median wall-clock seconds per 32-row request.",
    "cost": "List-price USD per 1,000 input rows at the release date's pricing, retries included."
  },
  "tracks": [
    {
      "key": "categorization",
      "name": "Transaction Categorization",
      "metric": "Exact category match",
      "url": "https://eval.tax/tracks/transaction-categorization/",
      "scored_rows": null,
      "leader": "Jupid Agent",
      "top_score": 95.4
    },
    {
      "key": "vendor",
      "name": "Vendor Normalization",
      "metric": "Vendor match F1",
      "url": "https://eval.tax/tracks/vendor-normalization/",
      "scored_rows": null,
      "leader": "Jupid Agent",
      "top_score": 94
    },
    {
      "key": "context",
      "name": "Client Context Fit",
      "metric": "Context-weighted accuracy",
      "url": "https://eval.tax/tracks/client-context-fit/",
      "scored_rows": null,
      "leader": "Jupid Agent",
      "top_score": 96.2
    },
    {
      "key": "coa",
      "name": "COA Mapping",
      "metric": "COA exact match",
      "url": "https://eval.tax/tracks/coa-mapping/",
      "scored_rows": null,
      "leader": "Jupid Agent",
      "top_score": 94.7
    },
    {
      "key": "splits",
      "name": "Split Detection",
      "metric": "Structured split exact match",
      "url": "https://eval.tax/tracks/split-detection/",
      "scored_rows": null,
      "leader": "GPT-5.6 Sol Preview",
      "top_score": 93.8
    },
    {
      "key": "tax",
      "name": "Tax Sensitivity",
      "metric": "Tax treatment correctness",
      "url": "https://eval.tax/tracks/tax-sensitivity/",
      "scored_rows": null,
      "leader": "Jupid Agent",
      "top_score": 94
    }
  ],
  "participants": [
    {
      "id": "jupid-agent",
      "rank": 1,
      "name": "Jupid Agent",
      "org": "Jupid",
      "type": "Agent",
      "access": null,
      "overall": 94.6,
      "categorization": 95.4,
      "vendor": 94,
      "context": 96.2,
      "coa": 94.7,
      "splits": 93,
      "tax": 94,
      "reliability": 92,
      "latency_s": 4.5,
      "cost_index": 7.6
    },
    {
      "id": "claude-mythos-5",
      "rank": 2,
      "name": "Claude Mythos 5",
      "org": "Anthropic",
      "type": "Model",
      "access": null,
      "overall": 93.3,
      "categorization": 93.7,
      "vendor": 93.8,
      "context": 92.4,
      "coa": 92.5,
      "splits": 93.5,
      "tax": 92.2,
      "reliability": 88,
      "latency_s": 6.4,
      "cost_index": 10.9
    },
    {
      "id": "gpt-5.6-sol-preview",
      "rank": 3,
      "name": "GPT-5.6 Sol Preview",
      "org": "OpenAI",
      "type": "Model",
      "access": null,
      "overall": 93,
      "categorization": 93.3,
      "vendor": 93.5,
      "context": 91.8,
      "coa": 92,
      "splits": 93.8,
      "tax": 92,
      "reliability": 86,
      "latency_s": 6,
      "cost_index": 12.2
    },
    {
      "id": "claude-fable-5",
      "rank": 4,
      "name": "Claude Fable 5",
      "org": "Anthropic",
      "type": "Model",
      "access": null,
      "overall": 92.8,
      "categorization": 93.2,
      "vendor": 93.4,
      "context": 91.8,
      "coa": 92,
      "splits": 93,
      "tax": 91.5,
      "reliability": 87,
      "latency_s": 5.7,
      "cost_index": 10.3
    },
    {
      "id": "gpt-5.5-pro",
      "rank": 5,
      "name": "GPT-5.5 Pro",
      "org": "OpenAI",
      "type": "Model",
      "access": null,
      "overall": 92.4,
      "categorization": 92.8,
      "vendor": 93,
      "context": 91.2,
      "coa": 91.7,
      "splits": 92.8,
      "tax": 91.7,
      "reliability": 86,
      "latency_s": 6.5,
      "cost_index": 13.8
    },
    {
      "id": "claude-opus-4.8",
      "rank": 6,
      "name": "Claude Opus 4.8",
      "org": "Anthropic",
      "type": "Model",
      "access": null,
      "overall": 91.9,
      "categorization": 92.2,
      "vendor": 92.8,
      "context": 90.8,
      "coa": 91.2,
      "splits": 92.1,
      "tax": 90.8,
      "reliability": 85,
      "latency_s": 5.2,
      "cost_index": 9.5
    },
    {
      "id": "gpt-5.5",
      "rank": 7,
      "name": "GPT-5.5",
      "org": "OpenAI",
      "type": "Model",
      "access": null,
      "overall": 91.5,
      "categorization": 91.9,
      "vendor": 92.4,
      "context": 90.5,
      "coa": 91,
      "splits": 91.9,
      "tax": 90.9,
      "reliability": 84,
      "latency_s": 4.4,
      "cost_index": 8
    },
    {
      "id": "gemini-3.1-pro-preview",
      "rank": 8,
      "name": "Gemini 3.1 Pro Preview",
      "org": "Google",
      "type": "Model",
      "access": null,
      "overall": 91.1,
      "categorization": 91.4,
      "vendor": 92,
      "context": 89.9,
      "coa": 90.7,
      "splits": 91.5,
      "tax": 90.3,
      "reliability": 83,
      "latency_s": 5,
      "cost_index": 8.5
    },
    {
      "id": "claude-opus-4.7",
      "rank": 9,
      "name": "Claude Opus 4.7",
      "org": "Anthropic",
      "type": "Model",
      "access": null,
      "overall": 90.3,
      "categorization": 90.8,
      "vendor": 91.5,
      "context": 89.1,
      "coa": 89.8,
      "splits": 90.6,
      "tax": 89.5,
      "reliability": 82,
      "latency_s": 5.1,
      "cost_index": 9
    },
    {
      "id": "deepseek-v4-pro-preview",
      "rank": 10,
      "name": "DeepSeek V4 Pro Preview",
      "org": "DeepSeek",
      "type": "Model",
      "access": null,
      "overall": 89.7,
      "categorization": 90.2,
      "vendor": 90.9,
      "context": 88.4,
      "coa": 89.3,
      "splits": 90.1,
      "tax": 88.9,
      "reliability": 81,
      "latency_s": 4.8,
      "cost_index": 5.9
    },
    {
      "id": "grok-4.3",
      "rank": 11,
      "name": "Grok 4.3",
      "org": "xAI",
      "type": "Model",
      "access": null,
      "overall": 89.1,
      "categorization": 89.6,
      "vendor": 90.4,
      "context": 87.9,
      "coa": 88.7,
      "splits": 89.6,
      "tax": 88.3,
      "reliability": 80,
      "latency_s": 4.7,
      "cost_index": 6.8
    },
    {
      "id": "qwen3.7-max",
      "rank": 12,
      "name": "Qwen3.7-Max",
      "org": "Alibaba",
      "type": "Model",
      "access": null,
      "overall": 88.7,
      "categorization": 89.2,
      "vendor": 90,
      "context": 87.4,
      "coa": 88.3,
      "splits": 89.1,
      "tax": 87.8,
      "reliability": 79,
      "latency_s": 4.2,
      "cost_index": 5.6
    },
    {
      "id": "gemini-3.5-flash",
      "rank": 13,
      "name": "Gemini 3.5 Flash",
      "org": "Google",
      "type": "Model",
      "access": null,
      "overall": 88.1,
      "categorization": 88.5,
      "vendor": 89.2,
      "context": 86.8,
      "coa": 87.6,
      "splits": 88.4,
      "tax": 87.1,
      "reliability": 77,
      "latency_s": 2.4,
      "cost_index": 3
    },
    {
      "id": "claude-sonnet-4.6",
      "rank": 14,
      "name": "Claude Sonnet 4.6",
      "org": "Anthropic",
      "type": "Model",
      "access": null,
      "overall": 87.8,
      "categorization": 88.2,
      "vendor": 89,
      "context": 86.7,
      "coa": 87.4,
      "splits": 88.1,
      "tax": 87,
      "reliability": 77,
      "latency_s": 3.4,
      "cost_index": 5.2
    },
    {
      "id": "gpt-5.5-instant",
      "rank": 15,
      "name": "GPT-5.5 Instant",
      "org": "OpenAI",
      "type": "Model",
      "access": null,
      "overall": 87.3,
      "categorization": 87.7,
      "vendor": 88.5,
      "context": 86,
      "coa": 86.8,
      "splits": 87.6,
      "tax": 86.4,
      "reliability": 76,
      "latency_s": 1.9,
      "cost_index": 2.4
    },
    {
      "id": "kimi-k2.6",
      "rank": 16,
      "name": "Kimi K2.6",
      "org": "Moonshot AI",
      "type": "Model",
      "access": null,
      "overall": 86.8,
      "categorization": 87.2,
      "vendor": 88,
      "context": 85.5,
      "coa": 86.4,
      "splits": 87,
      "tax": 85.9,
      "reliability": 75,
      "latency_s": 4,
      "cost_index": 4.8
    },
    {
      "id": "deepseek-v4-flash",
      "rank": 17,
      "name": "DeepSeek V4 Flash",
      "org": "DeepSeek",
      "type": "Model",
      "access": null,
      "overall": 86.3,
      "categorization": 86.8,
      "vendor": 87.6,
      "context": 85,
      "coa": 85.9,
      "splits": 86.6,
      "tax": 85.4,
      "reliability": 74,
      "latency_s": 3.1,
      "cost_index": 3.4
    },
    {
      "id": "qwen3.7-plus",
      "rank": 18,
      "name": "Qwen3.7-Plus",
      "org": "Alibaba",
      "type": "Model",
      "access": null,
      "overall": 85.9,
      "categorization": 86.4,
      "vendor": 87,
      "context": 84.6,
      "coa": 85.4,
      "splits": 86.2,
      "tax": 85,
      "reliability": 74,
      "latency_s": 3.4,
      "cost_index": 3.7
    },
    {
      "id": "grok-build-0.1",
      "rank": 19,
      "name": "Grok Build 0.1",
      "org": "xAI",
      "type": "Model",
      "access": null,
      "overall": 85.4,
      "categorization": 85.9,
      "vendor": 86.4,
      "context": 84.1,
      "coa": 84.9,
      "splits": 85.7,
      "tax": 84.4,
      "reliability": 73,
      "latency_s": 2.7,
      "cost_index": 3.3
    },
    {
      "id": "mistral-medium-3.5",
      "rank": 20,
      "name": "Mistral Medium 3.5",
      "org": "Mistral",
      "type": "Model",
      "access": null,
      "overall": 85,
      "categorization": 85.5,
      "vendor": 86.1,
      "context": 83.7,
      "coa": 84.6,
      "splits": 85.3,
      "tax": 84.1,
      "reliability": 72,
      "latency_s": 3.6,
      "cost_index": 4.3
    },
    {
      "id": "llama-4-maverick",
      "rank": 21,
      "name": "Llama 4 Maverick",
      "org": "Meta",
      "type": "Model",
      "access": null,
      "overall": 84.6,
      "categorization": 85.1,
      "vendor": 85.6,
      "context": 83.3,
      "coa": 84.2,
      "splits": 84.9,
      "tax": 83.8,
      "reliability": 71,
      "latency_s": 3.7,
      "cost_index": 4
    },
    {
      "id": "mistral-large-3",
      "rank": 22,
      "name": "Mistral Large 3",
      "org": "Mistral",
      "type": "Model",
      "access": null,
      "overall": 84.2,
      "categorization": 84.7,
      "vendor": 85.3,
      "context": 82.9,
      "coa": 83.8,
      "splits": 84.5,
      "tax": 83.4,
      "reliability": 70,
      "latency_s": 4.1,
      "cost_index": 4.7
    },
    {
      "id": "gemini-3-flash",
      "rank": 23,
      "name": "Gemini 3 Flash",
      "org": "Google",
      "type": "Model",
      "access": null,
      "overall": 83.8,
      "categorization": 84.2,
      "vendor": 84.9,
      "context": 82.5,
      "coa": 83.4,
      "splits": 84.1,
      "tax": 83,
      "reliability": 69,
      "latency_s": 2.2,
      "cost_index": 2.6
    },
    {
      "id": "claude-opus-4.6",
      "rank": 24,
      "name": "Claude Opus 4.6",
      "org": "Anthropic",
      "type": "Model",
      "access": null,
      "overall": 83.5,
      "categorization": 84,
      "vendor": 84.7,
      "context": 82.2,
      "coa": 83,
      "splits": 83.8,
      "tax": 82.7,
      "reliability": 69,
      "latency_s": 4.8,
      "cost_index": 8.6
    },
    {
      "id": "cohere-command-a-plus",
      "rank": 25,
      "name": "Cohere Command A+",
      "org": "Cohere",
      "type": "Model",
      "access": null,
      "overall": 83.1,
      "categorization": 83.5,
      "vendor": 84.1,
      "context": 81.7,
      "coa": 82.6,
      "splits": 83.3,
      "tax": 82.2,
      "reliability": 68,
      "latency_s": 3.7,
      "cost_index": 4.1
    },
    {
      "id": "amazon-nova-premier",
      "rank": 26,
      "name": "Amazon Nova Premier",
      "org": "Amazon",
      "type": "Model",
      "access": null,
      "overall": 82.7,
      "categorization": 83.1,
      "vendor": 83.7,
      "context": 81.3,
      "coa": 82.1,
      "splits": 82.9,
      "tax": 81.8,
      "reliability": 67,
      "latency_s": 3.4,
      "cost_index": 3.8
    },
    {
      "id": "claude-haiku-4.5",
      "rank": 27,
      "name": "Claude Haiku 4.5",
      "org": "Anthropic",
      "type": "Model",
      "access": null,
      "overall": 82.3,
      "categorization": 82.7,
      "vendor": 83.4,
      "context": 81,
      "coa": 81.8,
      "splits": 82.5,
      "tax": 81.5,
      "reliability": 66,
      "latency_s": 2.1,
      "cost_index": 2.5
    },
    {
      "id": "perplexity-sonar-reasoning-pro",
      "rank": 28,
      "name": "Perplexity Sonar Reasoning Pro",
      "org": "Perplexity",
      "type": "Model",
      "access": null,
      "overall": 82,
      "categorization": 82.4,
      "vendor": 83,
      "context": 80.6,
      "coa": 81.4,
      "splits": 82.1,
      "tax": 81.1,
      "reliability": 65,
      "latency_s": 4.5,
      "cost_index": 4.1
    },
    {
      "id": "mistral-small-4",
      "rank": 29,
      "name": "Mistral Small 4",
      "org": "Mistral",
      "type": "Model",
      "access": null,
      "overall": 81.5,
      "categorization": 81.9,
      "vendor": 82.5,
      "context": 80.2,
      "coa": 81,
      "splits": 81.7,
      "tax": 80.8,
      "reliability": 64,
      "latency_s": 2.7,
      "cost_index": 2.8
    },
    {
      "id": "gemini-3.1-flash-lite",
      "rank": 30,
      "name": "Gemini 3.1 Flash-Lite",
      "org": "Google",
      "type": "Model",
      "access": null,
      "overall": 81.1,
      "categorization": 81.5,
      "vendor": 82,
      "context": 79.8,
      "coa": 80.6,
      "splits": 81.2,
      "tax": 80.4,
      "reliability": 63,
      "latency_s": 1.7,
      "cost_index": 1.8
    }
  ]
}
