{
  "slug": "scabench",
  "name": "ScaBench and SCONE-bench",
  "tagline": "Contest-derived and Anthropic smart-contract benchmarks",
  "maintainer": "scabench-org; Anthropic",
  "url": "https://github.com/scabench-org/scabench",
  "repo": "https://github.com/anthropics/scone-bench",
  "category": "benchmarks",
  "targets": [
    "Solidity",
    "31 projects from Code4rena, Cantina, Sherlock"
  ],
  "approach": "Ground truth from public contest findings; SCONE-bench from Anthropic",
  "license": "Open source",
  "status": "Active",
  "summary": "ScaBench draws ground truth from 31 projects audited on Code4rena, Cantina and Sherlock and is the benchmark behind Hound's published recall; SCONE-bench is Anthropic's smart-contract benchmark.",
  "details": [
    "Contest-derived benchmarks have many human findings per project, which makes recall numbers harsher and more realistic."
  ],
  "strengths": [
    "Realistic ground truth.",
    "Open."
  ],
  "limits": [
    "Public findings are in training data.",
    "Solidity only."
  ],
  "fit": [
    "Use alongside EVMbench."
  ],
  "references": [
    [
      "ScaBench",
      "https://github.com/scabench-org/scabench"
    ],
    [
      "SCONE-bench",
      "https://github.com/anthropics/scone-bench"
    ]
  ],
  "category_name": "Benchmarks and research",
  "page": "https://agentsast.com/tools/scabench/",
  "updated": "2026-09-13"
}