{"version":"sfa.source-bundle.v1","content":{"id":"dataset_agent_evaluation_fixture","kind":"dataset","language":"en","url":"https://www.searchforagents.com/datasets/agent-evaluation-fixture","updatedAt":"2026-09-22"},"provenance":{"version":"sfa.provenance.v1","sourcePath":"content/datasets/agent-evaluation-fixture.md","canonicalMarkdownHash":"95eca2c9491d65e006c856c1936dccf0772159e18cb699f1effd624dd8151e8a","contentHash":"e3546322bbaa70b9e64c29cc4c923ce12b5515266dacb65c9c79a21e439afdde","schemaVersion":"dataset.v1","commit":"86eeb0c2cfe77a7d66370be2b282a2e8c1f0c055"},"sources":[{"title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","url":"https://arxiv.org/abs/2310.06770","publisher":"arXiv","publishedAt":"2023-10-10"}],"methodology":["https://www.searchforagents.com/methodology"],"files":[{"url":"/data/agent-evaluation-fixture.json","format":"application/json","sha256":"aa9e8654f73d958fbdbac007eef74dae029bc47f9232797c0f65a1aafacdd9a1","sizeBytes":562}],"license":"CC-BY-4.0","datasetVersion":"2026-09-22","limitations":["External source reachability is not asserted. Dataset integrity covers repository bytes only."]}