{"version":"sfa.source-bundle.v1","content":{"id":"research_evidence_first_agent_evaluation","kind":"research-page","language":"en","url":"https://www.searchforagents.com/research/evidence-first-agent-evaluation","updatedAt":"2026-09-22"},"provenance":{"version":"sfa.provenance.v1","sourcePath":"content/research/evidence-first-agent-evaluation.md","canonicalMarkdownHash":"3bf53a275f2e759e20f8864c2b602746a55123525d519891e2520b3b490c3ca2","contentHash":"9a46c0fbe105963bda7d485e6781da400ce4143996760fcfdee2a2a0d49fb0ce","schemaVersion":"research.v1","commit":"86eeb0c2cfe77a7d66370be2b282a2e8c1f0c055"},"sources":[{"title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","url":"https://arxiv.org/abs/2310.06770","publisher":"arXiv","publishedAt":"2023-10-10"},{"title":"AgentBench: Evaluating LLMs as Agents","url":"https://arxiv.org/abs/2308.03688","publisher":"arXiv","publishedAt":"2023-08-07"}],"methodology":["https://www.searchforagents.com/methodology"],"datasets":["dataset_agent_evaluation_fixture"],"limitations":["External source reachability is not asserted. Dataset integrity covers repository bytes only."]}