{"version":"1.0","type":"card","id":"373a54aa-196d-40c3-a3ca-4fb60cb05b43","url":"https://stacklist.com/card/373a54aa-196d-40c3-a3ca-4fb60cb05b43","title":"$τ$-bench: A Benchmark for Tool-Agent-User Interaction","source_url":"https://arxiv.org/abs/2406.12045","note":"This page presents the arXiv paper 2406.12045, which introduces $τ$-bench, a benchmark designed to evaluate interactions among tools, agents, and users in real-world domains. The paper aims to provide a comprehensive framework for assessing the effectiveness and efficiency of these interactions.","image":{"url":"https://ucarecdn.com/0b1ea719-c0fc-445d-953d-b0eabf04817c/","alt":"$τ$-bench: A Benchmark for Tool-Agent-User Interaction","width":336,"height":96},"stack":{"id":"29e01ac8-6202-4abf-ab14-a0613026186b","title":"AI Agent Evaluation Frameworks and Benchmarks","url":"https://stacklist.com/c/technology/stack/29e01ac8-6202-4abf-ab14-a0613026186b"},"created_at":"2026-07-02T09:48:54.151Z","updated_at":null,"aco":{"summary":"τ-bench is a benchmark for evaluating language agent interactions with simulated human users in real-world domains, testing agents' ability to use domain-specific API tools and follow policy guidelines. The paper introduces a pass^k reliability metric and finds that even state-of-the-art models like GPT-4o succeed on less than 50% of tasks, highlighting the need for more consistent and rule-following agent methods.","tags":["benchmark","language-agents","tool-use","human-agent-interaction","evaluation","function-calling","reliability"],"key_entities":[{"name":"τ-bench","type":"concept","confidence":1},{"name":"Shunyu Yao","type":"person","confidence":0.95},{"name":"Noah Shinn","type":"person","confidence":0.9},{"name":"Pedram Razavi","type":"person","confidence":0.9},{"name":"Karthik Narasimhan","type":"person","confidence":0.95},{"name":"GPT-4o","type":"technology","confidence":0.95},{"name":"arXiv","type":"organization","confidence":0.95},{"name":"pass^k metric","type":"concept","confidence":0.9},{"name":"function-calling agents","type":"concept","confidence":0.85}],"classification":"reference","language":"en","confidence":0.85,"provenance":{"model":"claude-opus-4-6","tool":"@stacklist/mcp-server@2.0.0","confidence":0.85,"timestamp":"2026-07-02T09:49:07.947Z"},"token_counts":{"approximate":988,"cl100k":947},"content_hash":"sha256:7ef81f098b387dbd454e75a834ee03961b3c1c3842b8da439bba72c34c190778","acp_version":"0.2","body_available":true,"body_tokens":988,"visibility":"public","agent_accessible":true,"status":"final"},"_links":{"self":"/api/public/card/373a54aa-196d-40c3-a3ca-4fb60cb05b43.json","html":"https://stacklist.com/card/373a54aa-196d-40c3-a3ca-4fb60cb05b43","md":"/api/public/card/373a54aa-196d-40c3-a3ca-4fb60cb05b43.md","stack_json":"/api/public/stack/29e01ac8-6202-4abf-ab14-a0613026186b.json"}}