{"version":"1.0","type":"card","id":"1d05f8cd-8fb8-43dc-9225-d9a5b3a49f25","url":"https://stacklist.com/card/1d05f8cd-8fb8-43dc-9225-d9a5b3a49f25","title":"AgentRewardBench","source_url":"https://agent-reward-bench.github.io/","note":"AgentRewardBench is a platform designed for evaluating and benchmarking reinforcement learning agents. It provides tools and metrics to assess agent performance in various environments.","image":{"url":"https://ucarecdn.com/da0ec6de-bdac-4d9d-986c-490675e41535/","alt":"AgentRewardBench","width":1280,"height":800},"stack":{"id":"29e01ac8-6202-4abf-ab14-a0613026186b","title":"AI Agent Evaluation Frameworks and Benchmarks","url":"https://stacklist.com/c/technology/stack/29e01ac8-6202-4abf-ab14-a0613026186b"},"created_at":"2026-07-02T09:49:06.061Z","updated_at":null,"aco":{"summary":"AgentRewardBench is a benchmark and Python library for evaluating automatic evaluators of web agent trajectories, including LLM judges. It provides tools, environments, evaluation metrics, a Hugging Face dataset, and a leaderboard for ranking automatic evaluators across different benchmarks.","tags":["agent-evaluation","web-agents","benchmark","llm-judges","trajectories","reward-model","python-library"],"key_entities":[{"name":"AgentRewardBench","type":"technology","confidence":1},{"name":"Xing Han Lù","type":"person","confidence":0.95},{"name":"Amirhossein Kazemnejad","type":"person","confidence":0.95},{"name":"Nicholas Meade","type":"person","confidence":0.9},{"name":"Siva Reddy","type":"person","confidence":0.9},{"name":"Christopher J. Pal","type":"person","confidence":0.9},{"name":"McGill-NLP","type":"organization","confidence":0.95},{"name":"Hugging Face Hub","type":"technology","confidence":0.9},{"name":"web agent trajectories","type":"concept","confidence":0.95},{"name":"LLM judges","type":"concept","confidence":0.9}],"classification":"framework","language":"en","confidence":0.85,"provenance":{"model":"claude-opus-4-6","tool":"@stacklist/mcp-server@2.0.0","confidence":0.85,"timestamp":"2026-07-02T09:49:15.134Z"},"token_counts":{"approximate":606,"cl100k":594},"content_hash":"sha256:9465a3e299b570373ee6c742dcf5e223105c8355f6747fa6a8f26fa4ec3fb1e8","acp_version":"0.2","body_available":true,"body_tokens":606,"visibility":"public","agent_accessible":true,"status":"final"},"_links":{"self":"/api/public/card/1d05f8cd-8fb8-43dc-9225-d9a5b3a49f25.json","html":"https://stacklist.com/card/1d05f8cd-8fb8-43dc-9225-d9a5b3a49f25","md":"/api/public/card/1d05f8cd-8fb8-43dc-9225-d9a5b3a49f25.md","stack_json":"/api/public/stack/29e01ac8-6202-4abf-ab14-a0613026186b.json"}}