{"version":"1.0","type":"card","id":"75394818-3927-400d-ab2f-56f509d6342d","url":"https://stacklist.com/card/75394818-3927-400d-ab2f-56f509d6342d","title":"Survey on Evaluation of LLM-based Agents","source_url":"https://arxiv.org/abs/2503.16416","note":"This survey provides the first comprehensive analysis of evaluation methods for LLM-based agents, examining core capabilities, application-specific benchmarks, generalist agent evaluation, benchmark dimensions, and evaluation frameworks.","image":{"url":"https://ucarecdn.com/22cd18df-b21b-4d8c-a68c-4f5aec4c1e4c/","alt":"Survey on Evaluation of LLM-based Agents","width":336,"height":96},"stack":{"id":"29e01ac8-6202-4abf-ab14-a0613026186b","title":"AI Agent Evaluation Frameworks and Benchmarks","url":"https://stacklist.com/c/technology/stack/29e01ac8-6202-4abf-ab14-a0613026186b"},"created_at":"2026-07-02T09:48:43.543Z","updated_at":null,"aco":{"summary":"This survey provides the first comprehensive analysis of evaluation methods for LLM-based agents, examining core capabilities, application-specific benchmarks, generalist agent evaluation, benchmark dimensions, and evaluation frameworks. The paper identifies trends toward more realistic evaluations and highlights critical gaps in assessing cost-efficiency, safety, robustness, and scalable evaluation methods.","tags":["llm-agents","evaluation","benchmarks","artificial-intelligence","survey","planning","tool-use"],"key_entities":[{"name":"Asaf Yehudai","type":"person","confidence":0.95},{"name":"Lilach Eden","type":"person","confidence":0.9},{"name":"Alan Li","type":"person","confidence":0.9},{"name":"Guy Uziel","type":"person","confidence":0.9},{"name":"Yilun Zhao","type":"person","confidence":0.9},{"name":"Roy Bar-Haim","type":"person","confidence":0.9},{"name":"Arman Cohan","type":"person","confidence":0.9},{"name":"Michal Shmueli-Scheuer","type":"person","confidence":0.9},{"name":"arXiv","type":"organization","confidence":0.95},{"name":"LLM-based agents","type":"concept","confidence":0.99},{"name":"agent evaluation","type":"concept","confidence":0.97},{"name":"benchmarks","type":"concept","confidence":0.92},{"name":"ACL Findings","type":"event","confidence":0.85},{"name":"SWE agents","type":"concept","confidence":0.8}],"classification":"reference","language":"en","confidence":0.85,"provenance":{"model":"claude-opus-4-6","tool":"@stacklist/mcp-server@2.0.0","confidence":0.85,"timestamp":"2026-07-02T09:48:53.489Z"},"token_counts":{"approximate":1024,"cl100k":983},"content_hash":"sha256:616cee134c87206809de54674ac11ee41be4a8063e883bedcdfd2589b634ab06","acp_version":"0.2","body_available":true,"body_tokens":1024,"visibility":"public","agent_accessible":true,"status":"final"},"_links":{"self":"/api/public/card/75394818-3927-400d-ab2f-56f509d6342d.json","html":"https://stacklist.com/card/75394818-3927-400d-ab2f-56f509d6342d","md":"/api/public/card/75394818-3927-400d-ab2f-56f509d6342d.md","stack_json":"/api/public/stack/29e01ac8-6202-4abf-ab14-a0613026186b.json"}}