{"version":"1.0","type":"card","id":"940529c5-7a06-4039-92cf-25ff8371d9d9","url":"https://stacklist.com/card/940529c5-7a06-4039-92cf-25ff8371d9d9","title":"Measure Performance with Evaluations - Phoenix","source_url":"https://arize.com/docs/phoenix/get-started/get-started-evaluations","note":"This page provides a comprehensive guide on how to measure the performance of machine learning models using the Evaluations feature in Phoenix. It covers the setup process, key metrics to consider, and best practices for effective evaluation.","image":{"url":"https://ucarecdn.com/4b30c74c-5a1d-426d-b0b2-2b32aaa76e2f/","alt":"Measure Performance with Evaluations - Phoenix","width":1280,"height":800},"stack":{"id":"d244b35a-f040-4bb7-96ae-187b792f699b","title":"AI agent evaluation frameworks","url":"https://stacklist.com/c/technology/stack/d244b35a-f040-4bb7-96ae-187b792f699b"},"created_at":"2026-07-02T10:02:51.915Z","updated_at":null,"aco":{"summary":"Phoenix Evaluations is a tutorial guide for setting up and running evaluations on existing trace data to measure LLM output quality in a repeatable way. It walks through defining an LLM-as-a-judge evaluation for completeness, using a financial analysis chatbot as the example application.","tags":["phoenix","evaluations","llm-as-a-judge","tracing","model-quality","observability","ai-evaluation"],"key_entities":[{"name":"Phoenix","type":"technology","confidence":0.99},{"name":"LLM-as-a-judge","type":"concept","confidence":0.95},{"name":"evaluations","type":"concept","confidence":0.95},{"name":"tracing","type":"concept","confidence":0.85},{"name":"Financial Analysis and Research Chatbot","type":"technology","confidence":0.8},{"name":"completeness evaluation","type":"concept","confidence":0.8}],"classification":"tutorial","language":"en","confidence":0.85,"provenance":{"model":"claude-opus-4-6","tool":"@stacklist/mcp-server@2.0.0","confidence":0.85,"timestamp":"2026-07-02T10:03:00.741Z"},"token_counts":{"approximate":2537,"cl100k":2250},"content_hash":"sha256:8a148c7e1536024478ed835e045d608d0ec99e2f2232fbcf438f9f728dc61721","acp_version":"0.2","body_available":true,"body_tokens":2537,"visibility":"public","agent_accessible":true,"status":"final"},"_links":{"self":"/api/public/card/940529c5-7a06-4039-92cf-25ff8371d9d9.json","html":"https://stacklist.com/card/940529c5-7a06-4039-92cf-25ff8371d9d9","md":"/api/public/card/940529c5-7a06-4039-92cf-25ff8371d9d9.md","stack_json":"/api/public/stack/d244b35a-f040-4bb7-96ae-187b792f699b.json"}}