{"version":"1.0","type":"card","id":"a65b7ae8-5b14-4af1-9c3f-0e2b3042dfc2","url":"https://stacklist.com/card/a65b7ae8-5b14-4af1-9c3f-0e2b3042dfc2","title":"Evaluation of LLM Applications - Langfuse","source_url":"https://langfuse.com/docs/evaluation/overview","note":"With Langfuse you can capture all your LLM evaluations in one place. You can combine a variety of different evaluation metrics like model-based evaluations (LLM-as-a-Judge), human annotations or fully custom evaluation workflows via API/SDKs. This allows you to measure quality, tonality, factual accuracy, completeness, and other dimensions of your LLM application.","image":{"url":"https://ucarecdn.com/70aedf0f-b964-4cad-9334-696a5c09f4a1/","alt":"Evaluation of LLM Applications - Langfuse","width":48,"height":48},"stack":{"id":"d244b35a-f040-4bb7-96ae-187b792f699b","title":"AI agent evaluation frameworks","url":"https://stacklist.com/c/technology/stack/d244b35a-f040-4bb7-96ae-187b792f699b"},"created_at":"2026-07-02T10:02:56.238Z","updated_at":null,"aco":{"summary":"Langfuse Evaluation provides a comprehensive framework for checking LLM application behavior through online production trace scoring and offline experimentation with datasets, experiments, and automated or manual evaluators. The documentation covers core features including annotation queues, LLM-as-a-Judge, score analytics, CI/CD experiment integration, and code evaluators to catch regressions before deployment.","tags":["langfuse","llm-evaluation","observability","datasets","experiments","llm-as-judge","ai-engineering"],"key_entities":[{"name":"Langfuse","type":"technology","confidence":0.99},{"name":"LLM Evaluation","type":"concept","confidence":0.97},{"name":"LLM-as-a-Judge","type":"concept","confidence":0.92},{"name":"Annotation Queues","type":"concept","confidence":0.85},{"name":"Datasets","type":"concept","confidence":0.8},{"name":"Experiments","type":"concept","confidence":0.85},{"name":"CI/CD","type":"concept","confidence":0.75}],"classification":"reference","language":"en","confidence":0.85,"provenance":{"model":"claude-opus-4-6","tool":"@stacklist/mcp-server@2.0.0","confidence":0.85,"timestamp":"2026-07-02T10:03:03.313Z"},"token_counts":{"approximate":500,"cl100k":384},"content_hash":"sha256:992e0a68abb847adac380b7b9de67e6e8576ae9e654f161d893236ec245229cd","acp_version":"0.2","body_available":true,"body_tokens":500,"visibility":"public","agent_accessible":true,"status":"final"},"_links":{"self":"/api/public/card/a65b7ae8-5b14-4af1-9c3f-0e2b3042dfc2.json","html":"https://stacklist.com/card/a65b7ae8-5b14-4af1-9c3f-0e2b3042dfc2","md":"/api/public/card/a65b7ae8-5b14-4af1-9c3f-0e2b3042dfc2.md","stack_json":"/api/public/stack/d244b35a-f040-4bb7-96ae-187b792f699b.json"}}