{"version":"1.0","type":"card","id":"34b2bb91-aa44-4fec-b823-e4dc45631bbb","url":"https://stacklist.com/card/34b2bb91-aa44-4fec-b823-e4dc45631bbb","title":"LMCache: Supercharge Your LLM with Fast KV Cache Layer","source_url":"https://github.com/lmcache/lmcache","note":"LMCache is designed to enhance the performance of large language models (LLMs) by providing the fastest key-value caching layer. This GitHub repository offers resources and documentation for integrating LMCache into your LLM projects.","image":{"url":"https://ucarecdn.com/ea3ddae8-6b99-429e-9eab-d40eecb17a79/","alt":"LMCache: Supercharge Your LLM with Fast KV Cache Layer","width":1200,"height":600},"stack":{"id":"c691055b-de15-47a1-ad6a-6c4ecec55f18","title":"GitHub repos to check out","url":"https://stacklist.com/stack/c691055b-de15-47a1-ad6a-6c4ecec55f18"},"created_at":"2026-07-13T18:10:31.466Z","updated_at":null,"aco":{"summary":"LMCache is a KV cache management layer for scalable LLM inference that converts temporary cache into reusable, persistent knowledge across multiple serving engines and storage backends. It reduces time-to-first-token and improves throughput for long-context and agentic workloads while maintaining vendor neutrality and production-level observability.","tags":["llm-inference","kv-cache","caching-library","gpu-optimization","distributed-systems","vendor-neutral","observability"],"key_entities":[{"name":"LMCache","type":"technology","confidence":0.99},{"name":"KV Cache","type":"technology","confidence":0.98},{"name":"vLLM","type":"technology","confidence":0.92},{"name":"Redis","type":"technology","confidence":0.9},{"name":"PyTorch","type":"technology","confidence":0.88},{"name":"PyTorch Foundation","type":"organization","confidence":0.85},{"name":"NVIDIA","type":"organization","confidence":0.9},{"name":"CoreWeave","type":"organization","confidence":0.85},{"name":"Cohere","type":"organization","confidence":0.85},{"name":"CacheBlend","type":"technology","confidence":0.8},{"name":"prefix-caching","type":"concept","confidence":0.87},{"name":"RAG","type":"concept","confidence":0.85},{"name":"GTC 2026","type":"event","confidence":0.8}],"classification":"reference","language":"en","confidence":0.85,"provenance":{"model":"claude-haiku-4-5","tool":"@stacklist/be@0.1.0","confidence":0.85,"timestamp":"2026-07-13T18:10:41.943Z"},"token_counts":{"approximate":1598,"cl100k":1449},"content_hash":"sha256:d8abb4bf431fdb6a419155b2ae819e6225b9d7fbf893fcc69e4e98adb967cfe7","acp_version":"0.2","body_available":true,"body_tokens":1598,"visibility":"public","agent_accessible":true,"status":"final"},"_links":{"self":"/api/public/card/34b2bb91-aa44-4fec-b823-e4dc45631bbb.json","html":"https://stacklist.com/card/34b2bb91-aa44-4fec-b823-e4dc45631bbb","md":"/api/public/card/34b2bb91-aa44-4fec-b823-e4dc45631bbb.md","stack_json":"/api/public/stack/c691055b-de15-47a1-ad6a-6c4ecec55f18.json"}}