{"version":"1.0","type":"card","id":"ac1c534e-0bf3-4bdd-8fc6-971b9c9bd3c5","url":"https://stacklist.com/card/ac1c534e-0bf3-4bdd-8fc6-971b9c9bd3c5","title":"Runs 405B LLMs on 8GB VRAM","source_url":"https://github.com/0xSojalSec/airllm","note":"This GitHub repository hosts the airllm project, which enables the running of 405 billion parameter large language models (LLMs) on systems with just 8GB of VRAM. Users can contribute to the development of airllm by creating an account on GitHub.","image":{"url":"https://ucarecdn.com/3979105e-da93-4e27-ba42-465977a97325/","alt":"Runs 405B LLMs on 8GB VRAM","width":1200,"height":600},"stack":{"id":"c691055b-de15-47a1-ad6a-6c4ecec55f18","title":"GitHub repos to check out","url":"https://stacklist.com/stack/c691055b-de15-47a1-ad6a-6c4ecec55f18"},"created_at":"2026-06-17T03:06:28.922Z","updated_at":null,"aco":{"summary":"AirLLM is an AI toolkit that optimizes inference memory usage, enabling large language models like 70B to run on single 4GB GPUs without quantization or pruning. It supports multiple model architectures and includes model compression features for up to 3x inference speed improvement.","tags":["large-language-models","memory-optimization","gpu-inference","model-compression","quantization","open-source","toolkit"],"key_entities":[{"name":"AirLLM","type":"technology","confidence":0.99},{"name":"Llama3.1","type":"technology","confidence":0.95},{"name":"Qwen2.5","type":"technology","confidence":0.9},{"name":"Llama3","type":"technology","confidence":0.95},{"name":"ChatGLM","type":"technology","confidence":0.85},{"name":"Mistral","type":"technology","confidence":0.85},{"name":"bitsandbytes","type":"technology","confidence":0.8},{"name":"block-wise quantization","type":"concept","confidence":0.85},{"name":"model compression","type":"concept","confidence":0.9},{"name":"Hugging Face","type":"organization","confidence":0.85}],"classification":"framework","language":"en","confidence":0.85,"provenance":{"model":"claude-haiku-4-5","tool":"@stacklist/be@0.1.0","confidence":0.85,"timestamp":"2026-06-17T03:06:34.724Z"},"token_counts":{"approximate":2292,"cl100k":2340},"content_hash":"sha256:6eefde2654dd02c287bd154d3f01e7920b1d00fc27377b98e66ebdfdbb96db14","acp_version":"0.2","body_available":true,"body_tokens":2292,"visibility":"public","agent_accessible":true,"status":"final"},"_links":{"self":"/api/public/card/ac1c534e-0bf3-4bdd-8fc6-971b9c9bd3c5.json","html":"https://stacklist.com/card/ac1c534e-0bf3-4bdd-8fc6-971b9c9bd3c5","md":"/api/public/card/ac1c534e-0bf3-4bdd-8fc6-971b9c9bd3c5.md","stack_json":"/api/public/stack/c691055b-de15-47a1-ad6a-6c4ecec55f18.json"}}