{"version":"1.0","type":"card","id":"0cbe0b53-0559-4f29-a5f6-523c257d6b46","url":"https://stacklist.com/card/0cbe0b53-0559-4f29-a5f6-523c257d6b46","title":"Building a Custom ChatGPT Model from Scratch","source_url":"https://www.linkedin.com/posts/alex-lotkov-0b0948243_llm-aiengineering-aiinfrastructure-ugcPost-7483362836237160449-G2re/?utm_source=share&utm_medium=member_ios&rcm=ACoAAAI21ZsBNnZPaKuTab7nquKLCveUW7o-1DE","note":"This page details the process of building a custom ChatGPT model from scratch, focusing on optimizing GPU memory and inference without using pretrained weights or APIs. Key results include impressive performance metrics and insights on hardware limitations in model inference.","image":{"url":"https://ucarecdn.com/c59c6b0b-4afb-4012-8707-f758c95e95ef/","alt":"Building a Custom ChatGPT Model from Scratch","width":1280,"height":800},"stack":{"id":"4aae218c-38d7-4c05-b7ae-f2db0029a2e8","title":"Local AI & GPUs","url":"https://stacklist.com/stack/4aae218c-38d7-4c05-b7ae-f2db0029a2e8"},"created_at":"2026-07-17T12:30:45.091Z","updated_at":null,"aco":{"summary":"AI Engineering project where Alex Lotkov built a 496M-parameter ChatGPT-like model from scratch in 24 hours for $20, achieving 230-340 tokens per second on a 4GB laptop GPU through aggressive memory optimization and custom CUDA kernels. The project demonstrates that model building, memory fitting, training, and efficient serving are distinct engineering challenges, with INT4 quantization providing optimal speed and memory usage.","tags":["llm-inference","gpu-optimization","cuda-kernels","quantization","ai-engineering","model-training","inference-optimization"],"key_entities":[{"name":"Alex Lotkov","type":"person","confidence":0.95},{"name":"CUDA","type":"technology","confidence":0.95},{"name":"FlashAttention","type":"technology","confidence":0.9},{"name":"PyTorch","type":"technology","confidence":0.85},{"name":"Transformer","type":"technology","confidence":0.95},{"name":"INT4 quantization","type":"technology","confidence":0.9},{"name":"RTX 3050","type":"technology","confidence":0.85},{"name":"Modal A10","type":"technology","confidence":0.85},{"name":"GPU memory optimization","type":"concept","confidence":0.9},{"name":"inference optimization","type":"concept","confidence":0.9},{"name":"RAG","type":"concept","confidence":0.85}],"classification":"analysis","language":"en","confidence":0.85,"provenance":{"model":"claude-haiku-4-5","tool":"@stacklist/be@0.1.0","confidence":0.85,"timestamp":"2026-07-17T12:30:51.965Z"},"token_counts":{"approximate":1058,"cl100k":955},"content_hash":"sha256:f7978eaa1c2da5bcc08408f819876ffcdf262dde9eb385d61f95a2ecadc4643e","acp_version":"0.2","body_available":true,"body_tokens":1058,"visibility":"public","agent_accessible":true,"status":"final"},"_links":{"self":"/api/public/card/0cbe0b53-0559-4f29-a5f6-523c257d6b46.json","html":"https://stacklist.com/card/0cbe0b53-0559-4f29-a5f6-523c257d6b46","md":"/api/public/card/0cbe0b53-0559-4f29-a5f6-523c257d6b46.md","stack_json":"/api/public/stack/4aae218c-38d7-4c05-b7ae-f2db0029a2e8.json"}}