{"version":"1.0","type":"card","id":"8521483e-0deb-40ab-bff1-4ef6d7a657cc","url":"https://stacklist.com/card/8521483e-0deb-40ab-bff1-4ef6d7a657cc","title":"GPU Performance Engineering Resources Curriculum","source_url":"https://github.com/wafer-ai/gpu-perf-engineering-resources","note":"This GitHub repository provides a comprehensive curriculum for learning about GPU performance engineering, covering everything from the basics to advanced techniques used by leading AI labs. It serves as a valuable resource for individuals looking to enhance their understanding and skills in this critical area of technology.","image":{"url":"https://ucarecdn.com/5b58bd0d-f281-4371-b527-ae38c92d7fe7/","alt":"GPU Performance Engineering Resources Curriculum","width":1200,"height":600},"stack":{"id":"c691055b-de15-47a1-ad6a-6c4ecec55f18","title":"GitHub repos to check out","url":"https://stacklist.com/stack/c691055b-de15-47a1-ad6a-6c4ecec55f18"},"created_at":"2026-06-29T22:09:20.388Z","updated_at":null,"aco":{"summary":"GPU Performance Engineering Resources is a comprehensive learning guide for engineers to master GPU kernel programming and optimization for high-performance AI systems. It covers fundamentals through production deployment with structured tiers of resources including architecture deep dives, matrix multiplication optimization, tensor cores, and distributed multi-GPU systems.","tags":["gpu-programming","performance-engineering","cuda","kernel-optimization","ai-infrastructure","tensor-cores"],"key_entities":[{"name":"NVIDIA","type":"organization","confidence":0.99},{"name":"Wafer","type":"organization","confidence":0.85},{"name":"DeepSeek","type":"organization","confidence":0.9},{"name":"PyTorch","type":"organization","confidence":0.95},{"name":"Colfax Research","type":"organization","confidence":0.85},{"name":"CUDA","type":"technology","confidence":0.99},{"name":"cuBLAS","type":"technology","confidence":0.98},{"name":"CUTLASS","type":"technology","confidence":0.98},{"name":"Triton","type":"technology","confidence":0.95},{"name":"ROCm","type":"technology","confidence":0.95},{"name":"FlashAttention","type":"technology","confidence":0.95},{"name":"Tensor Cores","type":"technology","confidence":0.99},{"name":"GPU kernel programming","type":"concept","confidence":0.99},{"name":"Matrix Multiplication Optimization","type":"concept","confidence":0.98},{"name":"Mixed Precision","type":"concept","confidence":0.95},{"name":"Distributed Multi-GPU","type":"concept","confidence":0.95},{"name":"Hwu","type":"person","confidence":0.8},{"name":"Kirk","type":"person","confidence":0.8},{"name":"El Hajj","type":"person","confidence":0.8},{"name":"Aleksa Gordić","type":"person","confidence":0.85},{"name":"Lei Mao","type":"person","confidence":0.85}],"classification":"reference","language":"en","confidence":0.85,"provenance":{"model":"claude-haiku-4-5","tool":"@stacklist/be@0.1.0","confidence":0.85,"timestamp":"2026-06-29T22:09:36.012Z"},"token_counts":{"approximate":3063,"cl100k":2673},"content_hash":"sha256:dd4be37353b7ebd5b58af0d7bac1555d12a459500c62db34a30b6b1e1f6a96ac","acp_version":"0.2","body_available":true,"body_tokens":3063,"visibility":"public","agent_accessible":true,"status":"final"},"_links":{"self":"/api/public/card/8521483e-0deb-40ab-bff1-4ef6d7a657cc.json","html":"https://stacklist.com/card/8521483e-0deb-40ab-bff1-4ef6d7a657cc","md":"/api/public/card/8521483e-0deb-40ab-bff1-4ef6d7a657cc.md","stack_json":"/api/public/stack/c691055b-de15-47a1-ad6a-6c4ecec55f18.json"}}