{"adoption": {"forks": 290, "observed_at": "2026-08-28T04:07:49.800931+00:00", "stars": 3223}, "canonical_url": "https://ross.abutalabs.com/products/how-to-optim-algorithm-in-cuda", "card": {"archived": false, "artifact_type": "learning-resource", "description": "how to optimize some algorithm in cuda.", "domain": ["gpu-computing", "deep-learning", "large-language-models", "developer-tools", "tutorials"], "enriched": true, "function": ["gpu-computing", "llm-inference", "llm-training", "benchmarking"], "health_score": 99, "homepage": null, "language": "Cuda", "license": null, "license_family": "other", "maturity": "active", "member_repos": ["BBuf/how-to-optim-algorithm-in-cuda"], "name": "BBuf/how-to-optim-algorithm-in-cuda", "platform": ["python", "cpp"], "pushed_at": "2026-08-22T06:39:47+00:00", "repo": "BBuf/how-to-optim-algorithm-in-cuda", "stars": 3223, "tags": ["cuda-kernels", "cutlass", "triton", "ptx-isa", "gemm-optimization", "study-notes", "ai-infrastructure", "gpu", "cuda"], "topics": ["cuda", "llm"], "urls": [], "use_cases": ["learn how to write optimized CUDA kernels for softmax and reduction", "understand GEMM optimization with CUTLASS and CuTe", "study Triton kernel examples for PyTorch", "optimize LLM inference performance on GPUs", "learn PTX ISA and GPU architecture internals", "find notes from the CUDA-MODE lecture series"], "what_it_is": "A curated collection of notes and hands-on code for optimizing algorithms on CUDA GPUs, covering handwritten kernels, CUTLASS/CuTe, Triton, PTX ISA, and PyTorch internals. It also includes LLM inference and training systems optimization material, serving as a public engineering notebook for GPU systems work.", "when_to_avoid": ["you need a production-ready kernel library to drop into your project", "you want a maintained software tool rather than study notes", "you need guaranteed correctness or support, since there is no license or formal releases"], "when_to_choose": ["you are learning GPU kernel optimization from first principles", "you want curated notes on CUTLASS, Triton, and PTX in one place", "you work on LLM inference or training systems and need performance background"]}, "data_as_of": "2026-08-30T08:39:29.467469+00:00", "members": [{"path": "/products/how-to-optim-algorithm-in-cuda", "repo": "BBuf/how-to-optim-algorithm-in-cuda", "role": "main", "score": 98}], "provenance": {"archived": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "artifact_type": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}, "description": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "domain": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}, "enriched": {"inputs": [], "kind": "computed", "method": "enrichment_status"}, "function": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}, "health_score": {"inputs": ["days_since_push", "days_since_release", "archived"], "kind": "computed", "method": "health_v1"}, "homepage": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "language": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "license": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "license_family": {"inputs": ["license"], "kind": "computed", "method": "license_family"}, "maturity": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}, "member_repos": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "name": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "platform": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}, "pushed_at": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "repo": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "stars": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "tags": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}, "topics": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "urls": {"kind": "observed", "observed_at": "2026-08-28T04:07:49.800931+00:00", "source": "github"}, "use_cases": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}, "what_it_is": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}, "when_to_avoid": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}, "when_to_choose": {"confidence": null, "enriched_at": "2026-08-29T18:44:18.158160+00:00", "kind": "inferred", "prompt_version": 1, "sources": [{"content_hash": "1d36abf96a8a35e44e02652fb0a8c4dc543c379a88ecb9c0d2d548884026038f", "fetched_at": "2026-08-28T04:07:49.800931+00:00", "kind": "readme", "missing": false, "url": "https://github.com/BBuf/how-to-optim-algorithm-in-cuda"}], "taxonomy_version": 1}}, "score": {"components": {"activity": 99, "longevity": 100, "rhythm": 97}, "computed_at": "2026-09-02T17:46:02.011165+00:00", "flags": ["no_license"], "formula": "round(0.45*activity + 0.35*rhythm + 0.20*longevity); archived -> min(score, 10)", "inputs": {"age_days": 2968, "days_push": 11, "days_rel": 22, "gap_med": 4, "n_releases_24m": 2}, "score": 98, "version": 2}, "staleness": {"enrichment_outdated": false, "low_confidence": false, "scrape_days": 9, "stale_scrape": false}}