56,966 blogs · [ { "id": "01a087e0-2c90-73f7-a398-df6b0e435038", "title": "Softmax and Cross-Entropy Backward Pass", "url": "https://shreyansh26.github.io/post/2026-07-06_softmax-cross-entropy-backprop/", "published_at": "2026-07-06T00:00:00+00:00" }, { "id": "01a087e0-2c90-73f7-a398-df6b0f073e75", "title": "Decompose-K: From torch.compile to Hand-Tuned Triton Kernels for Skinny Large‑K Matmuls", "url": "https://shreyansh26.github.io/post/2026-06-21_decompose-k-triton/", "published_at": "2026-06-21T00:00:00+00:00" }, { "id": "01a087e0-2c90-73f7-a398-df6b0f6ec287", "title": "KV Cache Compaction and Compression: From Attention Sinks to Learned Memory", "url": "https://shreyansh26.github.io/post/2026-06-01_kv-cache-compaction-compression/", "published_at": "2026-06-01T00:00:00+00:00" }, { "id": "01a087e0-2c90-73f7-a398-df6b0f8b2b5a", "title": "Paper Summary #17 - Engram", "url": "https://shreyansh26.github.io/post/2026-05-17_engram-layers/", "published_at": "2026-05-17T00:00:00+00:00" }, { "id": "01a087e0-2c90-73f7-a398-df6b0fd59df7", "title": "Paper Summary #16 - Canon Layers", "url": "https://shreyansh26.github.io/post/2026-05-16_canon-layers/", "published_at": "2026-05-16T00:00:00+00:00" }, { "id": "01a087e0-2c90-73f7-a398-df6b0ff23231", "title": "Paper Summary #15 - Hyper-Connections and mHC", "url": "https://shreyansh26.github.io/post/2026-05-15_hyper-connections-mhc/", "published_at": "2026-05-15T00:00:00+00:00" }, { "id": "01a087e0-2c90-73f7-a398-df6b10ed9663", "title": "Deep dive into CUDA Scan Kernels: Hierarchical and Single-Pass Variants", "url": "https://shreyansh26.github.io/post/2026-02-19_cuda-scan-kernels/", "published_at": "2026-02-19T00:00:00+00:00" }, { "id": "01a087e0-2c90-73f7-a398-df6b11c4e23f", "title": "Paper Summary #14 - Physics of Language Models: Part 3.1, Knowledge Storage and Extraction", "url": "https://shreyansh26.github.io/post/2026-01-17_physics-of-lms-3-1-knowledge-storage-and-extraction/", "published_at": "2026-01-17T00:00:00+00:00" }, { "id": "01a087e0-2c90-73f7-a398-df6b125c8389", "title": "Understanding Multi-Head Latent Attention (MLA)", "url": "https://shreyansh26.github.io/post/2025-11-08_multihead-latent-attention/", "published_at": "2025-11-08T00:00:00+00:00" }, { "id": "01a087e0-2c90-73f7-a398-df6b12f9b923", "title": "Deriving the Gradient for the Backward Pass of Layer Normalization", "url": "https://shreyansh26.github.io/post/2025-06-04_layernorm-gradients/", "published_at": "2025-06-04T00:00:00+00:00" } ] posts Claim your blog
Back to shreyansh26.github.io
Blog · corpus.blog/blogs/shreyansh26.github.io/posts

shreyansh26.github.io

shreyansh26.github.io

2026

2025