56,966 blogs · [ { "id": "01a0e2c0-45ce-70c7-8005-e355ea80e6bb", "title": "Optimization diaries: S/P ping-pong for FlashAttention-4 decode", "url": "https://research.colfax-intl.com/optimization-diaries-s-p-ping-pong-for-flashattention-4-decode/", "published_at": "2026-09-07T17:59:28+00:00" }, { "id": "01a0e2c0-45ce-70c7-8005-e355eb15f02a", "title": "Optimization diaries: Improving FlashAttention-4 backward pass kernel design for head dimension 64", "url": "https://research.colfax-intl.com/optimization-diaries-improving-flashattention-4-backward-for-head-dimension-64/", "published_at": "2026-09-06T21:45:28+00:00" }, { "id": "01a0e2c0-45ce-70c7-8005-e355eb208378", "title": "Optimizing an NVFP4 Blockscaled GEMM on RTX PRO 6000 Blackwell GPU (SM120)", "url": "https://research.colfax-intl.com/optimizing-an-nvfp4-blockscaled-gemm-on-rtx-pro-6000-blackwell-gpu-sm120/", "published_at": "2026-08-09T20:19:07+00:00" }, { "id": "01a0e2c0-45ce-70c7-8005-e355eb9e904e", "title": "NVFP4 Blockscaled GEMM on NVIDIA RTX Pro Blackwell GPUs (SM12x)", "url": "https://research.colfax-intl.com/cutlass-tutorial-nvfp4-blockscaled-gemm-on-nvidia-rtx-pro-blackwell-gpus-sm12x/", "published_at": "2026-06-20T17:46:55+00:00" }, { "id": "01a0e2c0-45ce-70c7-8005-e355ebea8145", "title": "Dynamic persistent tile scheduling with Cluster Launch Control (CLC) on NVIDIA Blackwell GPUs", "url": "https://research.colfax-intl.com/dynamic-persistent-tile-scheduling-with-cluster-launch-control-clc-on-nvidia-blackwell-gpus/", "published_at": "2026-05-09T17:25:55+00:00" }, { "id": "01a0e2c0-45ce-70c7-8005-e355ec02f530", "title": "GPU Mode: Fundamentals of CuTe Layout Algebra and Category-Theoretic Interpretation", "url": "https://research.colfax-intl.com/gpu-mode-fundamentals-of-cute-layout-algebra-and-category-theoretic-interpretation/", "published_at": "2026-04-20T18:37:47+00:00" }, { "id": "01a0e2c0-45ce-70c7-8005-e355ec321692", "title": "FlexAttention + FlashAttention-4: Fast and Flexible (External)", "url": "https://research.colfax-intl.com/flexattention-flashattention-4-fast-and-flexible-external/", "published_at": "2026-03-11T02:58:16+00:00" }, { "id": "01a0e2c0-45ce-70c7-8005-e355ecda8b96", "title": "CUTLASS Tutorial: Hardware-supported Block-scaling with NVIDIA Blackwell GPUs", "url": "https://research.colfax-intl.com/cutlass-tutorial-hardware-supported-block-scaling-with-nvidia-blackwell-gpus/", "published_at": "2026-03-05T17:57:02+00:00" }, { "id": "01a0e2c0-45ce-70c7-8005-e355edc7c2f2", "title": "FlashAttention-4: Algorithm and Kernel Pipelining Co-Design for Asymmetric Hardware Scaling", "url": "https://research.colfax-intl.com/flashattention-4-algorithm-and-kernel-pipelining-co-design-for-asymmetric-hardware-scaling/", "published_at": "2026-03-05T14:14:51+00:00" }, { "id": "01a0e2c0-45ce-70c7-8005-e355ee15333f", "title": "A User’s Guide to FlexAttention in FlashAttention CuTe DSL", "url": "https://research.colfax-intl.com/a-users-guide-to-flexattention-in-flash-attention-cute-dsl/", "published_at": "2025-11-15T03:11:38+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3ba9f75637b", "title": "Categorical Foundations for CuTe Layouts", "url": "https://research.colfax-intl.com/categorical-foundations-for-cute-layouts/", "published_at": "2025-09-21T22:36:34+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3baa02b308c", "title": "CUTLASS 3.x APIs: Orthogonal, Reusable, and Composable Abstractions for GEMM Kernel Design (External)", "url": "https://research.colfax-intl.com/cutlass-3-x-apis-orthogonal-reusable-and-composable-abstractions-for-gemm-kernel-design-external/", "published_at": "2025-07-20T03:57:52+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3baa0c78fe6", "title": "CUTLASS Tutorial: Sub-byte GEMM on NVIDIA® Blackwell GPUs", "url": "https://research.colfax-intl.com/cutlass-tutorial-sub-byte-gemm-on-nvidia-blackwell-gpus/", "published_at": "2025-06-07T20:51:29+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3baa127d65b", "title": "CUTLASS Tutorial: GEMM with Thread Block Clusters on NVIDIA® Blackwell GPUs", "url": "https://research.colfax-intl.com/cutlass-tutorial-gemm-with-thread-block-clusters-on-nvidia-blackwell-gpus/", "published_at": "2025-05-10T18:30:16+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3baa1e6f9ef", "title": "CUTLASS Tutorial: Writing GEMM Kernels Using Tensor Memory For NVIDIA® Blackwell GPUs", "url": "https://research.colfax-intl.com/cutlass-tutorial-writing-gemm-kernels-using-tensor-memory-for-nvidia-blackwell-gpus/", "published_at": "2025-04-19T16:00:33+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3baa28666cf", "title": "DeepSeek-R1 and FP8 Mixed-Precision Training", "url": "https://research.colfax-intl.com/deepseek-r1-and-fp8-mixed-precision-training/", "published_at": "2025-01-27T22:54:40+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3baa2fa543d", "title": "Protected: NVIDIA Data Center GPU Manager: An Intro to More Efficient and Granular GPU Management", "url": "https://research.colfax-intl.com/nvidia-data-center-gpu-manager/", "published_at": "2024-12-21T06:47:52+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3baa3a848bf", "title": "CUTLASS Tutorial: Persistent Kernels and Stream-K", "url": "https://research.colfax-intl.com/cutlass-tutorial-persistent-kernels-and-stream-k/", "published_at": "2024-12-20T07:13:39+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3baa3b503ce", "title": "FlashAttention-3 for Inference: INT8 Quantization and Query Head Packing for MQA/GQA (External)", "url": "https://research.colfax-intl.com/flashattention-3-for-inference-int8-quantization-and-query-head-packing-for-mqa-gqa-external/", "published_at": "2024-11-28T01:17:23+00:00" }, { "id": "01a0e2cf-1334-73a2-b1fe-c3baa3dd7d1a", "title": "GPU Mode: CUTLASS and FlashAttention-3", "url": "https://research.colfax-intl.com/gpu-mode-cutlass-and-flashattention-3/", "published_at": "2024-11-18T16:39:41+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a75415a4bd9d", "title": "Epilogue Fusion in CUTLASS with Epilogue Visitor Trees", "url": "https://research.colfax-intl.com/epilogue_visitor_tree/", "published_at": "2024-10-26T02:41:34+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a75416967fe8", "title": "GPU passthrough on Proxmox VE 8.2", "url": "https://research.colfax-intl.com/gpu-passthrough-on-proxmox-ve-8/", "published_at": "2024-10-24T00:20:41+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a75416f8bddf", "title": "CUTLASS Tutorial: Efficient GEMM kernel designs with Pipelining", "url": "https://research.colfax-intl.com/cutlass-tutorial-design-of-a-gemm-kernel/", "published_at": "2024-09-22T17:47:00+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a75417d494a1", "title": "CUTLASS Tutorial: Fast Matrix-Multiplication with WGMMA on NVIDIA® Hopper™ GPUs", "url": "https://research.colfax-intl.com/cutlass-tutorial-wgmma-hopper/", "published_at": "2024-08-06T16:33:04+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a7541838a3ae", "title": "FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision", "url": "https://research.colfax-intl.com/flashattention-3-fast-and-accurate-attention-with-asynchrony-and-low-precision/", "published_at": "2024-07-11T15:46:31+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a75418da8e3a", "title": "CUTLASS Tutorial: Mastering the NVIDIA® Tensor Memory Accelerator (TMA)", "url": "https://research.colfax-intl.com/tutorial-hopper-tma/", "published_at": "2024-06-24T16:39:24+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a7541996f843", "title": "Sharing NVIDIA® GPUs at the System Level: Time-Sliced and MIG-Backed vGPUs", "url": "https://research.colfax-intl.com/sharing-nvidia-gpus-at-the-system-level-time-sliced-and-mig-backed-vgpus/", "published_at": "2024-05-30T00:05:32+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a75419a4f28d", "title": "Tutorial: Matrix Transpose in CUTLASS", "url": "https://research.colfax-intl.com/tutorial-matrix-transpose-in-cutlass/", "published_at": "2024-05-06T01:50:59+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a75419dbedbb", "title": "Installing Ubuntu 22.04 LTS over the Network on Servers with the NVIDIA® Grace Hopper™ Superchip", "url": "https://research.colfax-intl.com/ubuntu-22-04-netboot-on-servers-with-the-nvidia-grace-hopper-superchip/", "published_at": "2024-04-18T23:12:44+00:00" }, { "id": "01a0e2e2-f1fa-72cf-8c47-a7541a7c63a8", "title": "Tutorial: Python bindings for CUDA libraries in PyTorch", "url": "https://research.colfax-intl.com/tutorial-python-binding-for-cuda-libraries-in-pytorch/", "published_at": "2024-03-13T22:56:25+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db61ec399d", "title": "Introduction to Transformers", "url": "https://research.colfax-intl.com/introduction-to-transformers/", "published_at": "2024-03-08T18:13:03+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db6281036c", "title": "Custom CUDA and Python", "url": "https://research.colfax-intl.com/custom-cuda-and-python/", "published_at": "2024-03-02T01:14:33+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db634785a3", "title": "Delivering 1 PFLOP/s of Performance with FP8 FlashAttention-2", "url": "https://research.colfax-intl.com/adding-fp8-to-flashattention/", "published_at": "2024-03-01T01:49:55+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db639b3cf3", "title": "Analog AI", "url": "https://research.colfax-intl.com/analog-ai/", "published_at": "2024-02-22T00:46:00+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db63ac3950", "title": "Narrow Bit-Width Formats for Deep Learning", "url": "https://research.colfax-intl.com/narrow-bit-width-formats-for-deep-learning/", "published_at": "2024-02-16T00:31:43+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db64a7f5dd", "title": "Proxmox VE", "url": "https://research.colfax-intl.com/proxmox-ve/", "published_at": "2024-02-08T00:15:00+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db65091c15", "title": "Software Pipelining in the NVIDIA Hopper Architecture", "url": "https://research.colfax-intl.com/software-pipelining-in-the-nvidia-h100-architecture/", "published_at": "2024-02-07T21:03:40+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db65507eb6", "title": "JSON Web Token Standard (JWT)", "url": "https://research.colfax-intl.com/json-web-token-standard-jwt/", "published_at": "2024-02-01T00:22:26+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db6620b42d", "title": "SPEC Benchmarks", "url": "https://research.colfax-intl.com/spec-benchmarks/", "published_at": "2024-01-25T00:59:45+00:00" }, { "id": "01a0e2f9-0f7b-7206-ae93-55db670dadf6", "title": "A note on the algebra of CuTe Layouts", "url": "https://research.colfax-intl.com/a-note-on-the-algebra-of-cute-layouts/", "published_at": "2023-12-15T01:56:54+00:00" }, { "id": "01a0e306-0b30-737a-a53d-f395ccbe9569", "title": "A Case Study in CUDA Kernel Fusion: Implementing FlashAttention-2 on NVIDIA Hopper Architecture using the CUTLASS Library", "url": "https://research.colfax-intl.com/nvidia-hopper-flashattention-2/", "published_at": "2023-12-05T08:09:34+00:00" }, { "id": "01a0e306-0b30-737a-a53d-f395cda6e28c", "title": "Developing CUDA Kernels for GEMM on NVIDIA Hopper Architecture using CUTLASS", "url": "https://research.colfax-intl.com/nvidia-hopper-gemm-cutlass/", "published_at": "2023-10-17T01:34:50+00:00" } ] posts Claim your blog
Back to research.colfax-intl.com
Blog · corpus.blog/blogs/research.colfax-intl.com/posts

research.colfax-intl.com

research.colfax-intl.com

2026

2025

2024

2023