41554 blogs · [ { "id": "01a08c61-ab22-72ab-bb07-636b4d8d415b", "title": "AI Commits Storage Violence", "url": "https://wes.today/storage-violence/", "published_at": "2026-05-22T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a28907e83", "title": "The training step", "url": "https://wes.today/series/training/train-from-scratch/training-step/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a28ee2f73", "title": "The training loop", "url": "https://wes.today/series/training/train-from-scratch/training-loop/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2964b0f5", "title": "Post-training", "url": "https://wes.today/series/training/train-from-scratch/post-training/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2a412d02", "title": "Evaluation", "url": "https://wes.today/series/training/train-from-scratch/evaluation/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2b4b27a9", "title": "Data mixture & curation", "url": "https://wes.today/series/training/train-from-scratch/training-data/data-mixture/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a32f8cd16", "title": "Why 16,384 GPUs?", "url": "https://wes.today/series/training/train-from-scratch/hardware-and-scale/why-scale/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a33c63f03", "title": "The units of distributed training", "url": "https://wes.today/series/training/train-from-scratch/training-step/training-units/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3cb5a4f6", "title": "Epochs", "url": "https://wes.today/series/training/train-from-scratch/training-loop/epochs/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3cb605eb", "title": "Batch size & gradient accumulation", "url": "https://wes.today/series/training/train-from-scratch/training-loop/batch-size/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3cd8bb45", "title": "Scaling laws & compute economics", "url": "https://wes.today/series/training/train-from-scratch/training-loop/scaling-laws/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3d1fe00b", "title": "Supervised fine-tuning (SFT)", "url": "https://wes.today/series/training/train-from-scratch/post-training/sft/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3d541e20", "title": "Preference training (RLHF, DPO)", "url": "https://wes.today/series/training/train-from-scratch/post-training/preference-training/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3e13064f", "title": "Reward hacking & objective mismatch", "url": "https://wes.today/series/training/train-from-scratch/post-training/reward-hacking/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3eed5402", "title": "Validation loss & benchmarks", "url": "https://wes.today/series/training/train-from-scratch/evaluation/validation-benchmarks/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3fcb4d89", "title": "Contamination & evaluation integrity", "url": "https://wes.today/series/training/train-from-scratch/evaluation/contamination/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b404de427", "title": "Why does the FFN hold 80% of the parameters?", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/ffn-hidden-dimension/ffn-parameters/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b41bba265", "title": "Mixture of Experts (MoE)", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/ffn-hidden-dimension/moe/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b429cc275", "title": "Synthetic data & distillation", "url": "https://wes.today/series/training/train-from-scratch/training-data/data-mixture/synthetic-data/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b43ab84fe", "title": "Long-context training & sequence packing", "url": "https://wes.today/series/training/train-from-scratch/training-data/data-mixture/long-context/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b47db774b", "title": "The parallelism orchestration stack", "url": "https://wes.today/series/training/train-from-scratch/training-step/forward-pass/parallelism-stack/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b49a2da80", "title": "The training objective: shifted tokens & loss masking", "url": "https://wes.today/series/training/train-from-scratch/training-step/loss-calculation/training-objective/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b49f6057d", "title": "Activation checkpointing", "url": "https://wes.today/series/training/train-from-scratch/training-step/backward-pass/activation-checkpointing/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b49fa883a", "title": "What actually changes in the weights", "url": "https://wes.today/series/training/train-from-scratch/training-loop/epochs/weight-changes/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4ad0a1db", "title": "Training stability: loss spikes, NaN, and precision", "url": "https://wes.today/series/training/train-from-scratch/training-loop/scaling-laws/training-stability/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4bc46fd7", "title": "ZeRO & memory optimization", "url": "https://wes.today/series/training/train-from-scratch/training-step/optimizer-step/zero/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4c12c067", "title": "Mixed precision training", "url": "https://wes.today/series/training/train-from-scratch/training-step/optimizer-step/mixed-precision/", "published_at": "2026-05-19T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a247f0bc6", "title": "What is the training data?", "url": "https://wes.today/series/training/train-from-scratch/training-data/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a27bb00e3", "title": "What is the model architecture?", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a281e7722", "title": "Hardware & scale", "url": "https://wes.today/series/training/train-from-scratch/hardware-and-scale/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2a41982c", "title": "What does the final training data actually look like?", "url": "https://wes.today/series/training/train-from-scratch/training-data/final-data-format/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2b2bf2c0", "title": "Why can't training data be pre-tokenized at the source?", "url": "https://wes.today/series/training/train-from-scratch/training-data/pre-tokenization/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2f81dd65", "title": "How do you decide the number of layers?", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/layer-count/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a303441fc", "title": "Vocabulary size: 128,000", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/vocabulary-size/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a30de7d96", "title": "Model dimension (d_model): 8,192", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/model-dimension/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a311a19a7", "title": "Attention heads: 64", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/attention-heads/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a314ac566", "title": "Key-value heads: 8 (Grouped Query Attention)", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/gqa/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a31f1d737", "title": "FFN hidden dimension: 28,672", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/ffn-hidden-dimension/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a326d191a", "title": "RMSNorm, RoPE, and SiLU", "url": "https://wes.today/series/training/train-from-scratch/model-architecture/rmsnorm-rope-silu/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a32af0def", "title": "Orchestration — Slurm vs Kubernetes", "url": "https://wes.today/series/training/train-from-scratch/hardware-and-scale/orchestration/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a33ddbe12", "title": "Phase 1 — Data loading", "url": "https://wes.today/series/training/train-from-scratch/training-step/data-loading/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b39ba7622", "title": "Phase 2 — Forward pass", "url": "https://wes.today/series/training/train-from-scratch/training-step/forward-pass/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3a4bc3d1", "title": "Phase 3 — Loss calculation", "url": "https://wes.today/series/training/train-from-scratch/training-step/loss-calculation/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3a926cfb", "title": "Phase 4 — Backward pass", "url": "https://wes.today/series/training/train-from-scratch/training-step/backward-pass/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3b212df7", "title": "Phase 5 — Gradient synchronization", "url": "https://wes.today/series/training/train-from-scratch/training-step/gradient-sync/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3bcd139f", "title": "Phase 6 — Optimizer step", "url": "https://wes.today/series/training/train-from-scratch/training-step/optimizer-step/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b3c3a29f1", "title": "Phase 7 — Checkpointing", "url": "https://wes.today/series/training/train-from-scratch/training-step/checkpointing/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4dbfa1cb", "title": "What happens when you train a model from scratch?", "url": "https://wes.today/series/training/train-from-scratch/", "published_at": "2026-05-14T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4ea82c6b", "title": "I Named This Post", "url": "https://wes.today/posts/i-named-this-post/", "published_at": "2026-05-07T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4eb48fc7", "title": "Claude Code Meets D&D", "url": "https://wes.today/posts/claude-code-meets-dnd/", "published_at": "2026-05-06T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4f8fbdee", "title": "Claude Code Meets Obsidian", "url": "https://wes.today/posts/claude-code-meets-obsidian/", "published_at": "2026-04-30T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4f98bed7", "title": "Hello World", "url": "https://wes.today/posts/hello-world/", "published_at": "2026-04-19T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a24ef7956", "title": "What are vectors?", "url": "https://wes.today/series/inference/what-happens/vectors/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a24ff24dc", "title": "What is a token?", "url": "https://wes.today/series/inference/what-happens/tokens/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a252fa613", "title": "What are embeddings and how are they created?", "url": "https://wes.today/series/inference/what-happens/embeddings/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a25ce0411", "title": "Prefill vs decode", "url": "https://wes.today/series/inference/what-happens/prefill-decode/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a25f4442a", "title": "How does \"thinking\" work?", "url": "https://wes.today/series/inference/what-happens/thinking/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a26eead61", "title": "How do tool calls work?", "url": "https://wes.today/series/inference/what-happens/tool-calls/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a27770e70", "title": "How does memory work?", "url": "https://wes.today/series/inference/what-happens/memory/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2a5e9368", "title": "How tokenization actually works", "url": "https://wes.today/series/inference/what-happens/tokens/tokenization/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2b390f9f", "title": "What are weights?", "url": "https://wes.today/series/inference/what-happens/embeddings/weights/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2be0ac6c", "title": "Gradients and gradient updates (how weights get their values)", "url": "https://wes.today/series/inference/what-happens/embeddings/gradients/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2c78b97b", "title": "\"Directions in the space encode relationships\"", "url": "https://wes.today/series/inference/what-happens/embeddings/directions/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2d3e15e1", "title": "What are model layers?", "url": "https://wes.today/series/inference/what-happens/embeddings/model-layers/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2db60168", "title": "How do layers transform vectors?", "url": "https://wes.today/series/inference/what-happens/embeddings/layer-transforms/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2e1349aa", "title": "What are hidden states?", "url": "https://wes.today/series/inference/what-happens/embeddings/hidden-states/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2e592929", "title": "What is a dot product?", "url": "https://wes.today/series/inference/what-happens/vectors/dot-product/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2e71ed72", "title": "KV cache and context memory costs", "url": "https://wes.today/series/inference/what-happens/prefill-decode/kv-cache/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2f1aec47", "title": "Skills: abstraction over tool calls", "url": "https://wes.today/series/inference/what-happens/tool-calls/skills/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab21-71e0-8401-4d6a2fbb62ec", "title": "Planning and multi-step execution", "url": "https://wes.today/series/inference/what-happens/thinking/planning/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b41259571", "title": "Tokenization performance: where does it run and what's the bottleneck?", "url": "https://wes.today/series/inference/what-happens/tokens/tokenization/tokenization-perf/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b427b215e", "title": "Dimension trade-offs: expressiveness vs. cost", "url": "https://wes.today/series/inference/what-happens/embeddings/layer-transforms/dimension-tradeoffs/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b433329e5", "title": "Attention deep dive: what does it mean for a token to \"pay attention\"?", "url": "https://wes.today/series/inference/what-happens/embeddings/model-layers/attention-deep-dive/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4444f1fa", "title": "FFN deep dive: the per-token thinking step", "url": "https://wes.today/series/inference/what-happens/embeddings/model-layers/ffn-deep-dive/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b447ec7e2", "title": "From final vector to predicted token", "url": "https://wes.today/series/inference/what-happens/embeddings/model-layers/final-vector-to-token/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b44b937e0", "title": "Sparse attention: skipping tokens you don't need", "url": "https://wes.today/series/inference/what-happens/prefill-decode/kv-cache/sparse-attention/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b45410c5e", "title": "Attention approximations: breaking the T² barrier differently", "url": "https://wes.today/series/inference/what-happens/prefill-decode/kv-cache/attention-approximations/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b45f47db1", "title": "What is quantization?", "url": "https://wes.today/series/inference/what-happens/prefill-decode/kv-cache/quantization/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4668b5d8", "title": "Paged attention: virtual memory for KV cache", "url": "https://wes.today/series/inference/what-happens/prefill-decode/kv-cache/paged-attention/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b471924c7", "title": "KV cache offloading: trading latency for capacity", "url": "https://wes.today/series/inference/what-happens/prefill-decode/kv-cache/kv-cache-offloading/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b48c6d17a", "title": "MQA and GQA: reducing cache size at the architecture level", "url": "https://wes.today/series/inference/what-happens/prefill-decode/kv-cache/mqa-gqa/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4c2ede0a", "title": "Multi-head attention: how it works and what it costs", "url": "https://wes.today/series/inference/what-happens/embeddings/model-layers/attention-deep-dive/multi-head-attention/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b4c9a56ea", "title": "How does the model know when to stop?", "url": "https://wes.today/series/inference/what-happens/embeddings/model-layers/final-vector-to-token/stopping/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b5057e2a1", "title": "What happens when you send a message to an LLM", "url": "https://wes.today/series/inference/what-happens/", "published_at": "2026-04-13T00:00:00+00:00" }, { "id": "01a08c61-ab22-72ab-bb07-636b50d8ed83", "title": "About", "url": "https://wes.today/about/", "published_at": null }, { "id": "01a08c61-ab22-72ab-bb07-636b515b5bff", "title": "Contact", "url": "https://wes.today/contact/", "published_at": null }, { "id": "01a08c61-ab22-72ab-bb07-636b521fc05d", "title": "Now", "url": "https://wes.today/now/", "published_at": null } ] posts Claim your blog
Back to wes.today
Blog · corpus.blog/blogs/wes.today/posts

wes.today

wes.today

2026

Undated