corpus.blog
Most cited
Talked about
Blogs
41554 blogs · [ { "id": "01a087ec-e4f4-72ad-a3dc-85b6c80832eb", "title": "A little experiment in evading AI detection", "url": "https://thegustafson.com/blog/evading-ai-detection", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6c8933b6f", "title": "My Throw Decides My Aim", "url": "https://thegustafson.com/blog/my-throw-decides-my-aim", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6c9443421", "title": "Vectors, Matrices, and the Spaces They Live In", "url": "https://thegustafson.com/blog/vectors-and-spaces", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6c9a8aea5", "title": "Norms, Dot Products, and Similarity", "url": "https://thegustafson.com/blog/norms-dots-similarity", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6c9ba109b", "title": "Distributions, Softmax, and the Chain Rule of Words", "url": "https://thegustafson.com/blog/probability-distributions", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6ca8673c1", "title": "Cross-Entropy, KL Divergence, and What Loss Functions Measure", "url": "https://thegustafson.com/blog/cross-entropy-and-loss", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6ca8dce68", "title": "Gradients and How Machines Learn", "url": "https://thegustafson.com/blog/gradients-and-learning", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6cb28871c", "title": "Optimizers: Momentum, Adam, and Learning Rate Schedules", "url": "https://thegustafson.com/blog/optimizers", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6cc14c603", "title": "GPUs, Floating Point, and Why Precision Matters", "url": "https://thegustafson.com/blog/gpus-and-floating-point", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6ccb80fbd", "title": "A Short Prehistory of Statistical NLP", "url": "https://thegustafson.com/blog/prehistory-of-nlp", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6cdb63ff5", "title": "Language Modeling as Next-Token Prediction", "url": "https://thegustafson.com/blog/next-token-prediction", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6cdebb27f", "title": "N-gram Models and the Curse of Sparsity", "url": "https://thegustafson.com/blog/ngrams-and-sparsity", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6cee6d8e3", "title": "Word2vec and the Embedding Revolution", "url": "https://thegustafson.com/blog/word2vec", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6cf71ea46", "title": "GloVe, FastText, and the Embedding Zoo", "url": "https://thegustafson.com/blog/glove-fasttext", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6cfdc3270", "title": "Recurrent Neural Networks and Sequence Modeling", "url": "https://thegustafson.com/blog/rnns", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d033b5f9", "title": "LSTMs, GRUs, and Gated Memory", "url": "https://thegustafson.com/blog/lstms-grus", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d081d528", "title": "Seq2seq, Bahdanau Attention, and Why Recurrence Hit a Wall", "url": "https://thegustafson.com/blog/seq2seq-and-bahdanau", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d130de4b", "title": "Unicode, Bytes, and What Text Actually Is", "url": "https://thegustafson.com/blog/unicode-and-bytes", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d144f1e9", "title": "Byte-Pair Encoding from Scratch", "url": "https://thegustafson.com/blog/bpe-from-scratch", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d171b401", "title": "WordPiece, Unigram, and SentencePiece", "url": "https://thegustafson.com/blog/wordpiece-unigram-sentencepiece", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d1c56c97", "title": "Vocabulary Size, Merge Order, and Fertility", "url": "https://thegustafson.com/blog/vocabulary-design", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d221b707", "title": "The Embedding Table and Its Geometry", "url": "https://thegustafson.com/blog/embedding-table", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d30de060", "title": "Special Tokens, Chat Templates, and Input Formatting", "url": "https://thegustafson.com/blog/special-tokens-chat-templates", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d3863e89", "title": "Packing, Masking, and Tokenization as a Model Interface", "url": "https://thegustafson.com/blog/packing-and-masking", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d3e5a2b0", "title": "Self-Attention: Q, K, V from First Principles", "url": "https://thegustafson.com/blog/attention-from-scratch", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d491d9a4", "title": "Multi-Head Attention and Representation Subspaces", "url": "https://thegustafson.com/blog/multi-head-attention", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d5194d45", "title": "Causal Masking and the Autoregressive Constraint", "url": "https://thegustafson.com/blog/causal-masking", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d55b7e51", "title": "Positional Encodings: Sinusoidal, Learned, RoPE, and ALiBi", "url": "https://thegustafson.com/blog/positional-encodings-and-rope", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d5c8e128", "title": "The Feed-Forward Block as Key-Value Memory", "url": "https://thegustafson.com/blog/ffn-as-memory", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d627cff2", "title": "Layer Normalization: Pre-Norm, Post-Norm, and RMSNorm", "url": "https://thegustafson.com/blog/layer-norms", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d643ebb8", "title": "Decoder-Only vs. Encoder-Decoder: Architecture Trade-offs", "url": "https://thegustafson.com/blog/decoder-vs-encoder-decoder", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d697d94e", "title": "A Close Reading of ‘Attention Is All You Need’", "url": "https://thegustafson.com/blog/reading-attention-is-all-you-need", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d7713429", "title": "Mixture-of-Experts Layers", "url": "https://thegustafson.com/blog/moe-layers", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d7fa91f2", "title": "Training View vs. Inference View", "url": "https://thegustafson.com/blog/training-vs-inference", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d8a22a32", "title": "Prefill vs. Decode: The Two Phases of Inference", "url": "https://thegustafson.com/blog/prefill-vs-decode", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d8ce0569", "title": "Why One New Token Means One New Row", "url": "https://thegustafson.com/blog/the-one-new-row", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d9bdc012", "title": "The KV Cache from First Principles", "url": "https://thegustafson.com/blog/kv-cache-from-first-principles", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6d9cda7af", "title": "Sampling Strategies: Temperature, Top-k, Top-p, and Min-p", "url": "https://thegustafson.com/blog/sampling-strategies", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6da362efa", "title": "Speculative Decoding", "url": "https://thegustafson.com/blog/speculative-decoding", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6da65da52", "title": "Continuous Batching", "url": "https://thegustafson.com/blog/continuous-batching", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6daeb8dc9", "title": "Prefix Caching and Prompt Reuse", "url": "https://thegustafson.com/blog/prefix-caching", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6db187390", "title": "Structured and Constrained Generation", "url": "https://thegustafson.com/blog/structured-generation", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6db51ff49", "title": "What an Inference Engine Actually Does", "url": "https://thegustafson.com/blog/what-an-inference-engine-does", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6dbd04411", "title": "Kernel Fusion and the Memory Wall", "url": "https://thegustafson.com/blog/kernel-fusion", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6dcbd61db", "title": "PagedAttention: Virtual Memory for the KV Cache", "url": "https://thegustafson.com/blog/paged-attention", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6dcc043c9", "title": "Memory Management: Fitting a Model and Deciding Concurrency", "url": "https://thegustafson.com/blog/memory-management", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6dcfd8d2b", "title": "Quantization: INT8, INT4, GPTQ, AWQ, and GGUF", "url": "https://thegustafson.com/blog/quantization", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6dd5be8aa", "title": "Tensor, Pipeline, and Expert Parallelism", "url": "https://thegustafson.com/blog/parallelism-strategies", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6de4b2d99", "title": "Throughput vs. Latency: Picking Your Tradeoff", "url": "https://thegustafson.com/blog/throughput-vs-latency", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6de7bc55f", "title": "Comparing Engines: vLLM, TGI, TensorRT-LLM, llama.cpp, SGLang", "url": "https://thegustafson.com/blog/comparing-engines", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6df334bdc", "title": "Benchmarking Your Own Inference Stack", "url": "https://thegustafson.com/blog/benchmarking-inference", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6dff2ac34", "title": "Pretraining Data: Mixtures, Curation, and What the Model Sees", "url": "https://thegustafson.com/blog/pretraining-data", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e065bc9f", "title": "Distributed Training: FSDP, DeepSpeed, and Megatron", "url": "https://thegustafson.com/blog/distributed-training", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e08cf8fa", "title": "Supervised Fine-Tuning", "url": "https://thegustafson.com/blog/sft", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e1220711", "title": "The RLHF Pipeline: Reward Models and PPO", "url": "https://thegustafson.com/blog/rlhf", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e17e5020", "title": "DPO and Its Variants: Skipping the Reward Model", "url": "https://thegustafson.com/blog/dpo-and-variants", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e1c26b5e", "title": "Constitutional AI and RLAIF", "url": "https://thegustafson.com/blog/constitutional-ai", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e28315ce", "title": "Tool-Use Fine-Tuning", "url": "https://thegustafson.com/blog/tool-use-finetuning", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e35c6ecf", "title": "Long-Context Training Techniques", "url": "https://thegustafson.com/blog/long-context-training", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e3bfc2ff", "title": "Synthetic Data and Distillation", "url": "https://thegustafson.com/blog/synthetic-data-distillation", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e45a79bd", "title": "LoRA and Parameter-Efficient Fine-Tuning", "url": "https://thegustafson.com/blog/lora-and-peft", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e4ae8282", "title": "Loss vs. Benchmarks: Two Languages for Model Quality", "url": "https://thegustafson.com/blog/loss-vs-benchmarks", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e5881499", "title": "Eval Harnesses: lm-eval-harness and HELM", "url": "https://thegustafson.com/blog/eval-harnesses", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e5a50542", "title": "Contamination and Leakage", "url": "https://thegustafson.com/blog/contamination", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e61d0f68", "title": "Evaluating Reasoning", "url": "https://thegustafson.com/blog/evaluating-reasoning", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e670c12c", "title": "Tool and Agent Evaluation", "url": "https://thegustafson.com/blog/tool-agent-evals", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e674daf4", "title": "Human Preference Evaluation", "url": "https://thegustafson.com/blog/human-preference-eval", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e6984da8", "title": "Calibration and Abstention", "url": "https://thegustafson.com/blog/calibration-and-abstention", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e6b83ebc", "title": "Embeddings from Scratch: From Word2Vec to E5", "url": "https://thegustafson.com/blog/embeddings-and-vector-search", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e787c5c8", "title": "Vector Search and Approximate Nearest Neighbors", "url": "https://thegustafson.com/blog/vector-search-and-ann", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e80b96f0", "title": "Dense, Sparse, and Hybrid Retrieval", "url": "https://thegustafson.com/blog/dense-sparse-hybrid", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e84f418f", "title": "Chunking Strategies", "url": "https://thegustafson.com/blog/chunking-strategies", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e89c155c", "title": "Rerankers and Cross-Encoders", "url": "https://thegustafson.com/blog/rerankers", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e93ade5d", "title": "RAG Architectures End to End", "url": "https://thegustafson.com/blog/rag-end-to-end", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6e9ffd4e1", "title": "GraphRAG and Knowledge Graphs", "url": "https://thegustafson.com/blog/graphrag", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6ea56d524", "title": "Context Windows: Advertised vs. Effective", "url": "https://thegustafson.com/blog/context-windows", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6eb2ae316", "title": "Context Engineering: Compaction, Clearing, and Memory", "url": "https://thegustafson.com/blog/context-engineering", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6eb8d3e39", "title": "A Short History of Agents: From ReAct to 2026", "url": "https://thegustafson.com/blog/history-of-agents", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6ec4baf67", "title": "Function Calling as Structured Generation", "url": "https://thegustafson.com/blog/function-calling", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6ecd45106", "title": "The Agent Loop: Model, Runtime, Tool, Resume", "url": "https://thegustafson.com/blog/the-agent-loop", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6ed419a78", "title": "Transcript Formats: How Providers Represent Tool Use", "url": "https://thegustafson.com/blog/transcript-formats", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6ee16436b", "title": "MCP: A Cross-System Standard for Tool Integration", "url": "https://thegustafson.com/blog/mcp", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6eee206c4", "title": "Remote MCP, OAuth, and Enterprise Auth", "url": "https://thegustafson.com/blog/remote-mcp-and-auth", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6ef6198b7", "title": "Planning and Reasoning in Agents", "url": "https://thegustafson.com/blog/planning-and-reasoning", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6effa7687", "title": "Computer Use and Browser Agents", "url": "https://thegustafson.com/blog/computer-use", "published_at": null }, { "id": "01a087ec-e4f4-72ad-a3dc-85b6f0ba47b9", "title": "Multi-Agent Systems and Agent-to-Agent Protocols", "url": "https://thegustafson.com/blog/multi-agent", "published_at": null } ] posts
Claim your blog
Back to thegustafson.com
Blog · corpus.blog/blogs/thegustafson.com/posts
thegustafson.com
thegustafson.com
Undated
A little experiment in evading AI detection
original ↗
—
My Throw Decides My Aim
original ↗
—
Vectors, Matrices, and the Spaces They Live In
original ↗
—
Norms, Dot Products, and Similarity
original ↗
—
Distributions, Softmax, and the Chain Rule of Words
original ↗
—
Cross-Entropy, KL Divergence, and What Loss Functions Measure
original ↗
—
Gradients and How Machines Learn
original ↗
—
Optimizers: Momentum, Adam, and Learning Rate Schedules
original ↗
—
GPUs, Floating Point, and Why Precision Matters
original ↗
—
A Short Prehistory of Statistical NLP
original ↗
—
Language Modeling as Next-Token Prediction
original ↗
—
N-gram Models and the Curse of Sparsity
original ↗
—
Word2vec and the Embedding Revolution
original ↗
—
GloVe, FastText, and the Embedding Zoo
original ↗
—
Recurrent Neural Networks and Sequence Modeling
original ↗
—
LSTMs, GRUs, and Gated Memory
original ↗
—
Seq2seq, Bahdanau Attention, and Why Recurrence Hit a Wall
original ↗
—
Unicode, Bytes, and What Text Actually Is
original ↗
—
Byte-Pair Encoding from Scratch
original ↗
—
WordPiece, Unigram, and SentencePiece
original ↗
—
Vocabulary Size, Merge Order, and Fertility
original ↗
—
The Embedding Table and Its Geometry
original ↗
—
Special Tokens, Chat Templates, and Input Formatting
original ↗
—
Packing, Masking, and Tokenization as a Model Interface
original ↗
—
Self-Attention: Q, K, V from First Principles
original ↗
—
Multi-Head Attention and Representation Subspaces
original ↗
—
Causal Masking and the Autoregressive Constraint
original ↗
—
Positional Encodings: Sinusoidal, Learned, RoPE, and ALiBi
original ↗
—
The Feed-Forward Block as Key-Value Memory
original ↗
—
Layer Normalization: Pre-Norm, Post-Norm, and RMSNorm
original ↗
—
Decoder-Only vs. Encoder-Decoder: Architecture Trade-offs
original ↗
—
A Close Reading of ‘Attention Is All You Need’
original ↗
—
Mixture-of-Experts Layers
original ↗
—
Training View vs. Inference View
original ↗
—
Prefill vs. Decode: The Two Phases of Inference
original ↗
—
Why One New Token Means One New Row
original ↗
—
The KV Cache from First Principles
original ↗
—
Sampling Strategies: Temperature, Top-k, Top-p, and Min-p
original ↗
—
Speculative Decoding
original ↗
—
Continuous Batching
original ↗
—
Prefix Caching and Prompt Reuse
original ↗
—
Structured and Constrained Generation
original ↗
—
What an Inference Engine Actually Does
original ↗
—
Kernel Fusion and the Memory Wall
original ↗
—
PagedAttention: Virtual Memory for the KV Cache
original ↗
—
Memory Management: Fitting a Model and Deciding Concurrency
original ↗
—
Quantization: INT8, INT4, GPTQ, AWQ, and GGUF
original ↗
—
Tensor, Pipeline, and Expert Parallelism
original ↗
—
Throughput vs. Latency: Picking Your Tradeoff
original ↗
—
Comparing Engines: vLLM, TGI, TensorRT-LLM, llama.cpp, SGLang
original ↗
—
Benchmarking Your Own Inference Stack
original ↗
—
Pretraining Data: Mixtures, Curation, and What the Model Sees
original ↗
—
Distributed Training: FSDP, DeepSpeed, and Megatron
original ↗
—
Supervised Fine-Tuning
original ↗
—
The RLHF Pipeline: Reward Models and PPO
original ↗
—
DPO and Its Variants: Skipping the Reward Model
original ↗
—
Constitutional AI and RLAIF
original ↗
—
Tool-Use Fine-Tuning
original ↗
—
Long-Context Training Techniques
original ↗
—
Synthetic Data and Distillation
original ↗
—
LoRA and Parameter-Efficient Fine-Tuning
original ↗
—
Loss vs. Benchmarks: Two Languages for Model Quality
original ↗
—
Eval Harnesses: lm-eval-harness and HELM
original ↗
—
Contamination and Leakage
original ↗
—
Evaluating Reasoning
original ↗
—
Tool and Agent Evaluation
original ↗
—
Human Preference Evaluation
original ↗
—
Calibration and Abstention
original ↗
—
Embeddings from Scratch: From Word2Vec to E5
original ↗
—
Vector Search and Approximate Nearest Neighbors
original ↗
—
Dense, Sparse, and Hybrid Retrieval
original ↗
—
Chunking Strategies
original ↗
—
Rerankers and Cross-Encoders
original ↗
—
RAG Architectures End to End
original ↗
—
GraphRAG and Knowledge Graphs
original ↗
—
Context Windows: Advertised vs. Effective
original ↗
—
Context Engineering: Compaction, Clearing, and Memory
original ↗
—
A Short History of Agents: From ReAct to 2026
original ↗
—
Function Calling as Structured Generation
original ↗
—
The Agent Loop: Model, Runtime, Tool, Resume
original ↗
—
Transcript Formats: How Providers Represent Tool Use
original ↗
—
MCP: A Cross-System Standard for Tool Integration
original ↗
—
Remote MCP, OAuth, and Enterprise Auth
original ↗
—
Planning and Reasoning in Agents
original ↗
—
Computer Use and Browser Agents
original ↗
—
Multi-Agent Systems and Agent-to-Agent Protocols
original ↗
—