57,803 blogs ยท [ { "id": "01a08c90-5fd9-7017-85ad-6da911bea4e2", "title": "Building data curation classifiers with agents, SetFit and Jobs", "url": "https://danielvanstrien.xyz/posts/2026/agents-data-curation/", "published_at": "2026-09-10T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7cd3b52d4", "title": "Distilling Qwen3.8 with datatrove on Hugging Face Jobs", "url": "https://danielvanstrien.xyz/posts/2026/distilling-qwen38-datatrove-jobs/", "published_at": "2026-08-17T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7cdbf58b1", "title": "A model release became a training recipe the same day. I mostly watched.", "url": "https://danielvanstrien.xyz/posts/2026/agent-trained-classifier/", "published_at": "2026-07-28T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7ce8d9f61", "title": "Benchmarking OCR fairly is harder than it looks", "url": "https://danielvanstrien.xyz/posts/2026/ocr-old-scans/", "published_at": "2026-07-07T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7cf2a296f", "title": "Is fine-tuning small models worth it? Structured extraction from index cards", "url": "https://danielvanstrien.xyz/posts/2026/nuextract-index-cards/", "published_at": "2026-06-24T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7cf4e7148", "title": "Filtering 1 TB of FineWeb-Edu for $2.40 using Buckets and Jobs", "url": "https://danielvanstrien.xyz/posts/2026/buckets-data-backend/", "published_at": "2026-06-02T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7cff3ca88", "title": "How to turn catalogue card images into structured JSON with a 4B open model", "url": "https://danielvanstrien.xyz/posts/2026/structured-records-from-cards/", "published_at": "2026-05-21T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d0258ba8", "title": "All four agents tied at F1 ~0.95. The number was a mirage.", "url": "https://danielvanstrien.xyz/posts/2026/agent-race-traces/", "published_at": "2026-05-06T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d1211456", "title": "Two agents, one prompt", "url": "https://danielvanstrien.xyz/posts/2026/agent-race/", "published_at": "2026-05-01T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d1f48bfd", "title": "AI providers have millions of agent sessions. The first 1,589 are public.", "url": "https://danielvanstrien.xyz/posts/2026/agent-sentiment/", "published_at": "2026-04-20T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d2ee0a93", "title": "Re-OCR Your Digitised Collections for ~$0.002/Page", "url": "https://danielvanstrien.xyz/posts/2026/re-ocr-collections/", "published_at": "2026-02-19T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d30d6c21", "title": "Train on Massive Datasets Without Downloading with Hugging Face Streaming and Unsloth", "url": "https://danielvanstrien.xyz/posts/2026/hf-streaming-unsloth/train-massive-datasets-without-downloading.html", "published_at": "2026-01-07T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d3ddd455", "title": "Fine-tuning Vision-Language Models for Art History: Iconclass Classification with TRL and HF Jobs", "url": "https://danielvanstrien.xyz/posts/2025/iconclass-vlm-sft/trl-vlm-fine-tuning-iconclass.html", "published_at": "2025-09-04T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d3ff859a", "title": "Efficient batch inference for LLMs with vLLM + UV Scripts on HF Jobs", "url": "https://danielvanstrien.xyz/posts/2025/hf-jobs/vllm-batch-inference.html", "published_at": "2025-07-30T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d44b9442", "title": "Who Benefits? Rethinking Library Data in the Age of AI", "url": "https://danielvanstrien.xyz/posts/2025/who-benefits-rethinking-library-data-ai/libraries-as-training-data.html", "published_at": "2025-06-09T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d50d5786", "title": "Efficient Inference for ModernBERT Classifiers Using vLLM", "url": "https://danielvanstrien.xyz/posts/2025/vllm/modern_inference_modernbert.html", "published_at": "2025-04-24T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d50f6364", "title": "Using QwQ to generate a reasoning dataset for structured data extraction", "url": "https://danielvanstrien.xyz/posts/2025/reasoning-models/generating-structured-data-extraction-dataset-with-qwq-and-curator.html", "published_at": "2025-03-11T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d5a5f3bb", "title": "Log GRPO Completions to ๐Ÿค— Datasets", "url": "https://danielvanstrien.xyz/posts/2025/grpo/trl_log_completions_to_hf_datasets.html", "published_at": "2025-02-20T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d63a9f04", "title": "Distiling DeepSeek reasoning to ModernBERT classifiers", "url": "https://danielvanstrien.xyz/posts/2025/deepseek/distil-deepseek-modernbert.html", "published_at": "2025-01-29T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d648763f", "title": "Hygge Data - Cozy Content Filtering for a finer Scandinavian FineWeb", "url": "https://danielvanstrien.xyz/posts/2025/FineWeb-c/scandinavian-content-filtering-fineweb.html", "published_at": "2025-01-09T00:00:00+00:00" }, { "id": "01a08767-6850-70c3-bce8-d8f7d720b6a0", "title": "How Fine is FineWeb2?", "url": "https://danielvanstrien.xyz/posts/2025/FineWeb-c/how-fine-is-fineweb2.html", "published_at": "2025-01-07T00:00:00+00:00" } ] posts Claim your blog
Back to danielvanstrien.xyz
Blog ยท corpus.blog/blogs/danielvanstrien.xyz/posts

danielvanstrien.xyz

danielvanstrien.xyz

2026

2025