41554 blogs · [ { "id": "01a0dcd4-4f97-716a-b4f5-f59c1d931a5b", "title": "Continual learning might make your blocking monitors nearly useless", "url": "https://blog.redwoodresearch.org/p/continual-learning-might-make-your", "published_at": "2026-09-25T01:15:10+00:00" }, { "id": "01a0d013-2ad9-70ad-b8c7-97720d208fc0", "title": "Latent reasoning architectures would undermine CoT, our strongest oversight tool", "url": "https://blog.redwoodresearch.org/p/latent-reasoning-architectures-would", "published_at": "2026-09-23T18:05:28+00:00" }, { "id": "01a0d013-2ad9-70ad-b8c7-97720d926970", "title": "Astra is much better at reasoning with filler tokens than previous models", "url": "https://blog.redwoodresearch.org/p/astra-is-much-better-at-reasoning", "published_at": "2026-09-23T00:25:40+00:00" }, { "id": "01a0953b-d2da-715b-b45f-488bfcc7ef2d", "title": "CoT controllability evals seem very under-elicited", "url": "https://blog.redwoodresearch.org/p/cot-controllability-evals-seem-very", "published_at": "2026-09-11T17:12:25+00:00" }, { "id": "01a08cdc-d4c4-7142-91a9-0c2eb8568c86", "title": "An operationalization of opaque serial depth", "url": "https://blog.redwoodresearch.org/p/an-operationalization-of-opaque-serial", "published_at": "2026-09-10T18:40:20+00:00" }, { "id": "01a08cdc-d4c4-7142-91a9-0c2eb92938c0", "title": "Proposal for tracking the effects of architecture on monitorability", "url": "https://blog.redwoodresearch.org/p/proposal-for-tracking-the-effects", "published_at": "2026-09-10T18:13:53+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dc522a96b", "title": "Brief independent investigation of agents’ behavior, reasoning and collaboration in the OpenAI / Hugging Face hacking incident", "url": "https://blog.redwoodresearch.org/p/brief-independent-investigation-of", "published_at": "2026-08-27T01:32:18+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dc60959b8", "title": "AI swarms are starting to pose indirect takeover risk", "url": "https://blog.redwoodresearch.org/p/ai-swarms-are-starting-to-pose-indirect", "published_at": "2026-08-12T16:45:59+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dc61caca4", "title": "SOTA alignment assessments don’t strongly update us against misalignment", "url": "https://blog.redwoodresearch.org/p/sota-alignment-assessments-dont-strongly", "published_at": "2026-07-31T23:09:20+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dc672cf7f", "title": "Untrusted advice for AI control: Short, strong advice significantly uplifts weak LLMs", "url": "https://blog.redwoodresearch.org/p/untrusted-advice-for-ai-control-short", "published_at": "2026-07-27T23:59:30+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dc716572d", "title": "An OpenAI model left notes about how to evade containment", "url": "https://blog.redwoodresearch.org/p/an-openai-model-left-notes-about", "published_at": "2026-07-26T03:53:29+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dc7930e5e", "title": "The OpenAI models that hacked Hugging Face weren’t just following instructions", "url": "https://blog.redwoodresearch.org/p/the-openai-models-that-hacked-hugging", "published_at": "2026-07-25T21:37:05+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dc818986a", "title": "The OpenAI/Huggingface incident | Redwood Research podcast episode 2", "url": "https://blog.redwoodresearch.org/p/the-openaihuggingface-incident-redwood", "published_at": "2026-07-23T17:26:00+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dc90ef175", "title": "Are we existentially threatened by the type of AI misalignment seen in the OpenAI Hugging Face attack?", "url": "https://blog.redwoodresearch.org/p/are-we-existentially-threatened-by", "published_at": "2026-07-23T03:54:59+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dc9ae7f10", "title": "AI Futurism Reading List", "url": "https://blog.redwoodresearch.org/p/ai-futurism-reading-list", "published_at": "2026-07-02T18:38:41+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dca98eeee", "title": "The distillation double bind: Distilling misaligned models either transfers misalignment or it doesn't", "url": "https://blog.redwoodresearch.org/p/the-distillation-double-bind-distilling", "published_at": "2026-06-18T21:45:27+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dcada868f", "title": "Estimating No-CoT Task-Completion Time Horizons of Frontier AI Models", "url": "https://blog.redwoodresearch.org/p/estimating-no-cot-task-completion", "published_at": "2026-06-10T17:52:45+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dcb776053", "title": "Efficient tradeoffs and the safety-usefulness tradeoff model", "url": "https://blog.redwoodresearch.org/p/efficient-tradeoffs-and-the-safety", "published_at": "2026-06-08T20:29:47+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dcbc2fcdc", "title": "Retrying vs Resampling in AI Control", "url": "https://blog.redwoodresearch.org/p/retrying-vs-resampling-in-ai-control", "published_at": "2026-05-29T17:01:53+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dcca7528b", "title": "Advice for making robust-to-training model organisms", "url": "https://blog.redwoodresearch.org/p/advice-for-making-robust-to-training", "published_at": "2026-05-28T17:00:58+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dcd45352f", "title": "Full automation of AI R&D probably yields a large speed up even without a software-only singularity", "url": "https://blog.redwoodresearch.org/p/full-automation-of-ai-r-and-d-probably", "published_at": "2026-05-27T18:27:27+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dcdc7be73", "title": "Incriminating misaligned AI models via distillation", "url": "https://blog.redwoodresearch.org/p/incriminating-misaligned-ai-models", "published_at": "2026-05-18T17:20:03+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dcdf29655", "title": "Risk reports need to address deployment-time spread of misalignment", "url": "https://blog.redwoodresearch.org/p/risk-reports-need-to-address-deployment", "published_at": "2026-05-15T18:59:44+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dce7bc9e4", "title": "How useful is the information you get from working inside an AI company?", "url": "https://blog.redwoodresearch.org/p/how-useful-is-the-information-you", "published_at": "2026-05-11T15:35:09+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dcea42d32", "title": "A review of “Investigating the consequences of accidentally grading CoT during RL”", "url": "https://blog.redwoodresearch.org/p/openai-cot", "published_at": "2026-05-07T18:01:27+00:00" }, { "id": "01a07d5d-9926-72c1-bdf5-2e8dcf0ccabd", "title": "Risk from fitness-seeking AIs: mechanisms and mitigations", "url": "https://blog.redwoodresearch.org/p/risk-from-fitness-seeking-ais-mechanisms", "published_at": "2026-05-01T17:55:50+00:00" } ] posts Claim your blog
Back to Redwood Research
Blog · corpus.blog/blogs/redwoodresearch.org/posts

Redwood Research

redwoodresearch.org

2026