56,966 blogs · [ { "id": "01a0e330-5771-7380-a435-77df2db4b1f1", "title": "Continual learning might make your blocking monitors nearly useless", "url": "https://www.alignmentforum.org/posts/QnDqGbKehEB3DxJAp/continual-learning-might-make-your-blocking-monitors-nearly", "published_at": "2026-09-24T23:45:34+00:00" }, { "id": "01a0d51d-41c4-73b1-846f-80d29ec87048", "title": "Latent reasoning architectures would undermine CoT, our strongest oversight tool", "url": "https://www.alignmentforum.org/posts/6m29SfjbittooYojj/latent-reasoning-architectures-would-undermine-cot-our", "published_at": "2026-09-23T18:01:49+00:00" }, { "id": "01a0d51d-41c5-709f-ad03-4e968a5ebfcb", "title": "Why I'm scared of RL", "url": "https://www.alignmentforum.org/posts/LcQ9x72eNji2gpS9b/why-i-m-scared-of-rl", "published_at": "2026-09-23T12:07:33+00:00" }, { "id": "01a0d51d-41c5-709f-ad03-4e968a7df3ce", "title": "WorkspaceBench: Evaluating Interpretability Methods for the Global Workspace", "url": "https://www.alignmentforum.org/posts/Zeg2JztbdhguL48uH/workspacebench-evaluating-interpretability-methods-for-the", "published_at": "2026-09-23T06:58:59+00:00" }, { "id": "01a0d51d-41c5-709f-ad03-4e968b16b7a7", "title": "[Paper] Stringological sequence prediction III", "url": "https://www.alignmentforum.org/posts/virawHQekJsozwotz/paper-stringological-sequence-prediction-iii", "published_at": "2026-09-18T16:52:53+00:00" }, { "id": "01a0d51d-41c5-709f-ad03-4e968b7e896a", "title": "A Defense of Gradual Disempowerment", "url": "https://www.alignmentforum.org/posts/jXBmrEQj7zKGgiaYh/a-defense-of-gradual-disempowerment", "published_at": "2026-09-17T21:04:05+00:00" }, { "id": "01a0d51d-41c5-709f-ad03-4e968bc42301", "title": "Shallow Beliefs: Midtraining does not inoculate against EM from reward hacking", "url": "https://www.alignmentforum.org/posts/khxvR2fgAeDvG5N2F/shallow-beliefs-midtraining-does-not-inoculate-against-em", "published_at": "2026-09-15T21:50:08+00:00" }, { "id": "01a0d51d-41c5-709f-ad03-4e968bd6cd13", "title": "Op-Ed: I Worked at Google DeepMind. You Should Listen to the Warnings About AI", "url": "https://www.alignmentforum.org/posts/YGTWfyZb9oE5EQPu6/op-ed-i-worked-at-google-deepmind-you-should-listen-to-the", "published_at": "2026-09-14T14:53:57+00:00" }, { "id": "01a094c3-3e32-720d-819e-6167fd91171b", "title": "CoT controllability evals seem very under-elicited", "url": "https://www.alignmentforum.org/posts/BbP2wCyDGdPWJ7PwP/cot-controllability-evals-seem-very-under-elicited", "published_at": "2026-09-11T17:12:04+00:00" }, { "id": "01a08c6e-03f4-7052-9e6d-ea9ea4553771", "title": "An operationalization of opaque serial depth", "url": "https://www.alignmentforum.org/posts/x8BvtWxtoajBGHS3g/an-operationalization-of-opaque-serial-depth", "published_at": "2026-09-10T17:26:39+00:00" }, { "id": "01a08c6e-03f4-7052-9e6d-ea9ea4a851fe", "title": "Proposal for tracking the effects of architecture on monitorability", "url": "https://www.alignmentforum.org/posts/hLPGv8QjPcNLtDp3A/proposal-for-tracking-the-effects-of-architecture-on", "published_at": "2026-09-10T17:18:38+00:00" }, { "id": "01a08c6e-03f4-7052-9e6d-ea9ea4c2cba7", "title": "Astra can do a concerning amount with no chain of thought", "url": "https://www.alignmentforum.org/posts/eRmzz8J8Qkzqvzrgg/astra-can-do-a-concerning-amount-with-no-chain-of-thought", "published_at": "2026-09-10T02:31:26+00:00" }, { "id": "01a08c6e-03f4-7052-9e6d-ea9ea5404256", "title": "How good are slop-vestigators?", "url": "https://www.alignmentforum.org/posts/wt4kk6vFPEhkXvF8Q/how-good-are-slop-vestigators", "published_at": "2026-09-08T22:13:19+00:00" }, { "id": "01a08214-4b2a-73b8-b38f-3a3311ddb2d0", "title": "Training on probes: Research ideas", "url": "https://www.alignmentforum.org/posts/HGfMBEFnsXYkKvbRo/training-on-probes-research-ideas", "published_at": "2026-09-08T17:06:23+00:00" }, { "id": "01a08214-4b2a-73b8-b38f-3a3312acdadb", "title": "Training on probes: What's going on", "url": "https://www.alignmentforum.org/posts/gHFCgrvfxQtaEnJye/training-on-probes-what-s-going-on", "published_at": "2026-09-08T17:06:18+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f58a482f", "title": "The Alignment Journal: Organization, Personnel, and Scope", "url": "https://www.alignmentforum.org/posts/9vm2wtAtb34pEkjje/the-alignment-journal-organization-personnel-and-scope", "published_at": "2026-09-01T20:49:18+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f59bd683", "title": "Training a Misaligned Reward Seeker", "url": "https://www.alignmentforum.org/posts/J76LZCC55RdHeqEhz/training-a-misaligned-reward-seeker", "published_at": "2026-09-01T01:41:44+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f5a173c1", "title": "Value generalisation Theory of Change: putting it into practice", "url": "https://www.alignmentforum.org/posts/kjPEH4KGEHd8iQ4Hn/value-generalisation-theory-of-change-putting-it-into", "published_at": "2026-08-31T10:51:46+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f5a25ed5", "title": "Value generalisation Theory of Change: the theory behind the approach", "url": "https://www.alignmentforum.org/posts/f79SNtqFD7SqJY2vf/value-generalisation-theory-of-change-the-theory-behind-the", "published_at": "2026-08-28T13:30:17+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f5b8670f", "title": "Brief independent investigation of agents’ behavior, reasoning and collaboration in the OpenAI / Hugging Face hacking incident", "url": "https://www.alignmentforum.org/posts/nB8KKapnWGBXtKKiM/brief-independent-investigation-of-agents-behavior-reasoning", "published_at": "2026-08-26T19:40:21+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f6909332", "title": "Debate Training Reduces Reward Hacking in RLAIF", "url": "https://www.alignmentforum.org/posts/BB8o7b8A4Aykeksvw/debate-training-reduces-reward-hacking-in-rlaif", "published_at": "2026-08-19T12:17:57+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f6ee5457", "title": "Does DiffusionGemma do latent reasoning?", "url": "https://www.alignmentforum.org/posts/QBuJ3suRZxrrxSTtv/does-diffusiongemma-do-latent-reasoning", "published_at": "2026-08-16T04:22:18+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f7508eff", "title": "AI swarms are starting to pose indirect takeover risk", "url": "https://www.alignmentforum.org/posts/8oFYZdXkTaNGRtcn8/ai-swarms-are-starting-to-pose-indirect-takeover-risk", "published_at": "2026-08-12T05:05:30+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f7632ee5", "title": "An anytime algorithm for mixing the computable measures", "url": "https://www.alignmentforum.org/posts/MgYCraoxMwfWwgWa5/an-anytime-algorithm-for-mixing-the-computable-measures", "published_at": "2026-08-12T00:57:09+00:00" }, { "id": "01a07df0-028e-7109-a519-2e75f8238ed5", "title": "Misaligned AIs could use killer robots to take over", "url": "https://www.alignmentforum.org/posts/9jKhqmFjMzdAvHANr/misaligned-ais-could-use-killer-robots-to-take-over", "published_at": "2026-08-11T19:03:26+00:00" }, { "id": "01a0b653-4459-70c9-b2bf-ef2d392c2255", "title": "Circuit discovery through chain of thought using policy gradients", "url": "https://www.alignmentforum.org/posts/N4P9LSa6B2KyQtG5d/circuit-discovery-through-chain-of-thought-using-policy", "published_at": "2025-11-28T00:00:00+00:00" }, { "id": "01a0b653-4461-71a7-869a-3eaa6fe0f896", "title": "Meta-agentic Prisoner's Dilemmas", "url": "https://www.alignmentforum.org/posts/abYDvsqmJkchzSY8q/meta-agentic-prisoner-s-dilemmas", "published_at": "2025-11-05T00:00:00+00:00" } ] posts Claim your blog
Back to AI Alignment Forum
Blog · corpus.blog/blogs/alignmentforum.org/posts

AI Alignment Forum

alignmentforum.org

2026

2025