| { |
| "schema_version": 1, |
| "title": "Reproduction: Taming Aleatoric Impulse in Off-Policy Reinforcement Learning", |
| "emoji": "🎯", |
| "space_id": "Sor0ush/repro-taming-aleatoric-impulse-off-policy-rl", |
| "paper": { |
| "openreview_id": "2tcZ8Vl30Q", |
| "arxiv_id": null, |
| "title": "Taming the Aleatoric Impulse in Off-Policy Reinforcement Learning", |
| "icml_number": 21650, |
| "github": "https://github.com/yzy-yuzhouyang/DSAC-AID" |
| }, |
| "tags": [ |
| "icml2026-repro", |
| "paper-2tcZ8Vl30Q" |
| ], |
| "updated_at": "2026-07-18T15:06:27+00:00", |
| "root": { |
| "slug": "index", |
| "title": "Reproduction: Taming Aleatoric Impulse in Off-Policy Reinforcement Learning", |
| "file": "pages/index.md", |
| "children": [ |
| { |
| "slug": "executive-summary", |
| "title": "Executive summary", |
| "file": "pages/executive-summary/page.md", |
| "children": [] |
| }, |
| { |
| "slug": "claim-1-sota-on-gym-mujoco-and-dmc", |
| "title": "Claim 1: SOTA on Gym-MuJoCo and DMC", |
| "file": "pages/claim-1-sota-on-gym-mujoco-and-dmc/page.md", |
| "children": [] |
| }, |
| { |
| "slug": "claim-2-epistemic-vs-aleatoric-disentanglement", |
| "title": "Claim 2: Epistemic vs aleatoric disentanglement", |
| "file": "pages/claim-2-epistemic-vs-aleatoric-disentanglement/page.md", |
| "children": [] |
| }, |
| { |
| "slug": "claim-3-lcb-pessimism-and-ucb-exploration", |
| "title": "Claim 3: LCB pessimism and UCB exploration", |
| "file": "pages/claim-3-lcb-pessimism-and-ucb-exploration/page.md", |
| "children": [] |
| }, |
| { |
| "slug": "conclusion", |
| "title": "Conclusion", |
| "file": "pages/conclusion/page.md", |
| "children": [] |
| } |
| ] |
| }, |
| "agent_view_tokens": 2556, |
| "revision": "1784387187161979687" |
| } |