<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<meta name="robots" content="noindex, nofollow">
<title>ICML 2026 field planner — Z. Tavares</title>
<link rel="preconnect" href="https://fonts.googleapis.com">
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
<link href="https://fonts.googleapis.com/css2?family=Space+Grotesk:wght@400;500;600&family=IBM+Plex+Mono:wght@400;500&family=Inter:wght@400;500&display=swap" rel="stylesheet">
<style>
:root{
--paper:#FAFAF7; --ink:#1A1D21; --graphite:#5C6470; --faint:#8B9099;
--hairline:#E3E4DE; --hairline2:#CFD2C9;
--cellon:#DCE7F7; --cobalt:#2563C7;
--world:#6D5AE6; --robot:#0E8A6D; --bayes:#2563C7; --spark:#D4562B;
--team:#C43D6B; --wild:#C98A12; --meta:#7A8089;
}
*{box-sizing:border-box}
html{-webkit-text-size-adjust:100%}
body{margin:0;background:var(--paper);color:var(--ink);font:15px/1.65 Inter,system-ui,sans-serif;-webkit-font-smoothing:antialiased}
.wrap{max-width:760px;margin:0 auto;padding:36px 20px 64px}
.eyebrow{font:500 11px/1 "IBM Plex Mono",monospace;letter-spacing:.14em;text-transform:uppercase;color:var(--graphite)}
h1{font:600 27px/1.2 "Space Grotesk",sans-serif;margin:10px 0 4px;letter-spacing:-.01em}
.sub{color:var(--graphite);font-size:14px;margin:0 0 26px}
.card{border:1px solid var(--hairline2);border-left:3px solid var(--team);background:#fff;padding:16px 18px;margin:0 0 28px}
.card h2{font:600 16px/1.35 "Space Grotesk",sans-serif;margin:0 0 6px}
.card p{margin:0 0 10px;font-size:13.5px;color:var(--graphite)}
.lnk{font:500 12px/1 "IBM Plex Mono",monospace;color:var(--cobalt);text-decoration:none;border:1px solid var(--hairline2);border-radius:99px;padding:5px 11px;display:inline-block;margin:2px 6px 2px 0;background:#fff}
.lnk:hover{border-color:var(--cobalt)}
.tabs{display:flex;gap:6px;flex-wrap:wrap;margin:0 0 16px}
button{font:500 12px/1 "IBM Plex Mono",monospace;letter-spacing:.04em;background:#fff;border:1px solid var(--hairline2);border-radius:99px;padding:8px 14px;cursor:pointer;color:var(--ink)}
button:hover{border-color:var(--graphite)}
button.on{background:var(--ink);color:#fff;border-color:var(--ink)}
button:focus-visible,.row:focus-visible{outline:2px solid var(--cobalt);outline-offset:2px}
.ctl{display:flex;align-items:center;gap:8px;flex-wrap:wrap;margin:0 0 10px}
.ctl .lab{font:500 11px/1 "IBM Plex Mono",monospace;letter-spacing:.12em;text-transform:uppercase;color:var(--faint);margin-right:2px}
.ctl .sp{flex:1}
.strip{display:flex;border:1px solid var(--hairline2);overflow:hidden;touch-action:none;user-select:none;-webkit-user-select:none;background:#fff}
.cell{flex:1;height:38px;border-right:1px solid var(--hairline);cursor:pointer;background:#fff}
.cell:last-child{border-right:none}
.cell.hr{border-right-color:var(--hairline2)}
.cell.on{background:var(--cellon);box-shadow:inset 0 -3px 0 var(--cobalt)}
.ruler{position:relative;height:16px;margin:4px 0 6px;font:400 10.5px/1 "IBM Plex Mono",monospace;color:var(--faint)}
.ruler span{position:absolute;transform:translateX(-2px)}
.sum{font:400 13px/1.5 "IBM Plex Mono",monospace;color:var(--graphite);margin:14px 0 8px;display:flex;align-items:center;gap:12px;flex-wrap:wrap}
#ics{padding:6px 11px}
.mapwrap{margin:14px 0 2px;border:1px solid var(--hairline2);background:#fff;padding:10px 10px 4px;overflow-x:auto;-webkit-overflow-scrolling:touch}
#map{min-width:620px}
.maphead{font:500 10.5px/1 "IBM Plex Mono",monospace;letter-spacing:.12em;text-transform:uppercase;color:var(--faint);margin:0 0 8px}
#map{width:100%;height:auto;display:block}
#map .zone rect{fill:#FDFDFB;stroke:var(--hairline2)}
#map .zone text{font:500 9.5px "IBM Plex Mono",monospace;fill:var(--faint);letter-spacing:.03em}
#map .zone.active rect{fill:#EFF4FB;stroke:var(--cobalt)}
#map .zone.active text{fill:var(--ink)}
#map .zone.pulse rect{stroke:var(--ink);stroke-width:1.8}
#map .fl{font:600 11px "Space Grotesk",sans-serif;fill:var(--graphite)}
#map .swing{fill:#F0F1EB;stroke:var(--hairline)}
#map .mall{fill:#3A3F46}
#map .malltxt{font:500 8.5px "IBM Plex Mono",monospace;fill:#D8DAD2;letter-spacing:.08em}
#map .cmp{font:500 9px "IBM Plex Mono",monospace;fill:var(--faint);letter-spacing:.05em}
.sumtog{color:var(--cobalt);text-decoration:none;font:500 11px "IBM Plex Mono",monospace}
.sumtog:hover{text-decoration:underline}
.psum{margin:7px 0 12px;padding:11px 14px;border-left:2px solid var(--cobalt);background:#F7F8F3;font-size:12.5px;line-height:1.62;color:var(--ink)}
.psum p{margin:0 0 8px}
.psum p:last-child{margin-bottom:0}
.psum ol{margin:0 0 8px 17px;padding:0}
.psum li{margin:0 0 6px}
.psum b{font-weight:600}
.psec{font:500 10px/1.5 "IBM Plex Mono",monospace;letter-spacing:.13em;text-transform:uppercase;color:var(--faint);margin:9px 0 1px;padding-top:8px;border-top:1px dashed var(--hairline2)}
.bgrow{color:var(--faint)}
.legend{display:flex;gap:14px;flex-wrap:wrap;font-size:12px;color:var(--graphite);margin:0 0 6px}
.legend i{width:8px;height:8px;border-radius:50%;display:inline-block;margin-right:5px;font-style:normal}
.plan{border-top:1px solid var(--hairline2)}
.row{display:grid;grid-template-columns:96px 1fr;gap:14px;padding:15px 6px 15px 10px;border-bottom:1px solid var(--hairline);cursor:pointer;border-left:3px solid transparent}
.row:hover{background:#F2F3EC}
.row.pinned{border-left-color:var(--ink);background:#F5F6F0}
.row.skip{opacity:.62}
.time{font:500 12.5px/1.6 "IBM Plex Mono",monospace;color:var(--graphite);padding-top:2px}
.time small{display:block;color:var(--faint);font-size:10.5px}
.tt{font:500 15px/1.4 "Space Grotesk",sans-serif;letter-spacing:-.005em}
.meta1{font:500 10.5px/1.7 "IBM Plex Mono",monospace;letter-spacing:.1em;text-transform:uppercase;color:var(--faint);margin:3px 0 5px}
.meta1 b{font-weight:500}
.note{font-size:13.5px;color:var(--graphite)}
.papers{margin-top:8px;border-top:1px dashed var(--hairline2);padding-top:7px}
.papers div{font-size:12.5px;color:var(--graphite);padding:2.5px 0;line-height:1.55}
.papers a{color:var(--cobalt);text-decoration:none;font:500 11.5px "IBM Plex Mono",monospace}
.papers a:hover{text-decoration:underline}
.links{margin-top:8px}
.gap{display:grid;grid-template-columns:96px 1fr;gap:14px;padding:10px 6px 10px 13px;border-bottom:1px solid var(--hairline);color:var(--faint);font-size:13px}
.gap .time{color:var(--faint)}
.skiphead{font:500 11px/1 "IBM Plex Mono",monospace;letter-spacing:.12em;text-transform:uppercase;color:var(--faint);margin:26px 0 6px}
.foot{font-size:12.5px;color:var(--faint);margin-top:22px;border-top:1px solid var(--hairline2);padding-top:12px}
.dot{width:9px;height:9px;border-radius:50%;display:inline-block;margin-right:7px;vertical-align:1px}
@media(max-width:560px){
.wrap{padding:24px 14px 48px}
.row,.gap{grid-template-columns:70px 1fr;gap:10px}
h1{font-size:22px}
.cell{height:42px}
button{padding:10px 13px}
.tabs{gap:5px}
.time{font-size:11.5px}
.maphead{font-size:10px}
.psum{padding:10px 10px}
.card{padding:13px 14px}
}
</style>
</head>
<body>
<div class="wrap">
<div class="eyebrow">ICML 2026 · COEX, Seoul · Jul 7–11 · all times KST</div>
<h1>Field planner — Z. Tavares, Basis</h1>
<p class="sub">Paint the hours you will be on site. Talks and workshops need their full window; poster sessions flow around whatever is locked in. Tap any row to pin it.</p>
<div class="card">
<h2>Your poster — Benchmarking world-model learning with environment-level queries</h2>
<p>Warrier, Nguyen, Naim, Jain, Liang, Schroeder, Yang, Tenenbaum, Vollmer, Ellis, Tavares. <b>Tuesday Jul 7 · 14:00–15:45 KST · Hall A, board #4313 · Poster Session 2</b> — your arrival day; it is scheduled into Tuesday below. Bring the interactive Autumn demo.</p>
<a class="lnk" href="https://icml.cc/virtual/2026/poster/64404">ICML poster page</a><a class="lnk" href="https://arxiv.org/abs/2510.19788">arXiv 2510.19788</a><a class="lnk" href="https://www.basis.ai/projects/mara/">MARA project</a>
</div>
<div class="tabs" id="tabs"></div>
<div class="ctl">
<span class="lab">Presets</span>
<button data-pre="all">All day</button>
<button data-pre="am">Morning</button>
<button data-pre="pm">Afternoon</button>
<button data-pre="none">Clear</button>
<span class="sp"></span>
<span class="lab">Pace</span>
<span id="pace"></span>
</div>
<div class="strip" id="strip"></div>
<div class="ruler" id="ruler"></div>
<div class="mapwrap">
<div class="maphead">COEX cross-section, after the official building diagram — south (Samsung Stn) left, north (Bongeunsa Stn) right · zones light up with the day's plan · hover a row to pulse its room · Hall A is directly below Hall C</div>
<svg id="map" viewBox="0 0 760 246" role="img" aria-label="COEX building cross-section with conference rooms">
<text class="fl" x="752" y="48" text-anchor="end">4F</text>
<text class="fl" x="752" y="90" text-anchor="end">3F</text>
<text class="fl" x="752" y="129" text-anchor="end">2F</text>
<text class="fl" x="752" y="170" text-anchor="end">1F</text>
<text class="fl" x="752" y="208" text-anchor="end">B1</text>
<g class="zone" id="z-R4"><rect x="20" y="26" width="128" height="36" rx="2"/><text x="28" y="48">CR 400-403</text></g>
<g class="zone" id="z-R3"><rect x="20" y="66" width="128" height="40" rx="2"/><text x="28" y="90">CR 300-328</text></g>
<g class="zone" id="z-C"><rect x="148" y="26" width="176" height="80" rx="2" data-dy="34"/><text x="156" y="44">HALL C · PLENARY</text></g>
<g class="zone" id="z-E14"><rect x="158" y="74" width="76" height="28" rx="2"/><text x="165" y="91">E1-E4</text></g>
<g class="zone" id="z-E56"><rect x="240" y="74" width="76" height="28" rx="2"/><text x="247" y="91">E5-E6</text></g>
<rect class="swing" x="324" y="66" width="32" height="40"/>
<g class="zone" id="z-D1"><rect x="356" y="66" width="110" height="40" rx="2"/><text x="364" y="90">HALL D1</text></g>
<g class="zone" id="z-D2"><rect x="470" y="66" width="110" height="40" rx="2"/><text x="478" y="90">HALL D2</text></g>
<g class="zone" id="z-AUD"><rect x="584" y="66" width="156" height="40" rx="2"/><text x="592" y="90">AUDITORIUM</text></g>
<g class="zone" id="z-ASEM"><rect x="584" y="110" width="156" height="28" rx="2"/><text x="592" y="127">ASEM · CR-N 201-211</text></g>
<g class="zone" id="z-A"><rect x="148" y="142" width="176" height="48" rx="2"/><text x="156" y="168">HALL A — POSTERS</text></g>
<rect class="swing" x="324" y="142" width="32" height="48"/>
<g class="zone" id="z-B2"><rect x="356" y="142" width="224" height="48" rx="2"/><text x="364" y="168">HALL B1-B2</text></g>
<g class="zone" id="z-GB101"><rect x="584" y="142" width="62" height="48" rx="2"/><text x="590" y="168">GB 101-2</text></g>
<g class="zone" id="z-GB103"><rect x="650" y="142" width="30" height="48" rx="2"/><text x="654" y="168">103</text></g>
<g class="zone" id="z-GB104"><rect x="684" y="142" width="56" height="48" rx="2"/><text x="690" y="168">104-5</text></g>
<rect class="mall" x="20" y="196" width="720" height="16" rx="2"/>
<text class="malltxt" x="380" y="207" text-anchor="middle">STARFIELD COEX MALL — B1</text>
<text class="cmp" x="20" y="236">← SOUTH · SAMSUNG STN (LINE 2)</text>
<text class="cmp" x="740" y="236" text-anchor="end">NORTH · BONGEUNSA STN (LINE 9) →</text>
</svg>
</div>
<div class="sum"><span id="sumtext"></span><button id="ics" title="Download the selected stops for this day as calendar events">⤓ day .ics</button></div>
<div class="legend" id="legend"></div>
<div class="plan" id="plan"></div>
<div class="skiphead" id="skiphead" style="display:none">Did not make the cut — tap to pin</div>
<div id="skipped"></div>
<div class="foot">Fixed talks and workshops require their full window painted; poster sessions, the expo and dinner fit any free 45 minutes inside their span. Main-conference poster sessions are unthemed — just Poster Session 1–8 in Hall A; the sweep names here are our filters. Lines starting with a board number are confirmed posters at that session; lines under a "background reading" divider are context, not ICML papers. The ICML app stores the schedule offline — COEX wifi is unreliable in crowded halls. Keynotes are streamed to overflow rooms. Everything you set — painted hours, pins, pace, open summaries — is saved in this browser and survives refresh; on a new conference day the planner opens on that day but keeps your settings. <a href="#" style="color:var(--cobalt);text-decoration:none" onclick="try{localStorage.removeItem('icml26planner:v1')}catch(e){};location.reload();return false">Reset saved preferences</a>.</div>
</div>
<script>
const APP="https://icml.cc/virtual/2026/search";
const WS="https://icml.cc/virtual/2026/events/workshop";
const W=id=>"https://icml.cc/virtual/2026/workshop/"+id;
const PP=id=>"https://icml.cc/virtual/2026/poster/"+id;
const SESS=id=>"https://icml.cc/virtual/2026/session/"+id;
const INV="https://icml.cc/virtual/2026/eventlistwithbios/invited%20talk";
const ORAL="https://icml.cc/virtual/2026/events/oral";
const AX=id=>"https://arxiv.org/abs/"+id;
const CATS={
world:{l:"World models",c:"var(--world)"},
robot:{l:"Robots",c:"var(--robot)"},
bayes:{l:"Bayes + probprog",c:"var(--bayes)"},
spark:{l:"Curiosity + creativity",c:"var(--spark)"},
team:{l:"Basis + you",c:"var(--team)"},
wild:{l:"Wildcard",c:"var(--wild)"},
meta:{l:"General",c:"var(--meta)"}
};
const SUMS={
wc:'<p><b>What it is.</b> A training recipe, not a new model — the artifacts are the recipe plus post-trained checkpoints of WorldPlay, the group's own open-source interactive video world model. That model is a video diffusion policy π<sub>θ</sub>(x<sub>n</sub> | x<sub>1:n−1</sub>, a<sub>n</sub>, c): it generates the next clip x<sub>n</sub> from the clips so far, a camera action a<sub>n</sub> issued by the user or agent (8 discrete translate/rotate moves), and a scene prompt c. WorldCompass adds an RL stage whose reward asks directly: <i>did x<sub>n</sub> do what a<sub>n</sub> said?</i></p><p><b>Why it exists.</b> Pixel-loss pretraining gives a gradient toward plausible video, not toward obedience. Judged strictly — every fourth frame checked against the commanded action — pretrained WorldPlay obeys composite action sequences about <b>20%</b> of the time.</p><p><b>The recipe, in order of novelty.</b></p><ol><li><i>Clip-level rollout.</i> One reward for a whole generated video averages away which clips obeyed; trained that way, accuracy lands at <b>12 — below the 20 it started from</b>. So: generate one shared history x<sub>1:n−1</sub>, then G=16 candidates for the single clip x<sub>n</sub>, each rewarded. Per-clip credit, rollout cost O(N+G) instead of O(N·G), and less exposure bias, since the history is the model's own imperfect output. A curriculum walks n from 1 to 16 (≈256 frames).</li><li><i>A reward pair that polices itself.</i> r<sub>IF</sub>: a 3D pose estimator reads the camera trajectory back off the generated pixels and checks it against a<sub>n</sub>. r<sub>VQ</sub>: an aesthetic score (HPSv3). Either alone reward-hacks — r<sub>IF</sub> collapses the visuals, r<sub>VQ</sub> makes attractive video in which the camera stops moving. The pair is the regularizer.</li><li><i>An optimizer diagnosis.</i> GRPO-for-diffusion explores by re-noising the same seed, which varies texture but not the camera path — no exploration in the rewarded dimension. Negative-aware fine-tuning (DiffusionNFT) over different seeds fixes that; keeping only each group's top-3 and bottom-3 halves iteration cost at equal accuracy.</li></ol><p><b>Evidence.</b> Composite actions <b>20→55</b>; basic ~60→mid-70s; visual quality rises too. Holds on two backbones (HunyuanVideo-1.5-8B, Wan2.2-5B) out to 381 frames, for 3 days on 64 H20s and 4k unlabeled images. The residual failures are switching <i>latency</i>, not disobedience.</p><p><b>Why it matters for MARA.</b> This lineage's “world model” is controllable video — no persistent state, no objects, no beliefs. The telling part is the evaluation they were forced to build: correctness as interventional consistency — <i>did the world respond correctly to the action?</i> — verified by a second model inferring what happened from pixels alone. That is AutumnBench's question restated without an environment to query, and the sharpest pixel-space foil for environment-level evaluation in the program.</p>',
intr:'<p><b>What it is.</b> A theory result, not a system — Herrmann and Schmidhuber formalize “interestingness” as a prospective estimate of future compression progress and ask, with Kolmogorov complexity and algorithmic statistics, when that estimate is predictable from the past at all. The object of study is the complexity–runtime profile K<sub>t</sub>(x): the length of the shortest program that outputs the data within t steps. A drop in the curve is a compression breakthrough; an object is interesting to the extent the curve will keep dropping beyond the current compute budget.</p><p><b>Why it exists.</b> Every standard curiosity signal — prediction error, information gain, learning progress, compression progress — is post-hoc: the agent must spend the training compute before finding out whether the object was worth it. The generation phase of an open-ended system needs the forecast beforehand, and raw prediction error rediscovers the noisy-TV trap. So: under what conditions does past progress license a bet on future progress?</p><p><b>The results.</b></p><ol><li><i>Recency is the signal.</i> Under both the Length and the Algorithmic (Solomonoff) prior, the probability of a further drop vanishes <b>exponentially</b> with the stagnation length Δ — the gap since the last observed breakthrough. Data size and the number or magnitude of earlier drops contribute negligibly — a pointed contrast with empirical curricula (Oudeyer-line learning progress, MAGELLAN) that rank tasks by how big the recent progress was.</li><li><i>The prior sets the optimism.</i> For the same observed profile, the Algorithmic prior expects more than the Length prior: the ratio of drop probabilities grows <b>linearly</b> and expected further progress <b>quadratically</b> in the remaining complexity gap. The Speed prior is the most conservative — penalizing runtime amounts to assuming any fast win would already have been found.</li></ol><p><b>Evidence.</b> Exhaustive enumeration across three Turing-complete paradigms: ≈<b>120 million</b> 2-Tag systems, ≈<b>34 million</b> Rule 110 tapes, ≈<b>2.3 billion</b> Brainfuck programs, each run up to <b>100k</b> steps. The importance-weighted empirical profiles reproduce the exponential decay and the prior ordering; the Speed prior's pessimism softens at physically realizable runtimes — the theory proper lives in the busy-beaver regime, a caveat the paper owns.</p><p><b>Why it matters for MARA.</b> This is the cleanest current answer to the question MARA's active-experimentation loop keeps asking: which environment deserves the next experiment? The prescription — bet on whatever yielded the most recent “aha!”, with confidence decaying exponentially through stagnation — is content-agnostic, needs no learned novelty model, and drops straight into an intrinsic-reward slot. Its quarrel with magnitude-based learning-progress heuristics is testable in AutumnBench-style environments, where how much structure remains to be learned is known by construction.</p>',
mz:'<p><b>What it is.</b> A training recipe from Tianmin Shu's lab — self-supervised RL that turns a small multimodal LLM into a fast online Theory-of-Mind engine with zero mental-state annotations. At every timestep the trained model emits, in one forward pass, a set of goal/belief hypotheses with probabilities: a running posterior over what the observed human is trying to do, ready to drive a helper agent.</p><p><b>Why it exists.</b> The two existing ToM routes fail real-time assistance from opposite ends. Bayesian inverse planning (including LLM hybrids AutoToM and ThoughtTracing) is robust but searches a hypothesis space at every step; bare LLMs are fast but flip their goal predictions constantly — in the assistance experiments GPT-5.2 and Gemini-3-Flash deliver <b>0.0%</b> speedup because the helper keeps reversing direction. Supervised ToM needs ground-truth mental-state labels that real domains lack.</p><p><b>The recipe.</b> Treat actions as evidence, not prediction targets.</p><ol><li>The model proposes hypotheses m<sub>i</sub> with weights p<sub>i</sub>; the reward is an ELBO for the inverse-planning posterior — the p-weighted log-likelihood of the human's observed actions under each hypothesis, plus a prior term and an entropy bonus. GRPO closes the loop: Bayesian inverse planning gets amortized into a single pass.</li><li>The action likelihood comes from an external evaluator — a model-based planner in GridWorld, an LLM in the household domain — so the signal stays self-supervised end to end.</li><li>The ablation (household, Qwen3-4B, full speedup <b>19.1</b>) shows the uncertainty machinery is most of the value: remove the LLM prior → <b>17.0</b> (raw action likelihood is hackable — stuffing every object into the goal scores high); remove multiple hypotheses → <b>10.3</b> (premature commitment); remove the entropy bonus → <b>5.2</b> (mode collapse).</li></ol><p><b>Evidence.</b> QA accuracy <b>1.7–2.5×</b> the base models at negligible extra inference cost, beating AutoToM and ThoughtTracing on both accuracy and compute. Proactive assistance: <b>23.0/24.5%</b> speedup in GridWorld (Qwen3-VL-4B/8B) where no baseline exceeds 1.4; <b>19.1%</b> on Online Watch-And-Help with Qwen3-4B at 201 TFLOPs, above Gemini-3-Flash (<b>17.7</b>) and Qwen3-235B (<b>12.3</b> at 1102 TFLOPs). A 12-participant IRB study with real humans: <b>19.7%</b> ± 6.3 speedup, statistically indistinguishable from Gemini-3-Flash.</p><p><b>Why it matters for MARA.</b> This is amortized inference in the Tenenbaum lineage executed with 2026 tools: keep the generative model (inverse planning) as the scorer, train a neural proposal to carry the posterior, and keep explicit hypotheses with calibrated uncertainty because the ablation shows they <i>are</i> the performance. Nothing in the recipe is ToM-specific — reward hypotheses by how well a simulator says they explain observed behavior — so the same loop applies where MARA needs it: latent environment dynamics in place of latent goals, an Autumn interpreter in place of the planner. The prior-term ablation doubles as a compact argument that probabilistic structure, not a scalar score, is what keeps hypothesis generation honest.</p>',
vs:'<p><b>What it is.</b> A diagnosis plus a prompting fix, not a training method. The diagnosis: mode collapse in aligned LLMs is driven by the preference data itself — <i>typicality bias</i>, annotators' systematic preference for familiar, fluent text — so it persists even under a perfect reward model and optimizer. The fix, Verbalized Sampling: ask the model to verbalize a distribution (“generate 5 responses with their probabilities”) instead of asking for one response.</p><p><b>Why it exists.</b> Prior accounts blamed the algorithm — a single reward model, KL-regularized majority-amplification. Writing the reward as r = r<sub>true</sub> + α·log π<sub>ref</sub>(y|x) and fitting it on HelpSteer pairs matched for correctness yields <b>α > 0</b> under two different base models: raters favor what the base model already finds typical, independent of task utility. Through the RLHF closed form the optimum becomes π<sub>ref</sub>(y)<sup>1+α/β</sup>e<sup>r/β</sup> — so wherever many answers tie on true utility (jokes, stories, personas), typicality is the tiebreaker and the aligned model collapses onto the base model's mode.</p><p><b>The idea.</b> Different prompts collapse to different modes. An instance-level prompt collapses to the modal single answer; a list-level prompt to a uniform list; a distribution-level prompt has as its mode an approximation of the pretraining distribution itself. Asked for US state names with probabilities, Claude-4-Sonnet's verbalized distribution sits at <b>KL 0.12</b> from a RedPajama reference; direct prompting repeats California and Texas. Training-free, no logit access, orthogonal to temperature and decoding tricks, and diversity becomes tunable by writing a probability threshold into the prompt.</p><p><b>Evidence.</b> Creative-writing diversity <b>1.6–2.1×</b> direct prompting, human raters <b>+25.7%</b>, quality flat. Across Tulu-3 post-training checkpoints, direct prompting retains <b>23.8%</b> of base-model diversity after DPO; VS retains <b>66.8%</b>. GPT-4.1 with VS matches a fine-tuned Llama-3.1-8B at reproducing human donation distributions in PersuasionForGood. Fine-tuning on 1K VS-generated math problems yields <b>37.5%</b> average downstream accuracy versus <b>30.6</b> for direct-prompt data — the latter below the <b>32.8</b> no-SFT baseline. Larger models gain 1.5–2× more than small ones; precision, factuality, and safety hold.</p><p><b>Why it matters for MARA.</b> An agent that learns world models by active experimentation is only as good as the spread of hypotheses it can propose, and any LLM-based proposer inherits mode collapse from alignment. VS is the cheapest available correction: posterior-shaped hypothesis sets with probabilities attached, from any API model, no fine-tuning. The state-names result is the deeper point — the aligned model still contains the pretraining distribution and can report it on request — which upgrades an LLM from a single-answer oracle to a queryable approximate prior over hypotheses, exactly the role probabilistic-programming pipelines (and MindZero-style hypothesis proposers) need it to play.</p>',
gpfn:'<p><b>What it is.</b> A new model plus the prior that makes it work — the first public prior-data fitted network (PFN) for graph node-level prediction, from Yandex Research. GraphPFN takes the tabular foundation model LimiX, inserts an adjacency-masked sparse-attention message-passing adapter into every transformer block, and pretrains on synthetic attributed graphs. One forward pass then performs in-context node classification or regression on an unseen graph — by the PFN theorem, an approximation of the Bayesian posterior predictive under the pretraining prior.</p><p><b>Why it exists.</b> Graphs are not one domain, so graph foundation models face feature/target heterogeneity that text and vision never did. The current best answer (G2T-FM, TabGFM) flattens the graph into hand-crafted tabular features — neighborhood aggregates, Laplacian positional encodings — and lets a tabular FM do the rest, which caps what graph structure the model can exploit. GraphPFN replaces the hand-crafted features with learned message passing.</p><p><b>The recipe.</b></p><ol><li><i>A structure prior tuned to look real.</i> Single stochastic block models give clusters that are too clean, so several first-level SBM graphs are merged by bijection into a second-level SBM (rough, overlapping communities), then a modified Barabási–Albert process with randomized initial degrees grafts on the low-degree periphery. Hyperparameters are resampled per dataset; graph-tool generates multiple graphs per second per CPU core.</li><li><i>A graph-aware causal model for attributes.</i> The TabPFN-style random SCM is extended so each hidden neuron is independently MLP-type or GNN-type (per-dataset mixing probability), with optional LapPE inputs — dialing how strongly features and targets depend on topology, down to the pure tabular prior at zero.</li><li><i>Adapter-only training.</i> LimiX stays frozen; only the graph adapters train, on a PFN cross-entropy plus masked-graph-modeling loss, for <b>500,000</b> steps on 8 A100s (~6 days).</li></ol><p><b>Evidence.</b> On GraphLand plus classic datasets up to <b>50,000</b> nodes, in-context GraphPFN is competitive with extensively tuned GNNs and beats G2T-FM ICL on several datasets; fine-tuned, it is best on <b>7 of 10</b> datasets it fits in memory, by <b>>2</b> points on three. Swapping pretrained adapters for random-init ones and fine-tuning drops performance significantly — the prior, not the architecture, carries the gain. Failures track prior mismatch: hm-prices has <b>10,730,995</b> edges against a pretraining maximum of <b>194,425</b>, and city-roads-M is a traffic topology the prior never generates.</p><p><b>Why it matters for MARA.</b> This is amortized Bayesian inference at foundation-model scale, and the prior is a legible generative program: when the model fails, the authors debug the program (add geometric graphs, raise the edge cap), not the dataset. That is the probabilistic-programming thesis operating in public. The transferable move for MARA is the template itself — write a generative prior over environments the way GraphPFN writes one over attributed graphs, pretrain an in-context learner against it, and get a model whose inductive bias is inspectable and editable at the source. An Autumn-program prior would slot exactly where the SBM prior sits.</p>',
awm:'<p><b>What it is.</b> A synthesis pipeline plus a released resource, not a learned model — the “world model” in the title is <b>1,000</b> generated executable environments, each a SQLite database (state), a generated Python/MCP tool layer (actions, observations, transitions), and per-task verification code (reward). Snowflake open-sources both the pipeline and the environments: <b>35,062</b> tools (≈35 per environment) and <b>10,000</b> tasks.</p><p><b>Why it exists.</b> Agentic RL is environment-starved: τ-bench ships 3 environments, TheMCPCompany 5, and LLM-simulated environments hallucinate state transitions while costing an LLM call per step. DeepSeek and Qwen built environment synthesis in-house and released neither pipeline nor environments. AWM's premise: tool-use environments share one skeleton — stateful backend, tool interface, success criteria — so an LLM can generate each part the way software is built, from requirements down.</p><p><b>The pipeline.</b></p><ol><li><i>Requirements-first generation.</i> 100 seed domain names expand into 1,000 CRUD-heavy scenarios; ten tasks per scenario act as functional requirements that dictate the SQLite schema (18.5 tables avg), the sample data, and the minimal MCP toolset; the generated backend averages ~<b>1,985</b> lines of Python.</li><li><i>Execution-based self-correction.</i> Every component is run in isolation and errors fed back for regeneration, up to five retries — over <b>85%</b> stage success at <b>1.13</b> attempts on average, ≈$57 per 100 samples on GPT-5.</li><li><i>Code-augmented reward.</i> Verification code diffs the database before and after the rollout; an LLM judge folds those state signals in with the trajectory, because rigid code checks false-negative on environment imperfections. GRPO training adds a step-level format gate with early termination and truncates history identically at training and inference.</li></ol><p><b>Evidence.</b> Qwen3-4B/8B/14B trained on 526 synthetic environments, never on benchmark ones: BFCLv3 8B overall <b>53.83→65.94</b>; MCP-Universe best at every scale (8B <b>6.70→11.17</b>); τ-bench competitive with EnvScaler — which synthesizes from existing task sets and regresses on the other two benchmarks (<b>−8.93</b> BFCLv3 average). The ablations carry the argument: code-augmented reward beats code-only beats LLM-only at every scale (<b>64.50/55.66/51.92</b> BFCLv3 at 4B), and the environment count scales monotonically — 10 environments hurt, 100 help, 526 help more. Bugs exist in 74–83% of environments, but only <b>14%</b> of tasks are blocked (EnvScaler: ~50%) and RL-time environment error holds near 4%.</p><p><b>Why it matters for MARA.</b> The title concedes MARA's point. When this team needed a world model agents could be trained against at scale, they wrote programs — database plus code, transitions correct by construction — and the Simulator baseline with GPT-5 as transition function loses while costing more per step. Worlds-as-executable-programs is the Autumn premise at industrial scale, and their reward — did the database state change correctly? — is environment-level evaluation in AutumnBench's sense, applied to training rather than testing. The scaling curve hands MARA a citable number for why environment diversity, not trajectory count, is the axis that matters.</p>',
doubt:'<p><b>What it is.</b> A position paper by Wagner and Abend (Hebrew University): reading world probabilities off token logprobs is unsound, and the only theoretically defensible route is second-order prediction — the model states probabilities as part of its output.</p><p><b>The argument.</b> Language models stopped being distributions over strings and became response predictors with textual inputs and outputs. Distribution estimation and response prediction are different objects, and the successive training phases pull the output distribution toward distinct, mutually conflicting targets — so a logprob is not an event probability, and treating it as one invites systematic pitfalls. The paper ends with directions for making verbalized probabilities probabilistically sound; the abstract reports no experiments.</p><p><b>Why it matters for MARA.</b> The theory-side companion to Verbalized Sampling (#1608, poster session 1): both conclude the distribution must be verbalized, not decoded. For MARA it marks the boundary of LLM-as-belief — an aligned model's logprobs cannot carry the uncertainty a world-modeling agent needs, so beliefs belong in an explicit probabilistic representation with the LLM as proposer, the probabilistic-programming division of labor. Same session as the AutumnBench poster; made for that conversation.</p><p><i>Abstract-only summary — full text not found.</i></p>',
cjepa:'<p><b>What it is.</b> An object-centric world model from the LeCun–Balestriero orbit. C-JEPA extends masked joint-embedding prediction from image patches to object slots, and the training move is object-level masking: the model must infer a masked object's latent state from the other objects. The authors formalize this as a latent intervention with counterfactual-like effect — it closes off shortcut solutions and makes interaction reasoning mandatory rather than optional.</p><p><b>Evidence (per the abstract).</b> About <b>+20%</b> absolute on counterfactual visual question answering over the identical architecture without object-level masking; on agent control tasks, planning with <b>1%</b> of the latent input features a patch-based world model needs, at comparable performance; and a formal analysis that object-level masking induces a causal inductive bias via latent interventions.</p><p><b>Why it matters for MARA.</b> The JEPA lineage arriving at MARA's premises from inside the LeCun program: causal competence needs objects as the unit and interventions as the learning signal. What C-JEPA simulates by masking latents, an active agent performs for real — AutumnBench-style environments are where the gap between latent pseudo-interventions and actual experiments becomes measurable. And the 1% figure is a clean datum for the claim that object abstraction buys planning efficiency.</p><p><i>Abstract-only summary — full text not found.</i></p>',
lmd:'<p><b>What it is.</b> A reframing plus an algorithm from the Macke lab: automated discovery of mechanistic simulator models is recast as Bayesian inference — sampling from a posterior p(M | D) over executable model programs — and instantiated as ModelSMC, a Sequential Monte Carlo sampler whose particles are candidate models an LLM proposes and mutates.</p><p><b>Why it exists.</b> Existing LLM discovery loops (FunSearch, evolutionary agents, sequential refiners) are defined operationally — agent roles, prompts, MSE/MMD/Wasserstein selection scores — with no explicit target distribution. That hides failure modes, forecloses convergence analysis, and makes it impossible to say which components are necessary. Casting the workflow as inference over programs supplies one probabilistic objective for proposal, refinement, and selection at once.</p><p><b>The method.</b></p><ol><li><i>Particles are programs.</i> Each particle carries a model implementation, a weight, and accumulated feedback. SMC's time index is refinement iteration, not dynamics — resample, propagate, weight.</li><li><i>LLM as proposal kernel.</i> Propagation is a mixture: with cloning probability the particle is copied unchanged (free — no re-simulation), otherwise the LLM samples a new implementation conditioned on the particle's ancestry, its evaluation feedback, and a task prompt. This anneals in the <i>proposal</i> rather than the target — broad early exploration, later concentration — the informal analogue of likelihood tempering.</li><li><i>Weights are a real likelihood.</i> Each model is weighted by the marginal likelihood of the data, parameters integrated out via neural likelihood estimation. The likelihood surrogate is NPE-PFN — the pretrained tabular foundation model TabPFN as a training-free density estimator, so no per-iteration estimator training and non-differentiable models need no parameter optimization. A consistency theorem ports a standard SMC result to model space under idealized assumptions (exact likelihoods, prior-matching proposal).</li></ol><p><b>Evidence.</b> On SIR, Hodgkin-Huxley, and an R-coded kidney pharmacology simulator, ModelSMC matches FunSearch+ and a single-particle ablation on point metrics (HH log-likelihood <b>256.00</b> vs 254.58 / 243.27) — the claim is that probabilistic framing costs nothing quantitatively while adding a posterior. The posterior is the product: across HH runs it concentrates on adding an M-type slow potassium current paired with a second channel (I<sub>M</sub> first in <b>9 of 10</b> seeds over <b>1,445</b> particles), reads posterior mass as a structural-adequacy diagnostic, and exposes ion-channel non-identifiability a single best-fit model would hide. The ablation across six axes (MSE vs likelihood weights, 200–1000 sim budget, pool size, GPT-5-mini backbone) shows no single change degrades it significantly.</p><p><b>Why it matters for MARA.</b> This is the MARA thesis in a neighboring lab's vocabulary: models are programs, discovery is posterior inference over them, and the LLM is a proposal kernel — not the reasoner, not the judge. The parts are ones Basis already owns — SMC over program particles is PoE-World's and probabilistic-programming machinery; TabPFN-as-likelihood is the amortized-inference story from the tabular-FM thread (GraphPFN, BayesFlow) put to work as the weighting step. Its sharpest contribution to the program is the reframe of what a discovery system should output: not one best model but a posterior whose mass ranks hypotheses, flags non-identifiability, and diagnoses structural mismatch — exactly the belief object an actively experimenting agent must maintain and update.</p>',
a2r:'<p><b>What it is.</b> A benchmark-generation paradigm, not a fixed benchmark — an automated pipeline (generate, expand, evaluate, analyze) that produces ARC-style abstract-reasoning tasks whose correctness is formally verifiable, so the test set scales without manual annotation.</p><p><b>The idea.</b> LLMs generate tasks that demand genuine rule extraction, then reuse validated rules over fresh input spaces to expand variations. Generation risks hallucinated tasks; the fix is a theoretical result — programmatic verification via cycle consistency (checking that an inverse operation perfectly reverses the forward operation) guarantees a unique solution. That turns benchmark construction into a checkable property rather than a trust exercise.</p><p><b>Evidence.</b> On a representative subset, top LLMs reach <b>39.8%</b> against <b>68.5%</b> for humans; generated 3D tasks fall well short of 2D and 1D in complexity, exposing a high-dimensional blind spot; and, counterintuitively, inputs with higher information complexity can <i>simplify</i> the reasoning.</p><p><b>Why it matters for MARA.</b> A generator of verifiable abstraction tasks is the evaluation counterpart to AutumnBench — where AutumnBench asks environment-level questions, A2RBench manufactures unbounded rule-induction problems with a formal correctness guarantee, closing the memorization loophole that dogs static ARC. The cycle-consistency guarantee is the transferable idea: a self-checking generator is how open-ended curricula avoid drifting into unsolvable or degenerate tasks.</p><p><i>Abstract-only summary — full text not found.</i></p>',
swm:'<p><b>What it is.</b> A training recipe plus a benchmark (Jiaxuan You's lab): a “social world model” that treats collective belief as the state, a news event as an exogenous action, and predicts the next belief — an event-conditioned transition p<sub>θ</sub>(s<sub>t+1</sub> | s<sub>t</sub>, e). State is grounded in prediction-market prices; SWM-Bench is curated from 3k+ Polymarket and Kalshi markets (Dec 2022–Jan 2026), the first benchmark that turns latent public opinion into a high-fidelity time series.</p><p><b>Why it exists.</b> Modeling belief dynamics hits three walls: beliefs are hard to quantify (solved by using market prices as the observable), transitions are semantic not symbolic (solved by an LLM backbone as the transition engine), and there are no event→shift attribution labels (solved by posterior guidance). The last is the crux — which news item caused a given move is latent.</p><p><b>The recipe.</b></p><ol><li><i>Latent event attribution.</i> The driving event is a categorical latent z over the day's candidate news set plus a null event. A prior attributor a<sub>φ</sub> scores candidates from pre-shift context only; the world model f<sub>ψ</sub> predicts the belief change conditioned on the chosen event. A parameter-free null branch fixes the persistence (efficient-market) baseline, so f<sub>ψ</sub> models only abnormal, event-driven moves.</li><li><i>Posterior guidance.</i> Attribution is far easier in hindsight, so a frozen LLM that sees the realized outcome produces a sharply peaked posterior q; the two forward models are trained to match it — pseudo-labeling, no human annotation. Read together the objective is a negative ELBO with q frozen and the prior learned; the authors call it posterior-guided distillation, not variational inference, since the bound never tightens.</li><li><i>Two inference modes.</i> Forecasting marginalizes over the prior attributor (persistence baseline plus attribution-weighted expected move); simulation bypasses the attributor to condition on a real or hypothetical event — a “what-if” interventional engine.</li></ol><p><b>Evidence.</b> The 8B SWM leads directional accuracy on both platforms (Kalshi <b>0.845</b>, Polymarket <b>0.685</b>), a <b>4%</b> DA gain over GPT-5.5 on Kalshi while ingesting the same news, and dominates Kalshi across metrics; on Polymarket it keeps the DA lead but trails frontier LLMs on magnitude — traced to Polymarket's 10% news-attribution rate and 42% mean-reversion (reflexive, flow-driven moves that break the exogeneity assumption) versus Kalshi's 19%/17%. The oracle-posterior upper bound (Kalshi DA <b>0.894</b>) localizes the bottleneck: the world model simulates transitions well once the event is known; the binding constraint is the prior attributor isolating the right shock. A precision–coverage knob: a 397B posterior gives top-1 mass 0.787 but attributes only 11.9% of transitions, a 32B posterior is flatter (0.603) but covers 47.8%.</p><p><b>Why it matters for MARA.</b> A world model whose actions are exogenous events rather than an agent's own — the passive-observation limit of the framework MARA pushes toward active experimentation. The transferable machinery is the posterior-guidance trick: when you cannot label which cause drove an observed effect, let a hindsight model that sees the outcome supply the attribution and distill it into a forward model that must act before the outcome — the same amortized inverse-inference move as MindZero and the ModelSMC discovery loop, here applied to causal attribution over a noisy event stream. The oracle-vs-prior gap is a clean statement of where such systems break: not the dynamics model, but proposing the right causal hypothesis under uncertainty.</p>',
ddojo:'<p><b>What it is.</b> A foundation world model for dexterous robot manipulation plus the dataset that trains it. DreamDojo is an action-conditioned video diffusion model (built on Cosmos-Predict2.5, 2B and 14B) that predicts future frames given a robot action; the headline asset is DreamDojo-HV, <b>44k hours</b> of egocentric human video — the largest corpus assembled for world-model pretraining, ~15× longer, 96× more skills, and 2,000× more scenes than the previous largest.</p><p><b>Why it exists.</b> Robot video world models plateau at discrete controls and stay confined to their training distribution, unresponsive to counterfactual actions, because teleoperated robot data is expensive and narrow and existing datasets are all expert demos with no intent stochasticity. Human video is cheap and vast, and the physics of contact transfers across the embodiment gap — but human video has no robot action labels.</p><p><b>The recipe.</b></p><ol><li><i>Latent actions as a universal label.</i> A VAE with an information bottleneck reads two consecutive frames and disentangles a 32-d continuous latent that captures the action between them, self-supervised, consistent across embodiments. This proxy label lets unlabeled human video teach action-conditioned dynamics — the ablation shows it beats action-free pretraining and matches retargeted/MANO ground-truth actions on simulation quality while being the only scalable option.</li><li><i>Causality-respecting conditioning.</i> Actions are rebaselined to relative form and injected as per-latent-frame chunks (4 actions per latent frame) rather than a global condition, so future actions cannot leak into present predictions; a temporal-consistency loss supplements per-frame flow matching.</li><li><i>Three phases.</i> Pretrain on human video (256 H100s), post-train on small target-robot data (resetting only the action layer — strong pretraining lets the target set stay small), then a Self-Forcing distillation to a 4-step causal-attention autoregressive student.</li></ol><p><b>Evidence.</b> Six OOD benchmarks (including counterfactual actions like reaching-and-missing) built on the Fourier GR-1; latent-action conditioning wins across PSNR/SSIM/LPIPS. The distilled student runs at real-time <b>10.81 FPS</b> (~4× the teacher's 2.72) for 1-minute rollouts with minor degradation. Policy evaluation is the standout: DreamDojo's simulated success rate tracks real-world AgiBot fruit-packing at Pearson <b>0.995</b>, rank violation MMRV <b>0.003</b>. Model-based planning (ensemble action proposals scored by a value model) lifts success <b>17%</b> over the best checkpoint and nearly <b>2×</b> over uniform sampling; live teleoperation of a virtual G1 runs real-time on a single RTX 5090. Limitation the authors flag: simulated success rates run optimistically high — the model under-generates nuanced failures.</p><p><b>Why it matters for MARA.</b> The pixel-space, video-diffusion route to a world model — controllable video with no persistent object state or beliefs, the same lineage as WorldCompass (#108) and WorldPlay. Two things earn MARA's attention: latent actions are an unsupervised inverse-dynamics move that recovers an action variable from raw observation, the concrete mechanism a curiosity-driven agent needs to act in environments it was not given a controller for; and the 0.995 policy-evaluation correlation is the strongest current evidence that a learned world model can stand in for real-robot rollouts — precisely the substitution the Droid stack would want, and precisely where AutumnBench-style environment-level probing would test whether the simulator's failure-mode blind spot (optimistic success rates) invalidates it as an evaluator.</p>',
live:'<p><b>What it is.</b> A one-trick training objective for autoregressive video world models: a cycle-consistency loss that bounds long-horizon error accumulation without a teacher model. LIVE rolls the model forward from ground-truth frames, then reverses the actions/camera and makes the model generate back to the original start, computing the diffusion loss on that recovery.</p><p><b>Why it exists.</b> Autoregressive video models drift: trained on ground-truth context but forced to condition on their own imperfect predictions at inference, small errors compound and quality collapses past the training horizon (exposure bias). Diffusion Forcing injects noise into context but noised ground truth still differs from real rollouts; Self-Forcing distills from a pretrained bidirectional teacher, which is expensive, caps the student at the teacher's capacity, induces mode-seeking, and still permits unbounded drift beyond the training length.</p><p><b>The idea.</b> A forward rollout produces semantically diverse futures that cannot be supervised against ground truth directly — two valid futures, no pixel loss between them. The fix is recoverability: require the model to reverse-generate the original prompt frames from its own imperfect rollout. Because forward quality degrades monotonically, reversal makes context quality improve monotonically — a shortcut where the model just attends to the best frame — so per-frame random-timestep noise is injected to force robust recovery. Minimizing recovery error implicitly caps forward distortion. Teacher Forcing and Diffusion Forcing fall out as special cases by varying the ground-truth-to-rollout ratio r, giving a progressive curriculum that lowers r during post-training.</p><p><b>Evidence.</b> FID holds near <b>10</b> across 32–200 frames while TF, DF, DFoT, and Geometry Forcing degrade sharply past 64. On RealEstate10K at <b>256 frames</b>, PSNR <b>13.89</b> vs the best real-time baseline's 11.51, with matching LPIPS/SSIM gains that widen with horizon; consistent wins on UE-Engine and Minecraft out to 400 frames, all real-time. The ablation confirms both moving parts: random per-frame noise beats fixed/none, and the progressive r-curriculum beats jumping straight to full rollout. 774M NFD backbone, 32 H100s, no teacher.</p><p><b>Why it matters for MARA.</b> The same cycle-consistency principle A2RBench uses for verifiable task generation, here turned on temporal drift — invertibility as a free supervision signal when forward outputs are unlabelable. For a world model an agent plans in, bounded error over long horizons is the difference between a usable rollout and a hallucination, the exact failure DreamDojo's optimistic success rates hint at. And eliminating the teacher matters for MARA's open-ended setting: a self-supervised objective that needs no pretrained oracle is what lets a world model keep improving in novel environments where no teacher exists.</p>',
dcr:'<p><b>What it is.</b> A benchmark (Jakob Foerster's lab, Meta) for multi-agent reasoning and theory of mind, built on the board game Decrypto: three roles — Alice (encoder) and Bob (decoder) share four secret keywords; Alice sends word-association hints for a 3-digit code that Bob must decode but Eve (interceptor) must not. It is the first platform for interactive ToM experiments — new experiments need only a prompt and a few lines of code.</p><p><b>Why it exists.</b> Existing ToM benchmarks are narrow (mostly Sally-Anne variants), leak into training data, saturate, and are non-interactive; embodied-scenario-to-text translation injects pragmatic artifacts. Decrypto is deliberately easy on every axis except multi-agent reasoning — no math, spatial, symbolic, tool-use, or tokenization confounds, just word associations — so what it measures is ToM. Formally it is a Rational Speech Act pragmatic-inference game: optimal play requires Bob to do second-order ToM (model Alice's beliefs about Eve's beliefs), shown in the appendix.</p><p><b>Evidence.</b> LLMs lag both humans and simple word-embedding (GloVe/Word2Vec) baselines: novice humans reach a <b>~90%</b> win rate against even the strongest LLM Eve, and the embedding baselines outperform all LLM agents in cooperation and competition. The diagnostic finding is an inverse-scaling result: on variants of two classic cognitive-science experiments (Smarties task for representational change and false belief; Three-Mountain for perspective taking), <b>Llama 3.1-70B beats newer reasoning models — Claude 3.7 with extended thinking and o1-high — on all three ToM abilities</b>. Every model scores near <b>0%</b> on the strong (self-consistency) variants even at temperature 0, and in perspective-taking all models except Llama predict Eve intercepts nearly every turn when the true rate is far lower — assuming Eve shares their privileged keyword knowledge even on turn 1, and persisting after the prompt is edited to stress that Eve does NOT know the keywords. A Hessian-determinant metric captures game-outcome sensitivity to player/prompt swaps; humans score much higher.</p><p><b>Why it matters for MARA.</b> Theory of mind is world-modeling pointed at other agents — inferring hidden mental state from behavior — and Decrypto is the interactive, leakage-resistant, non-saturating counterpart to AutumnBench in the social domain, with humans as the standing baseline models must reach. The inverse-scaling result is the sharp datum for the program's thesis: more RL-trained reasoning does not buy the perspective-taking that active social experimentation demands, and can erode it — the same skill MindZero (#412) trains directly rather than hopes emerges. Its cleanest transfer is methodological: porting genuine cognitive-science paradigms (false belief, perspective taking) into an interactive environment is exactly how MARA should be probing world models rather than scoring static QA.</p>',
tstr:'<p><b>What it is.</b> A representation-learning trick for JEPA world models, from Mengye Ren's and LeCun's groups: a curvature regularizer that straightens latent trajectories so gradient-based planning works. One idea, one loss term.</p><p><b>Why it exists.</b> Latent world models plan by rolling the model forward and minimizing a cost between predicted and goal states, usually Euclidean distance in the latent space. But pretrained encoders like DINOv2 carry planning-irrelevant detail and bend trajectories, so straight-line latent distance misrepresents geodesic (feasible-transition) distance and the planning objective turns non-convex — which is why the field falls back on expensive search planners (CEM, MPPI) instead of exploiting the model's differentiability.</p><p><b>The idea.</b> Borrowing the perceptual-straightening hypothesis from human vision (Hénaff et al.), jointly train a JEPA encoder and predictor while penalizing the curvature of encoded trajectories — maximizing the cosine similarity between consecutive latent velocity vectors. The JEPA prediction loss already straightens implicitly; the explicit regularizer strengthens it. Stop-gradient on the target branch prevents collapse; an aggregation head applies straightening to spatial features without over-flattening them. A theorem on linear latent dynamics shows an ε-straight transition controls the planning Hessian's condition number, so gradient descent converges faster.</p><p><b>Evidence.</b> On Wall, PointMaze (UMaze/Medium), and PushT, straightening lifts open-loop gradient-planning success <b>20–60%</b> and MPC <b>20–30%</b>. The starkest case: a ResNet trained from scratch on Wall goes from <b>4.67%→84.00%</b> open-loop and <b>10%→100%</b> MPC once straightening is added. After straightening, latent Euclidean distance to a goal tracks the A*-computed geodesic step count, and the action-space loss landscape visibly flattens toward convex.</p><p><b>Why it matters for MARA.</b> The geometry of the latent space is what makes a world model plannable — a JEPA-side complement to LIVE (bounding error over time) and DreamDojo (learning the dynamics at scale): here the payoff is that planning becomes cheap gradient descent instead of sampling-based search, which matters for any agent that must plan many experiments per step. For MARA's active-experimentation loop, a latent metric that faithfully reflects reachability is exactly what turns curiosity — “which state is far and worth reaching?” — into a well-posed optimization rather than a search. Same LeCun-lab JEPA lineage as Causal-JEPA (#1008), attacking the planning-usability of the representation rather than its causal content.</p>',
rpe:'<p><b>What it is.</b> A world model built to be an evaluator, not a planner — dWorldEval uses a discrete diffusion world model as a scalable proxy for real-robot policy evaluation. Vision, language, robot actions, and a task-progress signal are all mapped into one token space and denoised by a single transformer, trained from scratch on robot data rather than adapted from a video backbone.</p><p><b>Why it exists.</b> Evaluating policies across thousands of tasks by real execution is infeasible, and prior world-model evaluators (WorldEval, Ctrl-World, WorldGym) ride on pretrained video generators whose action controllability is too weak — the rollout drifts or ignores the action, severing the link between imagined and real success.</p><p><b>The design.</b> Three pieces: a unified discrete token space that makes actions first-class citizens rather than a conditioning afterthought; a sparse keyframe memory that bounds long-horizon drift (ablations show errors accumulate past 20 steps without it, and drift manufactures false negatives); and a <i>progress token</i> predicted jointly with future frames — success is declared automatically when progress hits 1, removing the human or VLM judge from the loop. An action-sensitive Δ-LPIPS metric diagnoses controllability, and a corruption study shows the causal chain: randomly swapping action tokens degrades Δ-LPIPS and collapses ranking correlation in tandem.</p><p><b>Evidence.</b> Imagined success rates track real execution at Pearson <b>r ≈ 0.9</b> across the board — LIBERO multi-view <b>0.910</b>, RoboTwin <b>0.927</b>, real-world tasks <b>0.918</b> — with rank violation MMRV <b>0.013</b> against baselines up to 0.039.</p><p><b>Why it matters for MARA.</b> Same claim as DreamDojo's policy-evaluation result (#3815, session 5), reached from the opposite architecture — discrete tokens from scratch versus a distilled video-diffusion giant — and the two agree that action faithfulness, not visual fidelity, is what makes a world model a valid evaluator. The progress token is the interesting move for AutumnBench thinking: it builds the evaluation query into the model's state rather than bolting a judge on afterwards. For the Droid stack, r≈0.9 policy ranking without real rollouts is the practical takeaway worth interrogating at the board.</p>',
mv4d:'<p><b>What it is.</b> A 4D world model for manipulation plus a fix for the inverse-dynamics step. From one single-view RGBD observation, MVISTA-4D imagines geometrically consistent RGBD video from arbitrary <i>other</i> viewpoints; back-projecting and fusing the imagined views assembles a more complete 3D structure over time — imagine-then-act with geometry instead of pixels.</p><p><b>Why it exists.</b> Image-space world models produce futures that look plausible while violating 3D constraints — fatal for contact-rich manipulation — while point-cloud dynamics models are geometric but sparse in appearance and semantics. Single-view RGBD models stay brittle under occlusion, with monocular depth drifting in scale and time. And downstream, plain inverse dynamics is ill-posed: many actions explain the same imagined transition.</p><p><b>The design.</b><ol><li><i>Cross-view and cross-modality fusion.</i> Epipolar-guided attention with learned deformable offsets enforces geometric alignment across generated views while RGB and depth exchange features — naive depth-channel concatenation performs worse because the pretrained video backbone expects 3-channel RGB.</li><li><i>Test-time action optimization.</i> The whole action trajectory is compressed to a latent code (TCN-VAE); given a predicted 4D future, the trajectory latent is optimized by backpropagating through the generative model, then a residual inverse-dynamics model predicts only a correction to that strong prior. Ablations knock out each piece — skipping latent optimization (Act-Head), the full-IDM route, and dropping the residual IDM all lose on RLBench.</li></ol></p><p><b>Evidence.</b> On RLBench, RoboTwin, and a self-collected 14-task real-robot 4D multiview dataset, the model beats UniPi, 4DGen (DUSt3R-style, capped at two views), TesserAct, and a point-based ACT on success rate over 100 episodes per task, and improves depth/point-cloud consistency metrics; better geometry also yields better appearance. Real-robot results again top TesserAct.</p><p><b>Why it matters for MARA.</b> A concrete argument that the representation a manipulation world model should imagine is 3D structure, not pixels — a step toward the object-and-geometry abstractions MARA holds necessary, from within the generative-video paradigm. The test-time latent optimization is the notable pattern: inference over a compact latent cause by inverting a generative model — amortized-inference thinking applied to action selection — and directly relevant to how the Droid stack might turn imagined futures into executable trajectories.</p>',
fuse:'<p><b>What it is.</b> A simulation-based inference method: FUSE pairs a multimodal flow-matching posterior estimator with Feynman–Kac steered sampling at inference. The architecture is a dual-track MM-DiT — parameters and observations keep separate token streams with joint attention rather than being crushed into one static embedding — and the sampler runs multiple generative trajectories in parallel, reweighting and resampling them mid-flow using simulator likelihoods.</p><p><b>Why it exists.</b> The SBI trade-off: MCMC is asymptotically exact but can take weeks per event; amortized neural posteriors (FMPE-style) are fast but fuse parameters and observations by brute force and leak probability mass into physically implausible regions when the learned transport is imperfect. FUSE attacks both ends — a fusion architecture that respects the structural disparity of the modalities, and a likelihood-based correction that spends test-time compute steering the amortized sampler back toward the true posterior.</p><p><b>The idea in one line.</b> FK steering is sequential Monte Carlo grafted onto a flow: propagate a population of trajectories, score them at intermediate times with the simulator likelihood and prior density, resample toward high-density regions — asymptotically-motivated correction at modest overhead on top of the amortized draw.</p><p><b>Evidence.</b> On the standard Lueckmann SBI benchmark suite, FUSE's posteriors match reference MCMC samples more closely (C2ST) than state-of-the-art neural baselines across tasks; on SLCP, switching FK steering on yields visibly more compact posteriors with higher mean and peak likelihoods; on real exoplanet orbital estimation it resolves parameter degeneracies where existing SBI methods fail to produce a meaningful posterior.</p><p><b>Why it matters for MARA.</b> The amortize-then-correct pattern in its cleanest current form: a learned proposal carries you fast, and a particle method with simulator-likelihood weights makes you honest — the same division of labor as ModelSMC (#3512) at the model level, here at the parameter level, and standard probabilistic-programming machinery (SMC, importance resampling) turning up inside a flow model. For any MARA pipeline that fits mechanistic simulators to data, this is the reference design for making fast neural posteriors trustworthy.</p>',
labo:'<p><b>What it is.</b> A Bayesian-optimization framework that treats the LLM as a cheap second fidelity, not as an oracle. LABO runs a dual-fidelity BO loop: LLM predictions explore the search space at negligible cost, real experiments are spent only where they are needed, and a gating rule decides which fidelity each candidate gets.</p><p><b>Why it exists.</b> Prior LLM+BO systems (LLAMBO, BOPRO, ReasoningBO, CAKE) embed the LLM in initialization, proposals, or the kernel — but never exploit the fact that an LLM evaluation costs orders of magnitude less than a wet-lab experiment. That cost asymmetry is the whole opportunity in formulation science, where each real evaluation is an expensive, slow experiment.</p><p><b>The design.</b> A Kennedy–O'Hagan joint Gaussian process decomposes the real objective into a scaled LLM-fidelity GP plus a discrepancy GP that models where the LLM is systematically wrong. The gating criterion is uncertainty-based: a candidate triggers a real experiment only when the discrepancy term dominates the predictive variance — that is, when the LLM cannot reduce uncertainty at that location. Warm-start uses LLM reasoning over domain priors for the first real measurements plus Latin-hypercube LLM-fidelity coverage of the space. The theory delivers a cumulative regret bound: sample-efficiency gains when LLM guidance is informative, and graceful degradation to vanilla-BO behavior when it is noisy or misaligned; the LLM-vs-real query partition provably stabilizes after finitely many steps.</p><p><b>Evidence.</b> Across diverse scientific formulation tasks, LABO finds better candidates than existing LLM-BO methods under identical real-experiment budgets when LLM predictions help, and matches vanilla BO when they do not — the robustness half of the claim is as load-bearing as the win.</p><p><b>Why it matters for MARA.</b> This is experiment selection — the core decision of an actively experimenting agent — given a clean probabilistic answer: query the world only where your model's uncertainty is dominated by what the cheap proposer cannot tell you. The architecture is MARA's preferred division of labor made precise: LLM as prior-laden proposer, GP posterior as the belief object, discrepancy modeling as the guard against trusting the proposer. Directly applicable to any Basis pipeline where simulator or LLM surrogates could stand in for costly real trials.</p>',
emw:'<p><b>What it is.</b> A “world model” for dynamic facial-expression recognition (DFER): EmWorld reframes scenario-incremental DFER as progressive Bayesian inference over latent states at two timescales — a slow component tracks scenario evolution with stochastic evolutionary priors, a fast component models frame-level expression dynamics, and joint inference decouples the expression signal from scenario drift.</p><p><b>Why it exists.</b> Deployed DFER models degrade as scenarios shift; prior fixes (feature alignment, domain-incremental learning) preserve old representations but never model the shift itself. EmWorld's pitch is turning passive feature discrimination into active probabilistic state inference.</p><p><b>Evidence.</b> Up to <b>3.84%</b> improvement over state of the art on FERV39k, DFEW, and MAFW, with claimed cross-scenario stability and long-term robustness.</p><p><b>Why it matters for MARA.</b> Marginal fit — the interest is terminological drift: “world model” here means a structured latent state-space model with Bayesian inference, applied to perception. The two-timescale latent decomposition (slow context, fast dynamics) is a reusable structure for environments whose rules shift mid-episode, AutumnBench's home territory. A quick walk-by, not a stop.</p><p><i>Abstract-only summary — full text not found.</i></p>',
wpl:'<p><b>What it is.</b> Tencent Hunyuan's streaming interactive video world model — the base model WorldCompass (#108, session 1) post-trains — built to hold both ends of the speed–memory trade-off: real-time <b>720p at 24 FPS</b> under live keyboard/mouse control, with scenes that stay geometrically coherent when the user walks back.</p><p><b>Why it exists.</b> Interactive video models split into two camps: distilled-for-speed (Oasis, Genie-line) forget the scene — revisit a room and it has changed — while memory-equipped models keep consistency but resist distillation, so they are slow. WorldPlay claims both at once.</p><p><b>The three ingredients.</b><ol><li><i>Dual action representation.</i> Discrete keys (WASD) are scale-adaptive but ambiguous for memory retrieval; continuous camera poses give exact locations but destabilize training under scene-scale variance. WorldPlay conditions on both — keys for control, poses for accurate location caching.</li><li><i>Reconstituted context memory.</i> Rather than passive retrieval, the context set is rebuilt each chunk by querying past frames on spatial and temporal proximity, then <i>temporal reframing</i> rewrites the retrieved frames' positional embeddings, pulling geometrically important but long-past frames close in time so transformer long-range decay cannot wash them out.</li><li><i>Context forcing.</i> Standard distillation trains a memory-aware autoregressive student against a memory-less bidirectional teacher — a distribution mismatch that erodes the student's memory. Aligning the memory context between teacher and student during distillation preserves long-range use at real-time speeds while damping error accumulation.</li></ol></p><p><b>Evidence.</b> Long-horizon streaming generation with revisit consistency across first- and third-person real and stylized scenes; supports 3D scene reconstruction from rollouts and text-triggered world events. Trained on a curated 320K-video real+synthetic corpus. The paper's comparison table positions it as the only general-domain system with long-horizon generation, flexible action control, real-time interactivity, and long-term geometric consistency simultaneously.</p><p><b>Why it matters for MARA.</b> The strongest current pixel-space answer to the question MARA poses structurally: how does a world model remember? Here memory is retrieved frames with rewritten timestamps — impressive engineering that still stores appearance, not state; nothing persists unless it was seen. Read together with WorldCompass's finding that this same model obeys only ~20% of composite actions before RL post-training, it is the clearest picture of what scale buys and what it does not — persistent object state, the thing an Autumn program gets by construction. Worth the board visit as the production-grade face of the video-world-model program.</p>',
nomad:'<p><b>What it is.</b> A lifelong trajectory-planning framework for autonomous driving: NOMAD couples a non-parametric Bayesian memory with diffusion-based trajectory generation so the planner keeps adapting to rare long-tail scenarios without forgetting how to drive.</p><p><b>The idea.</b> Continuous scene contexts are mapped to a dynamically growing set of discrete memory clusters (the non-parametric part — capacity grows with the data, no fixed number of modes); the clusters condition a diffusion model that behaves as a mixture of experts over driving behaviors. Forgetting is countered with generative replay — pseudo-experiences synthesized from previously learned clusters stand in for stored data during incremental learning.</p><p><b>Evidence.</b> Closed-loop nuPlan evaluation: state of the art on long-tail scenarios with a <b>+9.4%</b> interPlan score over the strongest baseline, competitive on regular benchmarks, and the highest average closed-loop score with <i>positive backward transfer</i> when long-tail scenarios arrive sequentially.</p><p><b>Why it matters for MARA.</b> A working example of Bayesian non-parametrics doing the job MARA assigns them — letting the hypothesis space grow as the world reveals new regimes — inside a modern generative planner. The clusters-as-experts structure is a coarse, learned cousin of symbolic mode discovery, and positive backward transfer is the metric worth stealing: adaptation that improves old skills is the signature open-ended learners need.</p><p><i>Abstract-only summary — full text not found.</i></p>',
dila:'<p><b>What it is.</b> A latent action world model trained end-to-end on unlabeled video, whose contribution is naming and attacking the “LAM Trade-off”: the tighter the bottleneck that makes inferred latent actions abstract and transferable, the worse the generation; relax it and actions absorb irrelevant visual detail. DiLA's resolution is content–structure disentanglement.</p><p><b>The idea.</b> Split the video representation into a structure pathway (dynamics-relevant spatial layout — positions, shapes) and a content pathway (appearance, texture). The latent action is only required to predict structure dynamics; content flows around the bottleneck to the generator. The two objectives co-evolve: the predictive bottleneck forces motion into the action and offloads statics to content, and that separation in turn makes the action space cleaner — a single-stage loop where prior LAMs (Genie-style VQ, variational bottlenecks) needed two-stage training with pretrained world models or settled for optical flow.</p><p><b>Evidence.</b> Across human activity, robot manipulation, and outdoor navigation datasets, DiLA reports superior video generation quality, cross-embodiment action transfer, MPC-based visual planning, and an interpretable, continuous latent-action manifold (tested out of distribution). The diagnostic ablations run both directions: remove the IDM+FDM bottleneck and disentanglement fails; remove disentanglement and the trade-off reappears.</p><p><b>Why it matters for MARA.</b> Third variation on the latent-action theme in this planner (DreamDojo #3815 uses them as pretraining labels, Olaf-World orients them): here the point is <i>factorization</i> — carving what changes from what merely appears, which is the neural shadow of the object/state abstraction MARA builds symbolically. The co-evolution claim is the interesting hypothesis: representational structure emerges because the action bottleneck demands it — a testable mechanism for how abstraction can arise without symbols, and a natural foil for Autumn-style models where the same factorization is explicit.</p>',
olaf:'<p><b>What it is.</b> A latent-action pretraining objective plus the world model built on it (Show Lab, NUS). Seq∆-REPA anchors latent actions to a shared coordinate system so they transfer across contexts; Olaf-World pretrains an action-conditioned video world model (SkyReels-V2-1.3B DiT backbone, 540p, 97-frame clips) from passive video using those actions.</p><p><b>Why it exists.</b> Latent actions learned within individual clips entangle scene-specific cues — the same “turn left” gets a different code in a different scene — because standard objectives provide no mechanism to align action semantics <i>across</i> contexts. The insight: actions are unobserved but their semantic effects are observable, so anchor the integrated latent action to temporal feature differences from a frozen self-supervised video encoder — effects as the shared reference frame.</p><p><b>Evidence.</b> The clean test is MIND (UE5), two disjoint first-/third-person subsets sharing one 8-action space. Linear probing of action semantics (Macro-F1): Olaf <b>0.81/0.83</b> in-domain vs AdaWorld's 0.60/0.48, and <b>0.63/0.59</b> cross-domain vs 0.48/0.50 — the latents are decodable and consistent across viewpoint shifts. Zero-shot action transfer avoids AdaWorld's wash-outs and agent drop-outs; with labeled adaptation budgets of ≈0, <b>1 minute</b>, or 2 hours of video (matched LoRA capacity), Olaf-World wins on VBench quality and relative-pose-error controllability.</p><p><b>Why it matters for MARA.</b> The most conceptually interesting of the latent-action trio in this planner: it individuates actions by their <i>effects</i> — a functional, interventional definition of an action, arrived at for engineering reasons. That is how a causal formalism defines actions too, and the 1-minute adaptation result is the practical payoff MARA cares about for the Droid stack: a pre-aligned action space makes grounding a new controller nearly free.</p>',
sbip:'<p><b>What it is.</b> Simulation-based inference applied to a real engineering problem (SRI): learn a posterior over the full space of eVTOL aircraft designs — discrete topologies <i>and</i> their continuous parameters — conditioned on desired performance (mass, lift, drag), so conceptual design becomes sampling from an inverted simulator rather than manual optimize-and-check.</p><p><b>Why it exists.</b> Conceptual design wants distributions of viable options, not one optimum — Bayesian by nature. But eVTOL design breaks standard SBI three ways: topology changes alter how many parameters exist (a one-wing craft has no second-wing parameters); the space mixes 144 discrete configurations with high-dimensional continuous parameters, well past typical SBI benchmarks; and designs have variable component counts.</p><p><b>The design.</b><ol><li><i>A probabilistic program as the prior.</i> Designs are typed trees — subsystems to components to atomic parameters — generated by a custom tree-based probabilistic program acting as a generative grammar; SUAVE (open-source multi-physics aircraft analysis) plus custom CAD interference and structural checks evaluates and filters them into the training set.</li><li><i>Hierarchical mixed diffusion.</i> MixeDiT (built on Riemannian Diffusion Language Modeling and Unified World Models) samples discrete topologies jointly with continuous observations, conditioned on performance targets; MaskeDiT then samples parameters conditioned on topology, its masking scheme handling variable-dimensional design vectors — extending Simformer-style any-subset conditioning beyond fixed-structure simulators.</li></ol></p><p><b>Evidence.</b> Posterior predictive checks through SUAVE plus qualitative case studies: the learned posterior rediscovers known trends and governing physical laws of aircraft design and reproduces known design hypotheses across topologies, while accelerating design generation substantially relative to simulate-and-filter. Dataset and code released.</p><p><b>Why it matters for MARA.</b> A full Basis-shaped pipeline in the wild: a symbolic generative program defines the design space, a simulator grounds it, and amortized inference inverts it — probabilistic programming and neural inference in exactly the division of labor MARA advocates, deployed on a problem with real physics. The variable-dimension masking is the technical piece worth taking: posterior inference over structures whose parameter count is itself uncertain is the same problem world-model induction faces when the number of objects or rules is unknown. Strong stop for the SBI conversation; Mandt-lab and Macke-lab adjacent.</p>',
rtwin:'<p><b>What it is.</b> A data generator plus benchmark for bimanual manipulation: RoboTwin 2.0 auto-generates expert trajectories in simulation and ships the stack — RoboTwin-OD object library (<b>731</b> instances, 147 categories, manipulation-relevant annotations), <b>100,000+</b> expert trajectories over <b>50</b> dual-arm tasks and <b>5</b> robot embodiments, generator, benchmark, and code all released.</p><p><b>Why it exists.</b> Existing synthetic pipelines fail three ways: no automated quality control (bad grasps pollute training data), superficial domain randomization (clean homogeneous scenes that do not transfer), and no cross-embodiment awareness (a low-DoF Piper needs lateral grasps where a Franka grasps top-down, and datasets encode neither).</p><p><b>The pipeline.</b> An MLLM writes task-execution code that is validated and refined by simulation-in-the-loop feedback — the simulator is the critic of generated code; structured domain randomization spans five axes (clutter, lighting, background, table height, language phrasing); embodiment-aware grasp adaptation generates robot-specific action candidates from annotated affordances.</p><p><b>Evidence.</b> Code-generation success improves <b>10.9%</b> with the feedback loop. Downstream, a VLA fine-tuned on large-scale synthetic data plus only 10 real demos gains <b>367%</b> relative over the 10-demo baseline; trained on synthetic data alone, zero-shot models still gain <b>228%</b> on real-world evaluation — the sim-to-real claim carried by hard numbers.</p><p><b>Why it matters for MARA.</b> The robotics twin of Agent World Model (#2202, session 1): environments and expert data synthesized by an LLM with execution-grounded verification, at benchmark scale. The simulation-in-the-loop refinement is the same self-correcting generator pattern that keeps synthetic worlds honest, and the five-axis randomization is a concrete recipe for the environment diversity that AWM's scaling curve says matters. Direct relevance to Droid: this is the current best open answer to how far synthetic bimanual data substitutes for real demonstrations.</p>',
d2c:'<p><b>What it is.</b> A multi-agent LLM framework for robot co-design: body and reward are optimized jointly through structured debate. A design agent proposes morphology edits, a control agent critiques them and writes reward code, and the pair iterates thesis–antithesis–synthesis; criterion-specific LLM judges give multi-objective feedback, and every selection decision is grounded in physics — candidates are trained in Brax and ranked by standardized task score, so rhetoric answers to simulation.</p><p><b>Why it exists.</b> A robot's body and its objective are coupled — lengthen a leg and the optimal gait changes; penalize energy and an agile body crawls — but existing LLM pipelines (Eureka-style reward writing, morphology editors) treat them separately, and single-agent generation reuses familiar patterns exactly where co-design needs diversity.</p><p><b>Evidence.</b> On five MuJoCo locomotion tasks, highest default-normalized score among LLM-based and black-box baselines — up to <b>3.2×</b> on Ant and nearly <b>9×</b> on Swimmer. The ablation is the point: iterative debate beats compute-matched zero-shot generation by <b>18–35%</b>, so the structure, not just the sampling budget, carries the gain. Cross-over tests show D2C-generated rewards transfer to unmodified default morphologies on <b>4/5</b> tasks — the shaping captures locomotion principles, not morphology-specific hacks.</p><p><b>Why it matters for MARA.</b> Adversarial multi-agent proposal generation with a simulator as arbiter — a social-dynamics answer to the hypothesis-diversity problem that Verbalized Sampling (#1608) solves with prompting and ModelSMC (#3512) with SMC, and a small working instance of the Schmidhuber-line claim that open-endedness can be decentralized into agent interaction (the position his interestingness paper contests). The debate-vs-zero-shot ablation gives that claim a number. Relevant to Droid whenever morphology or objective is on the table rather than fixed.</p>',
pis:'<p><b>What it is.</b> A training paradigm for embodied navigation: SAGE has a VLM agent learn inside a physics-grounded <i>semantic abstraction</i> — a sandbox — rather than a photorealistic simulator, explicitly modeled on human mental simulation: rehearse the plan in simplified physics, then execute in the world.</p><p><b>Why it exists.</b> Embodied navigation lacks aligned open-world vision-and-control data; photorealistic simulators transfer poorly; and RL from scratch on a real robot is sample-starved and cold-starting. The wager is that transfer lives in physics-constrained structure, not pixels — so abstract the environment, keep the physics.</p><p><b>The three phases.</b> <i>Genesis</i> synthesizes diverse physics-constrained semantic environments and distills VLM reasoning over them into structured embodied experiences; <i>Evolution</i> runs GRPO with hybrid prompt-augmented sampling and an asymmetric adaptive clipping mechanism that constrains experience-augmented samples differently from standard ones, keeping updates stable while priors are internalized; <i>Navigation</i> avoids end-to-end motor control — the policy selects frontier or memory nodes from structured visual buffers and a geometric planner executes them.</p><p><b>Evidence.</b> <b>53.21%</b> LLM-Match success on A-EQA, <b>+9.7%</b> over baseline, with a working transfer to a physical indoor robot; the abstraction-trained policy survives the jump to real deployment.</p><p><b>Why it matters for MARA.</b> The closest thing in this poster hall to an explicit endorsement of MARA's central bet: agents should plan in abstracted, physics-grounded models of the world, not photorealistic replicas — mental simulation as an architecture choice, with the sandbox playing the role Autumn-style symbolic environments play in the program. The subgoal-plus-geometric-planner split is also the right factoring for the Droid stack: semantics learned, control classical. Worth probing at the board how the sandbox is generated and how far its physics fidelity has to go.</p>',
iwb:'<p><b>What it is.</b> A benchmark for <i>interactive</i> world models — the video-generation kind — testing what current benchmarks skip: responsiveness to action sequences and memory. Assets: 330k standardized video clips (12 datasets plus 4 simulators, unified coordinates and camera parameters), 2.1k human-verified evaluation videos, an Action Generation Framework encoding <b>81</b> fundamental motions across control modalities (text, one-hot keys, camera intrinsics/extrinsics), six task types yielding <b>4.9k</b> test samples, nine metrics across visual quality, action following, and memory. Public leaderboard.</p><p><b>Why it exists.</b> World models take actions in incompatible formats — text prompts, keyboard one-hots, camera matrices — so they have never been evaluated on the same interaction tasks; and no existing benchmark tests memory (does the model remember what it saw when the trajectory loops back?). The memory-symmetry metric checks pixel consistency of symmetric frame pairs around a trajectory's temporal midpoint — loop closure as a score.</p><p><b>Evidence (14 models evaluated).</b> HY-World 1.5 leads at <b>0.7873</b> average, strongest on memory and trajectory following; its trajectory accuracy (<b>0.7472</b>) versus text-controlled CogVideoX-I2V (<b>0.5950</b>) quantifies the advantage of discrete action signals over text descriptions. Text-controlled models win visual consistency (0.8988 brightness consistency) while failing trajectories — the benchmark surfaces a quality-versus-controllability trade-off as a general pattern.</p><p><b>Why it matters for MARA.</b> The video-world-model community independently converging on AutumnBench's question — evaluate the model by interacting with it, including whether state persists — but scoring it in pixel space against reference videos rather than by environment-level queries. The gap between the two is a talking point the user owns; the action-unification framework and the memory tasks are the parts worth borrowing. Same session as A2RBench and Causal-JEPA.</p>',
pdwm:'<p><b>What it is.</b> An offline model-based RL algorithm, ROMBRL, that trains the world model and the policy jointly under one objective — a constrained maximin problem solved as a Stackelberg game, the first such formulation in offline MBRL.</p><p><b>Why it exists.</b> Standard offline MBRL is two-stage: fit a world model by transition likelihood, then optimize a policy inside it — an objective mismatch, since the likelihood-best model is not the policy-learning-best model. And deployment is brittle: Gaussian measurement noise at 5% of the state change craters state-of-the-art offline RL scores. Return-driven model adaptation (the online fix) fails offline — with no real environment as anchor, a model optimized to inflate imagined returns just diverges from the truth. So the adaptation runs the other way: the model is updated <i>adversarially</i>, and the policy maximizes worst-case return within an uncertainty set anchored to the maximum-likelihood model.</p><p><b>The idea.</b> RAMBO, the nearest prior work, alternates updates as if the game were symmetric and zero-sum, losing convergence guarantees and over-hedging. ROMBRL makes the asymmetry explicit: policy as leader, world model as constrained follower whose best response is an implicit function of the policy. Stackelberg learning dynamics with primal–dual constraint handling give local-equilibrium convergence and a bound on the policy's suboptimality gap; a Woodbury-identity trick keeps second-order gradients cheap, and a gradient mask keeps stale replay rollouts usable as the model shifts.</p><p><b>Evidence.</b> State-of-the-art on twelve noisy D4RL MuJoCo tasks and three stochastic Tokamak Control tasks; in clean settings it matches SOTA and beats RAMBO; significance reported via Cohen's d across seeds.</p><p><b>Why it matters for MARA.</b> A precise answer to a question MARA's world-model-for-control story must face: what should a world model optimize when it exists to serve a policy and the real world cannot arbitrate? The offline inversion — adversarial rather than sycophantic model adaptation, uncertainty set anchored at the likelihood solution — is the principled middle ground between frozen models and wishful ones, and the game-theoretic framing is the right language for agent-and-model co-adaptation generally.</p>',
lilo:'<p><b>What it is.</b> A Bayesian-optimization framework from Meta (FAIR) for objectives that live in someone's head: LILO lets the decision maker steer optimization with free-form natural-language feedback. An LLM translates the accumulated feedback into labeled pairwise preferences over observed outcomes; a pairwise Gaussian-process surrogate absorbs them; standard acquisition functions propose the next batch.</p><p><b>Why it exists.</b> Real optimization targets are often subjective and hard to write as closed-form objectives, and preferential BO forces the human into impoverished formats — scalars or bare A-versus-B choices. LLM-as-optimizer approaches accept rich language but throw away BO's calibrated uncertainty and sample efficiency. LILO's architectural claim: keep the LLM in a <i>supporting</i> role — translator, not optimizer — and both properties survive.</p><p><b>Evidence.</b> Across synthetic and real-world benchmarks, LILO consistently beats both conventional preference-based BO and LLM-only optimizers, with its largest margins in feedback-limited regimes — exactly where language's information density over pairwise clicks should pay, and does. Ablations cover acquisition functions and feedback formats; code released.</p><p><b>Why it matters for MARA.</b> The same design thesis as LABO (#1308, session 3) proven on the preference side: LLM as interface to structure, GP as the belief object, acquisition as the decision rule — language in, probability out. For MARA it is the missing channel for human-in-the-loop experiment design: a scientist's prose critique of an experiment's outcome becomes data a calibrated model can act on, without surrendering the loop to the LLM's own unquantified judgment.</p>',
tomap:'<p><b>What it is.</b> A training recipe for opponent-aware LLM persuaders. ToMAP bolts two theory-of-mind modules onto a small persuader before RL training: a counterclaim predictor that enumerates likely objections to the target claim, and an attitude predictor trained to estimate the opponent's current agreement with each objection — an explicit, updating model of the persuadee's mind fed into the policy at every turn.</p><p><b>Why it exists.</b> Human persuaders model their audience proactively and dynamically; bare LLM persuaders do not, so they repeat themselves and argue past the opponent. The bet: make the opponent model an explicit component rather than hoping RL induces one.</p><p><b>Evidence.</b> A <b>3B-parameter</b> ToMAP persuader beats much larger baselines including GPT-4o by <b>39.4%</b> relative across multiple persuadee models and corpora. The ablation singles out the attitude predictor as the crucial module; training with the ToM scaffold yields longer reasoning chains, less repetition, and more diverse arguments; the persuadee's judgments are validated against humans; gains hold against larger persuadees.</p><p><b>Why it matters for MARA.</b> Complementary to MindZero (#412) — where MindZero learns to infer mental states without labels, ToMAP shows what wiring an explicit belief-about-the-other into the policy buys in a live adversarial dialogue: a 3B model out-persuading GPT-4o is a sharp datum for structured opponent models over scale. It is also the intervention-capable sibling of Decrypto's (#3504, same session) diagnosis that frontier models lack perspective-taking — same session, natural pairing. The dual-use edge (trained persuaders) is worth noting in any conversation about it.</p>',
ansr:'<p><b>What it is.</b> An infrastructure fix that unlocks amortized symbolic regression: the authors identify the bottleneck holding neural SR back as <i>simplification</i> — reducing equivalent expressions to a normalized form — and ship SimpliPy, a rule-based engine <b>~100×</b> faster than SymPy at comparable quality, plus Flash-ANSR, the SR transformer trained on top of it.</p><p><b>Why it exists.</b> Amortized SR trains a transformer to map data to expressions, promising far greater efficiency than genetic programming — but randomly generated training expressions are full of redundancies, and when SymPy-grade simplification sits in the training loop, data generation runs orders of magnitude slower than gradient updates. Fast normalization unlocks three things at once: much larger on-the-fly training sets, no token budget wasted on redundant forms, and — the quietly damning part — systematic decontamination against symbolically equivalent test expressions, which virtually no prior work in the field performs.</p><p><b>Evidence.</b> Under a strict protocol (machine-precision recovery, physically motivated data domains from SRSD rather than toy intervals, conservative symbolic and numeric decontamination), Flash-ANSR dominates the inference-time-vs-recovery-rate Pareto frontier over static (NeSymReS) and unsimplified (E2E) baselines, and matches PySR — the genetic-programming state of the art — while returning <i>more concise</i> expressions as inference budget grows, where PySR returns more complex ones.</p><p><b>Why it matters for MARA.</b> Equation discovery is the terminal step of MARA-style science loops (FIRE-Bench, ModelSMC), and amortized SR is its scalable form; this paper both removes the engineering bottleneck and exposes an evaluation problem — equivalent-expression leakage — that silently inflates the field's numbers, an AutumnBench-flavored critique of benchmark hygiene. The parsimony-with-budget scaling is the result to remember: the amortized model gets simpler with more compute where search gets baroque.</p>',
gift:'<p><b>What it is.</b> A self-training framework for image-to-CAD program synthesis. GIFT (Geometric Inference Feedback Tuning) turns test-time compute into training data: sample many candidate programs, execute them, score the rendered geometry against the target, and feed the results back as supervision.</p><p><b>Why it exists.</b> The claimed bottleneck is not model or algorithm but data — verified image↔program pairs are scarce and expensive, so supervised fine-tunes go brittle as design complexity rises. But CAD has an executable checker: render the program, compare geometry. That feedback is free supervision.</p><p><b>The two mechanisms.</b> GIFT-REJECT is soft-rejection sampling — retain diverse high-fidelity programs beyond exact ground-truth matches, since many distinct programs realize near-identical geometry; GIFT-FAIL converts near-miss predictions into synthetic training examples targeted at the geometries the model finds hard. Together they amortize inference-time search into the weights.</p><p><b>Evidence.</b> Mean IoU <b>+12%</b> over a strong supervised baseline (CAD-Coder-SFT); matches the baseline's peak Pass@k while using <b>80% less</b> inference compute; solves <b>53% more</b> problems than the baseline; degrades more gracefully with program length; competitive with heavier multimodal systems, with no human annotation or architecture changes.</p><p><b>Why it matters for MARA.</b> Program synthesis with an executable world as the critic — the same verify-by-running loop as AWM and RoboTwin, applied to inverting geometry into symbolic programs, which is a sibling of inverting environments into Autumn programs. The soft-rejection insight travels: when many programs explain the observation, keeping the diverse near-equivalents rather than the single exact match is how a learner preserves the posterior instead of collapsing it — GIFT stumbles into the multi-hypothesis discipline MARA formalizes.</p>',
stein:'<p><b>What it is.</b> A training-free diffusion-guidance method: Stein Diffusion Guidance (SDG) corrects the posterior approximation that all Tweedie-based guidance relies on, making off-the-shelf-classifier steering work in low-density regions where standard guidance breaks.</p><p><b>Why it exists.</b> Training-free guidance approximates the posterior via Tweedie's formula — a one-point denoising estimate that is unreliable exactly where interesting targets live, off the data manifold's high-density core (novel molecules, rare conditions). Stochastic optimal control gives principled posterior sampling but is computationally prohibitive. SDG is the reconciliation: a surrogate SOC objective with a new theoretical bound on the value function showing that approximate posteriors <i>must</i> be corrected toward the true diffusion dynamics.</p><p><b>The idea.</b> Use Stein variational inference inside the sampler: at each step a set of particles is pushed along the steepest-descent direction of KL divergence between the approximate and true posteriors (a back-and-forth Stein correction), plus a novel running-cost functional — kernelized, gradient-based posterior repair with no retraining of the diffusion model or classifier.</p><p><b>Evidence.</b> Consistently beats standard training-free guidance across image-guidance tasks, and on the demanding case built to expose the gap — small-ligand sampling for protein docking, where hits are novel, low-density molecules — SDG finds them where Tweedie-based methods stall; particle-count ablations confirm the Stein correction carries the effect.</p><p><b>Why it matters for MARA.</b> A tool-level entry in the amortize-then-correct family running through this planner (FUSE's FK steering, ModelSMC): a pretrained generative model as prior, a particle-based variational correction as the honesty mechanism — SVGD where FUSE uses SMC. The low-density focus is the substantive part for scientific discovery: the samples worth finding are rarely where the prior puts its mass, and SDG is a principled way to steer there without retraining.</p>',
glsd:'<p><b>What it is.</b> An agent framework (Tsinghua CoAI) that turns scientific software into an embodied environment: EmbodiedAct grounds an LLM in MATLAB/Simulink with a tight perception–execution loop, so the agent monitors a running simulation as a continuous stream — stdout, error logs, system events, and time-series trajectories (voltages, velocities) — and can intervene mid-run.</p><p><b>Why it exists.</b> Code-as-action and MCP-style API agents are passive: feedback arrives only after execution terminates. But process-oriented science fails in transit — diverging oscillations, numerical instability, convergence stalls that raise no execution error yet invalidate the result and burn compute on doomed trials. Runtime perception is the missing sense.</p><p><b>The design.</b> A POMDP over the software's latent state with streaming observations, run by a modular architecture: a strategic planner decomposes intent into sub-tasks and constraints; a primitive generator emits software-specific simulation primitives (ode45 calls, Simulink block edits); a runtime monitor — backed by an asynchronous state-sync protocol over web sockets — watches the simulation lifecycle. Two loops: a fast inner loop fires hot-fixes on detected anomalies, a slow outer reflective loop re-plans against scientific intent.</p><p><b>Evidence.</b> On complex engineering-design and scientific-modeling tasks in MATLAB, EmbodiedAct significantly outperforms code-as-action and API baselines — state of the art on reliability and stability in long-horizon simulations and accuracy in modeling. Code released.</p><p><b>Why it matters for MARA.</b> The claim in the title is the interesting move: treating simulation software as a body — perception and intervention during a physical process, not queries to a black box. That is active experimentation in exactly MARA's sense, applied to the software instruments scientists already use; the fast/slow loop split (reflexive hot-fix vs deliberative re-plan) is a sensible control architecture for any experiment-running agent, Droid included. What it lacks is a belief state — anomaly detection is reactive, not model-based — which is where the probabilistic-programming version of this agent would differ.</p>',
krl:'<p><b>What it is.</b> A budget-allocation fix for LLM reinforcement learning: treat each task's exploration as a knapsack item with a learning value and a compute cost, and solve for the allocation that maximizes total learning under a fixed global rollout budget.</p><p><b>Why it exists.</b> Uniform rollout budgets (say 8 per prompt) starve GRPO of gradients at both ends: easy tasks go all-success, hard tasks all-failure, and either way the group-relative advantage is zero — no learning signal. The mismatch is between task difficulty and exploration effort, so the fix is heterogeneous allocation driven by the model's current learning status.</p><p><b>Evidence.</b> Applied to GRPO on Qwen models (1B–7B), the knapsack allocation raises the ratio of non-zero policy gradients by <b>20–40%</b> during training, funds up to <b>93</b> rollouts for the hardest problems (prohibitive under uniform budgets), and delivers <b>2–4</b> point average gains on math-reasoning benchmarks with peaks of <b>9</b>; matching it with uniform allocation would take roughly <b>2×</b> the compute — a reallocation “free lunch.”</p><p><b>Why it matters for MARA.</b> Exploration budgeting <i>is</i> the practical face of the interestingness question — where should the next unit of compute go? — and Knapsack RL answers it with expected learning progress per cost, an operational cousin of the compression-progress heuristics in the Schmidhuber paper (#4107) and the gating rule in LABO. The zero-gradient pathology is also worth remembering wherever MARA uses GRPO-style training (MindZero does): saturation and impossibility are equally silent failure modes, and allocation is the cheapest cure currently on offer.</p>',
jade:'<p><b>What it is.</b> A two-layer evaluation framework for agentic AI on open-ended professional tasks. Layer 1 encodes expert knowledge as a fixed set of evaluation skills (stable criteria); layer 2 does report-specific, claim-level assessment, with evidence-dependency gating that invalidates any conclusion built on a refuted claim.</p><p><b>Why it exists.</b> The rigor–flexibility dilemma: static rubrics are reproducible but cannot accommodate diverse valid strategies; LLM-as-judge adapts to each response but is unstable and biased. Human experts do both — fixed principles, dynamic claim-level checking — and JADE copies that.</p><p><b>Evidence.</b> On BizBench, improved evaluation stability and failure modes surfaced that holistic LLM judges miss; strong alignment with expert-authored rubrics; transfers to HealthBench and the 10-domain DR.BENCH.</p><p><b>Why it matters for MARA.</b> Peripheral, but the dependency-gated claim graph is a good idea: scoring a chain of reasoning by propagating refutation through its evidence structure is closer to how a world model should be judged than any holistic score — a lightweight cousin of environment-level evaluation applied to text-borne reasoning. Worth a glance for AutumnBench's open-response scoring.</p>',
sphd:'<p><b>What it is.</b> A text-to-3D-world generator: SphericalDreamer produces immersive outdoor environments that are both navigable over long ranges and complete over the full omnidirectional field of view — the combination prior methods cannot deliver (long-range methods leave holes; panoramic methods do not travel).</p><p><b>The idea.</b> Generate layered-depth panoramas along a route and fuse them pairwise, with harmonic blending to align depth across panoramas — stitching local omnidirectional views into one traversable scene rather than growing a scene frame by frame.</p><p><b>Evidence.</b> Evaluated against recent scene-generation baselines on visual and depth-consistency metrics; the paper's claim rests on being the first to satisfy navigability and full coverage simultaneously.</p><p><b>Why it matters for MARA.</b> Peripheral — content generation, not world modeling: there is no dynamics, state, or interaction beyond a movable camera. Useful mainly as a boundary marker for how the term “3D world” gets used in the generation community, and conceivably as scenery for embodied environments. A walk-by unless VR asset pipelines become relevant.</p><p><i>Note: summary from full text; thin fit to the program.</i></p>',
o_a2i:'<p><b>What it is.</b> A VLA architecture, BehaviorVLA, that inserts a learned <i>behavioral representation</i> between abstraction and execution: a causal Mamba-based Visuomotor Behavior Encoder aggregates long-horizon trajectory information into one behavior code, and a Phase-conditioned Behavior Decoder unfolds it into actions by aligning task-level priors with real-time execution progress.</p><p><b>Why it exists.</b> VLA models degrade under distribution shift because their action-centric latents are temporally fragmented (short-horizon) and statically aligned to execution — the behavior the model “means” to perform never exists as a coherent object. BehaviorVLA makes it one: encode the whole behavior first, instantiate it phase by phase.</p><p><b>Evidence.</b> State of the art on three benchmarks — RoboTwin 2.0 <b>58%</b>, LIBERO <b>98%</b>, CALVIN <b>4.36</b> average length — and the data-efficiency headline: in real-world sim-to-real transfer it matches OpenVLA-OFT with <b>50%</b> of the demonstration data. Two-phase training (behavior-manifold learning, then prior-guided policy tuning) with ablations on both components. Also poster <b>#802</b>, session 5.</p><p><b>Why it matters for MARA.</b> The abstraction-then-instantiation structure is a policy-side echo of MARA's modeling thesis: commit to a temporally extended, semantically coherent unit (a behavior, not a next action) and let execution be inference against it. The phase-conditioned decoding — tracking where in the behavior you are — is a small explicit state machine inside a neural policy, and the 50%-data result is the evidence that the structure, not scale, bought the generalization. Directly relevant to how the Droid stack should represent skills.</p>',
o_p2t:'<p><b>What it is.</b> A controlled study, not a new method: four representative ways of supervising VLA models with latent actions, instantiated under one unified baseline (shared backbone and action head) so the comparisons mean something. The axes: image-based latent actions (motion inferred from visual transitions) used to regularize the trajectory — via implicit alignment, direct decoding, or conditional decoding — versus action-based latent actions used to unify heterogeneous action spaces via action-to-token mapping.</p><p><b>Why it exists.</b> VLA training data spans incompatible robot platforms and human videos; latent actions are the field's favored intermediate representation, but integration strategies are fragmented across papers with incompatible baselines, so nobody knows which formulation does what.</p><p><b>Findings.</b> A formulation–task correspondence: image-based latents help long-horizon reasoning and scene-level generalization; action-based latents win on motorically complex coordination. Across strategies, directly supervising the VLM to predict <i>discrete</i> latent action tokens is most effective, and latent-action supervision consistently helps under mixed-data joint training. Benchmarks: LIBERO, RoboTwin 2.0, and real-robot runs. Also poster <b>#803</b>, session 5.</p><p><b>Why it matters for MARA.</b> The taxonomy the latent-action wave has been missing — this planner alone holds four latent-action systems (DreamDojo, DiLA, Olaf-World, and GR00T-line VLAs), and this paper says which formulation buys which capability. The image-vs-action split maps onto a MARA distinction: perception-derived abstractions serve reasoning and generalization, control-derived ones serve execution — evidence that a single latent vocabulary will not cover both, which is an argument for structured, multi-level action representations.</p>',
o_bhg:'<p><b>What it is.</b> A Bayesian structure-learning model for electronic health records: it reframes multi-disease risk around latent <i>risk-factor-modulated pathways</i>, where risk factors act on hyperedges — latent subsets of diseases sharing a risk pattern — so a disease can belong to several pathways and higher-order (beyond-pairwise) structure becomes explicit and interpretable.</p><p><b>Why it exists.</b> EHR modeling has many rare outcomes driven by shared factors; strong predictors either treat diseases independently or hide the structure in a black box, with no calibrated uncertainty over how risk organizes disease. This model recovers the grouping and quantifies confidence in it.</p><p><b>The idea.</b> A repulsion prior over hyperedges forces a parsimonious, identifiable set of pathways (competing explanations repel rather than duplicate); posterior inference yields calibrated uncertainty over both disease groupings and risk-factor influence. Because exact inference is intractable at EHR scale, a structured variational family preserves the logical dependencies among hyperedge existence, disease membership, and pathway effect — inference that respects the model's combinatorial constraints rather than mean-field-factorizing them away.</p><p><b>Evidence.</b> Structural-recovery and predictive experiments with a repulsion-prior ablation and CAVI-vs-MCMC calibration checks; on UK Biobank it disentangles real disease pathways and stays predictively stable in the rare-disease regime. Also poster <b>#3302</b>, session 5.</p><p><b>Why it matters for MARA.</b> A clean applied instance of the machinery MARA is built on: latent discrete structure, an identifiability-inducing prior, and structured variational inference that keeps logical dependencies intact — the same problem shape as inferring which rules or objects govern an environment. The repulsion prior is the transferable idea: parsimony as an explicit prior is how posterior inference over structures avoids the redundant-hypothesis collapse that plain likelihood invites.</p>',
o_tfbf:'<p><b>What it is.</b> A way to run a better particle filter for free: a pretrained diffusion emulator of a dynamical system is repurposed, with no additional training, to implement the fully-adapted auxiliary particle filter — an optimal variant that was known theoretically but impractical with classical numerical solvers.</p><p><b>Why it exists.</b> Particle filters are exact for nonlinear dynamics and observations but scale terribly in high dimensions; the fully-adapted APF (which proposes from the dynamics conditioned on the next observation) reduces the variance that kills them, but computing that proposal needs a tractable, conditionable forward model. A diffusion emulator already <i>is</i> one — its score lets you sample the transition conditioned on data — so the optimal filter falls out of a model people trained for forecasting.</p><p><b>Evidence.</b> Scales particle filtering to high-dimensional chaotic systems: Lorenz-63 up through medium-range weather with GenCast (Google DeepMind's diffusion weather model), against bootstrap PF, Ensemble Kalman, Ensemble Score, and Ensemble Flow baselines. Also poster <b>#1405</b>, session 5.</p><p><b>Why it matters for MARA.</b> A crisp instance of the pattern running through the whole probprog thread here (FUSE, Stein guidance, ModelSMC): a pretrained generative model becomes the proposal engine for principled inference, and classical Monte Carlo (here, exact particle filtering) becomes the correctness guarantee. State estimation from partial observations of a nonlinear system is exactly what a world-modeling agent does between experiments; that the optimal filter is now implementable on top of a learned emulator, training-free, is a directly usable building block for the Droid stack's perception loop.</p>',
o_cg:'<p><b>What it is.</b> A benchmark that casts scientific discovery as interactive games with a fixed, hidden structural causal model: the LLM agent must collect observational data, design and run experiments, and draw conclusions — and is scored on whether it actually identifies the causal mechanism, not on its narrative. The games deliberately inject selection bias, measurement error, and hidden confounders, the failure modes real discovery faces and other AI-Scientist benchmarks ignore.</p><p><b>Why it exists.</b> Building AI-Scientist agents is a hot direction, but scientific discovery is fundamentally about causation — distinguishing it from correlation, recognizing bias — and no prior benchmark tests that under adversarial confounding. CausalGame does, and grades by interventional outcome against the hidden SCM.</p><p><b>Evidence.</b> None of the evaluated agents shows reliable causal thinking. The best (Claude-Opus-4.5) reaches <b>68.0%</b> survival against analytical optima of <b>78–85%</b>; only <b>5–7%</b> of sessions earn credit on the causal-reasoning rubrics; simple non-LLM baselines overlap the lower agent range, so agentic interaction adds little without causal reasoning. Behavioral analysis shows agents under-explore the design space and drift away from correct configurations they had already found — threshold-clearing wins are trial-and-error. Causal thinking correlates weakly with general capability benchmarks, i.e. it is largely unmeasured elsewhere. Also poster <b>#3609</b>, session 7.</p><p><b>Why it matters for MARA.</b> This is AutumnBench's thesis proven in the causal-discovery domain: score the agent by intervening on a hidden ground-truth model, not by its self-report, and frontier models fall well short of the analytical optimum. The specific failure — under-exploration and drift, wins by luck rather than identification — is precisely the gap active experimentation is meant to close, and the “assess by interventional outcomes against a fixed hidden SCM” protocol is a design MARA should adopt wholesale. Made for the AutumnBench conversation; same session (5D) as the futures-benchmark position paper.</p>',
o_fut:'<p><b>What it is.</b> A position paper: AI's benchmark-centered selection environment, for all its success, <i>taxes exaptation</i> — the reuse of an idea evolved for one purpose to solve another — and so quietly forecloses the futures it cannot yet measure.</p><p><b>The argument.</b> Breakthroughs often come from ideas that looked uncompetitive by the metrics of their time; in biology this is exaptation (a trait evolved for one function becoming decisive for another), and scientific progress works the same way — but only if such ideas can survive the interval when they score badly. Benchmarking let the field sidestep unresolvable debates about the nature of intelligence by substituting a shared scoreboard, but when one selection rule dominates, ideas that do not fit it have nowhere to persist. The cost grows acute as the question shifts from <i>can machines behave intelligently?</i> to <i>can they do so while being aligned, interpretable, and safe?</i> — philosophically distinct questions that may demand discoveries no current benchmark can specify in advance.</p><p><b>The proposal.</b> Restore exaptive capacity without abandoning benchmarks: plural evaluation regimes, protected venues for non-comparable work, long-horizon funding, and training norms that teach researchers to question selection rules rather than only optimize within them.</p><p><b>Why it matters for MARA.</b> This is the meta-argument beneath the AutumnBench program stated in general form — a critique not of any one benchmark but of monoculture in evaluation itself, and the intellectual cover for building an environment-level, non-comparable evaluation that a leaderboard culture would otherwise starve. It is also a caution the user should turn on the program's own claims: AutumnBench is a new selection rule, and the paper's medicine (plural regimes, protected venues) applies to it too. The single most useful whetstone in this session for how the user frames what AutumnBench is <i>for</i> — and the ideal warm-up for Narayanan's evaluation-skeptic talk. Also poster <b>#4610</b>, session 7.</p><p><i>Abstract-only summary — full text not found.</i></p>',
o_rare:'<p><b>What it is.</b> An end-to-end framework for the systematic analysis of <i>rare events</i> in LLMs — behaviors far from typical but highly significant — treating the model as the probabilistic object it is and asking how to see, generate, and estimate the probability of outputs that essentially never appear in development yet surface at deployment scale.</p><p><b>The idea.</b> A practical pipeline spanning theory, efficient generation strategies to surface rare events, probability estimation, and error analysis, illustrated on concrete examples and pitched as general across models and contexts. This is the importance-sampling / rare-event-simulation tradition (the province of reliability engineering and particle methods) brought to bear on language models, where the tail is where safety failures live.</p><p><b>Why it matters for MARA.</b> Estimating the probability of events a model almost never emits is the same technical problem as posterior sampling in low-density regions (Stein guidance, #2512) and as forecasting rare long-tail regimes (NOMAD) — the tail is where both risk and discovery concentrate. For an agent whose competence must be characterized, not just averaged, a principled account of its rare behaviors is exactly the evaluation an expected-case benchmark cannot give. Also poster <b>#1508</b>, session 7.</p><p><i>Abstract-only summary — full text not found.</i></p>',
o_dirl:'<p><b>What it is.</b> A distributional framework for offline inverse RL: instead of recovering a single deterministic reward or matching only expected returns, it jointly models uncertainty over reward functions <i>and</i> the full distribution of returns, recovering both a reward distribution and a distribution-aware policy.</p><p><b>The idea.</b> Match expert behavior by minimizing first-order stochastic dominance (FSD) violations, which lets distortion risk measures enter policy learning — so the recovered policy is risk-aware rather than risk-neutral. Convergence at O(ε<sup>−2</sup>) iterations.</p><p><b>Evidence.</b> Synthetic benchmarks, real neurobehavioral data, and MuJoCo control: expressive reward representations and state-of-the-art imitation. Also poster <b>#212</b>, session 7.</p><p><b>Why it matters for MARA.</b> Inferring a reward as a <i>distribution</i>, not a point, is the reward-side instance of the multi-hypothesis discipline that runs through MARA — the same reason ModelSMC keeps a posterior over models rather than one best fit. The neurobehavioral application is the interesting tell: recovering a distribution over what an agent values is a tool for modeling minds (human or animal) from behavior, adjacent to the ToM thread (MindZero, ToMAP) and to Basis's inverse-modeling interests.</p>',
o_pp:'<p><b>What it is.</b> A pretraining study: expose a language model to <i>abstract structured data</i> — procedural data generated by formal languages and simple algorithms — before natural-language pretraining, on the hypothesis that learning simple logic and structure first eases later acquisition of semantic knowledge, the way humans learn arithmetic before higher reasoning.</p><p><b>The findings, in order.</b><ol><li><i>Which skills.</i> Different procedural data improve specific algorithmic skills, often dramatically — context recall (needle-in-a-haystack) jumps from <b>10%</b> to <b>98%</b> after pretraining on Dyck sequences (balanced brackets).</li><li><i>At scale.</i> Front-loading as little as <b>0.1%</b> procedural data beats standard pretraining on C4, CodeParrot, and DeepMind-Math at up to 1.3B parameters, reaching the same loss with only <b>55/67/86%</b> of the original data.</li><li><i>Mechanism.</i> Procedural pretraining instills non-trivial structure in both attention (matters most for structured domains like code) and MLP layers (matters most for language).</li></ol>The framing: disentangle knowledge acquisition from reasoning. Also poster <b>#1706</b>, session 8.</p><p><b>Why it matters for MARA.</b> A striking empirical handle on a question the Schmidhuber interestingness paper (#4107) poses theoretically — how much does abstract, compressible structure accelerate later learning? — answered here in data-efficiency terms: a trickle of formal-language data reorganizes the network so semantic learning comes faster. That structure precedes and scaffolds knowledge is close to MARA's bet that symbolic abstraction is the substrate reasoning should be built on, and the Dyck→recall result is a clean demonstration that the right formal prior installs a reusable capability rather than a memorized fact.</p>',
o_tau:'<p><b>What it is.</b> τ²-Bench, a benchmark for conversational agents in a <i>dual-control</i> environment (Sierra): unlike prior single-control benchmarks where only the AI uses tools and the user passively supplies information, here both agent and user act on a shared, dynamic world — modeled as a Dec-POMDP — as in technical support where the user must be guided to change state on their end.</p><p><b>The contributions.</b> A Telecom dual-control domain; a compositional task generator that builds verifiable tasks from atomic components for controlled coverage and complexity; a tool-and-state-constrained user simulator for higher-fidelity simulation; and fine-grained ablations that separate reasoning errors from communication/coordination errors.</p><p><b>Evidence.</b> Agents drop significantly moving from no-user to dual-control — guiding a user to act is a distinct, hard skill. Also poster <b>#2207</b>, session 8.</p><p><b>Why it matters for MARA.</b> A rare benchmark where the agent must model and steer <i>another actor</i> in a shared environment — theory of mind and coordination as measured competences, adjacent to Decrypto (#3504) and ToMAP. The reasoning-vs-communication error decomposition is the transferable idea: environment-level evaluation that attributes failure to a mechanism rather than a scalar.</p>',
o_prod:'<p><b>What it is.</b> The first systematic field study of production LLM agents (CAP): 20 in-depth case studies plus a survey of 306 practitioners across 26 domains, documenting why organizations build agents, how they build and evaluate them, and what breaks.</p><p><b>The findings.</b> Production agents are deliberately simple and controllable: <b>68%</b> take at most 10 steps before human intervention, <b>70%</b> prompt off-the-shelf models rather than tune weights, <b>74%</b> rely primarily on human evaluation. Reliability — consistent correct behavior over time — is the dominant challenge, addressed through system-level design rather than better models.</p><p><b>Why it matters for MARA.</b> A reality check, not a method: the gap between research agents and what survives deployment is a caution for any ambitious autonomous-agent program, and the finding that human evaluation dominates because automated evaluation is not trusted underlines why rigorous, environment-level evaluation (AutumnBench's remit) is an unmet need in practice. Also poster <b>#4211</b>, session 3.</p>',
o_omac:'<p><b>What it is.</b> OMAC, a framework for holistic optimization of LLM multi-agent systems. It identifies five optimization dimensions spanning agent functionality and collaboration structure, then optimizes them with two components — a Semantic Initializer and a Contrastive Comparator — first per-dimension, then jointly across dimensions.</p><p><b>Why it exists.</b> Multi-agent systems help on code generation and arithmetic reasoning, but are built by hand; systematic design and optimization of LLM-MAS is underexplored. OMAC automates it.</p><p><b>Evidence.</b> Outperforms recent multi-agent approaches across diverse tasks; code released. Also poster <b>#1806</b>, session 8.</p><p><b>Why it matters for MARA.</b> Peripheral — automated design of agent collectives, adjacent to Debate2Create's (#911) use of multi-agent debate as a search operator. The idea worth noting is treating collaboration structure itself as an optimization variable rather than a fixed scaffold, which is where open-ended systems that redesign their own organization would begin.</p>',
o_brm:'<p><b>What it is.</b> A reward-model design, BNRM, that folds non-negative factor analysis into the Bradley–Terry preference model to make RLHF reward learning robust to reward hacking — the exploitation of noisy annotations and systematic biases (response length, style) that plague preference-trained rewards.</p><p><b>The idea.</b> Represent reward through a sparse, non-negative latent-factor generative process at two levels: instance-specific latents induce disentangled reward representations, while sparsity over global factors acts as an implicit debiasing mechanism that suppresses spurious correlations. This <i>disentangle-then-debias</i> structure yields uncertainty-aware rewards; to scale it to modern LLMs, an amortized variational inference network conditioned on deep model representations trains the whole thing end to end.</p><p><b>Evidence.</b> Substantially reduces reward over-optimization, improves robustness under distribution shift, and produces more interpretable reward decompositions than strong baselines. Also poster <b>#315</b>, session 4.</p><p><b>Why it matters for MARA.</b> This is the probabilistic-programming toolkit — latent factor models, sparsity priors, amortized variational inference — brought to the exact place the alignment stack is most brittle: turning a scalar reward into a structured, uncertainty-aware, disentangled belief is the same upgrade Verbalized Sampling (#1608) and Express-Your-Doubts (#1709) argue for on the output side, here on the reward side. Reward hacking is a world-model failure in disguise (the policy exploits a mis-specified model of what is wanted); a debiasing prior over reward structure is a principled defense, and the amortized-inference implementation is directly reusable.</p>',
o_rubric:'<p><b>What it is.</b> A training method for long-form deep-research agents, RLER — Reinforcement Learning with Evolving Rubrics — where the evaluation rubrics co-evolve with the policy during training rather than being fixed in advance, and the resulting model, DR Tulu-8B, is the first fully open model trained directly for open-ended long-form research.</p><p><b>Why it exists.</b> Open deep-research agents are trained on short-form QA with verifiable rewards, which does not transfer to realistic long-form, well-attributed answers. Static rubrics cannot judge what the agent discovers mid-search; evolving ones incorporate newly found information and contrast model responses for sharper, on-policy fact-checking feedback.</p><p><b>Evidence.</b> Across four long-form benchmarks (science, healthcare, general), DR Tulu-8B beats open agents by <b>15.6%</b> over Tongyi DR on average, matches or exceeds proprietary agents (+0.7% over OpenAI DR) while being far smaller and <b>~1000×</b> cheaper per query. Also poster <b>#1800</b>, session 4.</p><p><b>Why it matters for MARA.</b> A moving reward that adapts to what exploration turns up is the RLHF-side echo of the active-experimentation loop — the evaluator cannot be fixed when the space of findings is open — and a direct cousin of A2RBench's and LIVE's self-updating checks. For a research agent the rubric <i>is</i> the world model of what counts as a good answer, and letting it co-evolve is how open-ended competence gets a learning signal.</p>',
o_vot:'<p><b>What it is.</b> A semi-supervised reward-learning method for preference-based RL, VOTP, that learns reward functions from only a handful of labels by using optimal transport to align visual trajectories inside the representation space of a video foundation model, generating high-fidelity pseudo-labels for masses of unlabeled data.</p><p><b>Why it exists.</b> Preference-based RL escapes hand-crafted reward engineering but is throttled by human labeling cost; a pretrained video model's representation is a cheap source of trajectory similarity that can propagate a few labels to many.</p><p><b>Evidence.</b> Outperforms state-of-the-art offline preference-based RL under limited feedback budgets across locomotion and manipulation, is robust to visual distractors, and learns meaningful rewards on real robot tasks with minimal human input. Also poster <b>#403</b>, session 4.</p><p><b>Why it matters for MARA.</b> Feedback efficiency is the binding constraint on any human-in-the-loop robot-learning program, and optimal-transport alignment in a foundation-model latent is a concrete way to stretch scarce labels — relevant to the Droid stack wherever reward or preference data is expensive. Adjacent to the amortize-scarce-supervision theme running through LABO and LILO.</p>',
w_like:'<p><b>What it is.</b> A SPIGM contributed talk (Zenn & Geiping, MPI-IS) that empirically dissects when a language model's sequence probability actually tracks correctness — the assumption underneath every mode-seeking decoding method (beam search, best-of-n, power sampling) and every verifier-free self-improvement scheme (self-consistency, self-distillation).</p><p><b>The study.</b> Correlation between sequence log-probability and correctness measured at four nested granularities — across decoding methods, across hyperparameters within a method, across prompt-answer pairs within a dataset, and across repeated responses to one prompt — over Qwen2.5/Qwen3/Olmo3 families and multiple benchmarks.</p><p><b>Findings.</b> There is <i>no uniform</i> probability–correctness relationship. Within a dataset, higher probability does predict correctness (a model can often tell right from wrong pairs) — but this signal does not transfer to the decisions decoding actually makes: tuning a method's hyperparameters to raise log-probability does <b>not</b> yield more correct answers, and methods that produce higher-probability sequences are not reliably more accurate. For a single prompt, probability across its own responses is uninformative (though more-correct samples show larger within-sample correlation).</p><p><b>Why it matters for MARA.</b> The empirical companion to Express Your Doubts (#1709) and Verbalized Sampling (#1608): all three converge on the conclusion that an LLM's token probabilities are not a trustworthy belief signal. Here it is measured rather than argued — the correctness signal exists at the population level but evaporates exactly where you would use it to make a decision. For any MARA pipeline tempted to read confidence off logprobs (verifier-free selection, self-consistency over a world model's outputs), this is the cautionary data, and the four-level decomposition is a reusable protocol for asking whether a probability means anything operationally.</p>',
w_nsode:'<p><b>What it is.</b> A SPIGM paper on discovering governing ODEs from data by generative modeling over a <i>grammar</i> of equations — neuro-symbolic model discovery, the workshop's closest paper to the Basis discovery thread. A grammar quantisation autoencoder (GQAE) embeds rule-based expressions into a discrete latent space; a discrete flow model samples in that space, steered by trained predictors.</p><p><b>The ideas.</b><ol><li><i>Grammar as representation.</i> Expressions are sequences of context-free-grammar production rules — structure and syntactic constraints built in, no token-sequence or explicit-tree learning — with a placeholder symbol for scalar constants fitted later by optimization.</li><li><i>Behavioural latent geometry.</i> The discrete latent is restructured by a Wasserstein behavioural distance so that ODEs which <i>act</i> alike sit near each other, not merely those that look alike syntactically — because structural similarity does not imply dynamical similarity.</li><li><i>Domain-knowledge-guided sampling.</i> Predictors enforce order and stability conditions on the ODEs, and coupling them to the discrete flow gives guided sampling in the rule space — reducing rejection and improving exploration.</li></ol></p><p><b>Why it matters for MARA.</b> A concrete instance of the program's core loop — inference over programs — with two ideas worth stealing: a discrete grammar as the hypothesis space (the same move Autumn and amortized symbolic regression make), and a <i>behavioural</i> metric on that space so the generative model organizes candidates by what they do, not how they read. That semantic-over-syntactic latent is exactly what a world-model discovery system needs to avoid drowning in expression variants that are formally distinct but dynamically identical.</p>',
w_epai:'<p><b>What it is.</b> The position paper behind this workshop (Cuzzolin): AI needs a paradigm shift toward <i>epistemic</i> AI — models that learn not only from what they know but from their own ignorance, representing and managing uncertainty as a first-class object rather than fitting data and hoping.</p><p><b>The argument.</b> Current ML is brittle where it matters — overconfident and wrong on out-of-distribution samples, natural fluctuations, and adversarial inputs (autonomous-vehicle incidents are the exhibit) — because the field overemphasizes data fitting and domain adaptation, both of which assume the future resembles the training distribution. The unknown-unknowns regime breaks that assumption by construction, and single-distribution uncertainty (a softmax, one posterior) cannot represent ignorance about which distribution you are even in. Epistemic AI proposes richer uncertainty representations (imprecise probabilities, possibility, credal sets) so a model can say not just how uncertain but how <i>ignorant</i> it is.</p><p><b>Why it matters for MARA.</b> This is the AutumnBench thesis stated as a general research program, and the intellectual home of the workshop the user should treat as home turf: humans beat frontier models precisely on the unknown-unknowns axis, and the diagnosis here — that data-fitting cannot produce the belief-updating an open world demands — is the argument for active experimentation and explicit probabilistic world models rather than scaled pattern recognition. Where MARA differs is prescription: Basis reaches for probabilistic programs and symbolic structure where this paper reaches for generalized uncertainty calculi — a productive disagreement to have in the room. The single best framing document in the Friday program for what AutumnBench is measuring.</p>',
w_ign:'<p><b>What it is.</b> A concrete mechanism for the unknown-unknowns problem: Structured Ignorance Certificates (SICs), a JSON output schema that forces a model, when a question exceeds its knowledge, to name the missing domain intersection, enumerate the concepts it would need, and propose a productive retrieval query — instead of hallucinating a fluent wrong answer.</p><p><b>The method.</b> A 7,347-sample Unknown-Unknown dataset is built by prompting Qwen3-14B to stitch questions from seven domains (physics, biology, engineering, CS, economics, medical, legal) into novel cross-domain queries no single-domain expert could answer, then training models to emit calibrated SICs on them — turning abstention from a refusal into a structured, actionable description of the gap.</p><p><b>Why it matters for MARA.</b> The operational counterpart to the epistemic-AI position: not just knowing that you do not know, but <i>characterizing</i> the ignorance well enough to act on it. That a model's response to a knowledge-boundary should be a structured object naming what is missing and a query that would resolve it is exactly the interface an actively experimenting agent needs — ignorance as a plan for the next experiment rather than a dead end. The cross-domain UU construction is also a reusable recipe for generating genuinely out-of-distribution probes, adjacent to how AutumnBench manufactures situations a model cannot have memorized.</p>',
w_neo:'<p><b>What it is.</b> A Compositional-Learning spotlight (Sungjin Ahn's lab) that reframes what a world model is <i>for</i>: not accurate future prediction but <i>theory-building</i>. Learning-to-Theorize infers explicit explanatory theories from raw, non-textual observations, instantiated as the Neural Theorizer (NEO) — a probabilistic model that induces latent programs as a learned Language of Thought and runs them through a shared transition model.</p><p><b>The argument.</b> Contemporary world models operationalize understanding as prediction in latent or observation space. Developmental cognitive science — the “baby as scientist,” theory-theory view — says children instead construct and revise structured, reusable, compositional explanations of how the world works, before language. NEO builds that in: a theory is an executable compositional program whose learned primitives recombine to explain novel phenomena, selected under a minimum-description-length principle.</p><p><b>Evidence.</b> Explanation-driven generalization — observations are understood in terms of the programs that generate them, and learned primitives transfer to explain phenomena unseen in training. A Compositional-Learning spotlight; the planner already flags it as Saturday's most Basis-shaped title. Full paper on arXiv (2605.03413).</p><p><b>Why it matters for MARA.</b> This is MARA's thesis arriving from the deep-learning side almost verbatim: world models as executable programs, induced under MDL, understood by theory-building rather than curve-fitting, with the Turing “simulate the child, not the adult” framing that Basis's curiosity/open-endedness thread lives in. NEO is the neural-program-induction cousin of Autumn — same commitment to compositional, executable theories as the object of learning, reached without the probabilistic-programming toolchain. The single most important paper to read alongside AutumnBench this week; the convergence (and the methodological divergence in how theories are represented and searched) is the conversation to have with the authors.</p>',
ws_eval:'<p><b>What it is.</b> A Hypothesis-Testing workshop paper (Sadhuka, Prinster, <b>Fannjiang</b>, Scalia, <b>Regev</b>, Wang — Genentech/Basis-adjacent authors) that turns any black-box agent verifier into a decision rule with a provable false-alarm guarantee. E-valuator frames “will this trajectory succeed?” as sequential hypothesis testing and rides an e-process, so the test stays statistically valid at <i>every</i> step of an arbitrarily long agent rollout — anytime-valid online monitoring.</p><p><b>Why it exists.</b> Verifiers — LLM judges, process-reward models — produce informative scores but no correctness guarantee, so a threshold on them can fire false alarms at an uncontrolled rate. Wrapping the raw score in an e-process converts a heuristic signal into a calibrated stopping decision; a quantile-estimation refinement raises statistical power.</p><p><b>Evidence.</b> Across <b>six datasets</b> and <b>three agents</b>, E-valuator gives both tighter false-alarm control and greater power than competing strategies; the paper shows a marginal-calibration baseline provably fails to control the false-alarm rate.</p><p><b>Why it matters for MARA.</b> Anytime-valid sequential testing is exactly the statistical layer an actively experimenting agent needs to decide, mid-trajectory, whether to trust its own rollout — and the authors are in Basis's orbit (Fannjiang and Regev). Verifying an agent's trajectory against a hidden success criterion with rigorous error control is the operational form of the AutumnBench question, and e-processes are the right tool for it: monitor continuously, stop when the evidence is decisive, never inflate the alarm rate by peeking. A directly adoptable primitive for the experiment-running loop.</p>',
ws_brain:'<p><b>What it is.</b> A SPIGM paper (Bracher, Intes, <b>Radev</b> — the BayesFlow author) that runs a brain foundation model <i>in reverse</i>: given a foundation model that emulates neural responses to stimuli, can simulation-based inference recover the stimulus properties from the predicted brain activity?</p><p><b>The setup.</b> Pair the brain emulator (TRIBEv2) with an LLM that generates news headlines from linguistic parameters — valence, arousal, dominance — then use amortized SBI to learn a probabilistic map from predicted brain maps back to those latent stimulus parameters. The LLM is a controllable stimulus generator; the emulator is the simulator; SBI inverts the pair.</p><p><b>Evidence.</b> The latent parameters are recoverable from predicted brain maps — validating the quality of the neural encodings and pointing toward decoding and inverse design with foundation brain models.</p><p><b>Why it matters for MARA.</b> A compact demonstration of the amortized-inference pattern (BayesFlow lineage, the same toolchain as FUSE and ModelSMC) turned on foundation models themselves: treat a large pretrained emulator as the forward simulator and invert it with SBI. The move — invert a foundation model to recover the latent cause of its outputs — is the general recipe MARA would use to interrogate any black-box world model, and the LLM-as-controllable-stimulus-generator idea is a neat way to manufacture the labeled simulations amortized inference needs.</p>',
ws_srsci:'<p><b>What it is.</b> An AI-for-Science workshop paper (also ICLR 2026) that promotes the LLM in symbolic regression from equation-<i>proposer</i> to autonomous experimenter. SR-Scientist wraps a code interpreter as tools for data analysis and equation evaluation, and the agent writes code, implements a candidate equation, submits it, reads the experimental feedback, and iterates over a long horizon with minimal hand-built pipeline.</p><p><b>Why it exists.</b> Prior LLM symbolic regression confines the model to proposing sympy strings or expression graphs inside a genetic-programming search — the LLM never touches the data or closes the loop. SR-Scientist gives it the instruments: analyze the data, fit, evaluate, refine.</p><p><b>Evidence.</b> Beats baselines by <b>6–35%</b> absolute across four science disciplines, with robustness to noise, generalization of discovered equations to out-of-domain data, and symbolic accuracy; an end-to-end RL framework further sharpens the agent. Code and a 30B model released.</p><p><b>Why it matters for MARA.</b> The agentic, tool-using face of the discovery loop that ModelSMC (#3512) and Flash-ANSR (#2707) attack from the probabilistic and amortized angles — same terminal task (recover the governing equation), different division of labor: here the LLM drives an executable evaluate-and-refine cycle rather than serving as a proposal kernel inside a sampler. The contrast with ModelSMC is the productive one to hold in mind: SR-Scientist optimizes toward a point estimate via feedback where ModelSMC maintains a posterior — the same tension between finding the best model and quantifying which models the data support.</p>',
ws_actflow:'<p><b>What it is.</b> A SPIGM/discovery paper (Lee, De Santi, Chatterjee, Yue, <b>Krause</b>) that reframes generative pretraining for design: standard flow/diffusion matches the data distribution, which covers only a sliver of the valid design space, but discovery needs valid <i>new-to-nature</i> designs the fitted model assigns negligible probability. ActFlow abandons distribution-matching for a different target — the model's <i>generable set</i>, the region it covers with non-negligible probability — and learns to enlarge it.</p><p><b>The idea.</b> Continued pretraining driven by verifier feedback: iteratively adapt a pretrained flow to synthetic data produced by active exploration in the learned flow representation, expanding coverage into new valid regions. The theory contributes first-of-their-kind statistical-learning guarantees for out-of-distribution flow modeling, analyzing generable-set expansion as a local-to-global reachability process over the representation.</p><p><b>Evidence.</b> Across small organic molecules, drug-like molecules, therapeutic peptides, and protein sequence design, ActFlow expands valid coverage far beyond the pretrained model's region and significantly beats standard synthetic-flow pretraining on OOD generative metrics.</p><p><b>Why it matters for MARA.</b> This is open-endedness given a generative-modeling formalization: the explicit objective is to reach valid designs <i>outside</i> what the data taught, using a verifier as the arbiter of validity and active exploration to push the frontier — curiosity as generable-set expansion. The generable-set-over-density reframing is the conceptual takeaway for the program: a model of the world should be judged by the space of valid possibilities it can reach, not by how tightly it fits the observed sample — precisely the distinction between memorization and the extrapolative competence AutumnBench probes.</p>',
ws_illus:'<p><b>What it is.</b> An Uncertainty-in-Agentic-Systems paper (Lin, Yun, Matarić, Canny, <b>Gretton</b>, <b>D'Amour</b>) with a sharp causal thesis: an experiment run on LLM-simulated users is not an intervention — it is an observational study wearing intervention's clothes.</p><p><b>The argument.</b> LLMs are trained on observational data, so when you intervene on an LLM-simulated user (change a policy, update an agent), the intervention also silently shifts the latent user attributes the persona induces — <i>user drift</i>. The simulated population is no longer the same across treatment and control, which is textbook confounding/selection bias, and it can inflate or attenuate the measured effect. They formalize this and, critically, offer diagnostics and a fix: <i>negative control outcomes</i> (attributes that should be invariant under the intervention) reveal the drift, and eliciting targeted, setting-relevant confounders into the persona specification substantially reduces the bias across survey-style and multi-turn agent evaluations.</p><p><b>Why it matters for MARA.</b> The cleanest statement of a trap the whole LLM-as-simulator wave falls into, and it is exactly MARA's distinction between intervention and observation — the difference between a world model you can act on and one you can only watch. Building Social World Models (#815) simulates belief shifts, CausalGame (#3609) tests causal thinking; this paper proves the substrate itself is observational unless you actively correct for drift. The negative-control diagnostic is a directly reusable check for any Basis pipeline that uses an LLM to stand in for a real experimental population — and a pointed reminder of why genuine active experimentation, not simulated intervention, is the program's core commitment.</p>',
ws_cawm:'<p><b>What it is.</b> An AI-for-Science position paper arguing that outcome-only evaluation of AI scientists is insufficient: task outcome, <i>mechanism fidelity</i>, and <i>epistemic honesty</i> must be measured separately, because an agent can reach the right answer while defending a mechanism its own data contradicts.</p><p><b>The evidence.</b> 28 episodes of a coding agent attempting to rediscover a known particle-identification observable in a Geant4 simulation, plus an 8-episode cross-model probe on two additional frontier models. In several episodes across primary and cross-model runs, the agent produced a correct-enough final answer while its stated mechanism was wrong or unsupported by the data it had itself generated — outcome-correct, mechanism-wrong, and unaware of the gap.</p><p><b>Why it matters for MARA.</b> A concrete, instrumented version of the argument threaded through CausalGame (#3609) and the AI-Verify-Not-Judge position: judging a discovery system by its final answer rewards lucky pattern-matching and hides the failure that matters — a model that cannot say <i>why</i>, or misreports it. For MARA, where the deliverable is an explanatory world model and not a prediction, mechanism fidelity and epistemic honesty are the evaluation axes; this paper turns them into measurable quantities and shows frontier agents failing them even when they look right. The right companion to AutumnBench's insistence on probing the model, not scoring its output.</p>',
ws_pac:'<p><b>What it is.</b> A Compositional-Learning theory paper that rebuts a folk belief: symbolic regression is <i>not</i> statistically intractable just because the space of expressions grows combinatorially with depth. It analyzes compositional function trees over a finite vocabulary of smooth operators ({+, ×, sin, exp}, affine maps) through the PAC-learning lens.</p><p><b>The result.</b> The generalization quantity — Rademacher complexity, hence excess risk — does <i>not</i> blow up exponentially with the number of distinct symbolic structures. It is controlled by (i) the tree depth and (ii) the Lipschitz constants of the base operators along the computation graph. Under mild Lipschitz conditions and bounded affine leaves, a finite-union bound over the vocabulary plus Maurer vector-contraction gives risk bounds that scale gracefully with depth and vocabulary size. The key move is separating computational hardness (search is NP-hard) from statistical learnability (the class is PAC-learnable with modest sample complexity) — two things the “SR is unlearnable” claim conflates.</p><p><b>Evidence.</b> A codebase training differentiable operator trees (not MLPs) on synthetic physics-like targets of controlled depth confirms the empirical generalization gap tracks the predicted complexity term — <i>compositional stability</i>, not symbol count, governs generalization.</p><p><b>Why it matters for MARA.</b> The theoretical license for the whole discovery program: if the object of learning is an executable compositional expression (Autumn programs, the equations ModelSMC and SR-Scientist and Flash-ANSR recover), this says the statistics are on your side — you do not need exponentially many samples to identify the right structure, you need to control its depth and smoothness. It reframes the discovery problem's difficulty as a search problem, not a sample-complexity one, which is exactly where amortized inference and good priors earn their keep — and it is a clean rebuttal to anyone who dismisses program-structured world models as statistically hopeless.</p>',
ws_cc:'<p><b>What it is.</b> A Compositional-Learning paper framing <i>causal world models</i> as systems that answer counterfactuals — predict how an environment would have evolved had a subset of events gone differently — and building one, the Causal Cartographer, for the domain LLMs fail on: real-world causal reasoning beyond memorized relationships.</p><p><b>The approach.</b> Two agents. A graph retrieval-augmented generation agent extracts causal relationships from data and assembles them into a large network of real-world causal links — a reusable repository of causal knowledge from which real-world counterfactuals can be built. A counterfactual-reasoning agent then does step-by-step causal inference <i>constrained</i> by that retrieved graph, so the chain of reasoning is grounded in explicit causal structure rather than the model's free association.</p><p><b>Evidence.</b> Extracts causal knowledge and improves LLM robustness on causal-reasoning tasks while cutting inference cost and spurious correlations; it also tackles the standing evaluation problem — only the factual world is observed — by constructing its counterfactuals from the extracted graph rather than relying solely on synthetic benchmarks.</p><p><b>Why it matters for MARA.</b> A direct instance of the program's definition of a world model — one that supports intervention and counterfactuals, not just prediction — implemented by making the causal structure explicit and external rather than hoping it emerges in weights. The pattern (retrieve an explicit causal graph, then reason constrained by it) is the neuro-symbolic compromise MARA lives in: LLM for extraction and proposal, structured causal object as the belief that keeps inference honest. Sits naturally beside CausalGame (#3609) and the Illusion-of-Intervention critique — where those diagnose LLMs' causal failures, this proposes the explicit-structure fix.</p>'
};
const DAYS=[
{id:"tue",tab:"Tue 7",date:"2026-07-07",events:[
{id:"t1",s:510,e:570,fx:1,t:"Invited talk — Pascale Fung: Towards AI agents in the real world",v:"Plenary hall (streamed to overflow rooms)",c:"world",p:9,n:"The world-modeling manifesto keynote: agents grounded in physical and mental world models. Her group wrote the companion position paper below.",L:[["talk + bio",INV],["embodied AI agents paper",AX("2506.22355")]]},
{id:"t2",s:600,e:660,fx:1,t:"Oral 1E — Interpretability and cognition",v:"Grand Ballroom 101-105",c:"wild",p:6,n:"Cognition-adjacent wildcard; parallel with 1D below.",P:[["10:00 position: a science of AI must study learning dynamics",0],["10:15 behavioral cloning for scientific data annotation",0],["10:30 AI Engram: in search of memory traces in AI",0],["10:45 guaranteed optimal compositional explanations for neurons",0]],L:[["session page","https://icml.cc/virtual/2026/session/68639"]]},
{id:"t3",s:600,e:660,fx:1,t:"Oral 1D — Representations and distributions",v:"Hall D1",c:"bayes",p:5,n:"Density estimation and active sensing; parallel with 1E above.",P:[["10:00 DiScoFormer: plug-in density and score estimation with transformers",0],["10:15 LASER: learning active sensing for continuum field reconstruction",0],["10:30 multimodal nested learning",0],["10:45 Riemannian metric matching",0]],L:[["session page","https://icml.cc/virtual/2026/session/68638"]]},
{id:"t4",s:630,e:735,fx:0,t:"Poster session 1 — your curated hit-list, boards confirmed",v:"Hall A · 10:30–12:15",c:"team",p:10,n:"Nine flagged boards, all in this one session. The Schmidhuber one is the curiosity-theory fight of the week. FIRE-Bench is in the program but still unscheduled.",P:[["#108 — WorldCompass: RL post-training recipe for video world models",AX("2602.09022"),SUMS.wc],["#412 — MindZero: online mental reasoning with zero annotations — Shu lab",AX("2606.00240"),SUMS.mz],["#1509 — Temporal straightening for latent planning — LeCun lab","https://icml.cc/virtual/2026/poster/64904",SUMS.tstr],["#1608 — Verbalized Sampling: mode collapse and LLM diversity",AX("2510.01171"),SUMS.vs],["#1614 — DDP-WM: disentangled dynamics prediction for efficient world models","https://icml.cc/virtual/2026/poster/62480"],["#2117 — WebWorld: large-scale world model for web agents",PP(65352)],["#2202 — Agent World Model: infinity synthetic environments",AX("2602.10090"),SUMS.awm],["#2411 — GraphPFN: a prior-data fitted graph foundation model",AX("2509.21489"),SUMS.gpfn],["#2701 — Calibrated test-time guidance for Bayesian inference — Mandt lab","https://icml.cc/virtual/2026/poster/64440"],["#4107 — Interestingness as an inductive heuristic for future compression progress — Herrmann, Schmidhuber",AX("2605.14831"),SUMS.intr],["#4310 — From shortcuts to reasoning: theory-of-mind post-training with RL",PP(64668)],["FIRE-Bench: rediscovering scientific insights — no slot assigned yet",PP(61975)]],L:[["session page",SESS(68685)]]},
{id:"t5",s:780,e:840,fx:0,t:"Expo hall + hallway track",v:"COEX exhibit floor",c:"meta",p:4,n:"Short jet-lag-friendly block before your own poster session.",L:[["exhibitors","https://icml.cc/virtual/2026/sponsor_list"]]},
{id:"t9",s:840,e:945,fx:1,t:"YOUR POSTER — AutumnBench world-model benchmarking",v:"Hall A · board #4313 · Poster Session 2",c:"team",p:10,n:"Benchmarking world-model learning with environment-level queries, 14:00–15:45. Set the interactive Autumn demo up early. Same-session boards worth a lull-sweep below — the express-your-doubts position paper is bait made for you.",P:[["— same session, sweep during lulls —",0],["#509 — hippocampus-entorhinal inspired world model",PP(65751)],["#1709 — position: probabilistic world modeling (express your doubts)",PP(67197),SUMS.doubt],["#2617 — boosting world models via latent-space value alignment",PP(66589)],["#2704 — factored latent action world models",PP(61299)],["#4000 — SCOPE: evolving symbolic worlds, open-ended planning",PP(64265)]],L:[["poster page","https://icml.cc/virtual/2026/poster/64404"],["arXiv 2510.19788",AX("2510.19788")],["session page",SESS(68686)]]},
{id:"t7",s:810,e:870,fx:1,t:"Oral 2F — robotics and multi-agent RL",v:"Auditorium · 13:30–14:30",c:"robot",p:7,n:"Collides with your poster: RoboMME (memory for robot generalist policies) is at 14:00, exactly when your session starts. Catch the first half, then run to Hall A. Oral 2E (RL theory) is parallel in the Grand Ballroom. There is no 16:00 oral block on Tuesday.",P:[["13:30 human-robot collaboration via heterogeneous-agent Lyapunov policies",0],["13:45 optimal and scalable MAPF via multi-marginal optimal transport",0],["14:00 RoboMME: benchmarking memory for robotic generalist policies",0],["14:15 unsupervised partner design enables robust ad-hoc teamwork",0]],L:[["session page","https://icml.cc/virtual/2026/session/68647"]]},
{id:"t8",s:1110,e:1230,fx:0,t:"Opening-week socials",v:"Various — see socials list",c:"meta",p:3,n:"Affinity and community socials cluster early in the week.",L:[["socials list","https://icml.cc/virtual/2026/events/social"]]}]},
{id:"wed",tab:"Wed 8",date:"2026-07-08",events:[
{id:"w1",s:510,e:570,fx:1,t:"Invited talk — Sham Kakade",v:"Plenary hall",c:"wild",p:7,n:"Learning theory meets frontier scale.",L:[["talk + bio",INV]]},
{id:"w2",s:600,e:660,fx:1,t:"Morning orals — 3B is the pick (10:00–11:00)",v:"Hall B2 (3B) · Hall C (3A)",c:"bayes",p:6,n:"3B opens with Bayesian non-negative reward modeling against reward hacking and closes with optimal-transport preference RL. 3A diffusion in Hall C is the alternative.",P:[["3B 10:00 mitigating reward hacking in RLHF via Bayesian non-negative reward modeling",PP(65437),SUMS.o_brm],["3B 10:15 RL with evolving rubrics for deep research",PP(65886),SUMS.o_rubric],["3B 10:45 video-based optimal transport for preference-based RL",PP(65169),SUMS.o_vot],["3A 10:00 any-order GPT as masked diffusion — Hall C",0]],L:[["session 3B","https://icml.cc/virtual/2026/session/68650"],["session 3A","https://icml.cc/virtual/2026/session/68649"]]},
{id:"w3",s:630,e:735,fx:0,t:"Poster session 3 — world-model sweep",v:"Hall A · 10:30–12:15",c:"world",p:8,n:"Confirmed hits below; filter the app for more.",P:[["#402 — Robot policy evaluation with discrete diffusion world models",PP(65898),SUMS.rpe],["#612 — MVISTA-4D: view-consistent 4D world models for robotics",PP(65571),SUMS.mv4d],["#1007 — EmWorld: emotion world model with latent state evolution",PP(62080),SUMS.emw],["#1308 — LABO: LLM-accelerated Bayesian optimization",PP(64679),SUMS.labo],["#1315 — FUSE: FK-steered flow-matching posterior estimation (SBI)",PP(62622),SUMS.fuse],["Search boards in the app",APP],["— background reading, not presented at ICML —",0],["World models for robot learning — survey",AX("2605.00080")],["Hallucination in world models is predictable and preventable",AX("2606.27326")],["What-If World: a causal benchmark for embodied world models",AX("2605.27589")],["Awesome world models — running literature list","https://github.com/leofan90/awesome-world-models"]],L:[["session page",SESS(68687)]]},
{id:"w4",s:810,e:870,fx:1,t:"Invited talk — Aviv Regev",v:"Plenary hall",c:"wild",p:6,n:"AI for biology at frontier scale — kin to the Basis science mission.",L:[["talk + bio",INV]]},
{id:"w5",s:870,e:900,fx:1,t:"Test of Time award talk",v:"Plenary hall",c:"meta",p:7,n:"Thirty minutes, reliably a highlight.",L:[["award page","https://icml.cc/virtual/2026/test-of-time-award"]]},
{id:"w6",s:870,e:975,fx:0,t:"Poster session 4 — robot + world-model sweep",v:"Hall A · 14:30–16:15",c:"robot",p:8,n:"Confirmed hits below — the LLM-based model discovery one is core Basis territory. Benchmark everything against your Droid stack.",P:[["#811 — NOMAD: Bayesian memory-adaptive diffusion planning",PP(62636),SUMS.nomad],["#815 — building social world models with LLMs",PP(63462),SUMS.swm],["#2411 — WorldPlay: long-term geometric consistency (WorldCompass's base model)",PP(65111),SUMS.wpl],["#3512 — a probabilistic framework for LLM-based model discovery",PP(66508),SUMS.lmd],["Search boards in the app",APP],["— background reading, not presented at ICML (ExoPredicator is yours; VisualPredicator was ICLR 25) —",0],["ExoPredicator: abstract models of dynamic worlds for robot planning — Liang, Tavares et al",AX("2509.26255")],["VisualPredicator: neuro-symbolic predicates for robot planning",AX("2410.23156")],["DROID: in-the-wild robot manipulation dataset",AX("2403.12945")],["SLAP: shortcut learning for abstract planning",AX("2511.01107")]],L:[["session page",SESS(68688)]]},
{id:"w7",s:960,e:1020,fx:1,t:"Orals 4B / 4D / 4E — the week's strongest oral hour",v:"Hall B2 (4B) · Hall D1 (4D) · Grand Ballroom (4E) · 16:00–17:00",c:"robot",p:8,n:"Three relevant parallels: 4B robotics VLA (abstraction-to-instantiation behavioral representations, latent action supervision, XR-1), 4D causal + probabilistic (Bayesian hypergraph inference, latent confounders), 4E dynamical systems closing with training-free Bayesian filtering via generative emulators.",P:[["4B 16:00 from abstraction to instantiation: behavioral representations for VLA — Hall B2",PP(66596),SUMS.o_a2i],["4B 16:15 pixels to tokens: latent action supervision for VLA",PP(63621),SUMS.o_p2t],["4D 16:30 disentangling latent risk pathways via Bayesian hypergraph inference — Hall D1",PP(60908),SUMS.o_bhg],["4E 16:45 training-free Bayesian filtering with generative emulators — Grand Ballroom",PP(62217),SUMS.o_tfbf]],L:[["session 4B","https://icml.cc/virtual/2026/session/68657"],["session 4D","https://icml.cc/virtual/2026/session/68659"],["session 4E","https://icml.cc/virtual/2026/session/68660"]]},
{id:"w8",s:1020,e:1125,fx:0,t:"Poster session 5 — the robot world-model motherlode",v:"Hall A · 17:00–18:45",c:"robot",p:7,n:"The densest robot-world-model cluster of the week hides in the evening slot. Worth staying for.",P:[["#310 — RoboTwin 2.0: data generator and benchmark",PP(62192),SUMS.rtwin],["#813 — Plan-in-sandbox: physics-grounded abstraction",PP(63554),SUMS.pis],["#911 — Debate2Create: robot co-design via LLM debate",PP(66635),SUMS.d2c],["#1009 — do diffusion models dream of electric planes? (SBI)",PP(66800),SUMS.sbip],["#1601 — DiLA: disentangled latent action world models",PP(65662),SUMS.dila],["#2811 — Olaf-World: latent actions for video world modeling",PP(66245),SUMS.olaf],["#3815 — DreamDojo: real-time robot world model from human videos",PP(65193),SUMS.ddojo]],L:[["session page",SESS(68689)]]}]},
{id:"thu",tab:"Thu 9",date:"2026-07-09",events:[
{id:"h1",s:510,e:570,fx:1,t:"Invited talk — Verena Rieser",v:"Plenary hall",c:"wild",p:5,n:"Responsible multimodal dialogue.",L:[["talk + bio",INV]]},
{id:"h2",s:600,e:660,fx:1,t:"Oral 5D — evaluation and methodology (10:00–11:00)",v:"Hall D1 (5D) · Hall D2 (5C)",c:"wild",p:7,n:"5D opens with a position talk — there are futures that benchmark-driven AI cannot see — then CausalGame on causal thinking of LLM agents. Made for the AutumnBench worldview, and perfect prep for Narayanan at 13:30. 5C RL (distributional IRL) is next door in Hall D2.",P:[["5D 10:00 position: there are futures that benchmark-driven AI cannot see",PP(67149),SUMS.o_fut],["5D 10:15 CausalGame: benchmarking causal thinking of LLM agents in games",PP(63530),SUMS.o_cg],["5D 10:45 rare event analysis of large language models",PP(66583),SUMS.o_rare],["5C 10:00 distributional inverse reinforcement learning — Hall D2",PP(63146),SUMS.o_dirl]],L:[["session 5D","https://icml.cc/virtual/2026/session/68666"],["session 5C","https://icml.cc/virtual/2026/session/68665"]]},
{id:"h3",s:630,e:735,fx:0,t:"Poster session 6 — Bayes + probprog sweep",v:"Hall A · 10:30–12:15",c:"bayes",p:8,n:"Confirmed boards below; filter the app for more amortized-inference work.",P:[["#1000 — GIFT: image-to-CAD program synthesis",PP(64635),SUMS.gift],["#1800 — ToMAP: opponent-aware LLM persuaders with theory of mind",PP(65704),SUMS.tomap],["#2512 — Stein diffusion guidance: posterior correction",PP(64586),SUMS.stein],["#2604 — LIVE: long-horizon interactive video world modeling",PP(62249),SUMS.live],["#2707 — Amortized neural symbolic regression",PP(64015),SUMS.ansr],["#3612 — LILO: Bayesian optimization with natural-language feedback",PP(60511),SUMS.lilo],["— background reading, NOT at ICML (verified against all 6,796 accepted papers) —",0],["BayesFlow 2: amortized Bayesian inference in Python",AX("2602.07098")],["Incremental computation for programmable inference — Mansinghka lab",AX("2606.05348")]],L:[["session page",SESS(68690)]]},
{id:"h4",s:810,e:870,fx:1,t:"Invited talk — Arvind Narayanan",v:"Plenary hall",c:"wild",p:8,n:"The evaluation skeptic. A whetstone for how you frame AutumnBench claims.",L:[["talk + bio",INV]]},
{id:"h5",s:870,e:930,fx:1,t:"ICML town hall",v:"Plenary hall",c:"meta",p:3,n:"Community business; skippable.",L:[["thursday program","https://icml.cc/virtual/2026/day/7/9"]]},
{id:"h6",s:870,e:975,fx:0,t:"Poster session 7 — curiosity sweep, two confirmed boards",v:"Hall A · 14:30–16:15",c:"spark",p:7,n:"Two boards confirmed for this session: iWorld-Bench #3605 and A2RBench #2107 (auto-generated, formally verifiable ARC-style benchmarks). Then filter exploration, intrinsic, diversity, creative.",P:[["#314 — policy-driven world-model adaptation for offline model-based RL",PP(62695),SUMS.pdwm],["#1008 — Causal-JEPA: world models via object-level latent interventions",PP(63623),SUMS.cjepa],["#2107 — A2RBench: formally verifiable ARC-style benchmark generation",PP(61871),SUMS.a2r],["#3504 — Decrypto: multi-agent reasoning and theory-of-mind benchmark",PP(62135),SUMS.dcr],["#3605 — iWorld-Bench: interactive world models benchmark",PP(63894),SUMS.iwb],["— background reading, not presented at ICML —",0],["Intrinsically motivated humans and agents in open-world exploration",AX("2503.23631")],["Open-endedness is essential for artificial superhuman intelligence",AX("2406.04268")],["Regularity as intrinsic reward for free play — Martius lab",AX("2312.01473")],["CURIOUS: intrinsically motivated multi-goal RL — Oudeyer lab",AX("1810.06284")]],L:[["session page",SESS(68691)]]},
{id:"h8",s:960,e:1020,fx:1,t:"Oral 6B — agentic systems",v:"Hall B2 · 16:00–17:00",c:"world",p:7,n:"Agent evaluation in the wild to close the main conference. In 6A (Hall C): how much can language models memorize, and procedural pretraining on abstract data.",P:[["16:00 tau2-Bench: evaluating conversational agents in a dual-control environment",PP(64377),SUMS.o_tau],["16:15 characterizing agents in production",PP(61834),SUMS.o_prod],["16:45 OMAC: optimization framework for LLM multi-agent collaboration",PP(66164),SUMS.o_omac],["6A 16:45 procedural pretraining: warming up LMs with abstract data — Hall C",PP(63432),SUMS.o_pp]],L:[["session 6B","https://icml.cc/virtual/2026/session/68671"],["session 6A","https://icml.cc/virtual/2026/session/68670"]]},
{id:"h9",s:1020,e:1125,fx:0,t:"Poster session 8 — evening sweep",v:"Hall A · 17:00–18:45",c:"spark",p:5,n:"Last main-conference poster block; runs right up to dinner.",P:[["#204 — Knapsack RL: exploration budget allocation",PP(60948),SUMS.krl],["#903 — JADE: open-ended professional task evaluation",PP(63884),SUMS.jade],["#1006 — SphericalDreamer: navigable 3D worlds",PP(62181),SUMS.sphd],["#1912 — Grounding LLMs for scientific discovery in embodied settings",PP(62124),SUMS.glsd]],L:[["session page",SESS(68692)]]},
{id:"h7",s:1110,e:1200,fx:0,t:"Dinner with the MARA orbit",v:"Off-site, Gangnam",c:"team",p:6,n:"Tenenbaum, Ellis and Silver orbits are all in town — lock a dinner before workshop-day chaos. Paint the evening cells to schedule it."}]},
{id:"fri",tab:"Fri 10",date:"2026-07-10",events:[
{id:"f1",s:540,e:720,fx:1,t:"Structured probabilistic inference + generative modeling — AM",v:"Hall D1 · opening 9:20",c:"bayes",p:10,n:"Home turf. The flagship probprog and structured-inference day; its framing of ML as uncertainty, latent structure and decisions under noise reads like a Basis manifesto.",P:[["09:30 invited — Mingyuan Zhou",0],["10:20 invited — Jiatao Gu",0],["11:10 invited — Pilar Cossio (cryo-EM inference)",0],["11:40 poster session + lunch",0],["— workshop papers worth hunting —",0],["contributed — When are likely answers right? On sequence probability and correctness in LLMs",AX("2606.27359"),SUMS.w_like],["Neuro-Symbolic ODE Discovery with Latent Grammar Flow",AX("2604.16232"),SUMS.w_nsode],["contributed — Exact Posterior Score Estimation for Solving Linear Inverse Problems",0],["contributed — ABC: Any-Subset Autoregression via Non-Markovian Diffusion Bridges",0],["contributed — Enhanced Diffusion Sampling: efficient rare-event sampling and free-energy calculation",0],["Position: Multi-Agent LLM Simulation as Approximate Posterior Inference Demands a Probabilistic Calibration Standard",0],["Position: Benchmark Method-Comparisons Are Posterior Identifiability Problems",0],["Inverting Foundation Models of Brain Function with Simulation-Based Inference",AX("2604.23865"),SUMS.ws_brain],["Your Autoregressive Model Already Reveals the Causal Graph",0],["Probabilistic Chain-of-Thought: Sequential Bayesian Inference over Latent Reasoning Correctness",0],["Active Flow Expansion for Out-of-Distribution Discovery: from Theory to Molecules",AX("2606.08802"),SUMS.ws_actflow],["Your GFlowNet Secretly Learns an Optimal Transport Plan",AX("2606.06272")]],L:[["workshop page",W(54089)],["accepted papers","https://spigmworkshop2026.github.io/papers/"]]},
{id:"f2",s:810,e:1020,fx:1,t:"Structured probabilistic inference — PM",v:"Hall D1",c:"bayes",p:7,n:"Panel then two more invited talks; closes with a poster session.",P:[["13:30 panel discussion",0],["14:30 invited — Jona Balle",0],["15:30 invited — Soojung Yang",0],["16:10 closing poster session",0]],L:[["workshop page",W(54089)]]},
{id:"f3",s:480,e:720,fx:1,t:"RLxF: RL from world feedback — AM",v:"Grand Ballroom 101-102 · 8:00 start",c:"robot",p:8,n:"World-grounded reward signals for RL and robotics — maps onto objective design for the physical robot.",P:[["08:10 Benjamin Eysenbach + Cathy Ji — the geometry of empowerment",0],["08:40 Jerry Tworek — fifty shades of self-improvement",0],["10:30 invited — Brian Zhan",0],["11:00 poster session 1",0],["— workshop papers worth hunting —",0],["ThoughtTrace: understanding user thoughts in real-world LLM interactions (best paper)",0],["ECHO: terminal agents learn world models for free",0],["Self-Improving World Modelling with Latent Actions",0],["Reinforcing VLAs in Task-Agnostic World Models",0],["Learning from World Feedback: why model uncertainty fails as a risk signal in model-based RL",0],["MAVRL: learning reward functions from multiple feedback types with amortized variational inference",0],["Bayesian Preference Learning for test-time steerable reward models",0],["FutureSim: replaying world events to evaluate adaptive agents",0],["Training AI Co-Scientists using rubric rewards",0]],L:[["workshop page",W(54067)]]},
{id:"f4",s:780,e:1020,fx:1,t:"RLxF: RL from world feedback — PM",v:"Grand Ballroom 101-102",c:"robot",p:9,n:"Chelsea Finn at 13:30 and Raileanu on superhuman scientific discovery at 15:30 — the strongest robot-learning afternoon of the week.",P:[["13:00 Jesse Zhang — beyond human feedback: synthetic rewards",0],["13:30 Chelsea Finn",0],["14:00 poster session 2",0],["15:30 Roberta Raileanu — superhuman scientific discovery",0],["16:00 panel",0]],L:[["workshop page",W(54067)]]},
{id:"f5",s:540,e:720,fx:1,t:"Epistemic intelligence: unknown unknowns — AM",v:"Room E5-E6, 3rd floor · opening 9:00",c:"world",p:7,n:"Belief updating under unknown unknowns — the exact axis where humans beat frontier models on AutumnBench. Bring the paper.",P:[["10:00 Jeremie Houssineau — computing with ignorance: a possibilistic view",0],["10:50 Fazl Barez — understanding model behaviour in the age of AGI",0],["11:28 paper talk — diffusion as prior construction, a Bayesian approach",0],["— workshop-theme papers worth hunting (accepted list login-gated) —",0],["Position: Epistemic AI is Essential for Models to Know When They Do Not Know — Cuzzolin (workshop framing)",AX("2505.04950"),SUMS.w_epai],["Structured Ignorance Certificates: diagnosing unknown unknowns in reasoning models",AX("2606.08571"),SUMS.w_ign]],L:[["workshop page",W(54075)],["AutumnBench",AX("2510.19788")]]},
{id:"f6",s:780,e:1020,fx:1,t:"Epistemic intelligence — PM",v:"Room E5-E6",c:"world",p:7,n:"A paper talk on the posterior-prior gap when world-model imagination fails — directly on the AutumnBench axis.",P:[["13:50 paper talk — the posterior-prior epistemic gap: when world model imagination fails",0],["14:05 Belinda Li — solving the specification problem through interpretability",0],["15:30 panel discussion",0]],L:[["workshop page",W(54075)]]},
{id:"f7",s:480,e:720,fx:1,t:"Continual adaptation at scale — AM",v:"Room 327 · 8:00 start",c:"bayes",p:6,n:"Emtiyaz Khan organizing; a Bayesian lens on continually adapting foundation models.",P:[["08:05 Continual Learning Bench — frontier systems on stateful tasks",0],["10:40 Tinne Tuytelaars",0],["11:20 toward continually improving long-horizon agents",0]],L:[["workshop page",W(54079)],["EvolvingAgent: continual world model",AX("2502.05907")]]},
{id:"f8",s:780,e:1020,fx:1,t:"Culture x AI — PM",v:"Room 307",c:"spark",p:6,n:"AI as cultural technology — a constructive vision of machine creativity. (AM had Lauren Klein 10:30 and the what-do-we-want panel 11:00.)",P:[["13:15 Ted Underwood",0],["13:50 Maria Antoniak",0],["14:25 Joel Leibo",0],["15:30 respondents discussion",0]],L:[["workshop page",W(54085)]]},
{id:"f9",s:480,e:720,fx:1,t:"DL4C: human-centered coding agents — AM",v:"Hall B2",c:"wild",p:5,n:"Wen-Ding Li, your ARC co-author, co-organizes; program-synthesis energy. Hour-by-hour program not posted yet — check the page.",L:[["workshop page",W(54074)]]},
{id:"f10",s:780,e:1020,fx:1,t:"Generative + agentic AI for biology — PM",v:"Hall D2",c:"wild",p:5,n:"Their topics include world models for multi-scale biology. Program not posted yet.",L:[["workshop page",W(54073)]]},
{id:"f11",s:480,e:720,fx:1,t:"Mechanistic interpretability — AM",v:"Hall C · opening 9:30",c:"wild",p:4,n:"Pragmatic vs ambitious interpretability debate.",P:[["10:00 keynote",0],["10:30 spotlight talks",0],["11:00 poster session 1",0]],L:[["workshop page",W(54071)]]}]},
{id:"sat",tab:"Sat 11",date:"2026-07-11",events:[
{id:"s1",s:480,e:720,fx:1,t:"AI for Science: AI Scientists — AM",v:"Hall C · 8:00 start",c:"team",p:10,n:"MARA is everyday science. Welling, Zitnik and Mengdi Wang on the tool, co-author, founder spectrum for AI-driven discovery.",P:[["08:15 Peter Clark — opening invited talk",0],["rest of the hour-by-hour program not posted yet",0],["— workshop papers worth hunting —",0],["Position: AI Should Verify, Not Judge, Scientific Work",0],["The Novelty Ceiling: PAC-theoretic bounds on autonomous scientific discovery and the minimum oversight rate",0],["SR-Scientist: scientific equation discovery with agentic AI",AX("2510.11661"),SUMS.ws_srsci],["Practical Bayesian Optimization for Scientific Discovery — Church, Snoek",0],["Asking the Right Question: epistemic inquiry as a learnable reasoning skill — Mansouri, Gulcehre, Schwaller",0],["Large Language Models as Generative Bayesian Policies — Rankovic, Schwaller",0],["Sibyl: multi-agent literature-based scientific discovery, with temporal backtesting",0],["Propose, Critique, Falsify: benchmarking self-verifying AI scientists",0],["Position: Correct Answer, Wrong Mechanism — when AI scientists defend claims their data contradicts",AX("2606.23175"),SUMS.ws_cawm],["Enabling Robust Epidemic Control via LLM-Elicited Causal Discovery",0]],L:[["workshop page",W(54099)]]},
{id:"s2",s:780,e:1020,fx:1,t:"AI for Science: AI Scientists — PM",v:"Hall C",c:"team",p:7,n:"Evaluation standards and governance for autonomous discovery.",L:[["workshop page",W(54099)]]},
{id:"s3",s:480,e:720,fx:1,t:"LM4Plan: planning in the era of LMs — AM",v:"Grand Ballroom 101-102 · 8:00 start",c:"world",p:9,n:"Planning with LMs: what they contribute, what can be guaranteed — the reasoning end of world models.",P:[["08:05 reasoning with LLMs — challenges and opportunities",0],["08:50 LLM planning success",0],["10:00 (how) do LLMs plan? a sober look",0],["10:45 towards causal artificial intelligence",0],["11:40 oral presentations",0],["— workshop papers worth hunting —",0],["LM-Landmarks: LM-guided landmark generation for classical planning, with formal soundness guarantees (oral)",0],["Reward Prediction with Factorized World States (oral)",0],["ICPRL: acquiring physical intuition from interactive control (oral)",0],["VeryTrace: verifying reasoning traces through compilable formalism and structured verification",0],["Theory of Mind beyond persuasion: inducing belief states via planning and action",0],["OrigamiBench: an interactive environment to synthesize flat-foldable origamis (oral)",0]],L:[["workshop page",W(54095)],["LAW: language, agent and world models",AX("2312.05230")],["text world models for LLM agents",AX("2606.09032")]]},
{id:"s4",s:780,e:1020,fx:1,t:"LM4Plan — PM",v:"Grand Ballroom 101-102",c:"world",p:8,n:"Model collapse and test-time compute through a planning lens; poster session closes the day.",P:[["13:00 model collapse and its implications on planning with LLMs",0],["13:45 implications of large-scale test-time compute",0],["14:40 oral presentations",0],["16:00 poster session",0]],L:[["workshop page",W(54095)],["Qwen-AgentWorld: language world models",AX("2606.24597")]]},
{id:"s5",s:480,e:720,fx:1,t:"Human-AI co-creativity — AM",v:"Room 401 · 8:00 start",c:"spark",p:8,n:"Design fixation, idea homogeneity, authorship — your creativity thread, plus the classic below for ammunition.",P:[["08:15 Ben Zhao — challenges facing generative-AI users",0],["08:45 Luba Elliott — the search for originality in the age of mainstream AI",0],["10:00 panel discussion",0],["10:45 Hwajung Hong — beyond the average: orchestrating human-AI co-creation",0],["11:15 John Chung — diversified and explorative language models",0]],L:[["workshop page",W(54083)],["Driven by compression progress — Schmidhuber 2008",AX("0812.4360")]]},
{id:"s6",s:780,e:1020,fx:1,t:"Human-AI co-creativity — PM",v:"Room 401",c:"spark",p:7,n:"Creativity-as-learning and language-as-vehicle talks, then lightning talks into the poster session.",P:[["13:00 Pronita Mehrotra — creativity as learning: designing AI to explore",0],["13:30 Lonneke van der Plas — modelling language as a vehicle for creativity",0],["14:00 poster lightning talks",0],["15:45 poster session",0]],L:[["workshop page",W(54083)]]},
{id:"s7",s:540,e:720,fx:1,t:"Foundation models for structured data — AM",v:"Grand Ballroom 103 · opening 9:00",c:"bayes",p:8,n:"The PFN and tabular-FM crowd — the nearest thing here to a Bayesian-foundation-models venue. PM highlights if you stay: what tabular FMs learn 13:00, TabICLv2 15:00.",P:[["09:15 invited — multimodal time-series foundation models",0],["09:45 industry spotlights — AWS Chronos-2, SAP, Layer6 TabDPT, Prior Labs",0],["10:25 SurvivalPFN — in-context Bayesian survival prediction",0],["11:05 poster session",0],["— workshop papers worth hunting —",0],["Objective and data-driven Bayesian inference using TabPFN models",0],["Causal Foundation Models for Time Series based on Prior-Data Fitted Networks",0],["CausalTab: pretraining across causal environments for tabular causal discovery",0],["Do Tabular Foundation Models Learn Rules or Memorize Exemplars?",0],["What You Pretrain On Matters: synthetic task distributions determine tabular-FM quality",0],["Foundation Models for Partial Causal Identification",0],["On the Uncertainty in Prior-Data Fitted Network Pretraining",0],["Where Computation Lives Inside TabPFN: causal localisation of attention-head function",0]],L:[["workshop page",W(54066)],["PFN literature list","https://github.com/Cloudy1225/Awesome-Prior-Data-Fitted-Networks"]]},
{id:"s8",s:780,e:1020,fx:1,t:"Statistical uncertainty in agentic systems — PM",v:"Room E1-E4, 3rd floor",c:"bayes",p:7,n:"Anytime-valid inference, e-values, risk bounds for agent pipelines. The heavyweights are in the AM block (Ying Jin 8:15, Yarin Gal 10:00, Romano 10:35, Fisch 11:20) if you want to defect from a morning pick.",P:[["13:00 panel discussion",0],["13:35 Seong Joon Oh invited talk",0],["14:25 Andreas Vlachos invited talk",0],["— workshop papers worth hunting —",0],["Goal-Optimal Agents Necessarily Learn Predictive World Models in POMDPs with episodic resets",0],["The Illusion of Intervention: your LLM-simulated experiment is an observational study — Gretton, D'Amour",AX("2605.20767"),SUMS.ws_illus],["Conformal Policy Control — Prinster, Fannjiang, Saria, Stanton",0],["Betting on Equilibrium: monitoring strategic behavior in multi-agent systems — Bach, Jordan",0],["Why Hierarchical Structure Matters for UQ in Agentic AI Evaluation — Summerfield lab",0],["Trace Length is a Simple Uncertainty Signal in Reasoning Models",0],["Model-Free Assessment of Simulator Fidelity via Quantile Curves",0],["When to Trust the Cheap Check: weak and strong verification for reasoning",0]],L:[["workshop page",W(54054)]]},
{id:"s9",s:480,e:720,fx:1,t:"Philosophy meets ML — AM",v:"Room 308 · opening 8:20",c:"wild",p:7,n:"Schoelkopf, Icard and Fortuin on what counts as trustworthy — epistemology for ML. Deeply your kind of wildcard.",P:[["08:30 surrogate-metric evaluation is a causal-inference problem",0],["09:05 75 years of Turing's AGI",0],["10:00 a Bayesian epistemology for LLM evaluation",0],["10:35 thinking about humans in the era of AI",0]],L:[["workshop page",W(54060)]]},
{id:"s10",s:780,e:1020,fx:1,t:"Compositional learning — PM",v:"Auditorium",c:"spark",p:6,n:"Compositional generalization for agents — PoE-World adjacent. (AM opener at 8:30, learning to theorize the world from observation, is the most Basis-shaped title of the day.)",P:[["13:25 reasoning with LLMs — challenges and opportunities",0],["14:00 discovering interpretable cognitive models from animal and human behavior",0],["14:35 mechanistic interpretability for scientific data",0],["15:10 panel discussion",0],["— workshop papers worth hunting —",0],["Learning to Theorize the World from Observation — NEO, latent programs as a Language of Thought (spotlight)",AX("2605.03413"),SUMS.w_neo],["Learning What's Missing: failure-driven skill discovery via predicate bridges",0],["Sample Complexity of Scientific Discovery: PAC learnability of compositional function trees",AX("2606.29331"),SUMS.ws_pac],["Causal-JEPA: learning world models through object-level latent masking (also poster #1008)",0],["COGITAO: procedural, object-centric evaluation of compositional and systematic generalization",0],["Compositional Neuro-Symbolic Reasoning",0],["Causal Cartographer: from mapping to reasoning over counterfactual worlds",AX("2505.14396"),SUMS.ws_cc],["Structure over Pixels: learning variable-length visual programs",0],["HINT: task demonstrations for hierarchical inference in abstract reasoning",0]],L:[["workshop page",W(54096)]]},
{id:"s11",s:540,e:720,fx:1,t:"Hypothesis testing workshop — AM",v:"Room 318 · opening 9:00",c:"bayes",p:6,n:"Your agents literally run hypothesis tests; e-values could formalize the experiment loop.",P:[["09:05 keynote 1",0],["10:05 keynote 2",0],["11:20 keynote 3",0],["— workshop papers worth hunting —",0],["E-valuator: reliable agent verifiers with sequential hypothesis testing — Fannjiang, Regev et al",AX("2512.03109"),SUMS.ws_eval],["Making Generative Models Know What They Don't Know via Hypothesis Testing — Vergari lab",0],["Membership Circuits: tractable membership testing via probabilistic circuits — Kersting lab",0],["Prediction-Powered Active Testing — Rainforth, Caron",0],["Posterior-Driven Actor-Critic for Active Hypothesis Testing — Fields, Javidi",0],["Tweedie's Formula for Testing: score identities from tests to diffusion models",0],["Sequential Kernel-based Conditional Independence Testing via Adaptive Betting",0],["Quantifying Ranking Uncertainty in LLM Benchmarks — Neuhof, Benjamini",0]],L:[["workshop page",W(54092)]]},
{id:"s12",s:780,e:1020,fx:1,t:"Decision-making: offline to online — PM",v:"Room via app · 8:00 start",c:"wild",p:5,n:"Bayesian optimization, bandits and closed-loop experiment design. AM is defect-worthy: Sergey Levine 8:15, Aarti Singh 8:45, Jacob Gardner 10:30, BayesOpt-for-LLM-training orals 11:00.",P:[["13:30 orals — the three regimes of offline-to-online RL",0],["14:30 poster + coffee",0],["15:30 Clara Wong-Fannjiang invited talk",0],["16:00 Wen Sun invited talk",0],["— workshop papers worth hunting —",0],["Practical Bayesian Optimization for Scientific Discovery — Church, Snoek, Pehlevan (oral; also AI4Sci)",0],["Position: Offline-Dataset Evaluation for Online Decision-Making Needs an Identification Standard",0],["Hidden Failure Modes in Latent World-Model Planning from Offline Data (oral)",0],["The Three Regimes of Offline-to-Online RL — Li, Ni, Sun, Bacon (oral)",0],["Freeze the Policy, Infer the Goal: cross-domain imitation with world models",0],["SALT: state- and temporally-abstracted world models for offline long-horizon decision-making",0],["FICReg: forward-inverse consistency regularization for latent world models",0],["In-context Learning for Latent Space Bayesian Optimization — Lähdesmäki, Martinelli",0],["Abstraction for Offline Goal-Conditioned RL — Foerster, Osborne",0]],L:[["workshop page",W(54062)]]}]}];
const N=25,T0=480,S=150;
const defAvail={tue:[4,16],wed:[1,18],thu:[1,18],fri:[0,17],sat:[0,17]};
let state={day:"tue",pins:{},avail:{},pace:{},sumOpen:{}};
DAYS.forEach(d=>{
const a=Array(N).fill(false);
for(let i=defAvail[d.id][0];i<=defAvail[d.id][1];i++)a[i]=true;
state.avail[d.id]=a;state.pace[d.id]="balanced";
});
let todayId=null;
try{
const kst=new Intl.DateTimeFormat("en-CA",{timeZone:"Asia/Seoul"}).format(new Date());
const hit=DAYS.find(d=>d.date===kst);if(hit)todayId=hit.id;
if(todayId)state.day=todayId;
}catch(e){}
const $=id=>document.getElementById(id);
const ft=m=>Math.floor(m/60)+":"+String(m%60).padStart(2,"0");
const fh=h=>(Math.round(h*100)/100)+"h";
const LSK="icml26planner:v1";
function kstToday(){try{return new Intl.DateTimeFormat("en-CA",{timeZone:"Asia/Seoul"}).format(new Date());}catch(e){return"";}}
function saveState(){
try{localStorage.setItem(LSK,JSON.stringify({date:kstToday(),day:state.day,pins:state.pins,avail:state.avail,pace:state.pace,sumOpen:state.sumOpen}));}catch(e){}
}
function loadState(){
try{
const s=JSON.parse(localStorage.getItem(LSK)||"null");
if(!s)return;
if(s.pins)state.pins=s.pins;
if(s.sumOpen)state.sumOpen=s.sumOpen;
if(s.pace)DAYS.forEach(d=>{if(s.pace[d.id])state.pace[d.id]=s.pace[d.id];});
if(s.avail)DAYS.forEach(d=>{const a=s.avail[d.id];if(Array.isArray(a)&&a.length===N)state.avail[d.id]=a.map(Boolean);});
const valid=s.day&&DAYS.some(d=>d.id===s.day);
if(valid&&(s.date===kstToday()||!todayId))state.day=s.day;
}catch(e){}
}
function schedule(day){
const cells=state.avail[day.id];
const bar={light:7,balanced:5,packed:0}[state.pace[day.id]];
const eligible=e=>state.pins[e.id]||e.p>=bar;
const order=arr=>arr.sort((a,b)=>(state.pins[b.id]?1:0)-(state.pins[a.id]?1:0)||b.p-a.p||a.s-b.s);
const sel=[],skip=[],win=[];
const overlaps=(s,e)=>win.some(w=>w[0]<e&&s<w[1]);
const need=ev=>{
const r=[];
for(let i=0;i<N;i++){
const cs=T0+i*30;
const ov=Math.min(cs+30,ev.e)-Math.max(cs,ev.s);
if(ov>=20)r.push(i);
}
return r;
};
for(const e of order(day.events.filter(x=>x.fx))){
if(!eligible(e)){skip.push([e,"below pace bar"]);continue;}
if(!need(e).every(i=>cells[i])){skip.push([e,"outside painted hours"]);continue;}
if(overlaps(e.s,e.e)){skip.push([e,"clashes with a pick"]);continue;}
sel.push({e:e,s:e.s,en:e.e});win.push([e.s,e.e]);
}
const free=()=>{
const f=Array(S).fill(false);
for(let i=0;i<S;i++){
const m=T0+i*5;
if(cells[Math.floor((m-T0)/30)]&&!win.some(w=>m>=w[0]&&m<w[1]))f[i]=true;
}
return f;
};
for(const e of order(day.events.filter(x=>!x.fx))){
if(!eligible(e)){skip.push([e,"below pace bar"]);continue;}
const f=free();let bs=-1,bl=0,cs=-1;
for(let i=0;i<=S;i++){
const m=T0+i*5;
const ok=i<S&&f[i]&&m>=e.s&&m<e.e;
if(ok&&cs<0)cs=i;
if((!ok||i===S)&&cs>=0){const len=i-cs;if(len>bl){bl=len;bs=cs;}cs=-1;}
}
if(bl*5>=45){
const ws=T0+bs*5,we=Math.min(T0+(bs+bl)*5,e.e);
sel.push({e:e,s:ws,en:we});win.push([ws,we]);
}else{
skip.push([e,cells.some(x=>x)?"no free 45 min window":"outside painted hours"]);
}
}
const f=free();const gaps=[];let cs=-1;
for(let i=0;i<=S;i++){
const ok=i<S&&f[i];
if(ok&&cs<0)cs=i;
if((!ok||i===S)&&cs>=0){const len=i-cs;if(len*5>=60)gaps.push([T0+cs*5,T0+i*5]);cs=-1;}
}
sel.sort((a,b)=>a.s-b.s);
return{sel:sel,skip:skip,gaps:gaps};
}
function paperHTML(e){
if(!e.P)return"";
let bg=false;
return '<div class="papers">'+e.P.map((p,i)=>{
if(p[1]===0&&/^—.*—$/.test(p[0].trim())){
const lbl=p[0].trim().replace(/^—\s*/,"").replace(/\s*—$/,"");
if(/background reading/i.test(lbl))bg=true;
return '<div class="psec">'+lbl+"</div>";
}
if(p[1]===0)return "<div"+(bg?' class="bgrow"':"")+">"+p[0]+"</div>";
const u=p[1]||APP;
const lbl=p[1]?(p[1].indexOf("arxiv")>-1?"arXiv":"page"):"app";
const k=e.id+"-"+i,open=state.sumOpen[k];
return '<div'+(bg?' class="bgrow"':"")+">"+p[0]+' — <a href="'+u+'" target="_blank" rel="noopener">'+lbl+"</a>"+
(p[2]?' · <a href="#" class="sumtog" data-k="'+k+'">'+(open?"summary ▾":"summary ▸")+"</a>":"")+
"</div>"+(p[2]&&open?'<div class="psum">'+p[2]+"</div>":"");
}).join("")+"</div>";
}
function linkHTML(e){
if(!e.L)return"";
return '<div class="links">'+e.L.map(l=>'<a class="lnk" href="'+l[1]+'" target="_blank" rel="noopener">'+l[0]+"</a>").join("")+"</div>";
}
function rowHTML(it,mode,reason){
const e=(it.e&&it.e.id)?it.e:it;const cat=CATS[e.c];const pin=state.pins[e.id];
const time=mode==="skip"?ft(e.s)+"–"+ft(e.e):ft(it.s)+"–"+ft(it.en);
const clipped=(!e.fx&&mode!=="skip"&&(it.s>e.s||it.en<e.e))?"<small>of "+ft(e.s)+"–"+ft(e.e)+"</small>":"";
return '<div class="row '+(mode==="skip"?"skip":"")+(pin?" pinned":"")+'" data-id="'+e.id+'" data-z="'+(venueZone(e.v)||"")+'" role="button" tabindex="0" aria-pressed="'+(pin?"true":"false")+'">'+
'<div class="time">'+time+clipped+"</div>"+
"<div>"+
'<div class="tt"><span class="dot" style="background:'+cat.c+'"></span>'+e.t+"</div>"+
'<div class="meta1">'+e.v+" · "+cat.l+(pin?" · pinned":"")+(reason?" · "+reason:"")+"</div>"+
'<div class="note">'+e.n+"</div>"+
(mode==="skip"?paperHTML(e):paperHTML(e)+linkHTML(e))+
"</div></div>";
}
function gapHTML(g){
return '<div class="gap"><div class="time">'+ft(g[0])+"–"+ft(g[1])+'</div><div>Open block — lunch, meetings, hallway track</div></div>';
}
function renderTabs(){
$("tabs").innerHTML=DAYS.map(d=>'<button data-day="'+d.id+'" class="'+(d.id===state.day?"on":"")+'">'+d.tab+(d.id===todayId?" · today":"")+"</button>").join("");
document.querySelectorAll("#tabs button").forEach(b=>b.onclick=()=>{
state.day=b.getAttribute("data-day");
renderTabs();renderStrip();renderPace();renderPlan();
});
}
function renderPace(){
const p=state.pace[state.day];
$("pace").innerHTML=["light","balanced","packed"].map(k=>'<button data-pace="'+k+'" class="'+(k===p?"on":"")+'" style="margin-left:4px">'+k+"</button>").join("");
document.querySelectorAll("#pace button").forEach(b=>b.onclick=()=>{
state.pace[state.day]=b.getAttribute("data-pace");
renderPace();renderPlan();
});
}
let painting=false,paintTo=true;
function renderStrip(){
const cells=state.avail[state.day];
$("strip").innerHTML=cells.map((on,i)=>'<div class="cell '+(on?"on ":"")+(i%4===3?"hr":"")+'" data-i="'+i+'" aria-hidden="true"></div>').join("");
$("ruler").innerHTML=[0,4,8,12,16,20,24].map(k=>'<span style="left:'+(Math.round(k/N*1000)/10)+'%">'+(8+k/2)+":00</span>").join("");
document.querySelectorAll(".cell").forEach(el=>{
const i=+el.getAttribute("data-i");
el.onpointerdown=ev=>{ev.preventDefault();painting=true;paintTo=!state.avail[state.day][i];apply(el,i);};
el.onpointerenter=()=>{if(painting)apply(el,i);};
});
}
function apply(el,i){
state.avail[state.day][i]=paintTo;
el.classList.toggle("on",paintTo);
}
document.addEventListener("pointerup",()=>{if(painting){painting=false;renderPlan();}});
document.querySelectorAll("[data-pre]").forEach(b=>b.onclick=()=>{
const k=b.getAttribute("data-pre");const a=state.avail[state.day];
for(let i=0;i<N;i++)a[i]=k==="all"?true:k==="none"?false:k==="am"?i<=8:(i>=9&&i<=18);
renderStrip();renderPlan();
});
function togglePin(id){state.pins[id]=!state.pins[id];renderPlan();}
function venueZone(v){
const s=v.toLowerCase();
if(s.indexOf("off-site")>-1||s.indexOf("various")>-1)return null;
if(s.indexOf("expo")>-1||s.indexOf("exhibit")>-1)return null;
if(s.indexOf("hall a")>-1)return "A";
if(s.indexOf("hall b")>-1)return "B2";
if(s.indexOf("hall c")>-1||s.indexOf("plenary")>-1)return "C";
if(s.indexOf("hall d1")>-1)return "D1";
if(s.indexOf("hall d2")>-1)return "D2";
if(s.indexOf("auditorium")>-1)return "AUD";
if(s.indexOf("asem")>-1)return "ASEM";
if(s.indexOf("101")>-1)return "GB101";
if(s.indexOf("103")>-1)return "GB103";
if(s.indexOf("104")>-1)return "GB104";
if(s.indexOf("e1")>-1)return "E14";
if(s.indexOf("e5")>-1)return "E56";
if(/room 4\d\d/.test(s))return "R4";
if(/room 3\d\d/.test(s))return "R3";
return null;
}
function renderMap(sel){
document.querySelectorAll("#map .zone").forEach(g=>{
g.classList.remove("active");
g.querySelectorAll("circle").forEach(c=>c.remove());
});
const by={};
sel.forEach(it=>{const z=venueZone(it.e.v);if(z)(by[z]=by[z]||[]).push(it.e);});
for(const z in by){
const g=$("z-"+z);if(!g)continue;
g.classList.add("active");
const r=g.querySelector("rect");
const x0=+r.getAttribute("x"),y0=+r.getAttribute("y"),h=+r.getAttribute("height"),w=+r.getAttribute("width");
const cy=r.hasAttribute("data-dy")?y0+ +r.getAttribute("data-dy"):y0+h-11;
const maxd=Math.max(1,Math.min(6,Math.floor((w-16)/12)));
by[z].slice(0,maxd).forEach((e,i)=>{
const c=document.createElementNS("http://www.w3.org/2000/svg","circle");
c.setAttribute("cx",x0+13+i*12);c.setAttribute("cy",cy);c.setAttribute("r",3.5);
c.setAttribute("style","fill:"+CATS[e.c].c);
g.appendChild(c);
});
}
}
let lastSel=[];
function icsPad(n){return String(n).padStart(2,"0");}
function icsEsc(s){return s.replace(/\\/g,"\\\\").replace(/[,;]/g,m=>"\\"+m);}
$("ics").onclick=()=>{
const day=DAYS.find(d=>d.id===state.day);
if(!lastSel.length)return;
const d8=day.date.replace(/-/g,"");
const T=m=>d8+"T"+icsPad(Math.floor(m/60))+icsPad(m%60)+"00";
const L=["BEGIN:VCALENDAR","VERSION:2.0","PRODID:-//Basis//ICML2026 planner//EN"];
lastSel.forEach(it=>{
L.push("BEGIN:VEVENT","UID:icml26-"+day.id+"-"+it.e.id+"@basis.ai",
"DTSTART;TZID=Asia/Seoul:"+T(it.s),"DTEND;TZID=Asia/Seoul:"+T(it.en),
"SUMMARY:"+icsEsc(it.e.t),"LOCATION:"+icsEsc(it.e.v),
"DESCRIPTION:"+icsEsc(it.e.n||""),"END:VEVENT");
});
L.push("END:VCALENDAR");
const a=document.createElement("a");
a.href=URL.createObjectURL(new Blob([L.join("\r\n")],{type:"text/calendar"}));
a.download="icml2026-"+day.id+".ics";a.click();
setTimeout(()=>URL.revokeObjectURL(a.href),4000);
};
function renderPlan(){
const day=DAYS.find(d=>d.id===state.day);
const r=schedule(day);
const painted=state.avail[state.day].filter(Boolean).length/2;
const booked=r.sel.reduce((a,it)=>a+(it.en-it.s)/60,0);
$("sumtext").textContent=painted===0
?"no hours painted — a rest day. Seoul has excellent naengmyeon."
:r.sel.length+" stops · "+fh(Math.round(booked*100)/100)+" booked of "+fh(painted)+" painted";
$("legend").innerHTML=Object.values(CATS).map(c=>"<span><i style=\"background:"+c.c+"\"></i>"+c.l+"</span>").join("");
const items=[...r.sel.map(it=>({s:it.s,h:rowHTML(it,"sel")})),...r.gaps.map(g=>({s:g[0],h:gapHTML(g)}))].sort((a,b)=>a.s-b.s);
$("plan").innerHTML=items.map(x=>x.h).join("");
r.skip.sort((a,b)=>b[0].p-a[0].p);
$("skiphead").style.display=r.skip.length?"block":"none";
$("skipped").innerHTML=r.skip.map(x=>rowHTML(x[0],"skip",x[1])).join("");
document.querySelectorAll(".row").forEach(el=>{
el.onclick=ev=>{if(ev.target.tagName==="A")return;togglePin(el.getAttribute("data-id"));};
el.onkeydown=ev=>{if(ev.key==="Enter"||ev.key===" "){ev.preventDefault();togglePin(el.getAttribute("data-id"));}};
const z=el.getAttribute("data-z");
if(z){
el.onmouseenter=()=>{const g=$("z-"+z);if(g)g.classList.add("pulse");};
el.onmouseleave=()=>{const g=$("z-"+z);if(g)g.classList.remove("pulse");};
}
});
document.querySelectorAll(".sumtog").forEach(a=>a.onclick=ev=>{
ev.preventDefault();ev.stopPropagation();
const k=a.getAttribute("data-k");state.sumOpen[k]=!state.sumOpen[k];renderPlan();
});
lastSel=r.sel;
renderMap(r.sel);
saveState();
}
loadState();renderTabs();renderStrip();renderPace();renderPlan();
</script>
</body>
</html>
Workflows from the Neura Market marketplace related to this DeepSeek resource