-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathindex.html
More file actions
665 lines (642 loc) · 42 KB
/
Copy pathindex.html
File metadata and controls
665 lines (642 loc) · 42 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8" />
<meta name="viewport" content="width=device-width, initial-scale=1" />
<title>VINE: Taming Generative Control Policies for Reinforcement Learning</title>
<link rel="preconnect" href="https://fonts.googleapis.com" />
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin />
<link href="https://fonts.googleapis.com/css2?family=Source+Sans+3:wght@400;500;600;700&display=swap" rel="stylesheet" />
<link rel="stylesheet" href="style.css?v=20260915-visual-details" />
</head>
<body>
<header class="site-header">
<div class="container">
<span class="venue">VINE</span>
</div>
</header>
<main class="container">
<h1 class="paper-title"><span class="paper-title-mark">VINE:</span> Taming Generative Control Policies for Reinforcement Learning</h1>
<p class="paper-subtitle">Stable value-gradient reinforcement learning through the full iterative sampling chain</p>
<ul class="authors">
<li>Rushuai Yang<sup>1,2</sup></li>
<li>Zhuo Han<sup>1</sup></li>
<li>Houlin Li<sup>1</sup></li>
<li>Rui Zhang<sup>1</sup></li>
<li>Hecheng Wang<sup>1</sup></li>
<li>Zhichao Wu<sup>1</sup></li>
<li>Zhaowei Zhang<sup>3</sup></li>
<li>Zihong Chen<sup>1</sup></li>
<li>Xiaohan Yan<sup>1</sup></li>
<li>Chuankang Li<sup>1</sup></li>
<li>Guanghui Ren<sup>1</sup></li>
<li>Chiming Liu<sup>1</sup></li>
<li>Yi Chen<sup>2</sup></li>
<li>Wei Shan<sup>1</sup></li>
<li>Maoqing Yao<sup>1</sup></li>
</ul>
<p class="affiliations affiliations--compact"><sup>1</sup> AgiBot <sup>2</sup> The Hong Kong University of Science and Technology <sup>3</sup> Peking University</p>
<div class="action-links">
<a href="https://arxiv.org/pdf/2607.10369" class="btn btn-primary">📄 Paper </a>
<a href="https://github.com/AgibotTech/vine-code" class="btn btn-secondary">💻 Code (Coming Soon)</a>
</div>
<section class="abstract">
<h2>Abstract</h2>
<p>
Generative control policies use diffusion and flow matching to iteratively transform noise into actions, enabling complex action distributions. Direct reinforcement learning (RL) optimization can nevertheless be unstable when critic value gradients propagate through the full generation chain. We introduce <strong>VINE (Value-gradient Iterative Noise Exploration)</strong>, a flow-matching sampling strategy that reconstructs the generation path at each iteration with fresh Gaussian noise. VINE changes only the vanilla flow-matching sampler while retaining iterative generation and end-to-end value-gradient optimization through the entire sampling chain. Across OGBench, online adaptation of the pretrained <strong>3B-parameter π<sub>0.5</sub> VLA</strong> on LIBERO, and real-world plug insertion, VINE supports stable value-gradient RL in the evaluated settings.
</p>
<div class="hero-figure overview-figure">
<img src="images/vine_overview.png" alt="VINE overview: stable end-to-end value-gradient fine-tuning for flow-matching policies." loading="eager" onerror="this.src='https://placehold.co/960x400/1a1a2e/eee?text=VINE+Overview'; this.onerror=null;" />
<p class="caption"><strong>VINE</strong> reconstructs intermediate sampling states with fresh Gaussian noise, while propagating value gradients through the full iterative generation chain.</p>
</div>
</section>
<section class="algorithm-section" id="algorithm">
<h2>Algorithm</h2>
<ul class="algorithm-points">
<li><strong>Reconstruct and refine.</strong> VINE reconstructs each intermediate state from the current endpoint estimate and fresh Gaussian noise, then predicts the next endpoint with the flow-matching velocity network.</li>
<li><strong>Keep the sampling budget.</strong> Each action uses exactly <span class="algo-math">K</span> velocity-network evaluations, matching Euler at the same step count. The network architecture is unchanged.</li>
<li><strong>Optimize through the full chain.</strong> The critic fits Bellman targets using VINE-generated next actions. Actor updates propagate value gradients through every sampling transition, with behavior regularization when used.</li>
<li><strong>Reuse pretrained policies.</strong> VINE uses the existing flow-matching velocity field without an additional prediction network. A frozen-checkpoint evaluation on LIBERO-10 tests this compatibility empirically.</li>
</ul>
<details class="algorithm-disclosure">
<summary>Click to see the full algorithm</summary>
<div class="algorithm-panel">
<p class="algorithm-caption"><strong>VINE: Actor-Critic Training.</strong> Here, <span class="algo-math">s</span> and <span class="algo-math">a</span> are the environment state and action, <span class="algo-math">x<sub>k</sub></span> is the intermediate sampling state, and <span class="algo-math">K</span> is the number of velocity-network evaluations. The hat denotes an action-endpoint estimate.</p>
<ol class="algorithm-code">
<li><span class="kw">function</span> <span class="fn">Generate</span>(<span class="var">s</span>)</li>
<li class="indent-1"><span class="hvar">a</span><sub>0</sub> ∼ <span class="dist">N</span>(0, <span class="var">I</span><sub>d</sub>)</li>
<li class="indent-1"><span class="kw">for</span> <span class="var">k</span> = 0, …, <span class="var">K</span> − 1 <span class="kw">do</span><span class="algo-comment">▹ Iterative generation</span></li>
<li class="indent-2"><span class="var">t</span><sub>k</sub> ← <span class="var">k</span>/<span class="var">K</span></li>
<li class="indent-2 strike-euler"><span class="var">x</span><sub>k+1</sub> ← <span class="var">x</span><sub>k</sub> + 1/<span class="var">K</span> <span class="var">v</span><sub>θ</sub>(<span class="var">x</span><sub>k</sub>, <span class="var">t</span><sub>k</sub>; <span class="var">s</span>)<span class="algo-comment">▹ Euler update</span></li>
<li class="indent-2 vine-line"><span class="var">z</span><sub>k</sub> ∼ <span class="dist">N</span>(0, <span class="var">I</span><sub>d</sub>)<span class="algo-comment">▹ VINE update</span></li>
<li class="indent-2 vine-line"><span class="var">x</span><sub>k</sub> ← <span class="var">t</span><sub>k</sub> <span class="hvar">a</span><sub>k</sub> + (1 − <span class="var">t</span><sub>k</sub>) <span class="var">z</span><sub>k</sub></li>
<li class="indent-2 vine-line"><span class="hvar">a</span><sub>k+1</sub> ← <span class="var">x</span><sub>k</sub> + (1 − <span class="var">t</span><sub>k</sub>) <span class="var">v</span><sub>θ</sub>(<span class="var">x</span><sub>k</sub>, <span class="var">t</span><sub>k</sub>; <span class="var">s</span>)</li>
<li class="indent-1"><span class="kw">end for</span></li>
<li class="indent-1"><span class="kw">return</span> <span class="hvar">a</span><sub>K</sub></li>
<li><span class="kw">end function</span></li>
<li><span class="kw">while</span> not converged <span class="kw">do</span></li>
<li class="indent-1">Collect transitions with <span class="fn">Generate</span>(<span class="var">s</span>) and add to <span class="dist">D</span><span class="algo-comment">▹ Optionally for online RL</span></li>
<li class="indent-1">Sample batch {(s, a, r, s′)} ∼ <span class="dist">D</span></li>
<li class="indent-1"><span class="var">a′</span> ← <span class="fn">Generate</span>(<span class="var">s′</span>)</li>
<li class="indent-1">Update <span class="var">φ</span> to minimize E[(<span class="var">Q</span><sub>φ</sub>(s, a) − r − γ <span class="var">Q</span><sub>φ̄</sub>(s′, a′))<sup>2</sup>]<span class="algo-comment">▹ Train critic Q<sub>φ</sub></span></li>
<li class="indent-1"><span class="hvar">a</span><sub>K</sub> ← <span class="fn">Generate</span>(<span class="var">s</span>)</li>
<li class="indent-1">Update <span class="var">θ</span> to minimize −<span class="var">Q</span><sub>φ</sub>(s, <span class="hvar">a</span><sub>K</sub>) + α‖<span class="hvar">a</span><sub>K</sub> − a‖<sup>2</sup><span class="algo-comment">▹ Train velocity v<sub>θ</sub> via BPTT</span></li>
<li><span class="kw">end while</span></li>
<li><span class="kw">return</span> policy <span class="var">π</span><sub>θ</sub>(s) ≡ <span class="fn">Generate</span>(<span class="var">s</span>)</li>
</ol>
<p class="section-desc">Terminal critic targets retain the observed reward and omit the next-state value term. During actor updates, critic parameters are fixed while their action gradients pass through every endpoint prediction and reconstruction; no intermediate state is detached.</p>
</div>
</details>
</section>
<section class="task-demos-and-results toy-section" aria-labelledby="toy-simulation">
<h2 id="toy-simulation">Toy Simulation: BC vs. RL</h2>
<p class="section-desc">We compare DDPM, Euler flow matching, and <strong>VINE</strong> on a two-dimensional five-mode action distribution with different rewards. The first row shows behavior-cloned policies; the second shows offline RL fine-tuning with a learned critic. In the paper, DDPM and Euler flow matching eventually produce increasingly large off-target actions, whereas VINE retains concentrated endpoints while shifting toward the highest-reward mode. The appendix additionally reports direct optimization of a known reward without a learned critic.</p>
<div class="td3bc-comparison" role="region" aria-label="BC and RL video comparison by sampler">
<table class="td3bc-grid" aria-label="Sampler comparison: BC in the first row and RL in the second row">
<colgroup>
<col class="td3bc-stage-column" />
<col span="3" />
</colgroup>
<thead>
<tr>
<th scope="col">Training</th>
<th scope="col">DDPM</th>
<th scope="col">Flow Matching (Euler)</th>
<th scope="col">VINE</th>
</tr>
</thead>
<tbody>
<tr>
<th scope="row">BC</th>
<td data-label="DDPM">
<figure>
<video src="videos/td3bc_experiments/ddpm_bc.mp4?v=20260709-2230" aria-label="DDPM after behavior cloning" controls muted playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption>Diffusion policy after behavior cloning.</figcaption>
</figure>
</td>
<td data-label="Flow Matching (Euler)">
<figure>
<video src="videos/td3bc_experiments/fm_bc.mp4?v=20260709-2230" aria-label="Flow Matching with Euler sampling after behavior cloning" controls muted playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption>Flow-matching policy with Euler sampling after behavior cloning.</figcaption>
</figure>
</td>
<td data-label="VINE">
<figure>
<video src="videos/td3bc_experiments/vine_bc.mp4?v=20260709-2230" aria-label="VINE sampling after behavior cloning" controls muted playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption>Flow-matching policy with <strong>VINE</strong> sampling after behavior cloning.</figcaption>
</figure>
</td>
</tr>
<tr>
<th scope="row">RL</th>
<td data-label="DDPM">
<figure>
<video src="videos/td3bc_experiments/ddpm_rl.mp4?v=20260709-2230" aria-label="DDPM after TD3+BC fine-tuning" controls muted playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption>Diffusion policy after value-gradient fine-tuning.</figcaption>
</figure>
</td>
<td data-label="Flow Matching (Euler)">
<figure>
<video src="videos/td3bc_experiments/fm_rl.mp4?v=20260709-2230" aria-label="Flow Matching with Euler sampling after TD3+BC fine-tuning" controls muted playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption>Flow-matching policy with Euler sampling after fine-tuning.</figcaption>
</figure>
</td>
<td data-label="VINE">
<figure>
<video src="videos/td3bc_experiments/vine_rl.mp4?v=20260709-2230" aria-label="VINE after TD3+BC fine-tuning" controls muted playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption><strong>VINE</strong> policy after value-gradient fine-tuning.</figcaption>
</figure>
</td>
</tr>
</tbody>
</table>
</div>
</section>
<section class="task-demos-and-results" id="experiments">
<h2>Experiments</h2>
<h3 class="subsection-title" id="real-world-experiments">Real-World Experiments</h3>
<p class="section-desc">Real-world manipulation with <strong>VINE</strong>, from precise insertion to dual-arm and tabletop tasks.</p>
<div class="real-world-showcase" aria-label="Four real-world manipulation demonstrations">
<figure>
<video src="videos/screw_insertion.mp4?v=20260710-1620" poster="images/real_world/screw_insertion.jpg" aria-label="Screw insertion" controls muted loop playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption><strong>Screw insertion</strong><span>Training time: <strong class="training-result">VINE — 5 min</strong> vs. HIL-SERL — 1 h.</span></figcaption>
</figure>
<figure>
<video src="videos/insertion.mp4" poster="images/real_world/insertion.jpg" aria-label="Plug insertion" controls muted loop playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption><strong>Plug insertion</strong><span>Training time: <strong class="training-result">VINE — 20 min</strong> vs. HIL-SERL — 1 h 40 min.</span></figcaption>
</figure>
<figure>
<video src="videos/real_world/VID_20260818_154320.mp4" poster="images/real_world/VID_20260818_154320.jpg" aria-label="Dual-arm panel pressing" controls muted loop playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption><strong>Dual-arm panel pressing</strong><span>Training time: <strong class="training-result">VINE — 1 h 1 min</strong> vs. HIL-SERL — 2 h 2 min.</span></figcaption>
</figure>
<figure>
<video src="videos/real_world/VID20260810171452.mp4" poster="images/real_world/VID20260810171452.jpg" aria-label="Material placement" controls muted loop playsinline autoplay disablePictureInPicture preload="metadata"></video>
<figcaption><strong>Material placement</strong><span>Online RL: <strong class="training-result">VINE-π<sub>0.5</sub> — 5 h</strong>; success rate: <strong>30% → 70%</strong>.</span></figcaption>
</figure>
</div>
<h3 class="subsection-title">Online Real-World Socket Insertion</h3>
<p class="section-desc">The robot must pick up a plug from the table and insert it into a fixed socket. All methods use images and proprioceptive observations to predict low-level actions or action chunks; human operators may intervene and provide corrective actions during online training. After <strong>20 minutes of online RL</strong>, VINE succeeds in <strong>20/20 evaluation trials</strong> and has a <strong>19.2%</strong> human-intervention ratio.</p>
<h3 class="subsection-title">Real-World Rollout Comparison</h3>
<p class="section-desc">Selected <strong>VINE</strong> successes and baseline failures on socket insertion. These clips illustrate behavior; the table below reports success over 20 evaluation trials per method.</p>
<h4 class="demo-row-title">VINE Success Cases</h4>
<div class="media-row-scroll" aria-label="VINE success rollout clips">
<figure>
<video src="videos/vine_success/vine2_23m34s_23m40s.mp4" controls muted loop playsinline autoplay disablePictureInPicture></video>
<figcaption><strong>VINE success case.</strong> The policy aligns the plug and completes the insertion.</figcaption>
</figure>
<figure>
<video src="videos/vine_success/vine2_27m44s_27m49s.mp4" controls muted loop playsinline autoplay disablePictureInPicture></video>
<figcaption><strong>VINE success case.</strong> Another successful rollout with stable contact-rich insertion.</figcaption>
</figure>
</div>
<h4 class="demo-row-title">Baseline Failure Cases</h4>
<div class="media-row-scroll" aria-label="Baseline failure rollout clips">
<figure>
<video src="videos/baseline_failure/sac-flow0706-eval_first29s.mp4" controls muted loop playsinline autoplay disablePictureInPicture></video>
<figcaption><strong>SAC-Flow failure case.</strong> The policy approaches the socket but cannot complete the insertion.</figcaption>
</figure>
<figure>
<video src="videos/baseline_failure/rlt-eval_58s-1m17s.mp4" controls muted loop playsinline autoplay disablePictureInPicture></video>
<figcaption><strong>RLT failure case.</strong> The rollout stalls before achieving a stable insertion.</figcaption>
</figure>
<figure>
<video src="videos/baseline_failure/expo_eval_3m24s-3m45s.mp4" controls muted loop playsinline autoplay disablePictureInPicture></video>
<figcaption><strong>EXPO failure case.</strong> The policy misses the precise alignment needed for contact-rich insertion.</figcaption>
</figure>
<figure>
<video src="videos/baseline_failure/dsrl_3m27s-3m55s.mp4" controls muted loop playsinline autoplay disablePictureInPicture></video>
<figcaption><strong>DSRL failure case.</strong> The robot fails to recover from misalignment during insertion.</figcaption>
</figure>
</div>
<h3 class="subsection-title">Online Learning with Human Feedback</h3>
<p class="section-desc">This recording shows VINE training on real-world socket insertion. A human operator can intervene during autonomous execution and provide corrective actions. The evaluation below reports task success, online training time, and intervention frequency separately.</p>
<div class="video-progression">
<figure>
<video src="videos/vine_fromscratch.mp4" controls muted loop playsinline autoplay disablePictureInPicture></video>
<figcaption>Online RL training with <strong>VINE</strong> on real-world socket insertion.</figcaption>
</figure>
<table class="metric-table" aria-label="Real-world socket insertion protocol">
<thead>
<tr>
<th>Protocol</th>
<th>Setting</th>
</tr>
</thead>
<tbody>
<tr>
<td>Task</td>
<td>Pick up a plug and insert it into a fixed socket.</td>
</tr>
<tr>
<td>Observations</td>
<td>Images and proprioceptive states.</td>
</tr>
<tr>
<td>Actions</td>
<td>End-effector control.</td>
</tr>
<tr>
<td>Human feedback</td>
<td>Corrective actions during online rollouts.</td>
</tr>
<tr>
<td>Evaluation</td>
<td>20 trials per method after the listed training budget.</td>
</tr>
</tbody>
</table>
</div>
<div class="mix-viz">
<table class="metric-table robot-results-table" aria-label="Real-world online RL socket insertion results">
<thead>
<tr>
<th scope="col">Method</th>
<th scope="col">Success (↑)</th>
<th scope="col">Online RL time</th>
<th scope="col">Human intervention (↓)</th>
</tr>
</thead>
<tbody>
<tr>
<th scope="row">BC init.</th>
<td data-label="Success">10/20</td>
<td data-label="Online RL time">--</td>
<td data-label="Human intervention">--</td>
</tr>
<tr>
<th scope="row">DSRL</th>
<td data-label="Success">17/20</td>
<td data-label="Online RL time">50 min</td>
<td data-label="Human intervention">--</td>
</tr>
<tr>
<th scope="row">RLT</th>
<td data-label="Success">17/20</td>
<td data-label="Online RL time">50 min</td>
<td data-label="Human intervention">29.7%</td>
</tr>
<tr>
<th scope="row">EXPO</th>
<td data-label="Success">19/20</td>
<td data-label="Online RL time">50 min</td>
<td data-label="Human intervention">56.3%</td>
</tr>
<tr>
<th scope="row">SAC-Flow</th>
<td data-label="Success">12/20</td>
<td data-label="Online RL time">20 min</td>
<td data-label="Human intervention">74.6%</td>
</tr>
<tr>
<th scope="row">HIL-SERL</th>
<td data-label="Success">16/20</td>
<td data-label="Online RL time">20 min</td>
<td data-label="Human intervention">55.9%</td>
</tr>
<tr class="method-highlight">
<th scope="row"><strong>VINE</strong></th>
<td data-label="Success"><strong>20/20</strong></td>
<td data-label="Online RL time"><strong>20 min</strong></td>
<td data-label="Human intervention"><strong>19.2%</strong></td>
</tr>
</tbody>
</table>
</div>
<p class="section-desc">
Success is measured over 20 evaluation trials. Human-intervention ratio is the fraction of online rollout steps under human control. Training times are the reported evaluation budgets, not times to a common convergence threshold. A dash denotes an unreported or inapplicable entry.
</p>
<h3 class="subsection-title" id="libero-experiments">Multi-Task Online RL with a Pretrained VLA</h3>
<p class="section-desc">We adapt the <strong>3B-parameter π<sub>0.5</sub></strong> VLA on LIBERO-10, LIBERO-Spatial, LIBERO-Object, and LIBERO-Goal, each with ten tabletop manipulation tasks. All variants share the same few-shot SFT initialization and differ only in the sampler: deterministic ODE, stochastic SDE, or VINE. In the evaluated setting, <strong>VINE yields more consistent policy improvement</strong>, while the ODE and SDE variants are unstable during multi-task fine-tuning. Each curve reports success averaged over the ten tasks in its suite.</p>
<figure class="hero-figure libero-figure">
<a href="images/libero_rl.pdf" target="_blank" rel="noopener">
<img src="images/libero_rl.png" alt="Online RL adaptation of π0.5 on LIBERO-10, Spatial, Object, and Goal, comparing ODE, SDE, and VINE sampling" width="3000" height="1010" loading="lazy" />
</a>
<figcaption class="caption"><strong>Online RL adaptation on LIBERO.</strong> Success is averaged over the ten tasks in each suite. Faint lines show the original observations; bold lines show five-point trailing means. Click the figure to open the original PDF.</figcaption>
</figure>
<p class="section-desc">In a separate frozen-checkpoint evaluation on LIBERO-10, Euler and VINE achieve <strong>44.34%</strong> and <strong>45.26%</strong> success, respectively, across ten evaluation seeds with 500 episodes per seed and sampler. These similar observed rates provide empirical evidence that the pretrained flow-matching checkpoint can be directly reused with VINE.</p>
<h3 class="subsection-title">Offline RL on OGBench</h3>
<p class="section-desc">We evaluate <strong>50 state-based tasks across ten OGBench domains</strong>, spanning long-horizon navigation and sparse-reward manipulation. Each task is trained for <strong>1M gradient updates with 12 random seeds</strong>; results report mean success with 95% confidence intervals.</p>
<div class="media-row media-row-wide final-results">
<figure class="plot-figure">
<a href="images/offline_rl_50_tasks.png" target="_blank" rel="noopener">
<img src="images/offline_rl_50_tasks.png" alt="Aggregate success on 50 OGBench tasks with 95% confidence intervals" loading="lazy" onerror="this.src='https://placehold.co/960x480/fafaf9/57534e?text=Offline+RL+Results'; this.onerror=null;" />
</a>
</figure>
</div>
<details class="algorithm-disclosure offline-domains-disclosure">
<summary>Click to expand offline results across 10 domains (50 tasks)</summary>
<div class="algorithm-panel offline-domains-panel">
<p class="offline-domains-caption"><strong>Offline results, 10 domains.</strong> Domain-level and overall mean success (%) with 95% confidence intervals in brackets, as reported in the paper. Each domain contains five tasks, with 12 training seeds per task. BPTT-free methods avoid propagating value gradients through the full sampling chain; Value-Gradient BPTT methods retain this optimization path.</p>
<div class="table-figure-scroll">
<table class="metric-table" aria-label="Current manuscript results across 50 OGBench tasks and ten domains">
<thead>
<tr>
<th rowspan="2" scope="col">Task category</th>
<th colspan="1" scope="colgroup">Gaussian</th>
<th colspan="11" scope="colgroup">BPTT-free</th>
<th colspan="3" scope="colgroup">Value-Gradient BPTT</th>
</tr>
<tr>
<th scope="col">ReBRAC</th>
<th scope="col">FQL</th>
<th scope="col">FAWAC</th>
<th scope="col">IFQL</th>
<th scope="col">QAM</th>
<th scope="col">DAC</th>
<th scope="col">QSM</th>
<th scope="col">CGQL</th>
<th scope="col">CGQL-M</th>
<th scope="col">CGQL-L</th>
<th scope="col">DSRL</th>
<th scope="col">FEdit</th>
<th scope="col">FBRAC</th>
<th scope="col">BAM</th>
<th scope="col">VINE</th>
</tr>
</thead>
<tbody>
<tr>
<th scope="row">antmaze-large</th>
<td data-label="ReBRAC">94 <small>[94, 95]</small></td>
<td data-label="FQL">76 <small>[72, 79]</small></td>
<td data-label="FAWAC">17 <small>[15, 19]</small></td>
<td data-label="IFQL">36 <small>[32, 39]</small></td>
<td data-label="QAM">81 <small>[78, 84]</small></td>
<td data-label="DAC">88 <small>[86, 90]</small></td>
<td data-label="QSM">91 <small>[89, 93]</small></td>
<td data-label="CGQL">76 <small>[73, 80]</small></td>
<td data-label="CGQL-M">71 <small>[68, 73]</small></td>
<td data-label="CGQL-L">65 <small>[62, 67]</small></td>
<td data-label="DSRL">61 <small>[56, 66]</small></td>
<td data-label="FEdit">58 <small>[53, 62]</small></td>
<td data-label="FBRAC">2 <small>[1, 4]</small></td>
<td data-label="BAM">84 <small>[82, 85]</small></td>
<td data-label="VINE"><strong>99</strong> <small>[98, 100]</small></td>
</tr>
<tr>
<th scope="row">antmaze-giant</th>
<td data-label="ReBRAC">57 <small>[53, 60]</small></td>
<td data-label="FQL">0 <small>[0, 0]</small></td>
<td data-label="FAWAC">0 <small>[0, 0]</small></td>
<td data-label="IFQL">1 <small>[0, 2]</small></td>
<td data-label="QAM">18 <small>[14, 22]</small></td>
<td data-label="DAC">16 <small>[11, 21]</small></td>
<td data-label="QSM">15 <small>[11, 17]</small></td>
<td data-label="CGQL">0 <small>[0, 2]</small></td>
<td data-label="CGQL-M">4 <small>[1, 8]</small></td>
<td data-label="CGQL-L">3 <small>[1, 6]</small></td>
<td data-label="DSRL">3 <small>[1, 4]</small></td>
<td data-label="FEdit">2 <small>[1, 3]</small></td>
<td data-label="FBRAC">0 <small>[0, 0]</small></td>
<td data-label="BAM">1 <small>[0, 2]</small></td>
<td data-label="VINE"><strong>76</strong> <small>[68, 83]</small></td>
</tr>
<tr>
<th scope="row">humanoidmaze-medium</th>
<td data-label="ReBRAC">69 <small>[65, 74]</small></td>
<td data-label="FQL">68 <small>[63, 73]</small></td>
<td data-label="FAWAC">24 <small>[22, 26]</small></td>
<td data-label="IFQL">86 <small>[85, 87]</small></td>
<td data-label="QAM">67 <small>[64, 69]</small></td>
<td data-label="DAC">83 <small>[81, 85]</small></td>
<td data-label="QSM">83 <small>[80, 86]</small></td>
<td data-label="CGQL">60 <small>[57, 62]</small></td>
<td data-label="CGQL-M">42 <small>[40, 43]</small></td>
<td data-label="CGQL-L">62 <small>[57, 67]</small></td>
<td data-label="DSRL">53 <small>[48, 57]</small></td>
<td data-label="FEdit">22 <small>[20, 23]</small></td>
<td data-label="FBRAC">39 <small>[37, 41]</small></td>
<td data-label="BAM">60 <small>[58, 62]</small></td>
<td data-label="VINE"><strong>87</strong> <small>[79, 93]</small></td>
</tr>
<tr>
<th scope="row">humanoidmaze-large</th>
<td data-label="ReBRAC">17 <small>[15, 20]</small></td>
<td data-label="FQL">9 <small>[7, 11]</small></td>
<td data-label="FAWAC">0 <small>[0, 0]</small></td>
<td data-label="IFQL">24 <small>[21, 27]</small></td>
<td data-label="QAM">11 <small>[9, 14]</small></td>
<td data-label="DAC">0 <small>[0, 0]</small></td>
<td data-label="QSM">10 <small>[9, 11]</small></td>
<td data-label="CGQL">5 <small>[4, 5]</small></td>
<td data-label="CGQL-M">6 <small>[3, 8]</small></td>
<td data-label="CGQL-L">6 <small>[5, 8]</small></td>
<td data-label="DSRL">3 <small>[2, 5]</small></td>
<td data-label="FEdit">3 <small>[2, 3]</small></td>
<td data-label="FBRAC">0 <small>[0, 0]</small></td>
<td data-label="BAM">5 <small>[4, 8]</small></td>
<td data-label="VINE"><strong>46</strong> <small>[36, 56]</small></td>
</tr>
<tr>
<th scope="row">scene-sparse</th>
<td data-label="ReBRAC">65 <small>[61, 69]</small></td>
<td data-label="FQL">78 <small>[77, 80]</small></td>
<td data-label="FAWAC">38 <small>[35, 41]</small></td>
<td data-label="IFQL">84 <small>[83, 85]</small></td>
<td data-label="QAM">97 <small>[96, 98]</small></td>
<td data-label="DAC">68 <small>[65, 70]</small></td>
<td data-label="QSM">86 <small>[84, 87]</small></td>
<td data-label="CGQL">38 <small>[36, 40]</small></td>
<td data-label="CGQL-M">74 <small>[72, 76]</small></td>
<td data-label="CGQL-L">88 <small>[85, 91]</small></td>
<td data-label="DSRL"><strong>99</strong> <small>[99, 100]</small></td>
<td data-label="FEdit">62 <small>[60, 65]</small></td>
<td data-label="FBRAC">50 <small>[43, 57]</small></td>
<td data-label="BAM">98 <small>[97, 99]</small></td>
<td data-label="VINE">73 <small>[60, 85]</small></td>
</tr>
<tr>
<th scope="row">puzzle-3x3-sparse</th>
<td data-label="ReBRAC">79 <small>[73, 84]</small></td>
<td data-label="FQL">70 <small>[60, 78]</small></td>
<td data-label="FAWAC">3 <small>[2, 3]</small></td>
<td data-label="IFQL"><strong>100</strong> <small>[100, 100]</small></td>
<td data-label="QAM"><strong>100</strong> <small>[99, 100]</small></td>
<td data-label="DAC">68 <small>[62, 75]</small></td>
<td data-label="QSM">53 <small>[49, 57]</small></td>
<td data-label="CGQL">48 <small>[39, 55]</small></td>
<td data-label="CGQL-M"><strong>100</strong> <small>[100, 100]</small></td>
<td data-label="CGQL-L">90 <small>[83, 96]</small></td>
<td data-label="DSRL">87 <small>[82, 92]</small></td>
<td data-label="FEdit">99 <small>[98, 100]</small></td>
<td data-label="FBRAC">0 <small>[0, 1]</small></td>
<td data-label="BAM">56 <small>[48, 64]</small></td>
<td data-label="VINE">95 <small>[93, 97]</small></td>
</tr>
<tr>
<th scope="row">puzzle-4x4-100M-sparse</th>
<td data-label="ReBRAC">0 <small>[0, 0]</small></td>
<td data-label="FQL">5 <small>[3, 7]</small></td>
<td data-label="FAWAC">0 <small>[0, 0]</small></td>
<td data-label="IFQL">0 <small>[0, 0]</small></td>
<td data-label="QAM">0 <small>[0, 0]</small></td>
<td data-label="DAC">0 <small>[0, 0]</small></td>
<td data-label="QSM">0 <small>[0, 0]</small></td>
<td data-label="CGQL">24 <small>[16, 33]</small></td>
<td data-label="CGQL-M">0 <small>[0, 0]</small></td>
<td data-label="CGQL-L">0 <small>[0, 0]</small></td>
<td data-label="DSRL">0 <small>[0, 0]</small></td>
<td data-label="FEdit"><strong>34</strong> <small>[30, 37]</small></td>
<td data-label="FBRAC">15 <small>[12, 19]</small></td>
<td data-label="BAM">0 <small>[0, 0]</small></td>
<td data-label="VINE">28 <small>[20, 35]</small></td>
</tr>
<tr>
<th scope="row">cube-double</th>
<td data-label="ReBRAC">9 <small>[8, 10]</small></td>
<td data-label="FQL">46 <small>[43, 49]</small></td>
<td data-label="FAWAC">2 <small>[2, 2]</small></td>
<td data-label="IFQL">11 <small>[10, 12]</small></td>
<td data-label="QAM">64 <small>[62, 66]</small></td>
<td data-label="DAC">35 <small>[33, 36]</small></td>
<td data-label="QSM">56 <small>[53, 58]</small></td>
<td data-label="CGQL">38 <small>[36, 41]</small></td>
<td data-label="CGQL-M">41 <small>[39, 43]</small></td>
<td data-label="CGQL-L">45 <small>[43, 47]</small></td>
<td data-label="DSRL"><strong>74</strong> <small>[72, 76]</small></td>
<td data-label="FEdit">40 <small>[37, 43]</small></td>
<td data-label="FBRAC">0 <small>[0, 0]</small></td>
<td data-label="BAM">47 <small>[44, 50]</small></td>
<td data-label="VINE">35 <small>[29, 41]</small></td>
</tr>
<tr>
<th scope="row">cube-triple</th>
<td data-label="ReBRAC">1 <small>[0, 1]</small></td>
<td data-label="FQL">3 <small>[2, 4]</small></td>
<td data-label="FAWAC">0 <small>[0, 0]</small></td>
<td data-label="IFQL">0 <small>[0, 0]</small></td>
<td data-label="QAM">3 <small>[3, 4]</small></td>
<td data-label="DAC">5 <small>[3, 6]</small></td>
<td data-label="QSM">3 <small>[3, 4]</small></td>
<td data-label="CGQL"><strong>8</strong> <small>[7, 9]</small></td>
<td data-label="CGQL-M"><strong>8</strong> <small>[7, 9]</small></td>
<td data-label="CGQL-L"><strong>8</strong> <small>[7, 9]</small></td>
<td data-label="DSRL">1 <small>[1, 2]</small></td>
<td data-label="FEdit">2 <small>[2, 3]</small></td>
<td data-label="FBRAC">0 <small>[0, 1]</small></td>
<td data-label="BAM">3 <small>[2, 5]</small></td>
<td data-label="VINE">6 <small>[4, 7]</small></td>
</tr>
<tr>
<th scope="row">cube-quadruple-100M</th>
<td data-label="ReBRAC">9 <small>[6, 11]</small></td>
<td data-label="FQL">2 <small>[1, 4]</small></td>
<td data-label="FAWAC">0 <small>[0, 0]</small></td>
<td data-label="IFQL">2 <small>[1, 3]</small></td>
<td data-label="QAM">3 <small>[2, 4]</small></td>
<td data-label="DAC">3 <small>[1, 5]</small></td>
<td data-label="QSM">19 <small>[19, 20]</small></td>
<td data-label="CGQL">0 <small>[0, 0]</small></td>
<td data-label="CGQL-M">1 <small>[0, 1]</small></td>
<td data-label="CGQL-L">0 <small>[0, 1]</small></td>
<td data-label="DSRL">2 <small>[2, 3]</small></td>
<td data-label="FEdit">5 <small>[3, 7]</small></td>
<td data-label="FBRAC">0 <small>[0, 0]</small></td>
<td data-label="BAM">0 <small>[0, 0]</small></td>
<td data-label="VINE"><strong>47</strong> <small>[38, 56]</small></td>
</tr>
<tr>
<th scope="row">all (50 tasks)</th>
<td data-label="ReBRAC">40 <small>[39, 41]</small></td>
<td data-label="FQL">36 <small>[34, 37]</small></td>
<td data-label="FAWAC">8 <small>[8, 9]</small></td>
<td data-label="IFQL">34 <small>[34, 35]</small></td>
<td data-label="QAM">44 <small>[44, 45]</small></td>
<td data-label="DAC">36 <small>[35, 38]</small></td>
<td data-label="QSM">42 <small>[41, 42]</small></td>
<td data-label="CGQL">30 <small>[28, 31]</small></td>
<td data-label="CGQL-M">35 <small>[34, 35]</small></td>
<td data-label="CGQL-L">37 <small>[36, 38]</small></td>
<td data-label="DSRL">38 <small>[38, 39]</small></td>
<td data-label="FEdit">33 <small>[32, 33]</small></td>
<td data-label="FBRAC">11 <small>[10, 12]</small></td>
<td data-label="BAM">35 <small>[34, 36]</small></td>
<td data-label="VINE"><strong>59</strong> <small>[56, 62]</small></td>
</tr>
</tbody>
</table>
</div>
</div>
</details>
</section>
<section class="citation">
<h2>Citation</h2>
<pre class="bibtex"><code>@misc{yang2026vinetaminggenerativecontrol,
title={VINE: Taming Generative Control Policies for Reinforcement Learning},
author={Rushuai Yang and Zhuo Han and Houlin Li and Rui Zhang and Hecheng Wang and Zhichao Wu and Zhaowei Zhang and Zihong Chen and Xiaohan Yan and Chuankang Li and Guanghui Ren and Chiming Liu and Yi Chen and Wei Shan and Maoqing Yao},
year={2026},
eprint={2607.10369},
archivePrefix={arXiv},
primaryClass={cs.RO},
url={https://arxiv.org/abs/2607.10369},
}
</code></pre>
</section>
<footer class="site-footer">
<p>Project page for VINE. Last updated: 2026.</p>
</footer>
</main>
<script>
document.querySelectorAll('video').forEach(function(v) {
var isToy = v.matches('.td3bc-comparison video');
var inView = !('IntersectionObserver' in window);
var restartTimer = null;
v.loop = !isToy;
v.defaultMuted = true;
v.muted = true;
v.volume = 0;
v.autoplay = true;
v.playsInline = true;
function forceMute() {
v.muted = true;
v.volume = 0;
}
function tryPlay() {
if (isToy && (!inView || restartTimer !== null)) return;
if (isToy && v.ended) v.currentTime = 0;
forceMute();
v.play().catch(function() {});
}
function cancelRestart() {
if (restartTimer !== null) {
clearTimeout(restartTimer);
restartTimer = null;
}
}
v.addEventListener('volumechange', forceMute);
v.addEventListener('play', function() {
forceMute();
if (isToy) {
cancelRestart();
if (!inView) v.pause();
}
});
if (isToy) v.addEventListener('seeking', cancelRestart);
v.addEventListener('loadedmetadata', forceMute);
v.addEventListener('ended', function() {
if (isToy) {
cancelRestart();
restartTimer = setTimeout(function() {
restartTimer = null;
tryPlay();
}, 2000);
} else {
v.currentTime = 0;
tryPlay();
}
});
if ('IntersectionObserver' in window) {
new IntersectionObserver(function(entries) {
entries.forEach(function(entry) {
inView = entry.isIntersecting;
if (inView) tryPlay();
else v.pause();
});
}, { threshold: 0.25 }).observe(v);
} else {
tryPlay();
}
});
</script>
</body>
</html>