"""RynnValue-4B — robot-manipulation value model demo.
Given a manipulation video and the task instruction, RynnValue predicts, for a
series of prefixes of the video, how many seconds of work are *still left*
before the instruction is complete, and writes a short textual analysis
(video description / does the video match the instruction / did it succeed).
The inference protocol mirrors `rynn_infer/inference.py` from the official
repo: prefix-uniform sampling (each score conditions only on frames seen so
far) plus a final generate() pass over the full-video prefix for the analysis
block.
"""
import os
os.environ.setdefault("PYTORCH_CUDA_ALLOC_CONF", "expandable_segments:True")
import spaces # noqa: E402 — must come before torch
import re # noqa: E402
import time # noqa: E402
import tempfile # noqa: E402
from concurrent.futures import ThreadPoolExecutor # noqa: E402
import gradio as gr # noqa: E402
import numpy as np # noqa: E402
import torch # noqa: E402
import imageio.v2 as imageio # noqa: E402
import matplotlib # noqa: E402
matplotlib.use("Agg")
import matplotlib.pyplot as plt # noqa: E402
from matplotlib import font_manager # noqa: E402
from PIL import Image, ImageDraw, ImageFont # noqa: E402
from transformers import AutoConfig, AutoModel, AutoProcessor # noqa: E402
MODEL_ID = "Alibaba-DAMO-Academy/RynnValue-4B"
# ----------------------------------------------------------------------------
# Defaults (kept in sync with the UI components so gr.Examples rows that only
# fill video+instruction behave exactly like pressing "Analyze").
# ----------------------------------------------------------------------------
DEFAULT_ROBOT = "a single-arm robot"
DEFAULT_CAMERA = "the main camera"
DEFAULT_NUM_STEPS = 32 # prefixes evaluated along the video
DEFAULT_NUM_FRAMES = 24 # frames resampled per prefix
DEFAULT_MAX_SIDE = 448 # longest image side fed to the model
DEFAULT_MAX_NEW_TOKENS = 128
DISPLAY_HEIGHT = 320 # height of the rendered video panel
MAX_RENDER_FRAMES = 480 # cap on frames written to the output video
WORK_BUDGET = 1100 # num_steps * num_frames ceiling (latency guard)
# ----------------------------------------------------------------------------
# Model (module scope, eager .to("cuda") — ZeroGPU packs the weights)
# ----------------------------------------------------------------------------
print(f"Loading {MODEL_ID} ...", flush=True)
_config = AutoConfig.from_pretrained(MODEL_ID, trust_remote_code=True)
# config.json does not persist the attention implementation, so force the
# custom prediction-slot isolation attention the value heads require.
_config._attn_implementation = "pred_slot_isolated_eager"
model = AutoModel.from_pretrained(
MODEL_ID,
config=_config,
trust_remote_code=True,
torch_dtype=torch.bfloat16,
)
# The value heads are built in __init__ with an explicit float32 dtype, so
# `torch_dtype=` alone leaves them fp32 and F.linear blows up on the bf16
# hidden states. The reference script casts the whole module the same way.
model = model.eval().to(device="cuda", dtype=torch.bfloat16)
processor = AutoProcessor.from_pretrained(MODEL_ID, trust_remote_code=True)
tokenizer = processor.tokenizer
EOS_TOKEN_ID = tokenizer.convert_tokens_to_ids("<|im_end|>")
_dtypes = {str(p.dtype) for p in model.parameters()}
print(
f"Loaded. attn={getattr(model.config, '_attn_implementation', '?')} "
f"dtypes={sorted(_dtypes)}",
flush=True,
)
try:
_FONT_PATH = font_manager.findfont("DejaVu Sans")
except Exception:
_FONT_PATH = None
def _font(size: int):
if _FONT_PATH:
try:
return ImageFont.truetype(_FONT_PATH, size)
except Exception:
pass
return ImageFont.load_default()
# ----------------------------------------------------------------------------
# Video I/O
# ----------------------------------------------------------------------------
def _resize_max_side(img: Image.Image, max_side: int) -> Image.Image:
w, h = img.size
if max_side <= 0 or max(w, h) <= max_side:
return img
scale = max_side / max(w, h)
return img.resize((max(1, round(w * scale)), max(1, round(h * scale))), Image.BICUBIC)
def _even(v: int) -> int:
v = int(round(v))
return v if v % 2 == 0 else v + 1
def _decode_video(path: str, max_side: int):
"""Decode a video into (display frames, model frames, output fps).
Scaling and frame decimation are pushed into ffmpeg (much cheaper than
doing them in Python), so at most ``MAX_RENDER_FRAMES`` frames come back
and the rendered video keeps the original wall-clock pace.
"""
probe = imageio.get_reader(path)
try:
meta = probe.get_meta_data()
finally:
probe.close()
src_fps = float(meta.get("fps") or 30.0)
src_w, src_h = meta.get("size") or (0, 0)
duration = float(meta.get("duration") or 0.0)
if not src_w or not src_h:
raise gr.Error("Could not read the video's dimensions.")
disp_h = DISPLAY_HEIGHT
disp_w = max(16, _even(src_w * disp_h / src_h))
n_total = int(duration * src_fps) if duration else 0
stride = max(1, int(np.ceil(n_total / MAX_RENDER_FRAMES))) if n_total else 1
out_fps = max(1.0, src_fps / stride)
# Decode at the smallest size that still satisfies both consumers.
long_needed = max(max(disp_w, disp_h), int(max_side))
scale = min(1.0, long_needed / max(src_w, src_h))
dec_w, dec_h = max(16, _even(src_w * scale)), max(16, _even(src_h * scale))
kwargs = dict(size=(dec_w, dec_h))
if stride > 1:
kwargs["fps"] = out_fps
reader = imageio.get_reader(path, **kwargs)
disp, model_frames = [], []
try:
for raw in reader:
img = Image.fromarray(raw).convert("RGB")
model_frames.append(_resize_max_side(img, max_side))
disp.append(
img if img.size == (disp_w, disp_h)
else img.resize((disp_w, disp_h), Image.BILINEAR)
)
if len(disp) >= MAX_RENDER_FRAMES:
break
finally:
reader.close()
if not disp:
raise gr.Error("Could not decode any frame from that video.")
return disp, model_frames, out_fps
def _sample_indices(total: int, num: int):
"""Uniformly pick ``num`` indices out of ``total`` (mirrors the repo helper)."""
if num <= 0 or num >= total:
return list(range(total))
if num == 1:
return [total - 1]
step = (total - 1) / (num - 1)
return sorted({int(round(j * step)) for j in range(num)})
# ----------------------------------------------------------------------------
# Trend rendering (same visual language as rynn_infer/plot_utils.py, but the
# static parts of the figure are rasterised once and the moving parts are drawn
# with PIL so we can render hundreds of frames in a couple of seconds).
# ----------------------------------------------------------------------------
def _format_time(seconds: float) -> str:
seconds = max(0.0, float(seconds))
return f"{int(seconds // 60):02d}:{int(seconds % 60):02d}.{int((seconds - int(seconds)) * 1000):03d}"
def _build_plot_background(x, y, remaining, size, task_title):
w, h = size
dpi = 100
fig, ax1 = plt.subplots(figsize=(w / dpi, h / dpi), dpi=dpi, constrained_layout=True)
# Legend proxies only — the blue curve itself is drawn per frame with PIL.
ax1.plot([], [], color="tab:blue", linewidth=2.0, label="predicted")
ax1.scatter([], [], color="red", s=28, label="current")
ax1.set_xlabel("Frame", fontsize=9)
ax1.set_ylabel("Predicted remaining (s)", color="tab:blue", fontsize=9)
ax1.tick_params(axis="x", labelsize=8)
ax1.tick_params(axis="y", labelcolor="tab:blue", labelsize=8)
ax1.grid(True, alpha=0.3)
ax1.set_xlim(0, max(float(x[-1]), 1.0))
y_min, y_max = float(np.min(y)), float(np.max(y))
if y_min == y_max:
y_min, y_max = y_min - 1.0, y_max + 1.0
margin = 0.05 * (y_max - y_min)
ax1.set_ylim(y_min - margin, y_max + margin)
ax2 = ax1.twinx()
ax2.plot(x, remaining, color="green", linestyle="--", linewidth=1.8, label="video timeline")
ax2.scatter([], [], color="green", s=22, label="current")
ax2.set_ylabel("Video remaining (s)", color="green", fontsize=9)
ax2.tick_params(axis="y", labelcolor="green", labelsize=8)
r_min, r_max = float(np.min(remaining)), float(np.max(remaining))
if r_min == r_max:
r_min, r_max = r_min - 1.0, r_max + 1.0
r_margin = 0.05 * (r_max - r_min)
ax2.set_ylim(r_min - r_margin, r_max + r_margin)
task_title = (task_title or "").strip()
if len(task_title) > 46:
task_title = task_title[:45] + "…"
ax1.set_title(f"{task_title}\nRemaining time" if task_title else "Remaining time", fontsize=10)
lines1, labels1 = ax1.get_legend_handles_labels()
lines2, labels2 = ax2.get_legend_handles_labels()
ax1.legend(lines1 + lines2, labels1 + labels2, loc="best", fontsize=7)
fig.canvas.draw()
arr = np.asarray(fig.canvas.buffer_rgba())[..., :3].copy()
bg = Image.fromarray(arr)
H = arr.shape[0]
pred_px = ax1.transData.transform(np.column_stack([x, y]))
ref_px = ax2.transData.transform(np.column_stack([x, remaining]))
plt.close(fig)
pred_pts = [(float(px), float(H - py)) for px, py in pred_px]
ref_pts = [(float(px), float(H - py)) for px, py in ref_px]
return bg, pred_pts, ref_pts
def _dot(draw, pt, color, r=4):
draw.ellipse([pt[0] - r, pt[1] - r, pt[0] + r, pt[1] + r], fill=color, outline=(255, 255, 255))
def _pad16(img: Image.Image) -> Image.Image:
nw = ((img.width + 15) // 16) * 16
nh = ((img.height + 15) // 16) * 16
if nw == img.width and nh == img.height:
return img
canvas = Image.new("RGB", (nw, nh), (255, 255, 255))
canvas.paste(img, (0, 0))
return canvas
def _render_trend_video(disp_frames, values, sampled_indices, fps, instruction, out_path):
n = len(disp_frames)
x = np.asarray(sampled_indices, dtype=float)
y = np.asarray(values, dtype=float)
remaining_ref = (n - 1 - x) / float(fps)
vid_w, vid_h = disp_frames[0].size
plot_w = int(min(520, max(300, vid_w * 0.62)))
bg, pred_pts, ref_pts = _build_plot_background(
x, y, remaining_ref, (plot_w, vid_h), instruction.strip()
)
idx_to_pos = {idx: pos for pos, idx in enumerate(sampled_indices)}
font = _font(15)
canvas_size = (vid_w + plot_w, vid_h)
writer = imageio.get_writer(
out_path, fps=max(1.0, fps), codec="libx264", macro_block_size=16, quality=7
)
try:
pos = 0
for i, frame in enumerate(disp_frames):
if i in idx_to_pos:
pos = idx_to_pos[i]
plot = bg.copy()
d = ImageDraw.Draw(plot)
if pos >= 1:
d.line(pred_pts[: pos + 1], fill=(31, 119, 180), width=3, joint="curve")
_dot(d, ref_pts[pos], (0, 128, 0), r=4)
_dot(d, pred_pts[pos], (220, 0, 0), r=5)
canvas = Image.new("RGB", canvas_size, (255, 255, 255))
canvas.paste(frame, (0, 0))
canvas.paste(plot, (vid_w, 0))
dd = ImageDraw.Draw(canvas)
lines = [
f"task: {instruction.strip()[:58]}",
f"predicted remaining: {y[pos]:.2f} s",
f"video remaining: {_format_time((n - 1 - i) / float(fps))}",
]
dd.rectangle([0, 0, vid_w, 8 + 20 * len(lines)], fill=(0, 0, 0))
ty = 6
for line in lines:
dd.text((10, ty), line, font=font, fill=(255, 120, 120))
ty += 20
writer.append_data(np.asarray(_pad16(canvas)))
finally:
writer.close()
return out_path
# ----------------------------------------------------------------------------
# Analysis-block parsing (from rynn_infer/inference.py)
# ----------------------------------------------------------------------------
_DESCRIPTION_RE = re.compile(r"-\s*Video Description:\s*(.+)", re.IGNORECASE)
_MATCH_RE = re.compile(r"-\s*Match:\s*(Yes|No)", re.IGNORECASE)
_SUCCESS_RE = re.compile(r"-\s*Success:\s*(Yes|No)", re.IGNORECASE)
def _parse_analysis(text: str):
def first(pattern):
m = pattern.search(text)
return m.group(1).strip() if m else None
return {
"description": first(_DESCRIPTION_RE),
"match": first(_MATCH_RE),
"success": first(_SUCCESS_RE),
}
def _reduce_pred_value(pred: torch.Tensor, n_samples: int) -> torch.Tensor:
"""Collapse a value-head output to one scalar per prefix sub-sample.
Verbatim from ``rynn_infer/inference.py``: ``pred_value`` is
``(num_heads, batch * slots)``, so it is folded back to ``(batch, slots)``
and the last slot (the prefix end) is read out per sample.
"""
if pred.dim() == 2 and pred.shape[0] == 1:
pred = pred.reshape(n_samples, -1)
if pred.dim() == 3:
pred = pred.mean(dim=0)
if pred.dim() == 2 and pred.shape[-1] > 1:
pred = pred[:, -1]
elif pred.dim() == 2:
pred = pred[:, 0]
return pred.float().reshape(-1)
def _badge(flag):
if flag is None:
return "—"
return "✅ Yes" if flag.lower() == "yes" else "❌ No"
# ----------------------------------------------------------------------------
# Inference
# ----------------------------------------------------------------------------
def _gpu_duration(*args, **kwargs):
"""Size the ZeroGPU reservation from the measured cost of one run.
Reference points measured on this Space (448 px, 24 frames/prefix,
batch 8): 32 prefixes over a 429-frame video = ~30 s wall clock end to end,
including decode and rendering. Cost is dominated by the value pass, which
scales with ``num_steps × num_frames`` and roughly with the square of the
image side (the eager attention is O(L²)).
"""
num_steps = kwargs.get("num_steps", DEFAULT_NUM_STEPS)
num_frames = kwargs.get("num_frames", DEFAULT_NUM_FRAMES)
side = kwargs.get("max_image_side", DEFAULT_MAX_SIDE)
if len(args) > 4:
num_steps = args[4]
if len(args) > 5:
num_frames = args[5]
if len(args) > 6:
side = args[6]
try:
work = min(int(num_steps) * int(num_frames), WORK_BUDGET)
factor = (float(side) / DEFAULT_MAX_SIDE) ** 2.5
except Exception:
work, factor = DEFAULT_NUM_STEPS * DEFAULT_NUM_FRAMES, 1.0
return int(min(180, max(30, 18 + work * 0.042 * factor)))
@spaces.GPU(duration=_gpu_duration)
def analyze(
video: str,
instruction: str,
robot_description: str = DEFAULT_ROBOT,
camera_description: str = DEFAULT_CAMERA,
num_steps: int = DEFAULT_NUM_STEPS,
num_frames: int = DEFAULT_NUM_FRAMES,
max_image_side: int = DEFAULT_MAX_SIDE,
max_new_tokens: int = DEFAULT_MAX_NEW_TOKENS,
# `progress` sits LAST on purpose: Gradio splices its Progress object into
# the argument list at this parameter's *positional index*, so the bound
# event must pass exactly the 8 preceding inputs. gr.Examples, which only
# fills 2 of them, therefore goes through `_run_example` instead.
progress=gr.Progress(),
):
"""Score how far a robot manipulation video is from completing an instruction.
Runs RynnValue-4B over prefixes of the video and returns the input video
rendered next to the predicted remaining-time curve, plus the model's
textual analysis (description / instruction match / success).
Args:
video: Path to a robot manipulation video (mp4/webm/avi).
instruction: The task the robot is supposed to accomplish.
robot_description: Embodiment phrase for the meta block, e.g. "a Franka single-arm robot".
camera_description: Viewpoint phrase for the meta block, e.g. "the main camera".
num_steps: How many prefixes along the video are scored.
num_frames: Frames uniformly resampled inside each prefix.
max_image_side: Longest image side fed to the vision encoder.
max_new_tokens: Token budget for the generated analysis block.
Returns:
A tuple of (path to the rendered mp4, markdown report).
"""
if not video:
raise gr.Error("Please upload or pick a video first.")
instruction = (instruction or "").strip()
if not instruction:
raise gr.Error("Please describe the task the robot should accomplish.")
robot_description = (robot_description or DEFAULT_ROBOT).strip() or DEFAULT_ROBOT
camera_description = (camera_description or DEFAULT_CAMERA).strip() or DEFAULT_CAMERA
num_steps = int(num_steps)
num_frames = int(num_frames)
max_image_side = int(max_image_side)
max_new_tokens = int(max_new_tokens)
notes = []
if num_steps * num_frames > WORK_BUDGET:
num_steps = max(8, WORK_BUDGET // num_frames)
notes.append(f"Reduced *evaluated prefixes* to **{num_steps}** to stay inside the GPU budget.")
t0 = time.perf_counter()
progress(0.02, desc="Decoding video…")
disp_frames, model_frames, out_fps = _decode_video(video, max_image_side)
total = len(disp_frames)
t_decode = time.perf_counter() - t0
eval_indices = _sample_indices(total, num_steps)
device = torch.device("cuda")
progress(0.15, desc="Preprocessing frames…")
def build_prefix(end_idx):
idx = np.linspace(0, end_idx, num_frames, dtype=int)
return processor.process_episode(
instruction=instruction,
images=[model_frames[j] for j in idx],
robot_description=robot_description,
camera_description=camera_description,
)
t1 = time.perf_counter()
with ThreadPoolExecutor(max_workers=4) as pool:
samples = list(pool.map(build_prefix, eval_indices))
t_prep = time.perf_counter() - t1
seq_len = int(samples[0]["input_ids"].shape[-1])
# `pred_slot_isolated_eager` materialises a full B×32×L×L attention matrix.
# Measured: batch 8 is no faster than batch 4 here (compute-bound), so keep
# the smaller batch and halve further on OOM (see the loop below).
batch_size = 4 if seq_len <= 3600 else (2 if seq_len <= 5400 else 1)
def run_batch(batch):
kwargs = dict(
input_ids=torch.cat([s["input_ids"] for s in batch], dim=0).to(device).long(),
attention_mask=torch.cat([s["attention_mask"] for s in batch], dim=0).to(device).long(),
pixel_values=torch.cat([s["pixel_values"].flatten(0, 1) for s in batch], dim=0).to(device),
image_grid_thw=torch.cat(
[s["image_grid_thw"].flatten(0, 1) for s in batch], dim=0
).to(device).long(),
)
with torch.inference_mode():
out = model(**kwargs)
return _reduce_pred_value(out.value.pred_value, len(batch)).tolist()
t2 = time.perf_counter()
values = []
while len(values) < len(samples):
chunk = samples[len(values) : len(values) + batch_size]
try:
values.extend(run_batch(chunk))
except torch.cuda.OutOfMemoryError:
torch.cuda.empty_cache()
if batch_size == 1:
raise gr.Error(
"Ran out of GPU memory. Try a smaller 'Max image side' or fewer "
"'Frames per prefix' in Advanced settings."
)
batch_size = max(1, batch_size // 2)
print(f"[oom] falling back to batch_size={batch_size}", flush=True)
continue
progress(
0.2 + 0.55 * len(values) / len(samples),
desc=f"Scoring prefix {len(values)}/{len(samples)}…",
)
t_value = time.perf_counter() - t2
# Analysis pass on the final prefix (the full video, uniformly sampled).
progress(0.78, desc="Writing analysis…")
t3 = time.perf_counter()
final = samples[-1]
input_ids = final["input_ids"].to(device).long()
with torch.inference_mode():
gen_out = model.generate(
input_ids=input_ids,
attention_mask=final["attention_mask"].to(device).long(),
pixel_values=final["pixel_values"].flatten(0, 1).to(device),
image_grid_thw=final["image_grid_thw"].flatten(0, 1).to(device).long(),
max_new_tokens=max_new_tokens,
do_sample=False,
num_beams=1,
eos_token_id=EOS_TOKEN_ID,
pad_token_id=EOS_TOKEN_ID,
use_cache=True,
)
raw_analysis = tokenizer.decode(gen_out[0, input_ids.shape[1] :], skip_special_tokens=True)
analysis = _parse_analysis(raw_analysis)
t_gen = time.perf_counter() - t3
progress(0.85, desc="Rendering trend video…")
t4 = time.perf_counter()
out_path = os.path.join(tempfile.mkdtemp(prefix="rynnvalue_"), "trend.mp4")
_render_trend_video(disp_frames, values, eval_indices, out_fps, instruction, out_path)
t_render = time.perf_counter() - t4
total_s = time.perf_counter() - t0
video_seconds = (total - 1) / max(out_fps, 1e-6)
report = [
"### Analysis",
f"**Video description** — {analysis['description'] or raw_analysis.strip() or '—'}",
"",
f"**Matches the instruction:** {_badge(analysis['match'])} • "
f"**Task completed:** {_badge(analysis['success'])}",
"",
"### Predicted remaining time",
f"- First evaluated prefix: **{values[0]:.2f} s**",
f"- Last evaluated prefix (full video): **{values[-1]:.2f} s**",
f"- Video length: {video_seconds:.2f} s ({total} frames @ {out_fps:.1f} fps)",
"",
f"{len(eval_indices)} prefixes × {num_frames} frames @ ≤{max_image_side}px · "
f"decode {t_decode:.1f}s · preprocess {t_prep:.1f}s · value {t_value:.1f}s · "
f"generate {t_gen:.1f}s · render {t_render:.1f}s · total {total_s:.1f}s",
]
if notes:
report.append("")
report.extend(f"⚠️ {n}" for n in notes)
print(
f"[timing] decode={t_decode:.2f} prep={t_prep:.2f} value={t_value:.2f} "
f"gen={t_gen:.2f} render={t_render:.2f} total={total_s:.2f} "
f"seq_len={seq_len} bs={batch_size} frames={total}",
flush=True,
)
return out_path, "\n".join(report)
def _run_example(video: str, instruction: str):
"""Two-argument entry point for gr.Examples (everything else stays default)."""
return analyze(video, instruction)
# ----------------------------------------------------------------------------
# UI
# ----------------------------------------------------------------------------
CSS = """
#col-container { max-width: 1180px; margin: 0 auto; }
.dark .gradio-container { color: var(--body-text-color); }
"""
HEADER = """# RynnValue-4B — how far is the robot from finishing?
[RynnValue-4B](https://huggingface.co/Alibaba-DAMO-Academy/RynnValue-4B) is a general-purpose
value model for robot manipulation. Give it a video and a task instruction and it predicts, along
the video, **how many seconds of work are still left** before the instruction is complete — plus a
short analysis of what it sees and whether the video actually matches the instruction.
[Model card](https://huggingface.co/Alibaba-DAMO-Academy/RynnValue-4B) ·
[GitHub](https://github.com/alibaba-damo-academy/RynnValue) ·
[Paper](https://arxiv.org/abs/2608.09853)
"""
EXAMPLES = [
["examples/put_box_in_drawer.mp4", "Put the box in the drawer and close it"],
["examples/soar_put_green_stick_in_brown_bowl.mp4", "Put green stick in brown bowl"],
["examples/berkeley_rpt_stack_cup.mp4", "Pick up the yellow cup and stack it on the other cup"],
["examples/jaco_play_pick_up_green_cup.mp4", "Pick up the green cup"],
["examples/soar_put_green_stick_in_brown_bowl.mp4", "Fold the towel and put it in the basket"],
]
with gr.Blocks(theme=gr.themes.Citrus(), css=CSS, title="RynnValue-4B") as demo:
with gr.Column(elem_id="col-container"):
gr.Markdown(HEADER)
with gr.Row():
with gr.Column(scale=1):
video_in = gr.Video(label="Manipulation video", height=300)
instruction_in = gr.Textbox(
label="Task instruction",
placeholder="e.g. Put the box in the drawer and close it",
lines=2,
)
run_btn = gr.Button("Analyze", variant="primary")
with gr.Column(scale=1):
video_out = gr.Video(
label="Video + predicted remaining-time curve",
height=300,
autoplay=True,
)
report_out = gr.Markdown()
with gr.Accordion("Advanced settings", open=False):
with gr.Row():
robot_in = gr.Textbox(
label="Robot description",
value=DEFAULT_ROBOT,
info='Meta block phrasing, e.g. "a Franka single-arm robot".',
)
camera_in = gr.Textbox(
label="Camera description",
value=DEFAULT_CAMERA,
info='e.g. "the main camera", "the wrist-mounted camera".',
)
with gr.Row():
steps_in = gr.Slider(
8, 48, value=DEFAULT_NUM_STEPS, step=1,
label="Evaluated prefixes",
info="Points on the predicted curve.",
)
frames_in = gr.Slider(
8, 32, value=DEFAULT_NUM_FRAMES, step=1,
label="Frames per prefix",
info="Temporal resolution the model sees.",
)
with gr.Row():
side_in = gr.Dropdown(
[320, 448, 640], value=DEFAULT_MAX_SIDE,
label="Max image side (px)",
)
tokens_in = gr.Slider(
32, 256, value=DEFAULT_MAX_NEW_TOKENS, step=8,
label="Analysis max new tokens",
)
gr.Examples(
examples=EXAMPLES,
inputs=[video_in, instruction_in],
outputs=[video_out, report_out],
fn=_run_example,
cache_examples=True,
cache_mode="lazy",
label="Examples (the last row deliberately mismatches the video)",
)
gr.Markdown(
"Blue = RynnValue's predicted remaining time. Green dashed = the video's own "
"remaining wall-clock time, i.e. the ground truth when the clip ends exactly at task "
"completion. Example clips come from the "
"[RynnValue](https://github.com/alibaba-damo-academy/RynnValue) repo (Apache-2.0) and "
"its bundled Robometer example videos (MIT)."
)
run_btn.click(
fn=analyze,
# Must be exactly the 8 parameters preceding `progress` in `analyze`.
inputs=[
video_in, instruction_in, robot_in, camera_in,
steps_in, frames_in, side_in, tokens_in,
],
outputs=[video_out, report_out],
api_name="analyze",
)
if __name__ == "__main__":
demo.launch(mcp_server=True)