Spaces:
Running on Zero
Running on Zero
| { | |
| "schema_version": "demo_quality_contract.v1", | |
| "model_card_promise": "Text-to-video and video-to-video generation in a streaming fashion with ~1s time to first frame, powered by a 14B autoregressive diffusion model distilled from Wan 2.1.", | |
| "demo_task": "text-to-video", | |
| "primary_user_flow": "User enters a text prompt, adjusts blocks and steps, clicks Generate, and receives an MP4 video.", | |
| "examples_provided": true, | |
| "examples": [ | |
| { | |
| "inputs": { | |
| "prompt": "a cat sitting on a boat", | |
| "num_blocks": 3, | |
| "num_inference_steps": 4, | |
| "seed": 42 | |
| }, | |
| "description": "Classic model-card example with conservative settings." | |
| }, | |
| { | |
| "inputs": { | |
| "prompt": "a futuristic city at sunset", | |
| "num_blocks": 3, | |
| "num_inference_steps": 4, | |
| "seed": 123 | |
| }, | |
| "description": "Sci-fi scene preset." | |
| }, | |
| { | |
| "inputs": { | |
| "prompt": "a panda playing guitar in a forest", | |
| "num_blocks": 2, | |
| "num_inference_steps": 4, | |
| "seed": 7 | |
| }, | |
| "description": "Lowest-cost preset suitable for a first call." | |
| } | |
| ], | |
| "canonical_smoke_example": { | |
| "api_name": "/generate", | |
| "inputs": { | |
| "prompt": "a cat sitting on a boat", | |
| "num_blocks": 1, | |
| "num_inference_steps": 2, | |
| "seed": 42 | |
| }, | |
| "expected_output_type": "video/mp4", | |
| "reason": "Single block (3 frames) with 2 diffusion steps is the smallest real forward-pass configuration that exercises the pipeline end-to-end while keeping duration as low as possible." | |
| }, | |
| "real_inference_required": true, | |
| "real_inference_implemented": true, | |
| "fallback_or_diagnostic_only": false, | |
| "limitations_disclosed": true, | |
| "promise_fulfillment_risk": "High. The model requires ~50+ GB of VRAM for its weights alone in bf16, exceeding standard HF Space GPU tiers (48 GB large / 96 GB xlarge). Even if it loads, the advertised 11 fps and interactive streaming rely on hardware (B200 / H100 with FP8) beyond typical Space capacity. The Space honestly attempts inference and surfaces OOM/load errors instead of faking output." | |
| } | |