diff --git a/phases/12-multimodal-ai/18-long-video-million-token/assets/long-video-paths.svg b/phases/12-multimodal-ai/18-long-video-million-token/assets/long-video-paths.svg new file mode 100644 index 000000000..d497fd6d8 --- /dev/null +++ b/phases/12-multimodal-ai/18-long-video-million-token/assets/long-video-paths.svg @@ -0,0 +1,67 @@ + + + + + + + + + Long-video understanding — four scaling paths + + + Path 1: brute context + Gemini 1.5 Pro: 1M tokens + Gemini 2.5 Pro: 10M+ tokens + Claude Opus 4.7: 1M tokens + engineering + custom attention hierarchy + MoE expert routing + closed-source + best recall, closed only + + + Path 2: ring attention + LWM: 1M-token training + LongVILA: 1400-frame videos + distributed ring pattern + engineering + each device holds chunk + rotates for attention passes + open-source + good open scaling, heavy compute + + + Path 3: token compression + Video-XL: one summary token + per clip (100s of frames -> 1) + LongVA: long-context transfer + VideoChat2: hierarchical pool + engineering + learned compressor pre-LLM + trades recall for scale + ~32k context sufficient + cheapest inference, weakest grounding + + + Path 4: agentic retrieval + VideoAgent: LLM as query planner + tool: find_clips(keyword) + VLM reads only matches + LLM composes final answer + engineering + retrieval quality is the bottleneck + 99% cheaper for single-event queries + worse for holistic understanding + best for 2h+ specific queries + diff --git a/phases/12-multimodal-ai/18-long-video-million-token/code/main.py b/phases/12-multimodal-ai/18-long-video-million-token/code/main.py new file mode 100644 index 000000000..9342e18f6 --- /dev/null +++ b/phases/12-multimodal-ai/18-long-video-million-token/code/main.py @@ -0,0 +1,115 @@ +"""Long-video token budget + needle-in-a-haystack simulator + agentic retrieval. + +Stdlib. Prints budget tables for long videos, runs a synthetic NIH recall test, +simulates a VideoAgent-style retrieval loop. +""" + +from __future__ import annotations + +import random +from dataclasses import dataclass + +random.seed(5) + + +def tokens(duration_s: float, fps: float, per_frame: int) -> int: + return int(duration_s * fps * per_frame) + + +def budget_table() -> None: + print("\nLONG-VIDEO TOKEN BUDGETS") + print("-" * 60) + print(f"{'duration':<14}{'FPS':>5}{'per_frame':>12}{'tokens':>12}{'fits in':>14}") + cases = [ + (60, 1, 81, "32k+"), + (300, 1, 81, "32k"), + (300, 2, 81, "128k"), + (1800, 1, 81, "256k"), + (3600, 1, 81, "1M / LongVILA"), + (7200, 1, 81, "Gemini 2.5 only"), + (7200, 1, 32, "agentic retrieval"), + ] + for dur, fps, pf, fits in cases: + t = tokens(dur, fps, pf) + print(f"{dur//60}min{' ':<8}{fps:>5}{pf:>12}{t:>12,} {fits}") + + +@dataclass +class Needle: + t: float + marker: str + + +def nih_trial(duration_s: float, model_recall_curve: list[tuple[float, float]]) -> dict: + needle_t = random.uniform(0, duration_s) + needle = Needle(t=needle_t, marker="unique sticker") + pct_into_video = needle_t / duration_s + for thresh, recall in model_recall_curve: + if pct_into_video <= thresh: + return {"needle_time": needle_t, + "pct_into_video": pct_into_video, + "recall_prob": recall} + return {"needle_time": needle_t, + "pct_into_video": pct_into_video, + "recall_prob": model_recall_curve[-1][1]} + + +def nih_simulation() -> None: + print("\nNEEDLE-IN-A-HAYSTACK SIMULATION (single trial per model)") + print("-" * 60) + models = [ + ("Qwen2.5-VL-72B @ 15min", 900, [(0.1, 0.98), (0.5, 0.90), (1.0, 0.85)]), + ("Qwen2.5-VL-72B @ 30min", 1800, [(0.1, 0.95), (0.5, 0.85), (1.0, 0.75)]), + ("Gemini 2.5 Pro @ 90min", 5400, [(0.1, 0.99), (0.5, 0.99), (1.0, 0.99)]), + ("VideoAgent (retrieval) 2h", 7200, [(0.1, 0.92), (0.5, 0.92), (1.0, 0.92)]), + ] + for name, dur, curve in models: + r = nih_trial(dur, curve) + print(f" {name:<32} needle@{r['needle_time']:>6.1f}s " + f"p(recall)={r['recall_prob']:.2f}") + + +def agentic_retrieval_sim(question: str, video_duration: float) -> dict: + """Simulate VideoAgent: LLM asks for clip, tool returns timestamps, VLM reads.""" + trace = [] + trace.append(("LLM ", f"reading question: '{question}'")) + query = question.split()[-1].lower() + trace.append(("LLM ", f"calling tool: find_clips(keyword='{query}')")) + hits = sorted([random.uniform(0, video_duration) for _ in range(3)]) + trace.append(("TOOL ", f"returned 3 clips: {[round(h,1) for h in hits]}")) + trace.append(("VLM ", f"encoding 3 x 30s clips (~7290 tokens total)")) + trace.append(("LLM ", "composing answer from clip descriptions")) + tokens_used = 3 * 30 * 81 + 200 + return {"steps": trace, "tokens": tokens_used} + + +def agentic_demo() -> None: + print("\nVIDEOAGENT-STYLE RETRIEVAL (2-hour video)") + print("-" * 60) + r = agentic_retrieval_sim("at what point does the cat jump", 7200) + for role, msg in r["steps"]: + print(f" [{role}] {msg}") + print(f"\n total tokens used: ~{r['tokens']:,}") + print(f" vs brute context 2h @ 1 FPS: ~583,000 tokens") + print(f" -> 99% cheaper inference for single-event queries") + + +def main() -> None: + print("=" * 60) + print("LONG-VIDEO UNDERSTANDING (Phase 12, Lesson 18)") + print("=" * 60) + + budget_table() + nih_simulation() + agentic_demo() + + print("\nSTRATEGY PICKER") + print("-" * 60) + print(" <15 min : brute context (Qwen2.5-VL-72B)") + print(" 15-60 min : LongVILA / Video-XL / Gemini 2.5") + print(" >1h general QA : Gemini 2.5 Pro (closed frontier)") + print(" >1h specific query : VideoAgent (agentic retrieval)") + + +if __name__ == "__main__": + main() diff --git a/phases/12-multimodal-ai/18-long-video-million-token/docs/en.md b/phases/12-multimodal-ai/18-long-video-million-token/docs/en.md new file mode 100644 index 000000000..9f689e2db --- /dev/null +++ b/phases/12-multimodal-ai/18-long-video-million-token/docs/en.md @@ -0,0 +1,138 @@ +# Long-Video Understanding at Million-Token Context + +> A 1-hour 4K video at 24 FPS, patched and embedded, produces on the order of 60 million tokens. A 2-hour podcast episode transcribed is 30,000 tokens. A full Blu-ray feature film, even compressed with aggressive pooling, is hundreds of thousands of tokens. Google's Gemini 1.5 (March 2024) opened this era with a 10-million-token context, doing reliable needle-in-a-haystack recall over hour-long videos. LWM (Liu et al., February 2024) showed ring attention's scaling path. LongVILA and Video-XL scaled ingestion further. VideoAgent swapped raw context for agentic retrieval. Each approach is a different trade-off on compute, recall, and engineering complexity. This lesson reads them side by side. + +**Type:** Build +**Languages:** Python (stdlib, needle-in-haystack simulator + agentic-retrieval router) +**Prerequisites:** Phase 12 · 17 (video temporal tokens) +**Time:** ~180 minutes + +## Learning Objectives + +- Compute total visual-token counts for long-form video at varying FPS and pooling. +- Explain the three scaling paths: brute context (Gemini 1.5), ring attention (LWM), token compression (LongVILA / Video-XL). +- Compare raw-context video VLMs vs agentic-retrieval video VLMs (VideoAgent) on accuracy and latency. +- Design a needle-in-a-haystack test for a 30-minute video and measure recall at a specific minute. + +## The Problem + +A single frame of Qwen2.5-VL-sized patches at 384 native resolution is ~729 tokens. At 3x3 pooling that's 81 tokens per frame. A 30-minute clip at 1 FPS = 1800 frames = 145,800 tokens. Doable by 2025 open VLMs, tight. At 2 FPS, 291,600 tokens — only the biggest contexts fit. + +A 2-hour movie at 1 FPS is 583k tokens. Beyond most 2026 open models; requires Gemini 2.5 Pro or pooling more aggressively. + +Three scaling paths emerged. + +## The Concept + +### Path 1: Brute context (Gemini 1.5, Claude Opus) + +Throw hardware at the problem. Scale context to millions of tokens, process everything in one forward pass. + +Gemini 1.5 Pro launched with 1M tokens; Gemini 1.5 Ultra to 10M; Gemini 2.5 Pro in 2026 does hours of video reliably. The paper (arXiv:2403.05530) documents needle-in-a-haystack recall at 99.7% up to ~9.5M tokens. + +Engineering: a custom attention implementation with memory hierarchy (local + global + sparse) plus MoE expert routing for long-context efficiency. Not published in full detail. Not open-source. + +### Path 2: Ring attention (LWM, LongVILA) + +Ring attention distributes long sequences across devices in a "ring" where each device holds a chunk. Attention across the full sequence happens by each device sending its chunk to the next in a ring pattern, computing partial attention, and aggregating. + +LWM (Liu et al., 2024) trained a 1M-token context model this way. Training compute scales linearly with context, not quadratically — the quadratic hit on attention is amortized across the ring's devices. + +LongVILA (arXiv:2408.10188) adapted the pattern to VLMs. 1400-frame videos at 192 tokens per frame = 268k context, trained with ring attention across 8-way parallelism. + +### Path 3: Token compression (Video-XL, LongVA) + +Cheaper than brute context: compress aggressively before the LLM sees the sequence. + +Video-XL (arXiv:2409.14485) uses a visual summary token: each clip of N frames produces a single "summary" token that attends over the N. At inference, the LLM sees one summary token per clip, drastically shrinking the context. + +LongVA extends LLM context from 200k to 2M with a "long context transfer" technique. Train on long-context text, transfer to long-context video via shared representation. + +Token compression trades off recall at specific timestamps for scalability. The model knows generally what happened but sometimes misses exact frames. + +### Path 4: Agentic retrieval (VideoAgent) + +Do not feed the full video to the LLM. Instead, treat the video as a database and use an LLM to query it. + +VideoAgent (arXiv:2403.10517): + +1. LLM reads the question. +2. LLM asks a retrieval tool for relevant clips ("show me segments with a cat"). +3. Tool returns matching clip timestamps. +4. LLM reads those clips via a VLM. +5. LLM composes the answer or asks follow-up queries. + +This is the LLM-as-agent pattern applied to long video. Cheaper inference (only relevant clips encoded), harder engineering (retrieval quality becomes the bottleneck). + +### Needle-in-a-haystack benchmarks + +The standard long-context test: insert a unique visual or textual marker at a random point in the video, then ask a query that requires recalling it. + +Metric: Recall@k across video length and marker position. + +Gemini 2.5 Pro scores >99% recall at up to 90-minute videos. Open 72B models (Qwen2.5-VL-72B, InternVL3-78B) score ~85-90% at 30 minutes and degrade past 60. + +VideoAgent can match or beat raw-context models at 2+ hours because retrieval hits the needle if the tool is good. + +### Which path to pick + +For a 15-minute clip at frontier accuracy: open 72B + native context usually works. Pick Qwen2.5-VL-72B. + +For 30-minute to 1-hour content: LongVILA or Video-XL for open; Gemini 2.5 Pro for closed. The quality bar matters — frontier goes closed. + +For 2+ hour content: VideoAgent or similar retrieval patterns. Alternatively, summarize to smaller chunks and feed hierarchical summaries. + +### 2026 production pattern + +In practice, production long-video pipelines are hybrid: + +1. Run dynamic-FPS sampling + aggressive pooling on the entire video (get a 100k-token global representation). +2. Pass to a 72B VLM for a global summary. +3. If user asks detailed questions, run agentic retrieval using the summary as an index. + +This combines brute-context for global understanding and retrieval for local detail. + +## Use It + +`code/main.py`: + +- Computes token budgets for videos from 1 minute to 3 hours at varying FPS + pooling. +- Simulates a needle-in-a-haystack run: inject a marker at a random timestamp, ask a question, score recall. +- Includes an agentic-retrieval router simulator that picks specific clips to feed to a downstream VLM. + +Run the budget table and feel the scale gap. + +## Ship It + +This lesson produces `outputs/skill-long-video-strategy-planner.md`. Given a video duration and query complexity, it picks between brute-context, compression, and agentic retrieval, and computes the latency + quality expectations. + +## Exercises + +1. A 45-minute lecture at 1 FPS, 81 tokens per frame. Total tokens? Fits in which models' contexts? + +2. Design a needle-in-a-haystack test: at what minute do you inject the marker, and what is the exact query format? + +3. Compare brute-context Qwen2.5-VL-72B (80k context) to VideoAgent (Claude 3.5 + retrieval) on a 1-hour video. Which wins on recall? Which wins on latency? + +4. Ring attention's memory cost scales linearly in sequence length and linearly in device count. Explain why and what fails if you drop the ring-rotation phase. + +5. Read Gemini 1.5 Section 5 on needle-in-a-haystack. What did the paper find about recall at the 1M vs 10M token boundary? + +## Key Terms + +| Term | What people say | What it actually means | +|------|-----------------|------------------------| +| Brute context | "Just more tokens" | Scale LLM context to millions of tokens; process everything in one pass | +| Ring attention | "LWM-style parallel" | Distributed attention pattern where each device holds a chunk and rotates | +| Token compression | "Summary tokens" | Reduce per-clip tokens via a learned compressor before the LLM | +| Needle-in-haystack | "NIH test" | Insert a unique marker at a random point, ask model to recall it at test time | +| Agentic retrieval | "LLM as query planner" | LLM asks a retrieval tool for relevant clips, reads them via a VLM, composes answer | +| VideoAgent | "Retrieval pattern for video" | Canonical agentic-retrieval design: question -> tool -> clip -> answer | + +## Further Reading + +- [Gemini Team — Gemini 1.5 (arXiv:2403.05530)](https://arxiv.org/abs/2403.05530) +- [Liu et al. — LWM / RingAttention (arXiv:2402.08268)](https://arxiv.org/abs/2402.08268) +- [Xue et al. — LongVILA (arXiv:2408.10188)](https://arxiv.org/abs/2408.10188) +- [Shu et al. — Video-XL (arXiv:2409.14485)](https://arxiv.org/abs/2409.14485) +- [Wang et al. — VideoAgent (arXiv:2403.10517)](https://arxiv.org/abs/2403.10517) diff --git a/phases/12-multimodal-ai/18-long-video-million-token/notebook/.gitkeep b/phases/12-multimodal-ai/18-long-video-million-token/notebook/.gitkeep new file mode 100644 index 000000000..e69de29bb diff --git a/phases/12-multimodal-ai/18-long-video-million-token/outputs/skill-long-video-strategy-planner.md b/phases/12-multimodal-ai/18-long-video-million-token/outputs/skill-long-video-strategy-planner.md new file mode 100644 index 000000000..e7f304088 --- /dev/null +++ b/phases/12-multimodal-ai/18-long-video-million-token/outputs/skill-long-video-strategy-planner.md @@ -0,0 +1,31 @@ +--- +name: long-video-strategy-planner +description: Pick brute-context, ring-attention, token-compression, or agentic-retrieval for a long-video understanding task and compute latency + recall expectations. +version: 1.0.0 +phase: 12 +lesson: 18 +tags: [long-video, gemini, ring-attention, videoagent, retrieval] +--- + +Given a video duration, query complexity (single event vs holistic summary), and open vs closed constraints, pick a long-video strategy and emit a config. + +Produce: + +1. Strategy pick. Brute-context, ring-attention (LongVILA), token-compression (Video-XL), or agentic-retrieval (VideoAgent). +2. Token budget. Duration * FPS * per-frame-tokens. Warn if > LLM context. +3. Expected recall. Needle-in-a-haystack recall at video-length percentiles. Cite Gemini 1.5 reports when relevant. +4. Latency. Prefill time for brute-context; retrieval + VLM for agentic. +5. Engineering path. Code snippet scaffold for the chosen strategy. +6. Fallback plan. Hybrid: brute-context global summary + agentic local detail. + +Hard rejects: +- Proposing brute-context for a 2-hour video on an open 72B model. Context does not fit. +- Claiming agentic retrieval always wins. For holistic-summary questions it loses to brute context. +- Recommending token compression without flagging the recall tax. + +Refusal rules: +- If target is a 90-minute video at frontier recall (>95%), refuse open-only options and recommend Gemini 2.5 Pro. +- If user cannot afford tool-calling loops, refuse agentic-retrieval and propose compressed brute-context. +- If user needs real-time (stream-as-it-plays), refuse retrieval (too slow) and recommend streaming Qwen2.5-VL. + +Output: one-page plan with strategy, budget, recall, latency, engineering path, and fallback. End with arXiv 2403.05530 (Gemini 1.5) and 2403.10517 (VideoAgent) for comparison.