cd /news/artificial-intelligence/google-gemini-video-intelligence-api · home topics artificial-intelligence article
[ARTICLE · art-126755] src=gist.github.com ↗ pub= topic=artificial-intelligence verified=true sentiment=· neutral

Google Gemini Video Intelligence API

A developer published a Python script that uses Google's Gemini API to analyze YouTube videos, extracting structured transcripts, concepts, timestamps, and relationship graphs into JSON. The tool supports an "agentic" processing mode and parses Gemini's reasoning steps, with a fallback for older API versions.

by read6 min views20 publishedSep 3, 2026

| | import json | | | import os | | | from typing import Any | | | from google import genai | | | # 1. Configuration & Input | | | # video_id = "88I6IidylGc" | | | video_id = "2lm1DFGFxPs" | | | YOUTUBE_URL = f"https://www.youtube.com/watch?v={video_id}" | | | processing_mode = "agentic" | | | # Use the script's directory as the base for all output files |

|  | _SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__)) | 
|  | OUTPUT_FILE = os.path.join(_SCRIPT_DIR, f"transcript_concepts_{video_id}_{processing_mode}.json") | 
|  | THOUGHTS_OUTPUT_FILE = os.path.join(_SCRIPT_DIR, f"transcript_concepts_thoughts_{video_id}_{processing_mode}.json") | 
|  | STEPS_OUTPUT_FILE = os.path.join(_SCRIPT_DIR, f"transcript_concepts_steps_{video_id}_{processing_mode}.json") | 

| | # 2. Schema Definition (Standard JSON Schema with lowercase types) | | | _YOUTUBE_RESPONSE_SCHEMA: dict[str, Any] = { | | | "type": "object", |

|  | "properties": { | 
|  | "videoTitle": {"type": "string"}, | 
|  | "contentType": {"type": "string", "enum": ["educational", "entertainment", "mixed"]}, | 
|  | "concepts": { | 

| | "type": "array", | | | "items": { | | | "type": "object", |

|  | "properties": { | 
|  | "label": {"type": "string"}, | 
|  | "timestamps": { | 

| | "type": "array", | | | "items": { | | | "type": "object", |

|  | "properties": { | 
|  | "seconds": {"type": "string"}, | 
|  | "description": {"type": "string"}, | 
|  | "quote": {"type": "string"}, | 
|  | "visual": {"type": "boolean"}, | 
|  | "visualDescription": {"type": "string"}, | 

| | }, | | | "required": ["seconds", "description", "quote"], | | | }, | | | }, | | | }, | | | "required": ["label", "timestamps"], | | | }, | | | }, | | | "edges": { | | | "type": "array", | | | "items": { | | | "type": "object", |

|  | "properties": { | 
|  | "fromLabel": {"type": "string"}, | 
|  | "toLabel": {"type": "string"}, | 
|  | "relation": {"type": "string", "enum": ["detail", "sibling", "abstraction", "related"]}, | 
|  | "rationale": {"type": "string"}, | 

| | }, | | | "required": ["fromLabel", "toLabel", "relation"], | | | }, | | | }, | | | "checkpoints": { | | | "type": "array", | | | "items": { | | | "type": "object", |

|  | "properties": { | 
|  | "conceptLabel": {"type": "string"}, | 
|  | "question": {"type": "string"}, | 
|  | "answer": {"type": "string"}, | 

| | }, | | | "required": ["conceptLabel", "question", "answer"], | | | }, | | | }, | | | }, | | | "required": ["videoTitle", "contentType", "concepts"], | | | } |

|  | def serialize_item(item: Any) -> Any: | 
|  | if hasattr(item, "model_dump"): | 
|  | return item.model_dump() | 
|  | elif hasattr(item, "__dict__"): | 

| | return { | | | k: str(v) if not isinstance(v, (dict, list, int, float, bool, type(None))) else v | | | for k, v in item.dict.items() | | | } |

|  | return str(item) | 
|  | def extract_thoughts(interaction: Any) -> list[str]: | 
|  | thoughts = [] | 

| | # May 2026 Steps Schema parsing |

|  | for step in getattr(interaction, "steps", []) or []: | 
|  | step_type = getattr(step, "type", None) or (step.get("type") if isinstance(step, dict) else None) | 
|  | if step_type == "thought": | 
|  | summary = getattr(step, "summary", []) or (step.get("summary") if isinstance(step, dict) else []) | 

| | for item in summary: | | | text = getattr(item, "text", None) or (item.get("text") if isinstance(item, dict) else None) | | | if text: | | | thoughts.append(text) | | | # Legacy fallback for older API versions | | | if not thoughts: |

|  | for candidate in getattr(interaction, "candidates", []) or []: | 
|  | for part in getattr(getattr(candidate, "content", None), "parts", []): | 
|  | if getattr(part, "thought", False): | 
|  | thoughts.append(part.text) | 

| | return thoughts | | | def get_interaction_text(interaction: Any) -> str: | | | if hasattr(interaction, "outputs") and interaction.outputs: |

|  | for output in reversed(interaction.outputs): | 
|  | if hasattr(output, "text") and output.text: | 

| | return output.text | | | if hasattr(output, "content") and getattr(output.content, "parts", None): | | | for part in output.content.parts: | | | if getattr(part, "text", None) and not getattr(part, "thought", False): | | | return part.text | | | if hasattr(interaction, "steps") and interaction.steps: |

|  | for step in reversed(interaction.steps): | 
|  | model_res = getattr(step, "model_response", None) or getattr(step, "response", None) | 
|  | if model_res and hasattr(model_res, "candidates"): | 
|  | for candidate in model_res.candidates or []: | 
|  | for part in getattr(getattr(candidate, "content", None), "parts", []): | 
|  | if getattr(part, "text", None) and not getattr(part, "thought", False): | 

| | return part.text |

|  | return getattr(interaction, "output_text", "{}") | 
|  | def analyze_youtube_video(url: str, output_path: str, thoughts_path: str, steps_path: str): | 

| | print("Sending request to Gemini via Interactions API...") |

|  | client = genai.Client() | 
|  | prompt = ( | 

| | f"Analyze the YouTube video ({url}) and identify its key structural units. " | | | "For each unit you must log the EXACT timecode in 'MM:SS' format as it first appears in the video's underlying audio stream. " | | | "Try to find at least 3 timestamps per concept label" | | | "Return in the seconds field in the response schema, every timestamp where a unit is first discussed in HH:MM:SS format or MM:SS if video is less than an hour." | | | ) | | | interaction = client.interactions.create( |

|  | model="gemini-3.5-flash-lite", | 
|  | input=[ | 

| | { | | | "type": "text", | | | "text": prompt, | | | }, | | | { | | | "type": "video", | | | "uri": url, | | | "processing": processing_mode | | | } | | | ], | | | response_format={ | | | "type": "text", | | | "mime_type": "application/json", | | | "schema": _YOUTUBE_RESPONSE_SCHEMA, | | | }, | | | generation_config={ | | | "temperature": 0.1, | | | "thinking_summaries": "auto", | | | }, | | | ) | | | # 1. Save thinking process |

|  | thoughts = extract_thoughts(interaction) | 
|  | with open(thoughts_path, "w", encoding="utf-8") as f: | 
|  | json.dump({"interaction_id": getattr(interaction, "id", None), "thoughts": thoughts}, f, indent=2) | 

| | # 2. Save steps |

|  | raw_steps = getattr(interaction, "steps", []) or [] | 
|  | serialized_steps = [serialize_item(step) for step in raw_steps] | 
|  | with open(steps_path, "w", encoding="utf-8") as f: | 
|  | json.dump({"interaction_id": getattr(interaction, "id", None), "steps": serialized_steps}, f, indent=2) | 

| | # 3. Extract and parse structured text |

|  | raw_text = get_interaction_text(interaction) | 
|  | data = json.loads(raw_text) | 
|  | with open(output_path, "w", encoding="utf-8") as f: | 
|  | json.dump(data, f, indent=2, ensure_ascii=False) | 
|  | print(f"Successfully saved structured output to '{output_path}'!") | 

| | return data | | | result = analyze_youtube_video(YOUTUBE_URL, OUTPUT_FILE, THOUGHTS_OUTPUT_FILE, STEPS_OUTPUT_FILE) | | | print(json.dumps(result, indent=2)) |

── more in #artificial-intelligence 4 stories · sorted by recency
── more on @google 3 stories trending now
sponsored brought to you by zahid.host 4,200+ EU-deployed projects
reading about agents? ship yours in a single git push.

Run your AI side-project on zahid.host

EU-based hosting, git-push deploys, automatic HTTPS, no cold starts. Free tier with a custom domain — perfect for shipping the agent you just read about.

$git push zahid main
Live at https://your-agent.zahid.host
Get free account → Pricing
from €0/mo · no card required
LIVE [news/google-gemini-video-…] indexed:0 read:6min 2026-09-03 ·