{
  "name": "whole-film-01-structure",
  "model": "gemini-3-flash",
  "transport": "inline_data (base64 bytes of the actual MP4 incl. audio)",
  "endpoint": "https://api.kie.ai/gemini/v1/models/gemini-3-flash-v1betamodels:streamGenerateContent",
  "prompt": "You are given the COMPLETE original film (325.68 seconds, vertical) with its soundtrack. Watch and listen to all of it before answering. This is an evidence-gathering pass for an atomic scene/shot/event analysis; do not summarize, do not evaluate health claims, do not invent performance data.\n\nReturn ONLY valid JSON (no markdown fences) with these keys:\n\n\"access\": {\"heard_audio\": bool, \"anchors\": [three distinctive items you actually heard or saw near the beginning, the middle and the very end, each with approximate time in seconds], \"captions_present_in_original\": bool, \"music_present\": \"none|continuous|intermittent|unsure\", \"ambience_or_sfx\": short description}\n\n\"transcript\": [every audible utterance in order: {\"start_s\", \"end_s\", \"speaker\": role label you infer from picture (e.g. doctor, man, woman, girl, older man, older woman, narrator, phone voice), \"text\": exact words, \"delivery\": 3-8 words on pace/intonation/volume/breath, \"listener_visible\": who is on screen listening, \"confidence\": 0-1}]. Transcribe every word, including interjections, repetitions, numbers and the closing offer. Never replace speech with ellipses.\n\n\"scenes\": [units where the situation changes consequentially: {\"id\": \"S1\"..., \"start_s\", \"end_s\", \"location\", \"present\": [roles], \"what_happens\": 1-3 sentences of observable action, \"information_changed\": what a viewer now knows that they did not before, \"question_opened\": the concrete question the viewer now holds, \"question_closed\": earlier question answered here or null, \"character_state_change\": who changes and how (observable), \"sound\": speech/silence/ambience notes, \"boundary_reason\": why this scene starts here (what changed), \"confidence\": 0-1}]\n\n\"shots\": [every visible hard cut or shot change you can detect across the whole film, in order: {\"start_s\", \"end_s\", \"scale\": ECU|CU|MCU|MS|MLS|LS|insert, \"angle_or_eyeline\": e.g. over-shoulder, frontal, profile, high, low, POV, \"subject\": who/what, \"action\": what visibly happens inside the shot including hands, gaze, body, props, \"speech_over_shot\": who speaks during it (may be off-screen), \"camera_motion\": static|push|pan|handheld|other, \"transition_in\": cut|dissolve|other, \"confidence\": 0-1}]. Cover the ENTIRE runtime; do not stop early. Timestamps are approximate; label them as such via confidence.\n\n\"text_on_screen\": [any on-screen text you see: {\"start_s\", \"end_s\", \"text\", \"position\", \"style\"}]\n\n\"uncertainties\": [things you could not resolve: unclear words, ambiguous cut boundaries, whether footage is generated, which speaker is which]\n",
  "prompt_sha256": "88d4fceb86d8b197d11fbc176f5b2999dab95991fc24191c95aac60a3db3729a",
  "media": {
    "path": "/workspaces/UGS/AdFactoryNarrativeFormat/trees/bf8f11bde4/analysis/original-atomic/proxy/whole-film-360p.mp4",
    "sha256": "c7ee1c36c8639490ee7aae79f6fefda4a5ac3d6dbfc54c51f40069f4cb68b495",
    "bytes": 14124934,
    "mime": "video/mp4"
  },
  "extra_media": [],
  "generation_config": {
    "thinkingConfig": {
      "includeThoughts": false,
      "thinkingLevel": "high"
    },
    "maxOutputTokens": 60000,
    "temperature": 0.2,
    "mediaResolution": "MEDIA_RESOLUTION_HIGH"
  },
  "meta": {
    "time_scale": 1.0,
    "source_start_s": 0.0,
    "coverage": "whole film via 360p proxy with full soundtrack"
  }
}
