{
 "S1sh3::signage": {
  "fp": "7ad460dd49caa3f7",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "era_assess::fc900d5ccc77d6e4": {
  "subjects": [],
  "subject_text": "절벽 상공의 구름 낀 하늘\n절벽과 숲 위로 펼쳐진 구름 덮인 낮 하늘. 두꺼운 구름층 사이로 확산된 창백한 자연광이 비치는 넓은 상공.",
  "identity": "canonical",
  "scope_id": "L06",
  "scope_role": "location_exterior",
  "scope_sha": "1337f9470de92a34"
 },
 "groupbg::forest_canopy_zone": {
  "input_fingerprint": "71158e4f2054f813",
  "meta": {
   "model": "gpt-image-2",
   "size": "1536x864",
   "pack": "11.202607220237",
   "contract": "bgfirst_full_v3",
   "group_sig": {
    "key": "forest_canopy_zone",
    "tags": [
     "S1sh3",
     "S1sh7",
     "S1sh8",
     "S2sh11",
     "S2sh12",
     "S2sh5"
    ]
   },
   "context_sig": "b47dee6951634367"
  },
  "prompt": "Create ONE empty live-action location background photograph — NO PEOPLE, no figures, no body parts, no silhouettes, no shadows or reflections of people anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods and produce in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch of a shot that happens at this location: use it ONLY as spatial evidence — what this place contains, how its ground, structures and landmarks are arranged and proportioned. Ignore the sketched people and arrows entirely, and do NOT copy its line style: render a fully photographic, physically plausible real place that fits THE LOCATION text below.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nTHE LOCATION — far-future Pennsylvania, United States; all people are American and English-speaking unless stated: Open exterior airspace just beyond the cliff, high above the dense forest canopy.\n\nSCENE EVIDENCE (verbatim quotes from the screenplay about this place — treat them as evidence of what the location physically contains and looks like; stage the PLACE those moments happen in, but do NOT depict the momentary actions, people or staged props themselves):\n- S1. EXT. 펜실베이니아 숲 상공 — 낮 (00:00–00:40)\n- 두 사람이 거대한 나뭇가지에 충돌하듯 착지한다.\n- S2. EXT. 펜실베이니아 숲 상공 — 낮 (00:40–01:25)\n- 윌마가 토니를 나무 뒤로 끌어당긴다.\n\nTIME OF DAY (lock): day.\n\nRender ONE photorealistic empty location photograph, 16:9, neutral enough that every shot of this place can be staged from it later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "effective_prompt": "Create ONE empty live-action location background photograph — NO PEOPLE, no figures, no body parts, no silhouettes, no shadows or reflections of people anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods and produce in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch of a shot that happens at this location: use it ONLY as spatial evidence — what this place contains, how its ground, structures and landmarks are arranged and proportioned. Ignore the sketched people and arrows entirely, and do NOT copy its line style: render a fully photographic, physically plausible real place that fits THE LOCATION text below.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nTHE LOCATION — far-future Pennsylvania, United States; all people are American and English-speaking unless stated: Open exterior airspace just beyond the cliff, high above the dense forest canopy.\n\nSCENE EVIDENCE (verbatim quotes from the screenplay about this place — treat them as evidence of what the location physically contains and looks like; stage the PLACE those moments happen in, but do NOT depict the momentary actions, people or staged props themselves):\n- S1. EXT. 펜실베이니아 숲 상공 — 낮 (00:00–00:40)\n- 두 사람이 거대한 나뭇가지에 충돌하듯 착지한다.\n- S2. EXT. 펜실베이니아 숲 상공 — 낮 (00:40–01:25)\n- 윌마가 토니를 나무 뒤로 끌어당긴다.\n\nTIME OF DAY (lock): day.\n\nRender ONE photorealistic empty location photograph, 16:9, neutral enough that every shot of this place can be staged from it later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/groupbg_forest_canopy_zone_67abf7.png",
  "asset_id": "29bdc411-d997-4693-ba25-910c6ce4aef6",
  "input_asset_ids": [
   "538710e3-018b-4dc4-848e-8da3919385e1"
  ],
  "origin_tag": "S1sh3",
  "place_text": "Open exterior airspace just beyond the cliff, high above the dense forest canopy.",
  "origin_inputs": {
   "place_text": "Open exterior airspace just beyond the cliff, high above the dense forest canopy.",
   "time_of_day_en": "day",
   "conti_asset_id": "538710e3-018b-4dc4-848e-8da3919385e1"
  }
 },
 "S1sh3::bgfirst_bg": {
  "input_fingerprint": "ed507211a2ad5419",
  "prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 발아래 울창한 나무 꼭대기들이 펼쳐진 허공을 가로지르는 비행 궤도의 한가운데, 옷자락이 비행 반대 방향으로 거세게 휘날리는 윌마와 토니\n\nLOCATION (lock): Open exterior airspace just beyond the cliff, high above the dense forest canopy.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the upper-center of the frame, midground, moves toward curving flight path; 토니(앤서니 로저스) in the upper-right of the frame, midground, moves toward curving flight path; treetops beneath the pair in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: treetops (dense beneath the flight path) — Seen diagonally downward as a continuous canopy directly below the airborne pair; used as Provides depth, danger, and the scale reference beneath the subjects.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight with restrained, moderate contrast keeps the airborne figures clearly separated from the forest below.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "effective_prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 발아래 울창한 나무 꼭대기들이 펼쳐진 허공을 가로지르는 비행 궤도의 한가운데, 옷자락이 비행 반대 방향으로 거세게 휘날리는 윌마와 토니\n\nLOCATION (lock): Open exterior airspace just beyond the cliff, high above the dense forest canopy.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the upper-center of the frame, midground, moves toward curving flight path; 토니(앤서니 로저스) in the upper-right of the frame, midground, moves toward curving flight path; treetops beneath the pair in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: treetops (dense beneath the flight path) — Seen diagonally downward as a continuous canopy directly below the airborne pair; used as Provides depth, danger, and the scale reference beneath the subjects.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight with restrained, moderate contrast keeps the airborne figures clearly separated from the forest below.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S1sh3__bgfirst_bg.png",
  "asset_id": "6e74e00e-4c81-4868-ae92-4916ae8ce4c0",
  "input_asset_ids": [
   "538710e3-018b-4dc4-848e-8da3919385e1",
   "29bdc411-d997-4693-ba25-910c6ce4aef6"
  ]
 },
 "S1sh3": {
  "input_fingerprint": "f981b131a72cb6df",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 발아래 울창한 나무 꼭대기들이 펼쳐진 허공을 가로지르는 비행 궤도의 한가운데, 옷자락이 비행 반대 방향으로 거세게 휘날리는 윌마와 토니\n\nLOCATION (lock): Open exterior airspace just beyond the cliff, high above the dense forest canopy. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the upper-center of the frame, midground, moves toward curving flight path; 토니(앤서니 로저스) in the upper-right of the frame, midground, moves toward curving flight path; treetops beneath the pair in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: treetops (dense beneath the flight path) — Seen diagonally downward as a continuous canopy directly below the airborne pair; used as Provides depth, danger, and the scale reference beneath the subjects.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight with restrained, moderate contrast keeps the airborne figures clearly separated from the forest below.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Dense treetops fill the space below in daylight. The cliff edge remains intact at this moment. 윌마 디어링: She is airborne on a curving trajectory, wearing her weighted jumper belt and carrying a gun. 토니(앤서니 로저스): He is airborne with stiffly extended legs, carrying his phone and rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Stage the shot. The FIRST attached image (SHOT BACKGROUND) is the finished empty background of this shot — keep it EXACTLY: its camera, perspective, architecture, lighting and every fixed feature stay untouched. The SECOND attached image (LAYOUT SKETCH) tells you ONLY where the people go: each sketched person's position, screen size, pose and the gaze/motion arrows. Ignore the sketch's background lines. The CHARACTER REFERENCE photographs show the real people.\nPlace the real people into the background at exactly the sketched positions, sizes and poses, following the arrow directions. No sketch lines or arrows may remain.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 발아래 울창한 나무 꼭대기들이 펼쳐진 허공을 가로지르는 비행 궤도의 한가운데, 옷자락이 비행 반대 방향으로 거세게 휘날리는 윌마와 토니\n\nLOCATION (lock): Open exterior airspace just beyond the cliff, high above the dense forest canopy. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the upper-center of the frame, midground, moves toward curving flight path; 토니(앤서니 로저스) in the upper-right of the frame, midground, moves toward curving flight path; treetops beneath the pair in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: treetops (dense beneath the flight path) — Seen diagonally downward as a continuous canopy directly below the airborne pair; used as Provides depth, danger, and the scale reference beneath the subjects.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight with restrained, moderate contrast keeps the airborne figures clearly separated from the forest below.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Dense treetops fill the space below in daylight. The cliff edge remains intact at this moment. 윌마 디어링: She is airborne on a curving trajectory, wearing her weighted jumper belt and carrying a gun. 토니(앤서니 로저스): He is airborne with stiffly extended legs, carrying his phone and rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 발아래 울창한 나무 꼭대기들이 펼쳐진 허공을 가로지르는 비행 궤도의 한가운데, 옷자락이 비행 반대 방향으로 거세게 휘날리는 윌마와 토니\n\nLOCATION (lock): Open exterior airspace just beyond the cliff, high above the dense forest canopy. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the upper-center of the frame, midground, moves toward curving flight path; 토니(앤서니 로저스) in the upper-right of the frame, midground, moves toward curving flight path; treetops beneath the pair in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: treetops (dense beneath the flight path) — Seen diagonally downward as a continuous canopy directly below the airborne pair; used as Provides depth, danger, and the scale reference beneath the subjects.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight with restrained, moderate contrast keeps the airborne figures clearly separated from the forest below.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Dense treetops fill the space below in daylight. The cliff edge remains intact at this moment. 윌마 디어링: She is airborne on a curving trajectory, wearing her weighted jumper belt and carrying a gun. 토니(앤서니 로저스): He is airborne with stiffly extended legs, carrying his phone and rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S1sh3__bgfirst_bg.png",
     "asset_id": "6e74e00e-4c81-4868-ae92-4916ae8ce4c0",
     "role": "bgfirst_bg"
    },
    {
     "label": "LAYOUT SKETCH (people placement only)",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/conti/conti_S1sh3.png",
     "asset_id": "538710e3-018b-4dc4-848e-8da3919385e1",
     "role": "conti_light"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ],
   "B": [
    {
     "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/groupbg_forest_canopy_zone_67abf7.png",
     "asset_id": "29bdc411-d997-4693-ba25-910c6ce4aef6",
     "role": "bgfirst_group_bg"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "judge_flip_skipped": "judge_fn_owns_order",
  "readings": [
   {
    "label": "A",
    "direction": "윌마는 정면 앞쪽을 향해 총을 겨누고 있고, 토니는 한 팔을 앞으로 뻗으며 날아가고 있음. 윌마의 옷자락이 비행 반대 방향인 뒤로 거세게 휘날림.",
    "built_space": "화면 왼쪽에 바위 절벽과 뻗어나온 나무 기둥이 있고, 그 아래 배경으로 울창한 숲의 수관이 넓게 펼쳐진 야외 상공임.",
    "entities": "윌마는 녹색 의상을 입고 총을 들고 있으나 레퍼런스에 없는 펄럭이는 긴 코트 형태가 추가됨. 토니는 헬멧을 착용하지 않았고 손에 폰이 없으며 다리를 뻗지 않은 채 수평으로 떠 있음.",
    "hard_violations": [
     "[gemini-pro] 물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
    ],
    "physics": "두 인물 모두 공중에 떠서 수평으로 비행 중이나, 이들의 몸을 띄우거나 지탱하는 물리적 지지 요소 및 도약 흔적이 전혀 보이지 않음 (nothing supports it)."
   },
   {
    "label": "B",
    "direction": "윌마는 정면 앞을 바라보며 비행하고, 토니는 손에 든 폰을 내려다보고 있음. 윌마의 짧은 머리카락과 옷자락이 비행 반대 방향으로 펄럭임.",
    "built_space": "화면 왼쪽 가장자리에 깎아지른 절벽 지형이 보이며, 두 인물의 발아래로 수많은 나무가 들어찬 울창한 숲이 펼쳐진 상공임.",
    "entities": "윌마는 레퍼런스와 일치하는 녹색 작업복 차림으로 총을 들고 있음. 토니는 헬멧과 전술복을 갖춰 입고 구조용 밧줄을 매단 채 폰을 손에 들고 다리를 꼿꼿하게 뻗고 있음.",
    "hard_violations": [
     "[gemini-pro] 물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
    ],
    "physics": "두 인물 모두 다리를 구부리거나 뻗은 포즈로 공중에 체공 중이나, 허공에서 몸을 지지하는 물리적 장치나 추진 요소, 도약점이 전혀 묘사되지 않음 (nothing supports it)."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 5,
        "verdict_ko": "복장, 소품(폰, 밧줄), 꼿꼿이 뻗은 다리 포즈 등 프롬프트의 지시를 대부분 훌륭히 구현했으나, 허공에 떠 있는 인물들의 몸을 지탱하는 물리적 요소가 전혀 묘사되지 않아 물리법칙 규정을 위반했습니다."
       },
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "토니의 헬멧과 폰이 누락되었고 윌마에게 레퍼런스에 없는 긴 옷자락이 추가되었으며, 두 인물 모두 지지대 없이 허공에 떠 있어 심각한 규정 위반이 발생했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마는 정면 앞쪽을 향해 총을 겨누고 있고, 토니는 한 팔을 앞으로 뻗으며 날아가고 있음. 윌마의 옷자락이 비행 반대 방향인 뒤로 거세게 휘날림.",
        "built_space": "화면 왼쪽에 바위 절벽과 뻗어나온 나무 기둥이 있고, 그 아래 배경으로 울창한 숲의 수관이 넓게 펼쳐진 야외 상공임.",
        "entities": "윌마는 녹색 의상을 입고 총을 들고 있으나 레퍼런스에 없는 펄럭이는 긴 코트 형태가 추가됨. 토니는 헬멧을 착용하지 않았고 손에 폰이 없으며 다리를 뻗지 않은 채 수평으로 떠 있음.",
        "hard_violations": [
         "물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
        ],
        "physics": "두 인물 모두 공중에 떠서 수평으로 비행 중이나, 이들의 몸을 띄우거나 지탱하는 물리적 지지 요소 및 도약 흔적이 전혀 보이지 않음 (nothing supports it)."
       },
       {
        "label": "B",
        "direction": "윌마는 정면 앞을 바라보며 비행하고, 토니는 손에 든 폰을 내려다보고 있음. 윌마의 짧은 머리카락과 옷자락이 비행 반대 방향으로 펄럭임.",
        "built_space": "화면 왼쪽 가장자리에 깎아지른 절벽 지형이 보이며, 두 인물의 발아래로 수많은 나무가 들어찬 울창한 숲이 펼쳐진 상공임.",
        "entities": "윌마는 레퍼런스와 일치하는 녹색 작업복 차림으로 총을 들고 있음. 토니는 헬멧과 전술복을 갖춰 입고 구조용 밧줄을 매단 채 폰을 손에 들고 다리를 꼿꼿하게 뻗고 있음.",
        "hard_violations": [
         "물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
        ],
        "physics": "두 인물 모두 다리를 구부리거나 뻗은 포즈로 공중에 체공 중이나, 허공에서 몸을 지지하는 물리적 장치나 추진 요소, 도약점이 전혀 묘사되지 않음 (nothing supports it)."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 5,
        "verdict_ko": "복장, 소품(폰, 밧줄), 꼿꼿이 뻗은 다리 포즈 등 프롬프트의 지시를 대부분 훌륭히 구현했으나, 허공에 떠 있는 인물들의 몸을 지탱하는 물리적 요소가 전혀 묘사되지 않아 물리법칙 규정을 위반했습니다."
       },
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "토니의 헬멧과 폰이 누락되었고 윌마에게 레퍼런스에 없는 긴 옷자락이 추가되었으며, 두 인물 모두 지지대 없이 허공에 떠 있어 심각한 규정 위반이 발생했습니다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "A",
        "direction": "윌마는 정면 앞쪽을 향해 총을 겨누고 있고, 토니는 한 팔을 앞으로 뻗으며 날아가고 있음. 윌마의 옷자락이 비행 반대 방향인 뒤로 거세게 휘날림.",
        "built_space": "화면 왼쪽에 바위 절벽과 뻗어나온 나무 기둥이 있고, 그 아래 배경으로 울창한 숲의 수관이 넓게 펼쳐진 야외 상공임.",
        "entities": "윌마는 녹색 의상을 입고 총을 들고 있으나 레퍼런스에 없는 펄럭이는 긴 코트 형태가 추가됨. 토니는 헬멧을 착용하지 않았고 손에 폰이 없으며 다리를 뻗지 않은 채 수평으로 떠 있음.",
        "hard_violations": [
         "물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
        ],
        "physics": "두 인물 모두 공중에 떠서 수평으로 비행 중이나, 이들의 몸을 띄우거나 지탱하는 물리적 지지 요소 및 도약 흔적이 전혀 보이지 않음 (nothing supports it)."
       },
       {
        "label": "B",
        "direction": "윌마는 정면 앞을 바라보며 비행하고, 토니는 손에 든 폰을 내려다보고 있음. 윌마의 짧은 머리카락과 옷자락이 비행 반대 방향으로 펄럭임.",
        "built_space": "화면 왼쪽 가장자리에 깎아지른 절벽 지형이 보이며, 두 인물의 발아래로 수많은 나무가 들어찬 울창한 숲이 펼쳐진 상공임.",
        "entities": "윌마는 레퍼런스와 일치하는 녹색 작업복 차림으로 총을 들고 있음. 토니는 헬멧과 전술복을 갖춰 입고 구조용 밧줄을 매단 채 폰을 손에 들고 다리를 꼿꼿하게 뻗고 있음.",
        "hard_violations": [
         "물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
        ],
        "physics": "두 인물 모두 다리를 구부리거나 뻗은 포즈로 공중에 체공 중이나, 허공에서 몸을 지지하는 물리적 장치나 추진 요소, 도약점이 전혀 묘사되지 않음 (nothing supports it)."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "두 인물이 같은 좌향 비행 흐름을 이루고 옷자락과 로프가 반대쪽으로 강하게 날리며, 배치와 절벽·수관의 장소 일치도도 가장 높다; 다만 토니의 헬멧과 휴대전화가 확인되지 않는다."
       },
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "정확한 인물 배치와 토니의 헬멧·전화·로프는 잘 맞지만, 윌마는 오른쪽으로, 토니는 왼쪽으로 향해 한 쌍의 곡선 비행 궤도로 읽히는 통일성이 B보다 약하다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마의 몸과 시선은 화면 오른쪽을 향하고 총구도 오른쪽 아래 허공을 향하며, 윌마의 긴 옷자락은 반대인 왼쪽으로 날린다. 토니는 뻗은 다리와 몸의 기울기로 보아 왼쪽을 향하는 듯하고 옷자락은 오른쪽으로 날리며, 시선은 손에 든 전화로 향한다. 따라서 각자의 역풍 표현은 있으나 두 사람의 이동 방향은 서로 반대로 읽힌다.",
        "built_space": "개방된 절벽 바깥 공중이며 화면 왼쪽에 암벽 가장자리, 아래와 원경에 연속된 울창한 수관이 보인다. 윌마는 상단 중앙의 중경, 토니는 상단 오른쪽의 중경에 있어 지정 배치와 맞는다. 인공 구조물이나 고정 설비, 반사는 없다.",
        "entities": "두 사람만 보이며 윌마는 젊은 미국인 여성, 토니는 30대 미국인 남성으로 읽힌다. 윌마는 녹색 복장과 벨트, 장총을 갖췄고 토니는 참조와 유사한 검은 장비·헬멧, 손에 든 전화와 허리에 매단 구조 로프를 갖췄다. 인물 크기가 작아 얼굴의 정확한 동일성은 제한적으로만 확인된다.",
        "hard_violations": [],
        "physics": "두 사람 모두 점프 벨트·하네스 장비가 몸을 지지하는 미래형 비행 장치로 읽히며, 옷과 머리카락의 후류가 실제 이동을 나타낸다. 윌마는 무릎을 굽힌 역동적 비행 자세이고 총을 양손으로 잡고 있다. 토니는 요구된 대로 두 다리를 뻣뻣하게 뻗고 전화는 손으로, 로프는 벨트에 고정해 지닌다. 토니의 다소 앉은 듯한 자세는 어색하지만 무지지 정지 부유로 단정할 정도는 아니다."
       },
       {
        "label": "B",
        "direction": "윌마와 토니 모두 화면 왼쪽을 향해 나란히 비행한다. 윌마의 시선과 권총 총구는 왼쪽의 절벽 부근 빈 공중을 향하며 명시된 표적은 없다. 토니의 시선과 뻗은 왼손도 윌마와 왼쪽 비행 방향을 향한다. 두 사람의 옷자락과 윌마의 머리카락, 토니의 로프는 모두 반대인 오른쪽으로 강하게 흘러 동일한 비행 방향을 명확히 만든다.",
        "built_space": "참조 장소와 같은 개방된 절벽 바깥 공중으로, 왼쪽에 암벽과 굵은 쓰러진 나무, 아래에는 대각선 아래로 내려다본 연속적인 울창한 수관이 펼쳐진다. 윌마는 상단 중앙 중경, 토니는 상단 오른쪽 중경에 정확히 놓였다. 인공 설비나 불가능한 반사는 없다.",
        "entities": "지정된 윌마와 토니 두 사람만 있다. 윌마는 녹색 비행복과 벨트, 손에 든 권총을 갖춘 젊은 여성으로 맞고, 토니는 검은 전술복·하네스와 구조 로프를 갖춘 30대 남성으로 읽힌다. 다만 토니는 참조의 헬멧을 쓰지 않았고 휴대전화도 화면에서 분명히 식별되지 않는다. 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "두 인물의 벨트와 하네스가 미래형 비행 장치로 몸을 지지하는 것으로 읽히며, 수평으로 기울인 몸과 강한 의복·로프 후류가 절벽 너머 곡선 비행의 진행을 설득력 있게 만든다. 윌마는 권총을 손으로 확실히 잡고 있고, 토니는 요구대로 두 다리를 뻣뻣하게 뻗으며 오른손으로 구조 로프를 잡는다. 로프의 나머지 부분은 오른쪽 후방으로 자연스럽게 끌려가며 별도로 떠 있는 물체는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "두 인물이 같은 좌향 비행 흐름을 이루고 옷자락과 로프가 반대쪽으로 강하게 날리며, 배치와 절벽·수관의 장소 일치도도 가장 높다; 다만 토니의 헬멧과 휴대전화가 확인되지 않는다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "정확한 인물 배치와 토니의 헬멧·전화·로프는 잘 맞지만, 윌마는 오른쪽으로, 토니는 왼쪽으로 향해 한 쌍의 곡선 비행 궤도로 읽히는 통일성이 B보다 약하다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "윌마의 몸과 시선은 화면 오른쪽을 향하고 총구도 오른쪽 아래 허공을 향하며, 윌마의 긴 옷자락은 반대인 왼쪽으로 날린다. 토니는 뻗은 다리와 몸의 기울기로 보아 왼쪽을 향하는 듯하고 옷자락은 오른쪽으로 날리며, 시선은 손에 든 전화로 향한다. 따라서 각자의 역풍 표현은 있으나 두 사람의 이동 방향은 서로 반대로 읽힌다.",
        "built_space": "개방된 절벽 바깥 공중이며 화면 왼쪽에 암벽 가장자리, 아래와 원경에 연속된 울창한 수관이 보인다. 윌마는 상단 중앙의 중경, 토니는 상단 오른쪽의 중경에 있어 지정 배치와 맞는다. 인공 구조물이나 고정 설비, 반사는 없다.",
        "entities": "두 사람만 보이며 윌마는 젊은 미국인 여성, 토니는 30대 미국인 남성으로 읽힌다. 윌마는 녹색 복장과 벨트, 장총을 갖췄고 토니는 참조와 유사한 검은 장비·헬멧, 손에 든 전화와 허리에 매단 구조 로프를 갖췄다. 인물 크기가 작아 얼굴의 정확한 동일성은 제한적으로만 확인된다.",
        "hard_violations": [],
        "physics": "두 사람 모두 점프 벨트·하네스 장비가 몸을 지지하는 미래형 비행 장치로 읽히며, 옷과 머리카락의 후류가 실제 이동을 나타낸다. 윌마는 무릎을 굽힌 역동적 비행 자세이고 총을 양손으로 잡고 있다. 토니는 요구된 대로 두 다리를 뻣뻣하게 뻗고 전화는 손으로, 로프는 벨트에 고정해 지닌다. 토니의 다소 앉은 듯한 자세는 어색하지만 무지지 정지 부유로 단정할 정도는 아니다."
       },
       {
        "label": "A",
        "direction": "윌마와 토니 모두 화면 왼쪽을 향해 나란히 비행한다. 윌마의 시선과 권총 총구는 왼쪽의 절벽 부근 빈 공중을 향하며 명시된 표적은 없다. 토니의 시선과 뻗은 왼손도 윌마와 왼쪽 비행 방향을 향한다. 두 사람의 옷자락과 윌마의 머리카락, 토니의 로프는 모두 반대인 오른쪽으로 강하게 흘러 동일한 비행 방향을 명확히 만든다.",
        "built_space": "참조 장소와 같은 개방된 절벽 바깥 공중으로, 왼쪽에 암벽과 굵은 쓰러진 나무, 아래에는 대각선 아래로 내려다본 연속적인 울창한 수관이 펼쳐진다. 윌마는 상단 중앙 중경, 토니는 상단 오른쪽 중경에 정확히 놓였다. 인공 설비나 불가능한 반사는 없다.",
        "entities": "지정된 윌마와 토니 두 사람만 있다. 윌마는 녹색 비행복과 벨트, 손에 든 권총을 갖춘 젊은 여성으로 맞고, 토니는 검은 전술복·하네스와 구조 로프를 갖춘 30대 남성으로 읽힌다. 다만 토니는 참조의 헬멧을 쓰지 않았고 휴대전화도 화면에서 분명히 식별되지 않는다. 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "두 인물의 벨트와 하네스가 미래형 비행 장치로 몸을 지지하는 것으로 읽히며, 수평으로 기울인 몸과 강한 의복·로프 후류가 절벽 너머 곡선 비행의 진행을 설득력 있게 만든다. 윌마는 권총을 손으로 확실히 잡고 있고, 토니는 요구대로 두 다리를 뻣뻣하게 뻗으며 오른손으로 구조 로프를 잡는다. 로프의 나머지 부분은 오른쪽 후방으로 자연스럽게 끌려가며 별도로 떠 있는 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.4,
    "B": 1.875
   },
   "adjusted": {
    "A": 1.15,
    "B": 1.625
   },
   "violations": {
    "A": [
     "[gemini-pro] 물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
    ],
    "B": [
     "[gemini-pro] 물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "agreed": false
  },
  "totals": {
   "B": 1625,
   "A": 1150
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 1625,
    "verdict_ko": "복장, 소품(폰, 밧줄), 꼿꼿이 뻗은 다리 포즈 등 프롬프트의 지시를 대부분 훌륭히 구현했으나, 허공에 떠 있는 인물들의 몸을 지탱하는 물리적 요소가 전혀 묘사되지 않아 물리법칙 규정을 위반했습니다.  ★위반: [gemini-pro] 물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
   },
   {
    "label": "A",
    "score": 1150,
    "verdict_ko": "토니의 헬멧과 폰이 누락되었고 윌마에게 레퍼런스에 없는 긴 옷자락이 추가되었으며, 두 인물 모두 지지대 없이 허공에 떠 있어 심각한 규정 위반이 발생했습니다.  ★위반: [gemini-pro] 물리적 지지대나 추진 장치, 도약 및 착지점 없이 허공에 떠 있는 두 인물"
   }
  ],
  "refs": [
   {
    "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/groupbg_forest_canopy_zone_67abf7.png",
    "asset_id": "29bdc411-d997-4693-ba25-910c6ce4aef6",
    "role": "bgfirst_group_bg"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc20-de97-7bc7-aa0d-f872445a270a",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S1sh3__bgfirst_bg.png",
   "bg_asset_id": "6e74e00e-4c81-4868-ae92-4916ae8ce4c0",
   "bg_record_key": "S1sh3::bgfirst_bg",
   "chain_winner": false,
   "authority": "groupbg",
   "group_key": "forest_canopy_zone",
   "groupbg_asset_id": "29bdc411-d997-4693-ba25-910c6ce4aef6"
  },
  "ref_mode": "그룹 배경+엔티티 (2택1: 무콘티 승)",
  "share_plan": {
   "ref_plan": "background"
  }
 },
 "S1sh3::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:22:08.634318+00:00",
  "fingerprint": "c554769041c0675c3c97e545c16e0bf6ba961b5a78c012ddd7700c5086983312",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S1sh3_sel.png",
  "source_sha256": "cd830a3e82e392cf05193f01fd3e19a2231cffd5aefa2621482393e5f71e8479",
  "file": "S1sh3_cine.png",
  "staged_sha256": "12107879832b23cc27105d6d3360fe5a500d42bff612f0f8d0e07768c15c9c0b",
  "latency_ms": 10811
 },
 "S1sh7::signage": {
  "fp": "230a2b1462647471",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S1sh7": {
  "input_fingerprint": "4b9ffda215f5386e",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 창백한 광선이 절벽 가장자리에 닿아 거대한 바위의 단면이 무자비하게 지워지는 중인 찰나\n\nLOCATION (lock): The exposed exterior edge of the forest cliff, where the pale beam is erasing the rock face. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: beam and cliff contact point in the middle-center of the frame, background; intervening treetops in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: distant cliff edge (a section is being cleanly erased at the beam contact point) — Its exposed edge and disappearing cross-section face the camera at an oblique angle; used as Primary environmental subject held at the end of the pan; pale deletion beam (contacting the cliff edge); used as Creates the exact focal contact point where the cliff disappears; intervening treetops (spread between camera and cliff); used as Low-frame distance layer establishing the beam as a remote threat.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight is interrupted only by the explicitly pale deletion beam, held in tense moderate contrast against the cliff.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A pale deletion beam is cleanly erasing the cliff edge. The giant branch below remains bent but unbroken after the landing.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 창백한 광선이 절벽 가장자리에 닿아 거대한 바위의 단면이 무자비하게 지워지는 중인 찰나\n\nLOCATION (lock): The exposed exterior edge of the forest cliff, where the pale beam is erasing the rock face. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: beam and cliff contact point in the middle-center of the frame, background; intervening treetops in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: distant cliff edge (a section is being cleanly erased at the beam contact point) — Its exposed edge and disappearing cross-section face the camera at an oblique angle; used as Primary environmental subject held at the end of the pan; pale deletion beam (contacting the cliff edge); used as Creates the exact focal contact point where the cliff disappears; intervening treetops (spread between camera and cliff); used as Low-frame distance layer establishing the beam as a remote threat.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight is interrupted only by the explicitly pale deletion beam, held in tense moderate contrast against the cliff.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A pale deletion beam is cleanly erasing the cliff edge. The giant branch below remains bent but unbroken after the landing.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 창백한 광선이 절벽 가장자리에 닿아 거대한 바위의 단면이 무자비하게 지워지는 중인 찰나\n\nLOCATION (lock): The exposed exterior edge of the forest cliff, where the pale beam is erasing the rock face. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: beam and cliff contact point in the middle-center of the frame, background; intervening treetops in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: distant cliff edge (a section is being cleanly erased at the beam contact point) — Its exposed edge and disappearing cross-section face the camera at an oblique angle; used as Primary environmental subject held at the end of the pan; pale deletion beam (contacting the cliff edge); used as Creates the exact focal contact point where the cliff disappears; intervening treetops (spread between camera and cliff); used as Low-frame distance layer establishing the beam as a remote threat.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight is interrupted only by the explicitly pale deletion beam, held in tense moderate contrast against the cliff.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A pale deletion beam is cleanly erasing the cliff edge. The giant branch below remains bent but unbroken after the landing.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "창백한 광선이 하늘에서 계곡 왼쪽의 멀리 있는 절벽 가장자리를 향해 대각선으로 내리꽂히고 있음.",
    "built_space": "넓은 숲 계곡과 화면 왼쪽의 절벽 지형이 레퍼런스 이미지의 공간 구조 및 비율과 정확히 일치함.",
    "entities": "창백한 광선, 멀리 있는 절벽, 숲의 나무들이 존재함. 텍스트가 요구한 '거대한 나뭇가지(giant branch)'가 보이지 않으며, 바위의 '지워지는 단면(cross-section)'이 평평하게 깎이는 묘사 없이 단순히 빛나는 충격파로만 처리됨.",
    "hard_violations": [],
    "physics": "광선이 직선으로 타겟을 향해 뻗어가며 숲과 지형에 중력이나 구조적으로 어긋나는 요소 없이 안정적으로 배치됨."
   },
   {
    "label": "B",
    "direction": "광선이 화면 왼쪽에서 오른쪽으로 거의 수평으로 뻗어나가 우측 근경에 있는 절벽 측면을 타격함.",
    "built_space": "기존 계곡 지형은 존재하나, 오른쪽 근경에 거대한 절벽 덩어리가 새롭게 추가되어 레퍼런스의 원래 공간 구조를 완전히 무너뜨림.",
    "entities": "창백한 광선, 평평하고 깔끔하게 지워진 절벽 단면, 그 아래 뻗어 있는 거대한 나뭇가지가 텍스트 설명대로 존재함.",
    "hard_violations": [
     "[gemini-pro] 배경 요소의 근경화 및 임의 구조물 창조: 광선이 닿는 타겟이 '원경의 절벽 가장자리'여야 하고 배경 요소를 근경으로 키우지 말라는 지시를 어긴 채, 오른쪽 가장자리에 거대한 바위를 임의로 배치하여 프레이밍을 크게 위반함."
    ],
    "physics": "광선에 의해 바위가 평평하게 잘려나간 단면은 표현되었으나, 우측의 거대한 바위 덩어리가 하단 지면과 연결되지 않은 채 화면 오른쪽 끝에서 튀어나와 공중에 위태롭게 떠 있는 듯한 물리적 모순을 띰."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "원경의 절벽에 광선이 닿는 구도와 공간의 깊이감을 프롬프트 지시대로 정확히 유지했으나, '깔끔하게 지워지는 단면'의 시각적 묘사와 유지 상태로 요구된 '거대한 나뭇가지'가 누락되었습니다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "지워진 단면과 거대한 나뭇가지의 디테일은 살렸으나, 원경이어야 할 타겟 절벽을 우측 근경에 형태가 불분명한 거대한 바위로 임의 창조하여 심각한 프레이밍 및 구조 위반을 범했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "창백한 광선이 하늘에서 계곡 왼쪽의 멀리 있는 절벽 가장자리를 향해 대각선으로 내리꽂히고 있음.",
        "built_space": "넓은 숲 계곡과 화면 왼쪽의 절벽 지형이 레퍼런스 이미지의 공간 구조 및 비율과 정확히 일치함.",
        "entities": "창백한 광선, 멀리 있는 절벽, 숲의 나무들이 존재함. 텍스트가 요구한 '거대한 나뭇가지(giant branch)'가 보이지 않으며, 바위의 '지워지는 단면(cross-section)'이 평평하게 깎이는 묘사 없이 단순히 빛나는 충격파로만 처리됨.",
        "hard_violations": [],
        "physics": "광선이 직선으로 타겟을 향해 뻗어가며 숲과 지형에 중력이나 구조적으로 어긋나는 요소 없이 안정적으로 배치됨."
       },
       {
        "label": "B",
        "direction": "광선이 화면 왼쪽에서 오른쪽으로 거의 수평으로 뻗어나가 우측 근경에 있는 절벽 측면을 타격함.",
        "built_space": "기존 계곡 지형은 존재하나, 오른쪽 근경에 거대한 절벽 덩어리가 새롭게 추가되어 레퍼런스의 원래 공간 구조를 완전히 무너뜨림.",
        "entities": "창백한 광선, 평평하고 깔끔하게 지워진 절벽 단면, 그 아래 뻗어 있는 거대한 나뭇가지가 텍스트 설명대로 존재함.",
        "hard_violations": [
         "배경 요소의 근경화 및 임의 구조물 창조: 광선이 닿는 타겟이 '원경의 절벽 가장자리'여야 하고 배경 요소를 근경으로 키우지 말라는 지시를 어긴 채, 오른쪽 가장자리에 거대한 바위를 임의로 배치하여 프레이밍을 크게 위반함."
        ],
        "physics": "광선에 의해 바위가 평평하게 잘려나간 단면은 표현되었으나, 우측의 거대한 바위 덩어리가 하단 지면과 연결되지 않은 채 화면 오른쪽 끝에서 튀어나와 공중에 위태롭게 떠 있는 듯한 물리적 모순을 띰."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "원경의 절벽에 광선이 닿는 구도와 공간의 깊이감을 프롬프트 지시대로 정확히 유지했으나, '깔끔하게 지워지는 단면'의 시각적 묘사와 유지 상태로 요구된 '거대한 나뭇가지'가 누락되었습니다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "지워진 단면과 거대한 나뭇가지의 디테일은 살렸으나, 원경이어야 할 타겟 절벽을 우측 근경에 형태가 불분명한 거대한 바위로 임의 창조하여 심각한 프레이밍 및 구조 위반을 범했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "창백한 광선이 하늘에서 계곡 왼쪽의 멀리 있는 절벽 가장자리를 향해 대각선으로 내리꽂히고 있음.",
        "built_space": "넓은 숲 계곡과 화면 왼쪽의 절벽 지형이 레퍼런스 이미지의 공간 구조 및 비율과 정확히 일치함.",
        "entities": "창백한 광선, 멀리 있는 절벽, 숲의 나무들이 존재함. 텍스트가 요구한 '거대한 나뭇가지(giant branch)'가 보이지 않으며, 바위의 '지워지는 단면(cross-section)'이 평평하게 깎이는 묘사 없이 단순히 빛나는 충격파로만 처리됨.",
        "hard_violations": [],
        "physics": "광선이 직선으로 타겟을 향해 뻗어가며 숲과 지형에 중력이나 구조적으로 어긋나는 요소 없이 안정적으로 배치됨."
       },
       {
        "label": "B",
        "direction": "광선이 화면 왼쪽에서 오른쪽으로 거의 수평으로 뻗어나가 우측 근경에 있는 절벽 측면을 타격함.",
        "built_space": "기존 계곡 지형은 존재하나, 오른쪽 근경에 거대한 절벽 덩어리가 새롭게 추가되어 레퍼런스의 원래 공간 구조를 완전히 무너뜨림.",
        "entities": "창백한 광선, 평평하고 깔끔하게 지워진 절벽 단면, 그 아래 뻗어 있는 거대한 나뭇가지가 텍스트 설명대로 존재함.",
        "hard_violations": [
         "배경 요소의 근경화 및 임의 구조물 창조: 광선이 닿는 타겟이 '원경의 절벽 가장자리'여야 하고 배경 요소를 근경으로 키우지 말라는 지시를 어긴 채, 오른쪽 가장자리에 거대한 바위를 임의로 배치하여 프레이밍을 크게 위반함."
        ],
        "physics": "광선에 의해 바위가 평평하게 잘려나간 단면은 표현되었으나, 우측의 거대한 바위 덩어리가 하단 지면과 연결되지 않은 채 화면 오른쪽 끝에서 튀어나와 공중에 위태롭게 떠 있는 듯한 물리적 모순을 띰."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 5,
        "verdict_ko": "광선은 절벽을 정확히 맞히고 굽은 가지도 보이지만, 절벽이 우측 전경을 거대하게 점유해 ‘중앙 배경의 먼 접촉점’이라는 핵심 와이드숏 구도를 어긴다."
       },
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "먼 절벽의 삭제 접촉점과 그 앞의 낮은 수관층을 중앙 배경에 배치해 권위 있는 숏 구도를 가장 충실히 구현했으나, 유지되어야 할 굽은 거대 가지는 보이지 않는다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "창백한 광선이 화면 왼쪽에서 오른쪽으로 수평 진행해 우측 절벽의 밝은 절단면을 정확히 맞힌다. 다만 접촉점은 중앙이 아니라 우측 중앙에 있다.",
        "built_space": "인공 구조물은 없다. 자연 절벽이 좌측 가장자리와 우측에 각각 보이며, 우측 절벽은 화면 밖으로 이어지는 암반에 붙어 있다. 우측 절벽 아래에는 굽었지만 부러지지 않은 큰 가지 한 개가 암벽 쪽에 연결되어 있다. 요구된 먼 절벽 대신 우측 암반이 전경 크기로 과도하게 확대되었다.",
        "entities": "사람·얼굴·인물은 없고 읽을 수 있는 글자도 없다. 숲, 절벽, 창백한 삭제 광선, 굽은 거대 가지가 모두 보인다. 삭제면은 지나치게 매끈한 흰 직사각형 단면처럼 보여 자연 암석이 소거되는 물질감은 다소 약하다.",
        "hard_violations": [],
        "physics": "광선은 절벽과 접촉하며 직선으로 이어져 있고, 접촉부가 밝아 삭제 작용의 원인이 읽힌다. 우측 절벽은 화면 밖의 암반에 연결되어 지지되며, 큰 가지도 절벽 쪽에 붙어 있어 떠 있지 않는다. 움직이거나 공중에 뜬 사람은 없다."
       },
       {
        "label": "B",
        "direction": "창백한 광선이 화면 우하단의 먼 숲 방향에서 좌상향해 중앙에 가까운 절벽 끝을 정확히 맞힌다. 광선의 끝과 밝은 접촉점이 일치하며 목표는 노출된 절벽 가장자리다.",
        "built_space": "인공 구조물은 없다. 좌측에서 중앙으로 이어지는 자연 암벽 능선과 여러 암주가 보이고, 삭제되는 끝부분은 능선에 연결되어 있다. 낮은 화면 중앙에는 카메라와 절벽 사이의 수관층이 넓게 펼쳐진다. 굽은 거대 가지는 이 프레임에서 보이지 않는다.",
        "entities": "사람·얼굴·인물과 읽을 수 있는 문자는 없다. 먼 숲 절벽, 노출된 암석 단면, 창백한 삭제 광선, 전경과 중경의 수관층이 보인다. 접촉부의 불규칙한 암석 가장자리와 밝은 소거 흔적이 실제 암반에 작용하는 효과로 읽힌다. 다만 지속 상태에 명시된 굽은 거대 가지는 확인할 수 없다.",
        "hard_violations": [],
        "physics": "광선은 먼 발생 방향에서 절벽 접촉점까지 연속적으로 진행하고, 접촉부의 발광이 암석 삭제 작용을 설명한다. 암주는 좌측 능선과 지반에 연결되어 지지되며 떠 있지 않는다. 공중에 뜬 몸이나 지지 없는 물체는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 5,
        "verdict_ko": "광선은 절벽을 정확히 맞히고 굽은 가지도 보이지만, 절벽이 우측 전경을 거대하게 점유해 ‘중앙 배경의 먼 접촉점’이라는 핵심 와이드숏 구도를 어긴다."
       },
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "먼 절벽의 삭제 접촉점과 그 앞의 낮은 수관층을 중앙 배경에 배치해 권위 있는 숏 구도를 가장 충실히 구현했으나, 유지되어야 할 굽은 거대 가지는 보이지 않는다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "창백한 광선이 화면 왼쪽에서 오른쪽으로 수평 진행해 우측 절벽의 밝은 절단면을 정확히 맞힌다. 다만 접촉점은 중앙이 아니라 우측 중앙에 있다.",
        "built_space": "인공 구조물은 없다. 자연 절벽이 좌측 가장자리와 우측에 각각 보이며, 우측 절벽은 화면 밖으로 이어지는 암반에 붙어 있다. 우측 절벽 아래에는 굽었지만 부러지지 않은 큰 가지 한 개가 암벽 쪽에 연결되어 있다. 요구된 먼 절벽 대신 우측 암반이 전경 크기로 과도하게 확대되었다.",
        "entities": "사람·얼굴·인물은 없고 읽을 수 있는 글자도 없다. 숲, 절벽, 창백한 삭제 광선, 굽은 거대 가지가 모두 보인다. 삭제면은 지나치게 매끈한 흰 직사각형 단면처럼 보여 자연 암석이 소거되는 물질감은 다소 약하다.",
        "hard_violations": [],
        "physics": "광선은 절벽과 접촉하며 직선으로 이어져 있고, 접촉부가 밝아 삭제 작용의 원인이 읽힌다. 우측 절벽은 화면 밖의 암반에 연결되어 지지되며, 큰 가지도 절벽 쪽에 붙어 있어 떠 있지 않는다. 움직이거나 공중에 뜬 사람은 없다."
       },
       {
        "label": "A",
        "direction": "창백한 광선이 화면 우하단의 먼 숲 방향에서 좌상향해 중앙에 가까운 절벽 끝을 정확히 맞힌다. 광선의 끝과 밝은 접촉점이 일치하며 목표는 노출된 절벽 가장자리다.",
        "built_space": "인공 구조물은 없다. 좌측에서 중앙으로 이어지는 자연 암벽 능선과 여러 암주가 보이고, 삭제되는 끝부분은 능선에 연결되어 있다. 낮은 화면 중앙에는 카메라와 절벽 사이의 수관층이 넓게 펼쳐진다. 굽은 거대 가지는 이 프레임에서 보이지 않는다.",
        "entities": "사람·얼굴·인물과 읽을 수 있는 문자는 없다. 먼 숲 절벽, 노출된 암석 단면, 창백한 삭제 광선, 전경과 중경의 수관층이 보인다. 접촉부의 불규칙한 암석 가장자리와 밝은 소거 흔적이 실제 암반에 작용하는 효과로 읽힌다. 다만 지속 상태에 명시된 굽은 거대 가지는 확인할 수 없다.",
        "hard_violations": [],
        "physics": "광선은 먼 발생 방향에서 절벽 접촉점까지 연속적으로 진행하고, 접촉부의 발광이 암석 삭제 작용을 설명한다. 암주는 좌측 능선과 지반에 연결되어 지지되며 떠 있지 않는다. 공중에 뜬 몸이나 지지 없는 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.125
   },
   "adjusted": {
    "A": 2.0,
    "B": 0.875
   },
   "violations": {
    "B": [
     "[gemini-pro] 배경 요소의 근경화 및 임의 구조물 창조: 광선이 닿는 타겟이 '원경의 절벽 가장자리'여야 하고 배경 요소를 근경으로 키우지 말라는 지시를 어긴 채, 오른쪽 가장자리에 거대한 바위를 임의로 배치하여 프레이밍을 크게 위반함."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 875
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "원경의 절벽에 광선이 닿는 구도와 공간의 깊이감을 프롬프트 지시대로 정확히 유지했으나, '깔끔하게 지워지는 단면'의 시각적 묘사와 유지 상태로 요구된 '거대한 나뭇가지'가 누락되었습니다."
   },
   {
    "label": "B",
    "score": 875,
    "verdict_ko": "지워진 단면과 거대한 나뭇가지의 디테일은 살렸으나, 원경이어야 할 타겟 절벽을 우측 근경에 형태가 불분명한 거대한 바위로 임의 창조하여 심각한 프레이밍 및 구조 위반을 범했습니다.  ★위반: [gemini-pro] 배경 요소의 근경화 및 임의 구조물 창조: 광선이 닿는 타겟이 '원경의 절벽 가장자리'여야 하고 배경 요소를 근경으로 키우지 말라는 지시를 어긴 채, 오른쪽 가장자리에 거대한 바위를 임의로 배치하여 프레이밍을 크게 위반함."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S1sh3_sel.png",
    "asset_id": "8defbed7-4c8f-484c-bfec-edaa9962327b",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc32-b91f-7ab9-a153-3c22962dd035",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S1sh3"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S1sh7::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:23:38.224475+00:00",
  "fingerprint": "51fbf4cf4b21493a70b4dbc47ad28e46f2495f094a13f9500fa8e9e041fcf7f3",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S1sh7_sel.png",
  "source_sha256": "31e54ea6a03f3d9c5c062d1e566039cf0f61fea32b8af8a164bf579e12ed048a",
  "file": "S1sh7_cine.png",
  "staged_sha256": "bcd45d51408a39eb9d5e5499e815941840033ebb9616860a02cd173ac613c923",
  "latency_ms": 12280
 },
 "S1sh8::signage": {
  "fp": "3254a23dd75c17d5",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S1sh8": {
  "input_fingerprint": "69946df9df87863f",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 거센 돌풍에 나뭇잎들이 수평으로 꺾여 날아가는 허공 속에서, 두 손으로 나뭇가지를 꽉 움켜쥐고 매달린 윌마와 토니\n\nLOCATION (lock): The exterior forest canopy around a massive branch, exposed to the vacuum-driven gale. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: medium shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-left of the frame, midground, reaches for overhead branch; 토니(앤서니 로저스) in the middle-right of the frame, midground, reaches for overhead branch; shared overhead branch in the upper-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: large tree branch (bent but unbroken) — Its length crosses the upper frame, viewed from below and slightly from one side; used as Load-bearing anchor for both characters and the upper compositional line; leaves (driven horizontally through the air by the gust); used as Visible motion layer crossing around the hanging bodies; open space behind the pair (visible behind their hanging bodies); used as Separates their silhouettes and conveys exposure beneath the branch.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight with tense moderate contrast preserves the hands, branch, and horizontally driven leaves as separate layers.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the cloudy daytime sky, the forest canopy, and the wind-darkened green and gray palette from the reference. Exclude any erasure beam, exposed cliff cut, or probe lights; add only the branch and storm-driven leaves required for this moment.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The giant branch is bent but not broken. Leaves and foliage are driven violently across the forest by the vacuum-created gale. 윌마 디어링: She remains on the giant branch with her weighted jumper belt and gun. 토니(앤서니 로저스): He remains on the giant branch, carrying his phone and rescue rope.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마와 토니 right now, so 윌마와 토니's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마와 토니: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 거센 돌풍에 나뭇잎들이 수평으로 꺾여 날아가는 허공 속에서, 두 손으로 나뭇가지를 꽉 움켜쥐고 매달린 윌마와 토니\n\nLOCATION (lock): The exterior forest canopy around a massive branch, exposed to the vacuum-driven gale. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: medium shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-left of the frame, midground, reaches for overhead branch; 토니(앤서니 로저스) in the middle-right of the frame, midground, reaches for overhead branch; shared overhead branch in the upper-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: large tree branch (bent but unbroken) — Its length crosses the upper frame, viewed from below and slightly from one side; used as Load-bearing anchor for both characters and the upper compositional line; leaves (driven horizontally through the air by the gust); used as Visible motion layer crossing around the hanging bodies; open space behind the pair (visible behind their hanging bodies); used as Separates their silhouettes and conveys exposure beneath the branch.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight with tense moderate contrast preserves the hands, branch, and horizontally driven leaves as separate layers.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the cloudy daytime sky, the forest canopy, and the wind-darkened green and gray palette from the reference. Exclude any erasure beam, exposed cliff cut, or probe lights; add only the branch and storm-driven leaves required for this moment.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The giant branch is bent but not broken. Leaves and foliage are driven violently across the forest by the vacuum-created gale. 윌마 디어링: She remains on the giant branch with her weighted jumper belt and gun. 토니(앤서니 로저스): He remains on the giant branch, carrying his phone and rescue rope.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마와 토니 right now, so 윌마와 토니's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마와 토니: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 거센 돌풍에 나뭇잎들이 수평으로 꺾여 날아가는 허공 속에서, 두 손으로 나뭇가지를 꽉 움켜쥐고 매달린 윌마와 토니\n\nLOCATION (lock): The exterior forest canopy around a massive branch, exposed to the vacuum-driven gale. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: medium shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-left of the frame, midground, reaches for overhead branch; 토니(앤서니 로저스) in the middle-right of the frame, midground, reaches for overhead branch; shared overhead branch in the upper-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: large tree branch (bent but unbroken) — Its length crosses the upper frame, viewed from below and slightly from one side; used as Load-bearing anchor for both characters and the upper compositional line; leaves (driven horizontally through the air by the gust); used as Visible motion layer crossing around the hanging bodies; open space behind the pair (visible behind their hanging bodies); used as Separates their silhouettes and conveys exposure beneath the branch.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated daylight with tense moderate contrast preserves the hands, branch, and horizontally driven leaves as separate layers.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the cloudy daytime sky, the forest canopy, and the wind-darkened green and gray palette from the reference. Exclude any erasure beam, exposed cliff cut, or probe lights; add only the branch and storm-driven leaves required for this moment.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The giant branch is bent but not broken. Leaves and foliage are driven violently across the forest by the vacuum-created gale. 윌마 디어링: She remains on the giant branch with her weighted jumper belt and gun. 토니(앤서니 로저스): He remains on the giant branch, carrying his phone and rescue rope.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마와 토니 right now, so 윌마와 토니's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마와 토니: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "나뭇잎들이 강풍에 의해 수평으로 날아가고 있으며, 두 인물의 시선은 각기 다른 곳을 향하고 있습니다.",
    "built_space": "화면 상단을 가로지르는 굵은 나뭇가지가 있으며, 배경에는 숲과 하늘이 보입니다.",
    "entities": "윌마와 토니 모두 레퍼런스의 외모와 의상(윌마의 녹색 점퍼와 총, 토니의 하네스와 밧줄)을 잘 갖추고 있습니다.",
    "hard_violations": [
     "[gemini-pro] 윌마가 두 손으로 나뭇가지를 움켜쥐고 매달려야 한다는 지시(두 손으로 나뭇가지를 꽉 움켜쥐고 매달린)를 어기고 한 손만 뻗고 있습니다.",
     "[gemini-pro] 윌마의 자세가 허공에 매달린 것이 아니라 프레임 밖 지면에 서 있는 물리적 형태를 취하고 있습니다."
    ],
    "physics": "토니는 두 손으로 나뭇가지를 잡고 허공에 매달려 있지만, 윌마는 한 손만 올린 채 바닥에 지탱하여 서 있는 자세를 취하고 있습니다."
   },
   {
    "label": "B",
    "direction": "나뭇잎들이 수평으로 빠르게 날아가고 있으며, 바람의 방향이 명확하게 표현되었습니다.",
    "built_space": "화면 상단에 굵은 나뭇가지가 배치되어 있고, 뒤쪽으로 숲의 배경과 열린 공간이 잘 드러납니다.",
    "entities": "윌마(총과 벨트 착용)와 토니(밧줄과 통신 장비 착용) 모두 레퍼런스의 외모와 지정된 소품을 정확히 묘사하고 있습니다.",
    "hard_violations": [],
    "physics": "두 인물 모두 바닥의 지지 없이 오직 두 손으로 나뭇가지를 꽉 쥐고 허공에 매달려 있는 자세를 자연스럽게 보여줍니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 10,
        "verdict_ko": "두 캐릭터 모두 지시된 대로 두 손으로 나뭇가지를 꽉 쥐고 허공에 매달린 모습을 정확하게 구현했으며, 레퍼런스의 의상과 배경 요소도 충실히 반영했습니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "윌마가 두 손이 아닌 한 손으로만 나뭇가지를 잡고 있으며, 허공에 매달리지 않고 바닥에 서 있는 듯한 자세를 취해 핵심 행동 지시를 위반했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "나뭇잎들이 강풍에 의해 수평으로 날아가고 있으며, 두 인물의 시선은 각기 다른 곳을 향하고 있습니다.",
        "built_space": "화면 상단을 가로지르는 굵은 나뭇가지가 있으며, 배경에는 숲과 하늘이 보입니다.",
        "entities": "윌마와 토니 모두 레퍼런스의 외모와 의상(윌마의 녹색 점퍼와 총, 토니의 하네스와 밧줄)을 잘 갖추고 있습니다.",
        "hard_violations": [
         "윌마가 두 손으로 나뭇가지를 움켜쥐고 매달려야 한다는 지시(두 손으로 나뭇가지를 꽉 움켜쥐고 매달린)를 어기고 한 손만 뻗고 있습니다.",
         "윌마의 자세가 허공에 매달린 것이 아니라 프레임 밖 지면에 서 있는 물리적 형태를 취하고 있습니다."
        ],
        "physics": "토니는 두 손으로 나뭇가지를 잡고 허공에 매달려 있지만, 윌마는 한 손만 올린 채 바닥에 지탱하여 서 있는 자세를 취하고 있습니다."
       },
       {
        "label": "B",
        "direction": "나뭇잎들이 수평으로 빠르게 날아가고 있으며, 바람의 방향이 명확하게 표현되었습니다.",
        "built_space": "화면 상단에 굵은 나뭇가지가 배치되어 있고, 뒤쪽으로 숲의 배경과 열린 공간이 잘 드러납니다.",
        "entities": "윌마(총과 벨트 착용)와 토니(밧줄과 통신 장비 착용) 모두 레퍼런스의 외모와 지정된 소품을 정확히 묘사하고 있습니다.",
        "hard_violations": [],
        "physics": "두 인물 모두 바닥의 지지 없이 오직 두 손으로 나뭇가지를 꽉 쥐고 허공에 매달려 있는 자세를 자연스럽게 보여줍니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 10,
        "verdict_ko": "두 캐릭터 모두 지시된 대로 두 손으로 나뭇가지를 꽉 쥐고 허공에 매달린 모습을 정확하게 구현했으며, 레퍼런스의 의상과 배경 요소도 충실히 반영했습니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "윌마가 두 손이 아닌 한 손으로만 나뭇가지를 잡고 있으며, 허공에 매달리지 않고 바닥에 서 있는 듯한 자세를 취해 핵심 행동 지시를 위반했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "나뭇잎들이 강풍에 의해 수평으로 날아가고 있으며, 두 인물의 시선은 각기 다른 곳을 향하고 있습니다.",
        "built_space": "화면 상단을 가로지르는 굵은 나뭇가지가 있으며, 배경에는 숲과 하늘이 보입니다.",
        "entities": "윌마와 토니 모두 레퍼런스의 외모와 의상(윌마의 녹색 점퍼와 총, 토니의 하네스와 밧줄)을 잘 갖추고 있습니다.",
        "hard_violations": [
         "윌마가 두 손으로 나뭇가지를 움켜쥐고 매달려야 한다는 지시(두 손으로 나뭇가지를 꽉 움켜쥐고 매달린)를 어기고 한 손만 뻗고 있습니다.",
         "윌마의 자세가 허공에 매달린 것이 아니라 프레임 밖 지면에 서 있는 물리적 형태를 취하고 있습니다."
        ],
        "physics": "토니는 두 손으로 나뭇가지를 잡고 허공에 매달려 있지만, 윌마는 한 손만 올린 채 바닥에 지탱하여 서 있는 자세를 취하고 있습니다."
       },
       {
        "label": "B",
        "direction": "나뭇잎들이 수평으로 빠르게 날아가고 있으며, 바람의 방향이 명확하게 표현되었습니다.",
        "built_space": "화면 상단에 굵은 나뭇가지가 배치되어 있고, 뒤쪽으로 숲의 배경과 열린 공간이 잘 드러납니다.",
        "entities": "윌마(총과 벨트 착용)와 토니(밧줄과 통신 장비 착용) 모두 레퍼런스의 외모와 지정된 소품을 정확히 묘사하고 있습니다.",
        "hard_violations": [],
        "physics": "두 인물 모두 바닥의 지지 없이 오직 두 손으로 나뭇가지를 꽉 쥐고 허공에 매달려 있는 자세를 자연스럽게 보여줍니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "윌마와 토니가 각각 두 손으로 같은 상단 가지를 붙잡고 매달린 핵심 순간과 좌우 배치를 정확히 구현했으며, 수평 돌풍의 잎 표현만 B보다 다소 약하다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "강풍에 수평으로 휘날리는 잎과 개방된 배경은 뛰어나지만, 윌마가 한 손만 가지를 잡고 있어 ‘두 손으로 꽉 움켜쥔’ 핵심 동작을 위반한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마는 오른쪽의 토니를 바라보고, 토니는 정면에서 약간 왼쪽 아래를 바라본다. 두 사람의 양팔과 손은 모두 머리 위의 공유 가지를 향하며 실제 접촉한다. 무기 조준이나 별도의 지시 대상은 없다. 공중의 잎은 여러 방향으로 기울고 흩어져 있어 강풍은 보이지만 일관된 수평 이동 방향은 B보다 약하다.",
        "built_space": "인공 구조물은 없다. 굽었지만 끊어지지 않은 거대한 가지 하나가 왼쪽 아래에서 상단 중앙과 오른쪽 상단을 가로지르며, 아래쪽과 약간 측면에서 보인다. 윌마는 중간 왼쪽, 토니는 중간 오른쪽에 있고 두 사람 뒤로 숲 계곡의 열린 공간이 보여 실루엣이 분리된다. 두 사람 모두 같은 가지 아래에 자연스럽게 배치되어 있다.",
        "entities": "미국인 20대 후반 여성 윌마와 30대 중반 남성 토니 두 명만 보이며, 얼굴·머리·체격과 녹색 점프수트 및 어두운 전술복이 각 인물 참고 이미지에 대체로 부합한다. 윌마의 허리 장비와 총, 토니의 허리 로프와 장비가 보인다. 토니의 휴대전화는 명확히 식별되지 않지만 가슴 장치와 여러 파우치가 있다. 거대한 가지, 숲 캐노피, 흐린 낮 하늘, 날리는 잎이 모두 존재한다.",
        "hard_violations": [],
        "physics": "윌마와 토니 모두 두 팔을 위로 뻗어 양손으로 두꺼운 가지를 감아 잡고 있으며, 그 손 접촉이 매달린 몸의 하중을 지지한다. 다리는 아래로 처지고 바람에 옆으로 조금 밀린 자세라 매달림의 물리적 결과로 성립한다. 가지는 굽었지만 연속되어 있고, 로프와 허리 장비도 몸에 고정되어 떠 있지 않는다."
       },
       {
        "label": "B",
        "direction": "윌마는 오른쪽의 토니를 바라보고, 토니는 왼쪽 아래의 윌마 쪽을 내려다본다. 토니의 두 손은 상단 가지를 향해 실제로 잡고 있지만, 윌마는 오른손만 가지를 잡고 왼팔은 왼쪽 바깥으로 뻗는다. 잎들은 화면을 거의 수평으로 횡단하는 강한 돌풍 방향을 명확히 보여준다.",
        "built_space": "인공 구조물은 없다. 굽었지만 끊어지지 않은 큰 가지 하나가 왼쪽에서 상단 중앙과 오른쪽 상단을 가로지르고 아래·측면에서 보인다. 윌마는 중간 왼쪽, 토니는 중간 오른쪽에 있으며, 뒤의 흐린 하늘과 먼 숲이 두 몸 뒤에 넓은 개방 공간을 만든다. 공유 가지의 위치와 인물 배치는 요구된 구도에 부합한다.",
        "entities": "윌마와 토니 두 명만 등장하며 성별·연령대·얼굴·머리·체격과 각자의 녹색 점프수트 및 어두운 전술복이 참고 이미지와 대체로 일치한다. 윌마의 총과 벨트, 토니의 구조 로프와 장비가 보이지만 휴대전화는 명확히 식별되지 않는다. 가지, 숲 캐노피, 흐린 낮 하늘과 다량의 바람에 날리는 잎이 모두 보인다.",
        "hard_violations": [],
        "physics": "토니는 양손으로 가지를 잡아 몸을 지지하며 다리가 아래로 떠 있어 자연스러운 매달림이다. 윌마도 한 손의 확실한 가지 접촉이 몸을 지지하므로 완전히 무지지 상태는 아니며, 자유로운 왼팔과 아래로 늘어진 몸은 강풍 속 한 손 매달림으로 물리적으로 가능하다. 다만 프롬프트가 요구한 양손 지지는 아니다. 로프와 허리 장비는 토니의 몸에 고정되어 있다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "윌마와 토니가 각각 두 손으로 같은 상단 가지를 붙잡고 매달린 핵심 순간과 좌우 배치를 정확히 구현했으며, 수평 돌풍의 잎 표현만 B보다 다소 약하다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "강풍에 수평으로 휘날리는 잎과 개방된 배경은 뛰어나지만, 윌마가 한 손만 가지를 잡고 있어 ‘두 손으로 꽉 움켜쥔’ 핵심 동작을 위반한다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "윌마는 오른쪽의 토니를 바라보고, 토니는 정면에서 약간 왼쪽 아래를 바라본다. 두 사람의 양팔과 손은 모두 머리 위의 공유 가지를 향하며 실제 접촉한다. 무기 조준이나 별도의 지시 대상은 없다. 공중의 잎은 여러 방향으로 기울고 흩어져 있어 강풍은 보이지만 일관된 수평 이동 방향은 B보다 약하다.",
        "built_space": "인공 구조물은 없다. 굽었지만 끊어지지 않은 거대한 가지 하나가 왼쪽 아래에서 상단 중앙과 오른쪽 상단을 가로지르며, 아래쪽과 약간 측면에서 보인다. 윌마는 중간 왼쪽, 토니는 중간 오른쪽에 있고 두 사람 뒤로 숲 계곡의 열린 공간이 보여 실루엣이 분리된다. 두 사람 모두 같은 가지 아래에 자연스럽게 배치되어 있다.",
        "entities": "미국인 20대 후반 여성 윌마와 30대 중반 남성 토니 두 명만 보이며, 얼굴·머리·체격과 녹색 점프수트 및 어두운 전술복이 각 인물 참고 이미지에 대체로 부합한다. 윌마의 허리 장비와 총, 토니의 허리 로프와 장비가 보인다. 토니의 휴대전화는 명확히 식별되지 않지만 가슴 장치와 여러 파우치가 있다. 거대한 가지, 숲 캐노피, 흐린 낮 하늘, 날리는 잎이 모두 존재한다.",
        "hard_violations": [],
        "physics": "윌마와 토니 모두 두 팔을 위로 뻗어 양손으로 두꺼운 가지를 감아 잡고 있으며, 그 손 접촉이 매달린 몸의 하중을 지지한다. 다리는 아래로 처지고 바람에 옆으로 조금 밀린 자세라 매달림의 물리적 결과로 성립한다. 가지는 굽었지만 연속되어 있고, 로프와 허리 장비도 몸에 고정되어 떠 있지 않는다."
       },
       {
        "label": "A",
        "direction": "윌마는 오른쪽의 토니를 바라보고, 토니는 왼쪽 아래의 윌마 쪽을 내려다본다. 토니의 두 손은 상단 가지를 향해 실제로 잡고 있지만, 윌마는 오른손만 가지를 잡고 왼팔은 왼쪽 바깥으로 뻗는다. 잎들은 화면을 거의 수평으로 횡단하는 강한 돌풍 방향을 명확히 보여준다.",
        "built_space": "인공 구조물은 없다. 굽었지만 끊어지지 않은 큰 가지 하나가 왼쪽에서 상단 중앙과 오른쪽 상단을 가로지르고 아래·측면에서 보인다. 윌마는 중간 왼쪽, 토니는 중간 오른쪽에 있으며, 뒤의 흐린 하늘과 먼 숲이 두 몸 뒤에 넓은 개방 공간을 만든다. 공유 가지의 위치와 인물 배치는 요구된 구도에 부합한다.",
        "entities": "윌마와 토니 두 명만 등장하며 성별·연령대·얼굴·머리·체격과 각자의 녹색 점프수트 및 어두운 전술복이 참고 이미지와 대체로 일치한다. 윌마의 총과 벨트, 토니의 구조 로프와 장비가 보이지만 휴대전화는 명확히 식별되지 않는다. 가지, 숲 캐노피, 흐린 낮 하늘과 다량의 바람에 날리는 잎이 모두 보인다.",
        "hard_violations": [],
        "physics": "토니는 양손으로 가지를 잡아 몸을 지지하며 다리가 아래로 떠 있어 자연스러운 매달림이다. 윌마도 한 손의 확실한 가지 접촉이 몸을 지지하므로 완전히 무지지 상태는 아니며, 자유로운 왼팔과 아래로 늘어진 몸은 강풍 속 한 손 매달림으로 물리적으로 가능하다. 다만 프롬프트가 요구한 양손 지지는 아니다. 로프와 허리 장비는 토니의 몸에 고정되어 있다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.15,
    "B": 2.0
   },
   "adjusted": {
    "A": 0.9,
    "B": 2.0
   },
   "violations": {
    "A": [
     "[gemini-pro] 윌마가 두 손으로 나뭇가지를 움켜쥐고 매달려야 한다는 지시(두 손으로 나뭇가지를 꽉 움켜쥐고 매달린)를 어기고 한 손만 뻗고 있습니다.",
     "[gemini-pro] 윌마의 자세가 허공에 매달린 것이 아니라 프레임 밖 지면에 서 있는 물리적 형태를 취하고 있습니다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "B": 2000,
   "A": 900
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "두 캐릭터 모두 지시된 대로 두 손으로 나뭇가지를 꽉 쥐고 허공에 매달린 모습을 정확하게 구현했으며, 레퍼런스의 의상과 배경 요소도 충실히 반영했습니다."
   },
   {
    "label": "A",
    "score": 900,
    "verdict_ko": "윌마가 두 손이 아닌 한 손으로만 나뭇가지를 잡고 있으며, 허공에 매달리지 않고 바닥에 서 있는 듯한 자세를 취해 핵심 행동 지시를 위반했습니다.  ★위반: [gemini-pro] 윌마가 두 손으로 나뭇가지를 움켜쥐고 매달려야 한다는 지시(두 손으로 나뭇가지를 꽉 움켜쥐고 매달린)를 어기고 한 손만 뻗고 있습니다. / [gemini-pro] 윌마의 자세가 허공에 매달린 것이 아니라 프레임 밖 지면에 서 있는 물리적 형태를 취하고 있습니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S1sh7_sel.png",
    "asset_id": "47543ebc-9931-42f5-91f7-dad6228de393",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc38-69b6-7c26-8e37-c8b8f4d5387c",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S1sh7"
  }
 },
 "S1sh8::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:24:47.779980+00:00",
  "fingerprint": "bb5041573db308a8bbbff69ebeb9a406a8b5332c9b51b14edd6fa361dfc990de",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S1sh8_sel.png",
  "source_sha256": "2c315e7c1d88c475fb4cfce64ab07d145bea8c6b231fc29231f893d4dc0ba98f",
  "file": "S1sh8_cine.png",
  "staged_sha256": "e7e076dc4b10b15c296b1e66be2549d496773be439ceb07020a3045bf7f5a4c2",
  "latency_ms": 16862
 },
 "S2sh5::signage": {
  "fp": "a3fccb6413460cec",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S2sh5": {
  "input_fingerprint": "55a5917a3430b79d",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 세 개의 작은 빛나는 탐색 침이 잿빛 하늘에서 숲속 나무 사이로 수직 하강하는 도중 허공에 떠 있는 찰나\n\nLOCATION (lock): Exterior airspace above and between the upper forest branches, directly beneath the gray sky. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 미지(MIDGE), three descending probe units in the upper-center of the frame, midground, moves toward forest space below; open corridor between trees in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: gray sky (visible through the trees); used as Upward-looking backdrop separating the three glowing probes; trees (spaced around the probes' vertical descent path) — Trunks are viewed steeply upward, with branches framing the open descent corridor; used as Creates vertical depth and brackets the descending formation.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Gray-sky daylight remains desaturated and restrained while the probes retain their explicitly glowing points.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Three small glowing search probes are descending from the sky between the trees. The passing beam has left the swept portion of the valley without mist.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 세 개의 작은 빛나는 탐색 침이 잿빛 하늘에서 숲속 나무 사이로 수직 하강하는 도중 허공에 떠 있는 찰나\n\nLOCATION (lock): Exterior airspace above and between the upper forest branches, directly beneath the gray sky. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 미지(MIDGE), three descending probe units in the upper-center of the frame, midground, moves toward forest space below; open corridor between trees in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: gray sky (visible through the trees); used as Upward-looking backdrop separating the three glowing probes; trees (spaced around the probes' vertical descent path) — Trunks are viewed steeply upward, with branches framing the open descent corridor; used as Creates vertical depth and brackets the descending formation.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Gray-sky daylight remains desaturated and restrained while the probes retain their explicitly glowing points.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Three small glowing search probes are descending from the sky between the trees. The passing beam has left the swept portion of the valley without mist.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 세 개의 작은 빛나는 탐색 침이 잿빛 하늘에서 숲속 나무 사이로 수직 하강하는 도중 허공에 떠 있는 찰나\n\nLOCATION (lock): Exterior airspace above and between the upper forest branches, directly beneath the gray sky. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 미지(MIDGE), three descending probe units in the upper-center of the frame, midground, moves toward forest space below; open corridor between trees in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: gray sky (visible through the trees); used as Upward-looking backdrop separating the three glowing probes; trees (spaced around the probes' vertical descent path) — Trunks are viewed steeply upward, with branches framing the open descent corridor; used as Creates vertical depth and brackets the descending formation.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Gray-sky daylight remains desaturated and restrained while the probes retain their explicitly glowing points.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Three small glowing search probes are descending from the sky between the trees. The passing beam has left the swept portion of the valley without mist.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "세 대의 탐사선이 아래를 향해 수직 일렬로 배열된 채 추진체를 분사하고 있음.",
    "built_space": "침엽수림 형태의 숲 배경. 카메라는 지시된 상향 각도가 아닌 수평에 가까운 각도로 숲을 바라보고 있음.",
    "entities": "빛나는 탐사선 3대. 사람은 존재하지 않음. 나무들이 레퍼런스(활엽수)와 전혀 다른 침엽수 종으로 그려짐.",
    "hard_violations": [
     "[gemini-pro] 프롬프트에 명시된 카메라 각도(Trunks are viewed steeply upward)를 완전히 위반하여 수평 앵글로 렌더링됨",
     "[gemini-pro] 레퍼런스 이미지의 장소적 특징(활엽수림)을 무시하고 지시되지 않은 다른 형태의 식생(침엽수)을 생성하여 위치 고정(LOCATION lock)을 위반함"
    ],
    "physics": "탐사선들이 하단의 추진력을 바탕으로 공중에 떠 있음."
   },
   {
    "label": "B",
    "direction": "세 대의 탐사선이 지면을 향해 빛줄기를 비추며 아래로 하강하고 있음.",
    "built_space": "레퍼런스와 일치하는 활엽수림 숲. 지면에서 하늘과 나무 기둥들을 가파르게 올려다보는(steeply upward) 프레임이 정확히 적용됨.",
    "entities": "빛과 빔을 방출하는 탐사선 3대, 레퍼런스의 환경과 이어지는 흩날리는 나뭇잎들. 사람은 등장하지 않음.",
    "hard_violations": [],
    "physics": "탐사선들이 공중 부양한 상태에서 안정적으로 빛을 쏘아내며 체공 중임."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "지면에서 하늘을 올려다보는 가파른 카메라 앵글(steeply upward)과 활엽수림의 식생, 하강하는 세 대의 탐사선과 잿빛 하늘의 분위기를 프롬프트와 레퍼런스의 지시에 맞춰 완벽에 가깝게 구현했습니다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "가파르게 위를 올려다보는 앵글(steeply upward) 지시를 무시하고 수평 앵글로 촬영했으며, 레퍼런스의 장소(활엽수림)를 침엽수림으로 임의 변경하는 심각한 오류를 범했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "세 대의 탐사선이 아래를 향해 수직 일렬로 배열된 채 추진체를 분사하고 있음.",
        "built_space": "침엽수림 형태의 숲 배경. 카메라는 지시된 상향 각도가 아닌 수평에 가까운 각도로 숲을 바라보고 있음.",
        "entities": "빛나는 탐사선 3대. 사람은 존재하지 않음. 나무들이 레퍼런스(활엽수)와 전혀 다른 침엽수 종으로 그려짐.",
        "hard_violations": [
         "프롬프트에 명시된 카메라 각도(Trunks are viewed steeply upward)를 완전히 위반하여 수평 앵글로 렌더링됨",
         "레퍼런스 이미지의 장소적 특징(활엽수림)을 무시하고 지시되지 않은 다른 형태의 식생(침엽수)을 생성하여 위치 고정(LOCATION lock)을 위반함"
        ],
        "physics": "탐사선들이 하단의 추진력을 바탕으로 공중에 떠 있음."
       },
       {
        "label": "B",
        "direction": "세 대의 탐사선이 지면을 향해 빛줄기를 비추며 아래로 하강하고 있음.",
        "built_space": "레퍼런스와 일치하는 활엽수림 숲. 지면에서 하늘과 나무 기둥들을 가파르게 올려다보는(steeply upward) 프레임이 정확히 적용됨.",
        "entities": "빛과 빔을 방출하는 탐사선 3대, 레퍼런스의 환경과 이어지는 흩날리는 나뭇잎들. 사람은 등장하지 않음.",
        "hard_violations": [],
        "physics": "탐사선들이 공중 부양한 상태에서 안정적으로 빛을 쏘아내며 체공 중임."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "지면에서 하늘을 올려다보는 가파른 카메라 앵글(steeply upward)과 활엽수림의 식생, 하강하는 세 대의 탐사선과 잿빛 하늘의 분위기를 프롬프트와 레퍼런스의 지시에 맞춰 완벽에 가깝게 구현했습니다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "가파르게 위를 올려다보는 앵글(steeply upward) 지시를 무시하고 수평 앵글로 촬영했으며, 레퍼런스의 장소(활엽수림)를 침엽수림으로 임의 변경하는 심각한 오류를 범했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "세 대의 탐사선이 아래를 향해 수직 일렬로 배열된 채 추진체를 분사하고 있음.",
        "built_space": "침엽수림 형태의 숲 배경. 카메라는 지시된 상향 각도가 아닌 수평에 가까운 각도로 숲을 바라보고 있음.",
        "entities": "빛나는 탐사선 3대. 사람은 존재하지 않음. 나무들이 레퍼런스(활엽수)와 전혀 다른 침엽수 종으로 그려짐.",
        "hard_violations": [
         "프롬프트에 명시된 카메라 각도(Trunks are viewed steeply upward)를 완전히 위반하여 수평 앵글로 렌더링됨",
         "레퍼런스 이미지의 장소적 특징(활엽수림)을 무시하고 지시되지 않은 다른 형태의 식생(침엽수)을 생성하여 위치 고정(LOCATION lock)을 위반함"
        ],
        "physics": "탐사선들이 하단의 추진력을 바탕으로 공중에 떠 있음."
       },
       {
        "label": "B",
        "direction": "세 대의 탐사선이 지면을 향해 빛줄기를 비추며 아래로 하강하고 있음.",
        "built_space": "레퍼런스와 일치하는 활엽수림 숲. 지면에서 하늘과 나무 기둥들을 가파르게 올려다보는(steeply upward) 프레임이 정확히 적용됨.",
        "entities": "빛과 빔을 방출하는 탐사선 3대, 레퍼런스의 환경과 이어지는 흩날리는 나뭇잎들. 사람은 등장하지 않음.",
        "hard_violations": [],
        "physics": "탐사선들이 공중 부양한 상태에서 안정적으로 빛을 쏘아내며 체공 중임."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "세 탐색 침이 상단 중앙의 잿빛 하늘 아래에서 나무 사이 열린 통로를 향해 하강하며, 가파른 상향 시점·작은 피사체 규모·녹색 숲의 장소 연속성이 가장 정확하다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "세 기체의 하강과 중앙 통로는 명확하지만 지나치게 일렬로 배치되고 크게 보이며, 낮은 산등성이와 가을빛 수목 때문에 상부 가지 사이를 올려다보는 잠긴 장소와 구도가 A보다 덜 충실하다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "세 기체 모두 상단 중앙에서 아래쪽의 숲속 열린 통로를 향하고 있다. 각 기체 아래의 빛기둥도 같은 하강 방향으로 뻗어 목표인 중앙 숲 공간에 닿는다.",
        "built_space": "인공 구조물은 없다. 여러 높은 줄기가 좌우와 가장자리에서 가파르게 위로 모이고, 중앙에는 잿빛 하늘이 보이는 넓은 수직 통로가 형성된다. 세 기체는 그 통로의 상단 중앙 중경에 작은 삼각 대형으로 놓였다.",
        "entities": "사람·얼굴·문자·로고는 없다. 이름 붙은 대상인 작은 탐색 침은 정확히 세 개이며 금속성 기체와 밝은 발광점으로 표현됐다. 숲은 이전 장면과 유사한 짙은 녹색 활엽수림이고 낮의 잿빛 하늘도 일치한다.",
        "hard_violations": [],
        "physics": "세 기체는 공중에 있으며 각 기체 하부의 발광 추진부와 아래로 뻗는 광선/배기 형태가 보인다. 모두 수직 통로 위에서 정상 자세를 유지해 동력 비행하며 하강하는 것으로 읽히고, 무동력으로 떠 있는 물체는 없다."
       },
       {
        "label": "B",
        "direction": "세 기체가 화면 중앙에서 세로로 정렬되어 아래쪽 숲과 먼 산등성이를 향한다. 하부의 밝은 배기 불꽃도 아래로 향해 동력 비행 방향은 명료하지만, 대형 자체가 상하로 길게 늘어서 있다.",
        "built_space": "인공 구조물은 없다. 굵은 줄기들이 좌우를 세로로 둘러싸고 중앙에 하늘 통로를 만들지만, 카메라는 A보다 덜 가파르게 위를 보며 프레임 하단에 먼 산등성이까지 드러낸다. 세 기체는 중앙 통로에 하나씩 수직 배열되어 있다.",
        "entities": "사람·얼굴·문자·로고는 없다. 탐색 기체는 정확히 세 개이고 금속 표면과 발광 추진점이 보인다. 다만 기체가 요구된 작은 중경 대상보다 다소 크며, 주황색 단풍과 침엽수가 섞인 숲은 참고 장면의 푸른 활엽수림과 계절감이 다르다.",
        "hard_violations": [],
        "physics": "각 기체 하부에 아래쪽으로 분사되는 불꽃이 있어 공중 체공을 지지하는 추진력이 명시적으로 보인다. 추진 중 감속하며 수직 하강하는 자세로 물리적으로 가능하고, 지지 없이 떠 있는 대상은 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "세 탐색 침이 상단 중앙의 잿빛 하늘 아래에서 나무 사이 열린 통로를 향해 하강하며, 가파른 상향 시점·작은 피사체 규모·녹색 숲의 장소 연속성이 가장 정확하다."
       },
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "세 기체의 하강과 중앙 통로는 명확하지만 지나치게 일렬로 배치되고 크게 보이며, 낮은 산등성이와 가을빛 수목 때문에 상부 가지 사이를 올려다보는 잠긴 장소와 구도가 A보다 덜 충실하다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "세 기체 모두 상단 중앙에서 아래쪽의 숲속 열린 통로를 향하고 있다. 각 기체 아래의 빛기둥도 같은 하강 방향으로 뻗어 목표인 중앙 숲 공간에 닿는다.",
        "built_space": "인공 구조물은 없다. 여러 높은 줄기가 좌우와 가장자리에서 가파르게 위로 모이고, 중앙에는 잿빛 하늘이 보이는 넓은 수직 통로가 형성된다. 세 기체는 그 통로의 상단 중앙 중경에 작은 삼각 대형으로 놓였다.",
        "entities": "사람·얼굴·문자·로고는 없다. 이름 붙은 대상인 작은 탐색 침은 정확히 세 개이며 금속성 기체와 밝은 발광점으로 표현됐다. 숲은 이전 장면과 유사한 짙은 녹색 활엽수림이고 낮의 잿빛 하늘도 일치한다.",
        "hard_violations": [],
        "physics": "세 기체는 공중에 있으며 각 기체 하부의 발광 추진부와 아래로 뻗는 광선/배기 형태가 보인다. 모두 수직 통로 위에서 정상 자세를 유지해 동력 비행하며 하강하는 것으로 읽히고, 무동력으로 떠 있는 물체는 없다."
       },
       {
        "label": "A",
        "direction": "세 기체가 화면 중앙에서 세로로 정렬되어 아래쪽 숲과 먼 산등성이를 향한다. 하부의 밝은 배기 불꽃도 아래로 향해 동력 비행 방향은 명료하지만, 대형 자체가 상하로 길게 늘어서 있다.",
        "built_space": "인공 구조물은 없다. 굵은 줄기들이 좌우를 세로로 둘러싸고 중앙에 하늘 통로를 만들지만, 카메라는 A보다 덜 가파르게 위를 보며 프레임 하단에 먼 산등성이까지 드러낸다. 세 기체는 중앙 통로에 하나씩 수직 배열되어 있다.",
        "entities": "사람·얼굴·문자·로고는 없다. 탐색 기체는 정확히 세 개이고 금속 표면과 발광 추진점이 보인다. 다만 기체가 요구된 작은 중경 대상보다 다소 크며, 주황색 단풍과 침엽수가 섞인 숲은 참고 장면의 푸른 활엽수림과 계절감이 다르다.",
        "hard_violations": [],
        "physics": "각 기체 하부에 아래쪽으로 분사되는 불꽃이 있어 공중 체공을 지지하는 추진력이 명시적으로 보인다. 추진 중 감속하며 수직 하강하는 자세로 물리적으로 가능하고, 지지 없이 떠 있는 대상은 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.111,
    "B": 2.0
   },
   "adjusted": {
    "A": 0.861,
    "B": 2.0
   },
   "violations": {
    "A": [
     "[gemini-pro] 프롬프트에 명시된 카메라 각도(Trunks are viewed steeply upward)를 완전히 위반하여 수평 앵글로 렌더링됨",
     "[gemini-pro] 레퍼런스 이미지의 장소적 특징(활엽수림)을 무시하고 지시되지 않은 다른 형태의 식생(침엽수)을 생성하여 위치 고정(LOCATION lock)을 위반함"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "B": 2000,
   "A": 861
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "지면에서 하늘을 올려다보는 가파른 카메라 앵글(steeply upward)과 활엽수림의 식생, 하강하는 세 대의 탐사선과 잿빛 하늘의 분위기를 프롬프트와 레퍼런스의 지시에 맞춰 완벽에 가깝게 구현했습니다."
   },
   {
    "label": "A",
    "score": 861,
    "verdict_ko": "가파르게 위를 올려다보는 앵글(steeply upward) 지시를 무시하고 수평 앵글로 촬영했으며, 레퍼런스의 장소(활엽수림)를 침엽수림으로 임의 변경하는 심각한 오류를 범했습니다.  ★위반: [gemini-pro] 프롬프트에 명시된 카메라 각도(Trunks are viewed steeply upward)를 완전히 위반하여 수평 앵글로 렌더링됨 / [gemini-pro] 레퍼런스 이미지의 장소적 특징(활엽수림)을 무시하고 지시되지 않은 다른 형태의 식생(침엽수)을 생성하여 위치 고정(LOCATION lock)을 위반함"
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S1sh8_sel.png",
    "asset_id": "098848a1-3e84-4d72-96e8-dfdfdf22420b",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc3d-0c20-7b1b-a130-eb105630dcc3",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S1sh8"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S2sh5::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:25:48.879287+00:00",
  "fingerprint": "a61341368341c94f8eecdd708c5b9a821638bd2ef2dc2ebcfe49d0cc2b8129e9",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S2sh5_sel.png",
  "source_sha256": "8d080d47e7f429b440e00a97ad1582d7b118ce46f65a658f4160d68293b37435",
  "file": "S2sh5_cine.png",
  "staged_sha256": "79f68af7d7f17b3a9dee63504d0d53a6abdf9ac5fde0c94a6ea1333c286ee711",
  "latency_ms": 15931
 },
 "S2sh11::signage": {
  "fp": "7bc0b3d26bc75333",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S2sh11": {
  "input_fingerprint": "45f20566857c4594",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 탐색 침에서 뻗어 나온 붉은 빛점이 금속 부품을 덮은 윌마의 손등 위에 머무르는 순간\n\nLOCATION (lock): A sheltered exterior spot beside a tree trunk within the forest, where a hovering search probe inspects them at close range. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: 윌마 디어링's covering hand in the middle-center of the frame, foreground; 미지(MIDGE), minimally indicated in the upper-right of the frame, midground, looks toward Wilma's covering hand.\n- KEY BACKGROUND ELEMENTS: metal part beneath Wilma's fingers (covered by her hand) — Only its edges remain visible beneath the enclosing fingers; used as Explains why the scanning point rests on the hand rather than directly on the metal; red light point (resting on the back of Wilma's hand); used as Central focal detail of the probe's observation.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Muted gray daylight surrounds the explicitly red light point, which supplies the shot's only localized color emphasis.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the gray daytime sky, dense forest foliage, bark, and cool muted lighting from the reference. Exclude the two distant probes not involved in this beat and any cliff-erasure effects; keep only the nearby probe and its red scanning light.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A search probe hovers directly ahead with its lens turned toward a metal zipper. The phone remains enclosed in the waterproof pouch and wrapped with a metal strap. 윌마 디어링: Her fingers cover the targeted metal zipper. Her weighted belt and gun are smeared with heat-eating gray moss.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 탐색 침에서 뻗어 나온 붉은 빛점이 금속 부품을 덮은 윌마의 손등 위에 머무르는 순간\n\nLOCATION (lock): A sheltered exterior spot beside a tree trunk within the forest, where a hovering search probe inspects them at close range. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: 윌마 디어링's covering hand in the middle-center of the frame, foreground; 미지(MIDGE), minimally indicated in the upper-right of the frame, midground, looks toward Wilma's covering hand.\n- KEY BACKGROUND ELEMENTS: metal part beneath Wilma's fingers (covered by her hand) — Only its edges remain visible beneath the enclosing fingers; used as Explains why the scanning point rests on the hand rather than directly on the metal; red light point (resting on the back of Wilma's hand); used as Central focal detail of the probe's observation.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Muted gray daylight surrounds the explicitly red light point, which supplies the shot's only localized color emphasis.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the gray daytime sky, dense forest foliage, bark, and cool muted lighting from the reference. Exclude the two distant probes not involved in this beat and any cliff-erasure effects; keep only the nearby probe and its red scanning light.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A search probe hovers directly ahead with its lens turned toward a metal zipper. The phone remains enclosed in the waterproof pouch and wrapped with a metal strap. 윌마 디어링: Her fingers cover the targeted metal zipper. Her weighted belt and gun are smeared with heat-eating gray moss.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 탐색 침에서 뻗어 나온 붉은 빛점이 금속 부품을 덮은 윌마의 손등 위에 머무르는 순간\n\nLOCATION (lock): A sheltered exterior spot beside a tree trunk within the forest, where a hovering search probe inspects them at close range. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: 윌마 디어링's covering hand in the middle-center of the frame, foreground; 미지(MIDGE), minimally indicated in the upper-right of the frame, midground, looks toward Wilma's covering hand.\n- KEY BACKGROUND ELEMENTS: metal part beneath Wilma's fingers (covered by her hand) — Only its edges remain visible beneath the enclosing fingers; used as Explains why the scanning point rests on the hand rather than directly on the metal; red light point (resting on the back of Wilma's hand); used as Central focal detail of the probe's observation.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Muted gray daylight surrounds the explicitly red light point, which supplies the shot's only localized color emphasis.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the gray daytime sky, dense forest foliage, bark, and cool muted lighting from the reference. Exclude the two distant probes not involved in this beat and any cliff-erasure effects; keep only the nearby probe and its red scanning light.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A search probe hovers directly ahead with its lens turned toward a metal zipper. The phone remains enclosed in the waterproof pouch and wrapped with a metal strap. 윌마 디어링: Her fingers cover the targeted metal zipper. Her weighted belt and gun are smeared with heat-eating gray moss.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "붉은 빛점이 인물의 오른손 손등 위에 정확히 머물고 있으며, 우측 상단에 배치된 인물은 자신의 손과 파우치 쪽으로 시선을 내리깔고 있음.",
    "built_space": "좌측에 이끼가 낀 굵은 나무 기둥이 화면의 깊이감을 형성하는 숲 속 야외. 인물은 나무 기둥 근처에서 손을 뻗고 있음.",
    "entities": "금속 지퍼가 달린 파우치를 덮고 있는 손은 윌마(여성)의 피부 톤과 연령에 부합함. 손등 위에 붉은 빛점이 선명하게 묘사됨. 우측 상단에는 레이아웃 지시에 따라 인물의 얼굴이 포함되었으나, 제공된 윌마의 레퍼런스 외모가 그대로 적용됨.",
    "hard_violations": [],
    "physics": "손은 나무 기둥 및 파우치와 자연스럽게 접촉하여 무게를 지탱하고 있음. 카메라를 향해 뻗은 오른팔의 구도와 원근감은 물리적으로 타당함."
   },
   {
    "label": "B",
    "direction": "공중에 떠 있는 탐색 침의 붉은 렌즈가 아래의 손을 직접 겨냥하고 있으며, 붉은 빛점이 손등 위에 맺혀 있음. 배경의 로봇은 화면 좌측을 향해 있음.",
    "built_space": "인물의 손 뒤편으로 숲의 나무 기둥과 나뭇잎들이 배치된 야외 공간. 우측 배경에 지시사항에 없는 로봇이 자리 잡고 있음.",
    "entities": "지퍼가 달린 금속 부품을 덮은 손은 굵은 털이 무성하여 윌마(여성)의 정체성을 심각하게 훼손함. 탐색 침과 붉은 빛점은 존재하나, 우측 배경에 프롬프트가 배제를 지시한 발명된 거대 기계/로봇이 추가됨.",
    "hard_violations": [
     "[gemini-pro] 지시되지 않은 발명된 물체 추가 (우측 배경의 대형 인간형 기계/로봇)",
     "[gpt] 서로 연결되지 않은 상단 렌즈 장치와 오른쪽 탐색 장치가 별개의 두 탐색체처럼 보여, 한 대만 남기라는 지시를 위반한 중복 물체로 읽힌다."
    ],
    "physics": "손은 바닥에 놓인 파우치 위에 안정적으로 얹혀 있음. 근거리에 떠 있는 탐색 침은 자체 동력으로 비행 중이며, 배경의 로봇은 지면에 서 있음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "프레임 레이아웃(중앙에 위치한 손, 우측 상단에서 손을 바라보는 인물)을 완벽히 준수했으며, 손등 위의 붉은 빛점과 환경 디테일이 프롬프트의 요구사항을 훌륭하게 충족합니다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "여성 캐릭터인 윌마의 손을 털이 많은 남성의 손처럼 잘못 묘사했으며, 프롬프트에서 금지한 형태의 정체불명 로봇을 배경에 추가하는 치명적인 위반(Hard Violation)을 범했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "붉은 빛점이 인물의 오른손 손등 위에 정확히 머물고 있으며, 우측 상단에 배치된 인물은 자신의 손과 파우치 쪽으로 시선을 내리깔고 있음.",
        "built_space": "좌측에 이끼가 낀 굵은 나무 기둥이 화면의 깊이감을 형성하는 숲 속 야외. 인물은 나무 기둥 근처에서 손을 뻗고 있음.",
        "entities": "금속 지퍼가 달린 파우치를 덮고 있는 손은 윌마(여성)의 피부 톤과 연령에 부합함. 손등 위에 붉은 빛점이 선명하게 묘사됨. 우측 상단에는 레이아웃 지시에 따라 인물의 얼굴이 포함되었으나, 제공된 윌마의 레퍼런스 외모가 그대로 적용됨.",
        "hard_violations": [],
        "physics": "손은 나무 기둥 및 파우치와 자연스럽게 접촉하여 무게를 지탱하고 있음. 카메라를 향해 뻗은 오른팔의 구도와 원근감은 물리적으로 타당함."
       },
       {
        "label": "B",
        "direction": "공중에 떠 있는 탐색 침의 붉은 렌즈가 아래의 손을 직접 겨냥하고 있으며, 붉은 빛점이 손등 위에 맺혀 있음. 배경의 로봇은 화면 좌측을 향해 있음.",
        "built_space": "인물의 손 뒤편으로 숲의 나무 기둥과 나뭇잎들이 배치된 야외 공간. 우측 배경에 지시사항에 없는 로봇이 자리 잡고 있음.",
        "entities": "지퍼가 달린 금속 부품을 덮은 손은 굵은 털이 무성하여 윌마(여성)의 정체성을 심각하게 훼손함. 탐색 침과 붉은 빛점은 존재하나, 우측 배경에 프롬프트가 배제를 지시한 발명된 거대 기계/로봇이 추가됨.",
        "hard_violations": [
         "지시되지 않은 발명된 물체 추가 (우측 배경의 대형 인간형 기계/로봇)"
        ],
        "physics": "손은 바닥에 놓인 파우치 위에 안정적으로 얹혀 있음. 근거리에 떠 있는 탐색 침은 자체 동력으로 비행 중이며, 배경의 로봇은 지면에 서 있음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "프레임 레이아웃(중앙에 위치한 손, 우측 상단에서 손을 바라보는 인물)을 완벽히 준수했으며, 손등 위의 붉은 빛점과 환경 디테일이 프롬프트의 요구사항을 훌륭하게 충족합니다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "여성 캐릭터인 윌마의 손을 털이 많은 남성의 손처럼 잘못 묘사했으며, 프롬프트에서 금지한 형태의 정체불명 로봇을 배경에 추가하는 치명적인 위반(Hard Violation)을 범했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "붉은 빛점이 인물의 오른손 손등 위에 정확히 머물고 있으며, 우측 상단에 배치된 인물은 자신의 손과 파우치 쪽으로 시선을 내리깔고 있음.",
        "built_space": "좌측에 이끼가 낀 굵은 나무 기둥이 화면의 깊이감을 형성하는 숲 속 야외. 인물은 나무 기둥 근처에서 손을 뻗고 있음.",
        "entities": "금속 지퍼가 달린 파우치를 덮고 있는 손은 윌마(여성)의 피부 톤과 연령에 부합함. 손등 위에 붉은 빛점이 선명하게 묘사됨. 우측 상단에는 레이아웃 지시에 따라 인물의 얼굴이 포함되었으나, 제공된 윌마의 레퍼런스 외모가 그대로 적용됨.",
        "hard_violations": [],
        "physics": "손은 나무 기둥 및 파우치와 자연스럽게 접촉하여 무게를 지탱하고 있음. 카메라를 향해 뻗은 오른팔의 구도와 원근감은 물리적으로 타당함."
       },
       {
        "label": "B",
        "direction": "공중에 떠 있는 탐색 침의 붉은 렌즈가 아래의 손을 직접 겨냥하고 있으며, 붉은 빛점이 손등 위에 맺혀 있음. 배경의 로봇은 화면 좌측을 향해 있음.",
        "built_space": "인물의 손 뒤편으로 숲의 나무 기둥과 나뭇잎들이 배치된 야외 공간. 우측 배경에 지시사항에 없는 로봇이 자리 잡고 있음.",
        "entities": "지퍼가 달린 금속 부품을 덮은 손은 굵은 털이 무성하여 윌마(여성)의 정체성을 심각하게 훼손함. 탐색 침과 붉은 빛점은 존재하나, 우측 배경에 프롬프트가 배제를 지시한 발명된 거대 기계/로봇이 추가됨.",
        "hard_violations": [
         "지시되지 않은 발명된 물체 추가 (우측 배경의 대형 인간형 기계/로봇)"
        ],
        "physics": "손은 바닥에 놓인 파우치 위에 안정적으로 얹혀 있음. 근거리에 떠 있는 탐색 침은 자체 동력으로 비행 중이며, 배경의 로봇은 지면에 서 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "탐색 렌즈가 손등의 붉은 점을 정확히 겨누지만, 손이 윌마의 여성 손으로 보이지 않고 별개의 탐색 장치가 하나 더 보여 치명적으로 불리하다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "탐색 침 자체는 프레임 밖이지만, 중앙 전경의 윌마 손과 손등의 붉은 점, 손가락 아래 지퍼, 우상단에서 손을 보는 얼굴이라는 핵심 인서트 구성을 가장 충실히 구현했다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "상단의 붉은 렌즈가 아래쪽 윌마로 의도된 손을 향하며, 그 광축은 손등의 붉은 점에 대체로 도달한다. 오른쪽의 별도 장치도 손 쪽을 향하는 듯하지만 초점이 흐려 정확한 목표는 불분명하다. 사람의 시선은 보이지 않는다.",
        "built_space": "굵은 나무줄기 1개와 축축한 숲 바닥 및 foliage가 보이며, 이전 장면의 차갑고 흐린 숲 재질과 조명은 잘 이어진다. 손은 녹색 옷 또는 파우치 위에 놓여 있고 손가락 아래 지퍼 이빨과 넓은 금속판 가장자리가 보인다. 상단 렌즈 장치와 오른쪽 기계 장치가 분리된 두 개의 탐색 장치처럼 보인다.",
        "entities": "붉은 광점, 탐색 렌즈, 나무줄기, 녹색 복장, 손가락 아래의 금속 지퍼 및 금속 부품은 존재한다. 그러나 손은 크고 털이 많은 성인 남성의 손처럼 보여 20대 후반 여성 윌마의 신체로 명확히 읽히지 않는다. 우상단에 최소한으로 보여야 하는 미지의 표시는 없으며, 대신 기계 장치가 차지한다.",
        "hard_violations": [
         "서로 연결되지 않은 상단 렌즈 장치와 오른쪽 탐색 장치가 별개의 두 탐색체처럼 보여, 한 대만 남기라는 지시를 위반한 중복 물체로 읽힌다."
        ],
        "physics": "손과 팔은 녹색 복장 또는 파우치 위에 안정적으로 놓여 있어 지지된다. 상단 탐색 렌즈는 프레임 위로 이어지는 하우징에 연결되어 있고 오른쪽 장치도 화면 가장자리 밖 몸체로 이어져 완전히 무지지 상태로 보이지는 않는다. 광점은 렌즈와 손등 사이의 정렬로 물리적 원인이 설명된다."
       },
       {
        "label": "B",
        "direction": "우상단 얼굴의 두 눈은 중앙 전경의 손과 붉은 점을 향한다. 탐색 침이나 렌즈는 프레임 밖이라 그 자체의 조준축은 확인할 수 없지만, 붉은 점은 손등에 정확히 머문다.",
        "built_space": "왼쪽에 이끼 낀 나무줄기 1개가 크게 보이고 뒤에는 흐린 숲이 이어져 잠긴 장소와 낮의 차가운 회색 조명을 잘 유지한다. 손은 나무 옆의 녹색 파우치 또는 복장 위에 놓이며, 금속 지퍼가 손가락 아래에서 부분적으로 드러난다. 탐색체 본체나 고정 설비는 화면에 보이지 않으며 불가능한 반사도 없다.",
        "entities": "중앙의 여성 손, 손등의 붉은 광점, 손가락 아래 금속 지퍼, 녹색 복장과 나무줄기가 확인된다. 우상단 얼굴은 윌마 참고 이미지와 유사한 젊은 백인 여성으로 보이며 같은 몸에 연결되어 있어 별도 인물이 중복된 것은 아니다. 다만 프레임 지시가 이름 붙인 미지로 확정할 수 없고, 가까이 떠 있는 탐색체는 화면에 보이지 않는다. 지퍼도 가장자리만 남기기보다는 비교적 긴 구간이 노출된다.",
        "hard_violations": [],
        "physics": "손가락은 파우치와 지퍼 위를 눌러 덮고 있고 손목과 팔은 소매 및 몸으로 이어져 자연스럽게 지지된다. 얼굴과 상체는 나무 뒤에 몸을 낮춘 자세로 연결되어 있으며 공중에 뜬 신체나 물체는 없다. 붉은 점의 광원인 탐색체는 프레임 밖에 있다고 해석해야 한다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "탐색 렌즈가 손등의 붉은 점을 정확히 겨누지만, 손이 윌마의 여성 손으로 보이지 않고 별개의 탐색 장치가 하나 더 보여 치명적으로 불리하다."
       },
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "탐색 침 자체는 프레임 밖이지만, 중앙 전경의 윌마 손과 손등의 붉은 점, 손가락 아래 지퍼, 우상단에서 손을 보는 얼굴이라는 핵심 인서트 구성을 가장 충실히 구현했다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "상단의 붉은 렌즈가 아래쪽 윌마로 의도된 손을 향하며, 그 광축은 손등의 붉은 점에 대체로 도달한다. 오른쪽의 별도 장치도 손 쪽을 향하는 듯하지만 초점이 흐려 정확한 목표는 불분명하다. 사람의 시선은 보이지 않는다.",
        "built_space": "굵은 나무줄기 1개와 축축한 숲 바닥 및 foliage가 보이며, 이전 장면의 차갑고 흐린 숲 재질과 조명은 잘 이어진다. 손은 녹색 옷 또는 파우치 위에 놓여 있고 손가락 아래 지퍼 이빨과 넓은 금속판 가장자리가 보인다. 상단 렌즈 장치와 오른쪽 기계 장치가 분리된 두 개의 탐색 장치처럼 보인다.",
        "entities": "붉은 광점, 탐색 렌즈, 나무줄기, 녹색 복장, 손가락 아래의 금속 지퍼 및 금속 부품은 존재한다. 그러나 손은 크고 털이 많은 성인 남성의 손처럼 보여 20대 후반 여성 윌마의 신체로 명확히 읽히지 않는다. 우상단에 최소한으로 보여야 하는 미지의 표시는 없으며, 대신 기계 장치가 차지한다.",
        "hard_violations": [
         "서로 연결되지 않은 상단 렌즈 장치와 오른쪽 탐색 장치가 별개의 두 탐색체처럼 보여, 한 대만 남기라는 지시를 위반한 중복 물체로 읽힌다."
        ],
        "physics": "손과 팔은 녹색 복장 또는 파우치 위에 안정적으로 놓여 있어 지지된다. 상단 탐색 렌즈는 프레임 위로 이어지는 하우징에 연결되어 있고 오른쪽 장치도 화면 가장자리 밖 몸체로 이어져 완전히 무지지 상태로 보이지는 않는다. 광점은 렌즈와 손등 사이의 정렬로 물리적 원인이 설명된다."
       },
       {
        "label": "A",
        "direction": "우상단 얼굴의 두 눈은 중앙 전경의 손과 붉은 점을 향한다. 탐색 침이나 렌즈는 프레임 밖이라 그 자체의 조준축은 확인할 수 없지만, 붉은 점은 손등에 정확히 머문다.",
        "built_space": "왼쪽에 이끼 낀 나무줄기 1개가 크게 보이고 뒤에는 흐린 숲이 이어져 잠긴 장소와 낮의 차가운 회색 조명을 잘 유지한다. 손은 나무 옆의 녹색 파우치 또는 복장 위에 놓이며, 금속 지퍼가 손가락 아래에서 부분적으로 드러난다. 탐색체 본체나 고정 설비는 화면에 보이지 않으며 불가능한 반사도 없다.",
        "entities": "중앙의 여성 손, 손등의 붉은 광점, 손가락 아래 금속 지퍼, 녹색 복장과 나무줄기가 확인된다. 우상단 얼굴은 윌마 참고 이미지와 유사한 젊은 백인 여성으로 보이며 같은 몸에 연결되어 있어 별도 인물이 중복된 것은 아니다. 다만 프레임 지시가 이름 붙인 미지로 확정할 수 없고, 가까이 떠 있는 탐색체는 화면에 보이지 않는다. 지퍼도 가장자리만 남기기보다는 비교적 긴 구간이 노출된다.",
        "hard_violations": [],
        "physics": "손가락은 파우치와 지퍼 위를 눌러 덮고 있고 손목과 팔은 소매 및 몸으로 이어져 자연스럽게 지지된다. 얼굴과 상체는 나무 뒤에 몸을 낮춘 자세로 연결되어 있으며 공중에 뜬 신체나 물체는 없다. 붉은 점의 광원인 탐색체는 프레임 밖에 있다고 해석해야 한다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 0.762
   },
   "adjusted": {
    "A": 2.0,
    "B": 0.512
   },
   "violations": {
    "B": [
     "[gemini-pro] 지시되지 않은 발명된 물체 추가 (우측 배경의 대형 인간형 기계/로봇)",
     "[gpt] 서로 연결되지 않은 상단 렌즈 장치와 오른쪽 탐색 장치가 별개의 두 탐색체처럼 보여, 한 대만 남기라는 지시를 위반한 중복 물체로 읽힌다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 512
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "프레임 레이아웃(중앙에 위치한 손, 우측 상단에서 손을 바라보는 인물)을 완벽히 준수했으며, 손등 위의 붉은 빛점과 환경 디테일이 프롬프트의 요구사항을 훌륭하게 충족합니다."
   },
   {
    "label": "B",
    "score": 512,
    "verdict_ko": "여성 캐릭터인 윌마의 손을 털이 많은 남성의 손처럼 잘못 묘사했으며, 프롬프트에서 금지한 형태의 정체불명 로봇을 배경에 추가하는 치명적인 위반(Hard Violation)을 범했습니다.  ★위반: [gemini-pro] 지시되지 않은 발명된 물체 추가 (우측 배경의 대형 인간형 기계/로봇) / [gpt] 서로 연결되지 않은 상단 렌즈 장치와 오른쪽 탐색 장치가 별개의 두 탐색체처럼 보여, 한 대만 남기라는 지시를 위반한 중복 물체로 읽힌다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S2sh5_sel.png",
    "asset_id": "42298d9d-acd7-4df6-98c9-e65eacd7d571",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc40-cf51-7fe6-9a75-9e61e60374fb",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S2sh5"
  }
 },
 "S2sh11::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:27:42.228532+00:00",
  "fingerprint": "374c6b03b3090ed3d7c54412e754d1a524358d95aee53f1f67f2a6449a416c5f",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S2sh11_sel.png",
  "source_sha256": "db42b2527c501ea5ea2da4b8a0a86a3391987a04188f55ef2b8d1d6820ef4d68",
  "file": "S2sh11_cine.png",
  "staged_sha256": "063cb61748e6c4b8ee030616b85f0043cb6c96e9fe6b9d1efe2e66b5daba0ab8",
  "latency_ms": 16711
 },
 "S2sh12::signage": {
  "fp": "26d29279ba062709",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S2sh12": {
  "input_fingerprint": "b25ee01dce87337e",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 탐색 침의 빛점이 다른 방향을 향한 가운데, 가슴에 윌마의 손이 얹힌 채 안도하며 눈을 감은 토니의 표정\n\nLOCATION (lock): The exterior hiding spot behind a forest tree, with the search probe drifting nearby among the trees. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-center of the frame, foreground; 미지(MIDGE), peripheral in the middle-right of the frame, midground, looks toward direction away from Tony.\n- KEY BACKGROUND ELEMENTS: redirected light point (aimed away from Tony); used as Peripheral evidence that the immediate scan has moved away from him; tree cover (the characters remain sheltered beside it) — Only a limited section is visible behind the close grouping; used as Provides minimal spatial context without competing with Tony's expression.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained gray daylight and low-to-moderate contrast hold the relief in Tony's face while the redirected point remains a small peripheral accent.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 윌마 디어링 — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the cloudy daylight, dense treetops, and cool gray-green forest palette from the reference. Exclude the other two descending probes and any cliff-erasure effects; retain only the nearby probe light turning away.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The probe's light has turned away and is moving on. The phone remains sealed inside the metal-wrapped waterproof pouch. 윌마 디어링: Her hand remains over the metal zipper as the probe passes. Her weighted belt and gun remain coated with gray moss. 토니(앤서니 로저스): He remains still behind the tree, breathing shallowly and carrying his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 탐색 침의 빛점이 다른 방향을 향한 가운데, 가슴에 윌마의 손이 얹힌 채 안도하며 눈을 감은 토니의 표정\n\nLOCATION (lock): The exterior hiding spot behind a forest tree, with the search probe drifting nearby among the trees. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-center of the frame, foreground; 미지(MIDGE), peripheral in the middle-right of the frame, midground, looks toward direction away from Tony.\n- KEY BACKGROUND ELEMENTS: redirected light point (aimed away from Tony); used as Peripheral evidence that the immediate scan has moved away from him; tree cover (the characters remain sheltered beside it) — Only a limited section is visible behind the close grouping; used as Provides minimal spatial context without competing with Tony's expression.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained gray daylight and low-to-moderate contrast hold the relief in Tony's face while the redirected point remains a small peripheral accent.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 윌마 디어링 — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the cloudy daylight, dense treetops, and cool gray-green forest palette from the reference. Exclude the other two descending probes and any cliff-erasure effects; retain only the nearby probe light turning away.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The probe's light has turned away and is moving on. The phone remains sealed inside the metal-wrapped waterproof pouch. 윌마 디어링: Her hand remains over the metal zipper as the probe passes. Her weighted belt and gun remain coated with gray moss. 토니(앤서니 로저스): He remains still behind the tree, breathing shallowly and carrying his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 탐색 침의 빛점이 다른 방향을 향한 가운데, 가슴에 윌마의 손이 얹힌 채 안도하며 눈을 감은 토니의 표정\n\nLOCATION (lock): The exterior hiding spot behind a forest tree, with the search probe drifting nearby among the trees. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-center of the frame, foreground; 미지(MIDGE), peripheral in the middle-right of the frame, midground, looks toward direction away from Tony.\n- KEY BACKGROUND ELEMENTS: redirected light point (aimed away from Tony); used as Peripheral evidence that the immediate scan has moved away from him; tree cover (the characters remain sheltered beside it) — Only a limited section is visible behind the close grouping; used as Provides minimal spatial context without competing with Tony's expression.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained gray daylight and low-to-moderate contrast hold the relief in Tony's face while the redirected point remains a small peripheral accent.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 윌마 디어링 — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the cloudy daylight, dense treetops, and cool gray-green forest palette from the reference. Exclude the other two descending probes and any cliff-erasure effects; retain only the nearby probe light turning away.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The probe's light has turned away and is moving on. The phone remains sealed inside the metal-wrapped waterproof pouch. 윌마 디어링: Her hand remains over the metal zipper as the probe passes. Her weighted belt and gun remain coated with gray moss. 토니(앤서니 로저스): He remains still behind the tree, breathing shallowly and carrying his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "배경의 드론이 왼쪽으로 하얀 빛을 비추고 있으나, 토니의 가슴에 얹힌 윌마의 손등에는 붉은 레이저 점이 맺혀 있음. 토니는 눈을 감은 채 정면을 향하고, 우측의 여성은 왼쪽(토니 방향)을 바라봄.",
    "built_space": "숲 속 환경. 토니의 바로 뒤에 굵은 나무 기둥이 위치하여 배경을 이룸.",
    "entities": "토니는 레퍼런스와 일치하게 헬멧, 복면, 로프 등을 완벽히 착용함. 화면 왼쪽 아래에서 녹색 소매를 입은 윌마의 팔이 뻗어 나옴. 우측에는 윌마의 복장을 한 여성이 미지의 역할로 배치됨. 배경에 드론이 존재함.",
    "hard_violations": [
     "[gpt] 오른쪽에 서 있는 윌마와 별개로 화면 왼쪽 밖의 인물에게 연결된 팔과 손이 추가되어 윌마의 신체가 중복된다.",
     "[gpt] 왼쪽 팔을 오른쪽 윌마의 몸과 연결할 수 없는 물리적으로 불가능한 인물 배치다."
    ],
    "physics": "토니는 나무에 기대어 체중을 지탱함. 윌마의 손은 토니의 가슴 장비 위에 얹혀 지지를 받음. 드론은 추진력에 의해 공중에 떠 있음."
   },
   {
    "label": "B",
    "direction": "배경의 드론이 왼쪽으로 붉은 빛을 비추고 있음. 왼쪽의 윌마는 토니를 바라보고, 토니는 눈을 감은 채 정면을 향함. 우측의 여성은 오른쪽(토니의 반대 방향)을 바라봄.",
    "built_space": "숲 속 환경. 토니의 바로 뒤에 굵은 나무 기둥이 위치함.",
    "entities": "토니는 헬멧과 복면을 모두 착용하지 않아 레퍼런스와 크게 불일치함. 화면 왼쪽에 윌마의 얼굴과 어깨가 나타나며, 우측에도 윌마와 동일한 헤어 및 복장의 여성이 있음.",
    "hard_violations": [
     "[gemini-pro] duplicated or extra bodies (동일한 외형과 복장을 한 여성이 화면 왼쪽과 오른쪽에 중복되어 나타남)",
     "[gpt] 윌마로 보이는 여성이 화면 왼쪽 전경과 오른쪽 중경에 두 몸으로 중복 등장한다."
    ],
    "physics": "토니는 나무에 기대어 서 있음. 토니의 가슴에 얹힌 손은 몸에 의해 지지됨. 드론은 공중에 안정적으로 떠 있음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "지정된 프레이밍(중앙의 토니, 우측의 미지, 화면 밖에서 뻗은 윌마의 손)과 토니의 복장 레퍼런스를 훌륭하게 구현했으나, 빛점이 다른 곳을 향해야 한다는 지시와 달리 손 위에 붉은 점이 남아있어 감점되었습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "탐색 침의 빛이 다른 방향을 향하는 지시는 따랐으나, 프레이밍을 어기고 왼쪽 전경에 윌마의 얼굴을 크게 배치했으며 토니의 핵심 복장(헬멧, 마스크)을 완전히 누락했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "배경의 드론이 왼쪽으로 하얀 빛을 비추고 있으나, 토니의 가슴에 얹힌 윌마의 손등에는 붉은 레이저 점이 맺혀 있음. 토니는 눈을 감은 채 정면을 향하고, 우측의 여성은 왼쪽(토니 방향)을 바라봄.",
        "built_space": "숲 속 환경. 토니의 바로 뒤에 굵은 나무 기둥이 위치하여 배경을 이룸.",
        "entities": "토니는 레퍼런스와 일치하게 헬멧, 복면, 로프 등을 완벽히 착용함. 화면 왼쪽 아래에서 녹색 소매를 입은 윌마의 팔이 뻗어 나옴. 우측에는 윌마의 복장을 한 여성이 미지의 역할로 배치됨. 배경에 드론이 존재함.",
        "hard_violations": [],
        "physics": "토니는 나무에 기대어 체중을 지탱함. 윌마의 손은 토니의 가슴 장비 위에 얹혀 지지를 받음. 드론은 추진력에 의해 공중에 떠 있음."
       },
       {
        "label": "B",
        "direction": "배경의 드론이 왼쪽으로 붉은 빛을 비추고 있음. 왼쪽의 윌마는 토니를 바라보고, 토니는 눈을 감은 채 정면을 향함. 우측의 여성은 오른쪽(토니의 반대 방향)을 바라봄.",
        "built_space": "숲 속 환경. 토니의 바로 뒤에 굵은 나무 기둥이 위치함.",
        "entities": "토니는 헬멧과 복면을 모두 착용하지 않아 레퍼런스와 크게 불일치함. 화면 왼쪽에 윌마의 얼굴과 어깨가 나타나며, 우측에도 윌마와 동일한 헤어 및 복장의 여성이 있음.",
        "hard_violations": [
         "duplicated or extra bodies (동일한 외형과 복장을 한 여성이 화면 왼쪽과 오른쪽에 중복되어 나타남)"
        ],
        "physics": "토니는 나무에 기대어 서 있음. 토니의 가슴에 얹힌 손은 몸에 의해 지지됨. 드론은 공중에 안정적으로 떠 있음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "지정된 프레이밍(중앙의 토니, 우측의 미지, 화면 밖에서 뻗은 윌마의 손)과 토니의 복장 레퍼런스를 훌륭하게 구현했으나, 빛점이 다른 곳을 향해야 한다는 지시와 달리 손 위에 붉은 점이 남아있어 감점되었습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "탐색 침의 빛이 다른 방향을 향하는 지시는 따랐으나, 프레이밍을 어기고 왼쪽 전경에 윌마의 얼굴을 크게 배치했으며 토니의 핵심 복장(헬멧, 마스크)을 완전히 누락했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "배경의 드론이 왼쪽으로 하얀 빛을 비추고 있으나, 토니의 가슴에 얹힌 윌마의 손등에는 붉은 레이저 점이 맺혀 있음. 토니는 눈을 감은 채 정면을 향하고, 우측의 여성은 왼쪽(토니 방향)을 바라봄.",
        "built_space": "숲 속 환경. 토니의 바로 뒤에 굵은 나무 기둥이 위치하여 배경을 이룸.",
        "entities": "토니는 레퍼런스와 일치하게 헬멧, 복면, 로프 등을 완벽히 착용함. 화면 왼쪽 아래에서 녹색 소매를 입은 윌마의 팔이 뻗어 나옴. 우측에는 윌마의 복장을 한 여성이 미지의 역할로 배치됨. 배경에 드론이 존재함.",
        "hard_violations": [],
        "physics": "토니는 나무에 기대어 체중을 지탱함. 윌마의 손은 토니의 가슴 장비 위에 얹혀 지지를 받음. 드론은 추진력에 의해 공중에 떠 있음."
       },
       {
        "label": "B",
        "direction": "배경의 드론이 왼쪽으로 붉은 빛을 비추고 있음. 왼쪽의 윌마는 토니를 바라보고, 토니는 눈을 감은 채 정면을 향함. 우측의 여성은 오른쪽(토니의 반대 방향)을 바라봄.",
        "built_space": "숲 속 환경. 토니의 바로 뒤에 굵은 나무 기둥이 위치함.",
        "entities": "토니는 헬멧과 복면을 모두 착용하지 않아 레퍼런스와 크게 불일치함. 화면 왼쪽에 윌마의 얼굴과 어깨가 나타나며, 우측에도 윌마와 동일한 헤어 및 복장의 여성이 있음.",
        "hard_violations": [
         "duplicated or extra bodies (동일한 외형과 복장을 한 여성이 화면 왼쪽과 오른쪽에 중복되어 나타남)"
        ],
        "physics": "토니는 나무에 기대어 서 있음. 토니의 가슴에 얹힌 손은 몸에 의해 지지됨. 드론은 공중에 안정적으로 떠 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "토니의 중앙 클로즈업과 눈을 감은 안도 표정은 맞지만, 윌마가 좌우에 중복 등장하고 탐색광도 토니 쪽으로 향해 사용 불가하다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "토니의 헬멧과 중앙 클로즈업은 참고 이미지에 가깝지만, 중복된 윌마의 손에 붉은 빛점이 직접 찍혀 ‘탐색광이 다른 방향으로 이동했다’는 핵심 순간을 정면으로 위반한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니는 눈을 감아 시선이 없고, 화면 오른쪽의 윌마는 토니에게서 벗어난 오른쪽 숲 방향을 본다. 그러나 탐색 프로브의 붉은 광선은 프로브에서 화면 왼쪽 아래, 즉 토니의 오른쪽 어깨와 몸통 쪽으로 향하며 ‘토니와 다른 방향’으로 확실히 빗나간 것으로 읽히지 않는다.",
        "built_space": "인공 구조물은 없으며, 중앙 뒤의 굵고 이끼 낀 나무 한 그루와 제한된 회색빛 숲이 은신처를 이룬다. 토니는 나무에 등을 기대고 중앙 전경에 있고, 오른쪽 윌마는 숲의 중경에 서 있다. 다만 왼쪽 전경에도 별도의 윌마 머리·상체가 있어 요구된 인물 배치와 수가 성립하지 않는다.",
        "entities": "중앙 인물은 30대 미국인 남성 토니로 보이며 얼굴·체격과 낡은 전술복, 구조용 밧줄은 대체로 맞지만 참고 이미지의 헬멧이 빠졌다. 오른쪽 여성은 윌마의 금발 묶음머리와 녹색 복장을 갖췄으나, 왼쪽에도 같은 외형의 여성이 추가로 보인다. 윌마의 손은 토니의 가슴에 실제로 닿아 있다. 프로브는 하나만 보이지만 붉은 광선이 작은 주변부 빛점보다 지나치게 크고 두드러진다.",
        "hard_violations": [
         "윌마로 보이는 여성이 화면 왼쪽 전경과 오른쪽 중경에 두 몸으로 중복 등장한다."
        ],
        "physics": "토니는 나무에 등을 기대어 지지되고 있으며, 윌마의 손은 토니의 가슴에 접촉하고 팔은 왼쪽 전경 인물에게 연결된다. 오른쪽 윌마는 지면에 서 있는 자세로 보인다. 프로브는 공중을 비행하는 장치로 묘사되어 자체 추진으로 떠 있는 것으로 읽히며 별도의 비정상적 부유 인체는 없다."
       },
       {
        "label": "B",
        "direction": "토니는 눈을 감고 있고, 오른쪽 윌마는 토니에게서 벗어난 화면 왼쪽 숲과 프로브 쪽을 본다. 프로브의 흰 광선은 화면 왼쪽 아래, 토니 쪽으로 향하며, 별도의 붉은 빛점이 토니 가슴 위 윌마의 손등에 직접 닿아 있다. 따라서 탐색광이 토니에게서 다른 방향으로 돌아섰다는 핵심 방향 관계가 명백히 실패한다.",
        "built_space": "인공 구조물은 없고, 중앙의 이끼 낀 큰 나무와 빽빽한 회색·녹색 숲이 은신처를 형성한다. 토니는 나무에 기대 중앙 전경에 있고 오른쪽 윌마는 중경에 배치된다. 하지만 토니의 가슴을 누르는 팔은 화면 왼쪽 밖의 별도 몸에서 들어오므로 오른쪽 윌마의 위치와 연결될 수 없으며, 요구된 단일 윌마 배치가 깨진다.",
        "entities": "토니는 30대 미국인 남성으로 보이고 참고 이미지와 유사한 얼굴, 검은 헬멧, 낡은 전술복과 구조용 밧줄을 갖췄다. 오른쪽 여성은 윌마의 금발 묶음머리와 녹색 복장에 대체로 부합한다. 그러나 왼쪽에서 들어오는 여성형 팔과 손은 오른쪽 윌마의 몸에 해부학적으로 이어지지 않아 추가 인물을 만든다. 프로브는 하나 보이며, 빛점이 윌마의 손에 남아 있다.",
        "hard_violations": [
         "오른쪽에 서 있는 윌마와 별개로 화면 왼쪽 밖의 인물에게 연결된 팔과 손이 추가되어 윌마의 신체가 중복된다.",
         "왼쪽 팔을 오른쪽 윌마의 몸과 연결할 수 없는 물리적으로 불가능한 인물 배치다."
        ],
        "physics": "토니는 나무에 기대어 지지되고 눈을 감은 정지 자세도 가능하다. 손은 토니의 가슴에 닿아 있지만, 그 팔은 왼쪽 화면 밖으로 이어지고 오른쪽 윌마와 연결될 수 없어 해당 손의 주체가 별도 몸이어야 한다. 오른쪽 윌마는 지면에 서거나 웅크린 자세로 지지되는 것으로 보이며, 프로브는 자체 추진 비행 장치로 읽힌다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "토니의 중앙 클로즈업과 눈을 감은 안도 표정은 맞지만, 윌마가 좌우에 중복 등장하고 탐색광도 토니 쪽으로 향해 사용 불가하다."
       },
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "토니의 헬멧과 중앙 클로즈업은 참고 이미지에 가깝지만, 중복된 윌마의 손에 붉은 빛점이 직접 찍혀 ‘탐색광이 다른 방향으로 이동했다’는 핵심 순간을 정면으로 위반한다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "토니는 눈을 감아 시선이 없고, 화면 오른쪽의 윌마는 토니에게서 벗어난 오른쪽 숲 방향을 본다. 그러나 탐색 프로브의 붉은 광선은 프로브에서 화면 왼쪽 아래, 즉 토니의 오른쪽 어깨와 몸통 쪽으로 향하며 ‘토니와 다른 방향’으로 확실히 빗나간 것으로 읽히지 않는다.",
        "built_space": "인공 구조물은 없으며, 중앙 뒤의 굵고 이끼 낀 나무 한 그루와 제한된 회색빛 숲이 은신처를 이룬다. 토니는 나무에 등을 기대고 중앙 전경에 있고, 오른쪽 윌마는 숲의 중경에 서 있다. 다만 왼쪽 전경에도 별도의 윌마 머리·상체가 있어 요구된 인물 배치와 수가 성립하지 않는다.",
        "entities": "중앙 인물은 30대 미국인 남성 토니로 보이며 얼굴·체격과 낡은 전술복, 구조용 밧줄은 대체로 맞지만 참고 이미지의 헬멧이 빠졌다. 오른쪽 여성은 윌마의 금발 묶음머리와 녹색 복장을 갖췄으나, 왼쪽에도 같은 외형의 여성이 추가로 보인다. 윌마의 손은 토니의 가슴에 실제로 닿아 있다. 프로브는 하나만 보이지만 붉은 광선이 작은 주변부 빛점보다 지나치게 크고 두드러진다.",
        "hard_violations": [
         "윌마로 보이는 여성이 화면 왼쪽 전경과 오른쪽 중경에 두 몸으로 중복 등장한다."
        ],
        "physics": "토니는 나무에 등을 기대어 지지되고 있으며, 윌마의 손은 토니의 가슴에 접촉하고 팔은 왼쪽 전경 인물에게 연결된다. 오른쪽 윌마는 지면에 서 있는 자세로 보인다. 프로브는 공중을 비행하는 장치로 묘사되어 자체 추진으로 떠 있는 것으로 읽히며 별도의 비정상적 부유 인체는 없다."
       },
       {
        "label": "A",
        "direction": "토니는 눈을 감고 있고, 오른쪽 윌마는 토니에게서 벗어난 화면 왼쪽 숲과 프로브 쪽을 본다. 프로브의 흰 광선은 화면 왼쪽 아래, 토니 쪽으로 향하며, 별도의 붉은 빛점이 토니 가슴 위 윌마의 손등에 직접 닿아 있다. 따라서 탐색광이 토니에게서 다른 방향으로 돌아섰다는 핵심 방향 관계가 명백히 실패한다.",
        "built_space": "인공 구조물은 없고, 중앙의 이끼 낀 큰 나무와 빽빽한 회색·녹색 숲이 은신처를 형성한다. 토니는 나무에 기대 중앙 전경에 있고 오른쪽 윌마는 중경에 배치된다. 하지만 토니의 가슴을 누르는 팔은 화면 왼쪽 밖의 별도 몸에서 들어오므로 오른쪽 윌마의 위치와 연결될 수 없으며, 요구된 단일 윌마 배치가 깨진다.",
        "entities": "토니는 30대 미국인 남성으로 보이고 참고 이미지와 유사한 얼굴, 검은 헬멧, 낡은 전술복과 구조용 밧줄을 갖췄다. 오른쪽 여성은 윌마의 금발 묶음머리와 녹색 복장에 대체로 부합한다. 그러나 왼쪽에서 들어오는 여성형 팔과 손은 오른쪽 윌마의 몸에 해부학적으로 이어지지 않아 추가 인물을 만든다. 프로브는 하나 보이며, 빛점이 윌마의 손에 남아 있다.",
        "hard_violations": [
         "오른쪽에 서 있는 윌마와 별개로 화면 왼쪽 밖의 인물에게 연결된 팔과 손이 추가되어 윌마의 신체가 중복된다.",
         "왼쪽 팔을 오른쪽 윌마의 몸과 연결할 수 없는 물리적으로 불가능한 인물 배치다."
        ],
        "physics": "토니는 나무에 기대어 지지되고 눈을 감은 정지 자세도 가능하다. 손은 토니의 가슴에 닿아 있지만, 그 팔은 왼쪽 화면 밖으로 이어지고 오른쪽 윌마와 연결될 수 없어 해당 손의 주체가 별도 몸이어야 한다. 오른쪽 윌마는 지면에 서거나 웅크린 자세로 지지되는 것으로 보이며, 프로브는 자체 추진 비행 장치로 읽힌다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.667,
    "B": 1.571
   },
   "adjusted": {
    "A": 1.417,
    "B": 1.321
   },
   "violations": {
    "B": [
     "[gemini-pro] duplicated or extra bodies (동일한 외형과 복장을 한 여성이 화면 왼쪽과 오른쪽에 중복되어 나타남)",
     "[gpt] 윌마로 보이는 여성이 화면 왼쪽 전경과 오른쪽 중경에 두 몸으로 중복 등장한다."
    ],
    "A": [
     "[gpt] 오른쪽에 서 있는 윌마와 별개로 화면 왼쪽 밖의 인물에게 연결된 팔과 손이 추가되어 윌마의 신체가 중복된다.",
     "[gpt] 왼쪽 팔을 오른쪽 윌마의 몸과 연결할 수 없는 물리적으로 불가능한 인물 배치다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "agreed": false
  },
  "totals": {
   "A": 1417,
   "B": 1321
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1417,
    "verdict_ko": "지정된 프레이밍(중앙의 토니, 우측의 미지, 화면 밖에서 뻗은 윌마의 손)과 토니의 복장 레퍼런스를 훌륭하게 구현했으나, 빛점이 다른 곳을 향해야 한다는 지시와 달리 손 위에 붉은 점이 남아있어 감점되었습니다.  ★위반: [gpt] 오른쪽에 서 있는 윌마와 별개로 화면 왼쪽 밖의 인물에게 연결된 팔과 손이 추가되어 윌마의 신체가 중복된다. / [gpt] 왼쪽 팔을 오른쪽 윌마의 몸과 연결할 수 없는 물리적으로 불가능한 인물 배치다."
   },
   {
    "label": "B",
    "score": 1321,
    "verdict_ko": "탐색 침의 빛이 다른 방향을 향하는 지시는 따랐으나, 프레이밍을 어기고 왼쪽 전경에 윌마의 얼굴을 크게 배치했으며 토니의 핵심 복장(헬멧, 마스크)을 완전히 누락했습니다.  ★위반: [gemini-pro] duplicated or extra bodies (동일한 외형과 복장을 한 여성이 화면 왼쪽과 오른쪽에 중복되어 나타남) / [gpt] 윌마로 보이는 여성이 화면 왼쪽 전경과 오른쪽 중경에 두 몸으로 중복 등장한다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The clothing, hair and overall look of 윌마 디어링 — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S2sh11_sel.png",
    "asset_id": "1abbd4e0-dc45-4761-95db-c10b2a250fd6",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc47-f0bd-721f-be05-ba3f2220a2a3",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S2sh11"
  }
 },
 "S2sh12::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:30:03.154724+00:00",
  "fingerprint": "3b808cea9c7093d460b182120884c676d2ddb8b9e5e4ad8ccbab607ca64dd911",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S2sh12_sel.png",
  "source_sha256": "a7775dc21ec29902c9b22f32c7cae485164d7e4a0581adf8a39be13c7eb6124e",
  "file": "S2sh12_cine.png",
  "staged_sha256": "cf3997a4d0a80e103bd561cace70cb6764c2d97bad166179ac161337fa92f33e",
  "latency_ms": 17338
 },
 "S3sh5::signage": {
  "fp": "993c956e30cde3b1",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "era_assess::964b937551be18e4": {
  "subjects": [],
  "subject_text": "거대 수목이 우거진 숲속 능선\n거대한 나무들이 빽빽하게 선 낮의 숲속 능선. 넓게 벌어진 수관과 굵은 가지, 나무 사이의 좁은 골짜기와 덤불이 이어진다.",
  "identity": "canonical",
  "scope_id": "L04",
  "scope_role": "location_exterior",
  "scope_sha": "24ac46c326f3f438"
 },
 "groupbg::forest_training_slope": {
  "input_fingerprint": "a3f0d028c71c2ab5",
  "meta": {
   "model": "gpt-image-2",
   "size": "1536x864",
   "pack": "11.202607220237",
   "contract": "bgfirst_full_v3",
   "group_sig": {
    "key": "forest_training_slope",
    "tags": [
     "S3sh5",
     "S3sh8",
     "S4sh11",
     "S4sh5"
    ]
   },
   "context_sig": "0864a44d1bf30933"
  },
  "prompt": "Create ONE empty live-action location background photograph — NO PEOPLE, no figures, no body parts, no silhouettes, no shadows or reflections of people anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods and produce in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch of a shot that happens at this location: use it ONLY as spatial evidence — what this place contains, how its ground, structures and landmarks are arranged and proportioned. Ignore the sketched people and arrows entirely, and do NOT copy its line style: render a fully photographic, physically plausible real place that fits THE LOCATION text below.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nTHE LOCATION — far-future Pennsylvania, United States; all people are American and English-speaking unless stated: The exterior forest floor on a wooded ridge, where the two stand over the belt and sealed phone pouch.\n\nSCENE EVIDENCE (verbatim quotes from the screenplay about this place — treat them as evidence of what the location physically contains and looks like; stage the PLACE those moments happen in, but do NOT depict the momentary actions, people or staged props themselves):\n- S3. EXT. 숲 바닥 — 잠시 후 (01:25–02:10)\n- 윌마가 쓰러진 추격자에게서 가져온 점퍼 벨트를 토니 앞에 던진다.\n- S4. EXT. 경사진 숲 — 연속 (02:10–03:10)\n- 토니가 뛴다.\n과하게 높이 솟아 나뭇가지 사이에 거꾸로 처박힌다.\n\nTIME OF DAY (lock): day.\n\nRender ONE photorealistic empty location photograph, 16:9, neutral enough that every shot of this place can be staged from it later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "effective_prompt": "Create ONE empty live-action location background photograph — NO PEOPLE, no figures, no body parts, no silhouettes, no shadows or reflections of people anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods and produce in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch of a shot that happens at this location: use it ONLY as spatial evidence — what this place contains, how its ground, structures and landmarks are arranged and proportioned. Ignore the sketched people and arrows entirely, and do NOT copy its line style: render a fully photographic, physically plausible real place that fits THE LOCATION text below.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nTHE LOCATION — far-future Pennsylvania, United States; all people are American and English-speaking unless stated: The exterior forest floor on a wooded ridge, where the two stand over the belt and sealed phone pouch.\n\nSCENE EVIDENCE (verbatim quotes from the screenplay about this place — treat them as evidence of what the location physically contains and looks like; stage the PLACE those moments happen in, but do NOT depict the momentary actions, people or staged props themselves):\n- S3. EXT. 숲 바닥 — 잠시 후 (01:25–02:10)\n- 윌마가 쓰러진 추격자에게서 가져온 점퍼 벨트를 토니 앞에 던진다.\n- S4. EXT. 경사진 숲 — 연속 (02:10–03:10)\n- 토니가 뛴다.\n과하게 높이 솟아 나뭇가지 사이에 거꾸로 처박힌다.\n\nTIME OF DAY (lock): day.\n\nRender ONE photorealistic empty location photograph, 16:9, neutral enough that every shot of this place can be staged from it later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/groupbg_forest_training_slope_f39462.png",
  "asset_id": "6b920ab3-07aa-46e7-8953-c58df18fb015",
  "input_asset_ids": [
   "d884add5-303c-436e-bb81-50175e27f60f"
  ],
  "origin_tag": "S3sh5",
  "place_text": "The exterior forest floor on a wooded ridge, where the two stand over the belt and sealed phone pouch.",
  "origin_inputs": {
   "place_text": "The exterior forest floor on a wooded ridge, where the two stand over the belt and sealed phone pouch.",
   "time_of_day_en": "day",
   "conti_asset_id": "d884add5-303c-436e-bb81-50175e27f60f"
  }
 },
 "S3sh5::bgfirst_bg": {
  "input_fingerprint": "fe6641200756d825",
  "prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 주머니를 향해 뻗은 토니의 손목을 윌마의 억센 손이 강하게 움켜쥔 클로즈업\n\nLOCATION (lock): The exterior forest floor on a wooded ridge, where the two stand over the belt and sealed phone pouch.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스)'s reaching arm in the middle-left of the frame, foreground, reaches for Tony's pocket; 윌마 디어링's restraining hand in the middle-center of the frame, foreground; Tony's pocket in the middle-right of the frame, midground.\n- KEY BACKGROUND ELEMENTS: Tony's pocket (still closed as his reach is intercepted) — The pocket opening is angled away from the camera, immediately beyond his halted fingertips; used as Visible destination of the stopped reach; forest floor (visible only as limited context beneath the close action); used as Keeps the confrontation grounded without distracting from the wrist grip.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is kept desaturated with tense, moderate contrast around the overlapping hands.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "effective_prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 주머니를 향해 뻗은 토니의 손목을 윌마의 억센 손이 강하게 움켜쥔 클로즈업\n\nLOCATION (lock): The exterior forest floor on a wooded ridge, where the two stand over the belt and sealed phone pouch.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스)'s reaching arm in the middle-left of the frame, foreground, reaches for Tony's pocket; 윌마 디어링's restraining hand in the middle-center of the frame, foreground; Tony's pocket in the middle-right of the frame, midground.\n- KEY BACKGROUND ELEMENTS: Tony's pocket (still closed as his reach is intercepted) — The pocket opening is angled away from the camera, immediately beyond his halted fingertips; used as Visible destination of the stopped reach; forest floor (visible only as limited context beneath the close action); used as Keeps the confrontation grounded without distracting from the wrist grip.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is kept desaturated with tense, moderate contrast around the overlapping hands.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S3sh5__bgfirst_bg.png",
  "asset_id": "16a263e8-8d61-4a21-9203-f506908b59bb",
  "input_asset_ids": [
   "d884add5-303c-436e-bb81-50175e27f60f",
   "6b920ab3-07aa-46e7-8953-c58df18fb015"
  ]
 },
 "S3sh5": {
  "input_fingerprint": "97525414d7188600",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 주머니를 향해 뻗은 토니의 손목을 윌마의 억센 손이 강하게 움켜쥔 클로즈업\n\nLOCATION (lock): The exterior forest floor on a wooded ridge, where the two stand over the belt and sealed phone pouch. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스)'s reaching arm in the middle-left of the frame, foreground, reaches for Tony's pocket; 윌마 디어링's restraining hand in the middle-center of the frame, foreground; Tony's pocket in the middle-right of the frame, midground.\n- KEY BACKGROUND ELEMENTS: Tony's pocket (still closed as his reach is intercepted) — The pocket opening is angled away from the camera, immediately beyond his halted fingertips; used as Visible destination of the stopped reach; forest floor (visible only as limited context beneath the close action); used as Keeps the confrontation grounded without distracting from the wrist grip.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is kept desaturated with tense, moderate contrast around the overlapping hands.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The captured jumper belt lies in front of Tony, with a flat black plate and layered heavy silver weights inside. The phone remains inside the metal-wrapped waterproof pouch with three percent battery reported. 윌마 디어링: She grips a wrist reaching toward the pouch. Her own weighted belt and gun remain smeared with gray moss. 토니(앤서니 로저스): His hand is extended toward the pouch but stopped at the wrist. He still carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Stage the shot. The FIRST attached image (SHOT BACKGROUND) is the finished empty background of this shot — keep it EXACTLY: its camera, perspective, architecture, lighting and every fixed feature stay untouched. The SECOND attached image (LAYOUT SKETCH) tells you ONLY where the people go: each sketched person's position, screen size, pose and the gaze/motion arrows. Ignore the sketch's background lines. The CHARACTER REFERENCE photographs show the real people.\nPlace the real people into the background at exactly the sketched positions, sizes and poses, following the arrow directions. No sketch lines or arrows may remain.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 주머니를 향해 뻗은 토니의 손목을 윌마의 억센 손이 강하게 움켜쥔 클로즈업\n\nLOCATION (lock): The exterior forest floor on a wooded ridge, where the two stand over the belt and sealed phone pouch. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스)'s reaching arm in the middle-left of the frame, foreground, reaches for Tony's pocket; 윌마 디어링's restraining hand in the middle-center of the frame, foreground; Tony's pocket in the middle-right of the frame, midground.\n- KEY BACKGROUND ELEMENTS: Tony's pocket (still closed as his reach is intercepted) — The pocket opening is angled away from the camera, immediately beyond his halted fingertips; used as Visible destination of the stopped reach; forest floor (visible only as limited context beneath the close action); used as Keeps the confrontation grounded without distracting from the wrist grip.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is kept desaturated with tense, moderate contrast around the overlapping hands.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The captured jumper belt lies in front of Tony, with a flat black plate and layered heavy silver weights inside. The phone remains inside the metal-wrapped waterproof pouch with three percent battery reported. 윌마 디어링: She grips a wrist reaching toward the pouch. Her own weighted belt and gun remain smeared with gray moss. 토니(앤서니 로저스): His hand is extended toward the pouch but stopped at the wrist. He still carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 주머니를 향해 뻗은 토니의 손목을 윌마의 억센 손이 강하게 움켜쥔 클로즈업\n\nLOCATION (lock): The exterior forest floor on a wooded ridge, where the two stand over the belt and sealed phone pouch. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스)'s reaching arm in the middle-left of the frame, foreground, reaches for Tony's pocket; 윌마 디어링's restraining hand in the middle-center of the frame, foreground; Tony's pocket in the middle-right of the frame, midground.\n- KEY BACKGROUND ELEMENTS: Tony's pocket (still closed as his reach is intercepted) — The pocket opening is angled away from the camera, immediately beyond his halted fingertips; used as Visible destination of the stopped reach; forest floor (visible only as limited context beneath the close action); used as Keeps the confrontation grounded without distracting from the wrist grip.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is kept desaturated with tense, moderate contrast around the overlapping hands.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The captured jumper belt lies in front of Tony, with a flat black plate and layered heavy silver weights inside. The phone remains inside the metal-wrapped waterproof pouch with three percent battery reported. 윌마 디어링: She grips a wrist reaching toward the pouch. Her own weighted belt and gun remain smeared with gray moss. 토니(앤서니 로저스): His hand is extended toward the pouch but stopped at the wrist. He still carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S3sh5__bgfirst_bg.png",
     "asset_id": "16a263e8-8d61-4a21-9203-f506908b59bb",
     "role": "bgfirst_bg"
    },
    {
     "label": "LAYOUT SKETCH (people placement only)",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/conti/conti_S3sh5.png",
     "asset_id": "d884add5-303c-436e-bb81-50175e27f60f",
     "role": "conti_light"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ],
   "B": [
    {
     "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/groupbg_forest_training_slope_f39462.png",
     "asset_id": "6b920ab3-07aa-46e7-8953-c58df18fb015",
     "role": "bgfirst_group_bg"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "judge_flip_skipped": "judge_fn_owns_order",
  "initial_roll_all_fail": true,
  "readings": [
   {
    "label": "A",
    "direction": "왼쪽에서 화면 우측의 주머니를 향해 팔이 수평으로 뻗어 있으며, 위쪽에서 내려온 손이 그 손목을 움켜쥐고 있다.",
    "built_space": "흙과 이끼가 덮인 숲 바닥이 배경에 보여 로케이션 사진의 환경과 일치한다.",
    "entities": "뻗은 팔(토니)은 참조와 다른 단순한 녹색 셔츠를 입고 있다. 주머니가 있는 우측 몸통 역시 갈색 바지를 입어 토니의 의상과 불일치한다. 손목을 잡은 위쪽 팔(윌마)은 맨손이다.",
    "hard_violations": [
     "[gemini-pro] duplicated or extra bodies",
     "[gemini-pro] physically impossible staging"
    ],
    "physics": "위에서 뻗어 나온 손이 왼쪽 팔의 손목을 물리적으로 단단히 붙잡고 있다."
   },
   {
    "label": "B",
    "direction": "왼쪽에서 뻗은 팔이 우측 인물 허리에 찬 파우치를 향하고 있으며, 중앙 인물의 손이 뻗은 손목을 붙잡고 있다.",
    "built_space": "숲의 나무와 바닥이 배경으로 보여 지정된 숲 지형과 잘 부합한다.",
    "entities": "왼쪽 뻗은 팔은 토니의 더러운 수트와 장갑, 밧줄을 정확히 묘사했다. 우측 몸통 역시 토니의 수트와 밧줄을 착용하고 있다. 중앙의 손(윌마)은 녹색 수트를 입고 있으나 장갑을 끼지 않은 맨손이다.",
    "hard_violations": [
     "[gemini-pro] duplicated or extra bodies",
     "[gpt] 명시된 소품으로 식별되지 않는 금속 원통 두 개가 든 열린 장비 파우치를 핵심 목적지로 발명해 넣었다."
    ],
    "physics": "중앙 인물의 손이 뻗어오는 손목을 단단히 잡고 있으며, 금속으로 감싸진 파우치가 우측 인물의 벨트에 잘 고정되어 있다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "의상과 소품의 질감은 참조와 유사하지만, 토니의 신체와 고유 장비(구조용 밧줄)가 화면 양쪽의 두 명에게 복제되어 나타나는 치명적인 위반이 있습니다."
       },
       {
        "label": "A",
        "score": 1,
        "verdict_ko": "토니가 자신의 주머니로 손을 뻗는 동작을 서로 마주 보는 두 명의 인물로 잘못 연출했으며, 의상마저 참조와 전혀 일치하지 않습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "왼쪽에서 화면 우측의 주머니를 향해 팔이 수평으로 뻗어 있으며, 위쪽에서 내려온 손이 그 손목을 움켜쥐고 있다.",
        "built_space": "흙과 이끼가 덮인 숲 바닥이 배경에 보여 로케이션 사진의 환경과 일치한다.",
        "entities": "뻗은 팔(토니)은 참조와 다른 단순한 녹색 셔츠를 입고 있다. 주머니가 있는 우측 몸통 역시 갈색 바지를 입어 토니의 의상과 불일치한다. 손목을 잡은 위쪽 팔(윌마)은 맨손이다.",
        "hard_violations": [
         "duplicated or extra bodies",
         "physically impossible staging"
        ],
        "physics": "위에서 뻗어 나온 손이 왼쪽 팔의 손목을 물리적으로 단단히 붙잡고 있다."
       },
       {
        "label": "B",
        "direction": "왼쪽에서 뻗은 팔이 우측 인물 허리에 찬 파우치를 향하고 있으며, 중앙 인물의 손이 뻗은 손목을 붙잡고 있다.",
        "built_space": "숲의 나무와 바닥이 배경으로 보여 지정된 숲 지형과 잘 부합한다.",
        "entities": "왼쪽 뻗은 팔은 토니의 더러운 수트와 장갑, 밧줄을 정확히 묘사했다. 우측 몸통 역시 토니의 수트와 밧줄을 착용하고 있다. 중앙의 손(윌마)은 녹색 수트를 입고 있으나 장갑을 끼지 않은 맨손이다.",
        "hard_violations": [
         "duplicated or extra bodies"
        ],
        "physics": "중앙 인물의 손이 뻗어오는 손목을 단단히 잡고 있으며, 금속으로 감싸진 파우치가 우측 인물의 벨트에 잘 고정되어 있다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "의상과 소품의 질감은 참조와 유사하지만, 토니의 신체와 고유 장비(구조용 밧줄)가 화면 양쪽의 두 명에게 복제되어 나타나는 치명적인 위반이 있습니다."
       },
       {
        "label": "A",
        "score": 1,
        "verdict_ko": "토니가 자신의 주머니로 손을 뻗는 동작을 서로 마주 보는 두 명의 인물로 잘못 연출했으며, 의상마저 참조와 전혀 일치하지 않습니다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "A",
        "direction": "왼쪽에서 화면 우측의 주머니를 향해 팔이 수평으로 뻗어 있으며, 위쪽에서 내려온 손이 그 손목을 움켜쥐고 있다.",
        "built_space": "흙과 이끼가 덮인 숲 바닥이 배경에 보여 로케이션 사진의 환경과 일치한다.",
        "entities": "뻗은 팔(토니)은 참조와 다른 단순한 녹색 셔츠를 입고 있다. 주머니가 있는 우측 몸통 역시 갈색 바지를 입어 토니의 의상과 불일치한다. 손목을 잡은 위쪽 팔(윌마)은 맨손이다.",
        "hard_violations": [
         "duplicated or extra bodies",
         "physically impossible staging"
        ],
        "physics": "위에서 뻗어 나온 손이 왼쪽 팔의 손목을 물리적으로 단단히 붙잡고 있다."
       },
       {
        "label": "B",
        "direction": "왼쪽에서 뻗은 팔이 우측 인물 허리에 찬 파우치를 향하고 있으며, 중앙 인물의 손이 뻗은 손목을 붙잡고 있다.",
        "built_space": "숲의 나무와 바닥이 배경으로 보여 지정된 숲 지형과 잘 부합한다.",
        "entities": "왼쪽 뻗은 팔은 토니의 더러운 수트와 장갑, 밧줄을 정확히 묘사했다. 우측 몸통 역시 토니의 수트와 밧줄을 착용하고 있다. 중앙의 손(윌마)은 녹색 수트를 입고 있으나 장갑을 끼지 않은 맨손이다.",
        "hard_violations": [
         "duplicated or extra bodies"
        ],
        "physics": "중앙 인물의 손이 뻗어오는 손목을 단단히 잡고 있으며, 금속으로 감싸진 파우치가 우측 인물의 벨트에 잘 고정되어 있다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "손목 제지와 닫힌 주머니 직전의 손끝을 정확한 클로즈업으로 보여 A보다 우수하지만, 주머니가 토니가 아니라 오른쪽 윌마의 옷에 달려 있어 핵심 소유·방향 관계는 실패한다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "손목을 강하게 붙잡는 순간은 보이지만 손끝이 닫힌 토니의 주머니가 아닌 윌마 쪽의 열린 장비 파우치와 노출된 금속 원통을 향해 핵심 목적지와 소품 상태가 크게 어긋난다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "왼쪽 토니의 장갑 낀 손은 오른쪽으로 뻗고, 윌마의 손은 중앙에서 그 손목을 아래로 눌러 붙잡는다. 멈춘 손끝이 향하는 대상은 토니의 닫힌 주머니가 아니라 오른쪽 윌마의 허리에 달린 열린 장비 파우치와 그 안의 금속 원통이다.",
        "built_space": "실외 숲 능선 바닥이며 인공 구조물이나 고정 설비는 없다. 왼쪽에 토니의 일부 몸과 팔, 중앙부터 오른쪽에 윌마의 몸과 팔이 있고, 아래에는 흙·이끼·바위가 제한적으로 보인다. 장소의 숲 재질과 주간광은 참조와 대체로 맞지만, 목적 파우치는 바닥이 아니라 윌마의 허리에 장착되어 있다.",
        "entities": "토니는 왼쪽의 오염된 어두운 전술복, 장갑, 아래쪽 구조용 로프로 식별되고 윌마는 중앙의 녹색 전술복과 맨손으로 식별된다. 얼굴은 프레임 밖이라 얼굴 정체성은 확인할 수 없다. 토니의 닫힌 주머니는 보이지 않으며, 대신 열린 파우치와 전화기나 명시된 층상 은색 추로 읽히지 않는 금속 원통들이 보인다.",
        "hard_violations": [
         "명시된 소품으로 식별되지 않는 금속 원통 두 개가 든 열린 장비 파우치를 핵심 목적지로 발명해 넣었다."
        ],
        "physics": "토니의 팔은 왼쪽 몸에 연결되어 있고 윌마의 손도 중앙 상단의 소매와 팔에 연결되어 있다. 윌마의 손가락이 토니의 손목을 실제로 감싸며 제동하는 접촉이 보여 동작은 지지된다. 파우치와 원통도 허리 장비에 고정되어 있어 떠 있는 물체는 없다."
       },
       {
        "label": "B",
        "direction": "왼쪽 토니의 맨손은 오른쪽으로 뻗고 윌마의 손이 중앙에서 손목을 단단히 감싼다. 손끝은 바로 오른쪽의 닫힌 바지 주머니 입구를 향해 시각적 조준은 매우 정확하지만, 그 주머니는 왼쪽 토니의 몸이 아니라 오른쪽 윌마의 엉덩이에 달려 있다.",
        "built_space": "인공 구조물이나 고정 설비가 없는 숲 능선 바닥이다. 왼쪽에는 로프를 찬 토니의 몸 일부와 팔, 오른쪽에는 별도 인물인 윌마의 허리와 바지 주머니, 중앙에는 손목 제지가 배치된다. 흙·이끼·나무가 좁은 배경 문맥으로만 보여 요구된 클로즈업 규모와 장소감은 잘 맞는다.",
        "entities": "왼쪽 인물은 구조용 로프 때문에 토니로 읽히지만 참조의 검고 오염된 전술복과 장갑 대신 걷어 올린 녹색 소매와 맨손이다. 오른쪽 인물은 윌마로 읽히나 참조의 녹색 비행복 대신 검은 상의와 갈색 바지를 입었다. 닫힌 주머니 자체는 명확하지만 토니의 주머니가 아니며, 전화 파우치와 포획 벨트는 프레임에 보이지 않는다.",
        "hard_violations": [],
        "physics": "두 팔은 각각 프레임 밖 몸으로 자연스럽게 이어지고, 윌마의 손가락과 엄지가 토니의 손목 둘레를 실제로 감싸 제지력을 전달한다. 토니의 손은 주머니 직전에서 멈춘 자연스러운 긴장 자세이며, 떠 있거나 지지되지 않은 신체·물체는 없다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "손목 제지와 닫힌 주머니 직전의 손끝을 정확한 클로즈업으로 보여 A보다 우수하지만, 주머니가 토니가 아니라 오른쪽 윌마의 옷에 달려 있어 핵심 소유·방향 관계는 실패한다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "손목을 강하게 붙잡는 순간은 보이지만 손끝이 닫힌 토니의 주머니가 아닌 윌마 쪽의 열린 장비 파우치와 노출된 금속 원통을 향해 핵심 목적지와 소품 상태가 크게 어긋난다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "왼쪽 토니의 장갑 낀 손은 오른쪽으로 뻗고, 윌마의 손은 중앙에서 그 손목을 아래로 눌러 붙잡는다. 멈춘 손끝이 향하는 대상은 토니의 닫힌 주머니가 아니라 오른쪽 윌마의 허리에 달린 열린 장비 파우치와 그 안의 금속 원통이다.",
        "built_space": "실외 숲 능선 바닥이며 인공 구조물이나 고정 설비는 없다. 왼쪽에 토니의 일부 몸과 팔, 중앙부터 오른쪽에 윌마의 몸과 팔이 있고, 아래에는 흙·이끼·바위가 제한적으로 보인다. 장소의 숲 재질과 주간광은 참조와 대체로 맞지만, 목적 파우치는 바닥이 아니라 윌마의 허리에 장착되어 있다.",
        "entities": "토니는 왼쪽의 오염된 어두운 전술복, 장갑, 아래쪽 구조용 로프로 식별되고 윌마는 중앙의 녹색 전술복과 맨손으로 식별된다. 얼굴은 프레임 밖이라 얼굴 정체성은 확인할 수 없다. 토니의 닫힌 주머니는 보이지 않으며, 대신 열린 파우치와 전화기나 명시된 층상 은색 추로 읽히지 않는 금속 원통들이 보인다.",
        "hard_violations": [
         "명시된 소품으로 식별되지 않는 금속 원통 두 개가 든 열린 장비 파우치를 핵심 목적지로 발명해 넣었다."
        ],
        "physics": "토니의 팔은 왼쪽 몸에 연결되어 있고 윌마의 손도 중앙 상단의 소매와 팔에 연결되어 있다. 윌마의 손가락이 토니의 손목을 실제로 감싸며 제동하는 접촉이 보여 동작은 지지된다. 파우치와 원통도 허리 장비에 고정되어 있어 떠 있는 물체는 없다."
       },
       {
        "label": "A",
        "direction": "왼쪽 토니의 맨손은 오른쪽으로 뻗고 윌마의 손이 중앙에서 손목을 단단히 감싼다. 손끝은 바로 오른쪽의 닫힌 바지 주머니 입구를 향해 시각적 조준은 매우 정확하지만, 그 주머니는 왼쪽 토니의 몸이 아니라 오른쪽 윌마의 엉덩이에 달려 있다.",
        "built_space": "인공 구조물이나 고정 설비가 없는 숲 능선 바닥이다. 왼쪽에는 로프를 찬 토니의 몸 일부와 팔, 오른쪽에는 별도 인물인 윌마의 허리와 바지 주머니, 중앙에는 손목 제지가 배치된다. 흙·이끼·나무가 좁은 배경 문맥으로만 보여 요구된 클로즈업 규모와 장소감은 잘 맞는다.",
        "entities": "왼쪽 인물은 구조용 로프 때문에 토니로 읽히지만 참조의 검고 오염된 전술복과 장갑 대신 걷어 올린 녹색 소매와 맨손이다. 오른쪽 인물은 윌마로 읽히나 참조의 녹색 비행복 대신 검은 상의와 갈색 바지를 입었다. 닫힌 주머니 자체는 명확하지만 토니의 주머니가 아니며, 전화 파우치와 포획 벨트는 프레임에 보이지 않는다.",
        "hard_violations": [],
        "physics": "두 팔은 각각 프레임 밖 몸으로 자연스럽게 이어지고, 윌마의 손가락과 엄지가 토니의 손목 둘레를 실제로 감싸 제지력을 전달한다. 토니의 손은 주머니 직전에서 멈춘 자연스러운 긴장 자세이며, 떠 있거나 지지되지 않은 신체·물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.5,
    "B": 1.5
   },
   "adjusted": {
    "A": 1.25,
    "B": 1.25
   },
   "violations": {
    "A": [
     "[gemini-pro] duplicated or extra bodies",
     "[gemini-pro] physically impossible staging"
    ],
    "B": [
     "[gemini-pro] duplicated or extra bodies",
     "[gpt] 명시된 소품으로 식별되지 않는 금속 원통 두 개가 든 열린 장비 파우치를 핵심 목적지로 발명해 넣었다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "agreed": false
  },
  "totals": {
   "B": 1250,
   "A": 1250
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 1250,
    "verdict_ko": "의상과 소품의 질감은 참조와 유사하지만, 토니의 신체와 고유 장비(구조용 밧줄)가 화면 양쪽의 두 명에게 복제되어 나타나는 치명적인 위반이 있습니다.  ★위반: [gemini-pro] duplicated or extra bodies / [gpt] 명시된 소품으로 식별되지 않는 금속 원통 두 개가 든 열린 장비 파우치를 핵심 목적지로 발명해 넣었다."
   },
   {
    "label": "A",
    "score": 1250,
    "verdict_ko": "토니가 자신의 주머니로 손을 뻗는 동작을 서로 마주 보는 두 명의 인물로 잘못 연출했으며, 의상마저 참조와 전혀 일치하지 않습니다.  ★위반: [gemini-pro] duplicated or extra bodies / [gemini-pro] physically impossible staging"
   }
  ],
  "refs": [
   {
    "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/groupbg_forest_training_slope_f39462.png",
    "asset_id": "6b920ab3-07aa-46e7-8953-c58df18fb015",
    "role": "bgfirst_group_bg"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "needs_reshoot": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc50-c9fc-7829-ad08-7b6e05fb8e64",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S3sh5__bgfirst_bg.png",
   "bg_asset_id": "16a263e8-8d61-4a21-9203-f506908b59bb",
   "bg_record_key": "S3sh5::bgfirst_bg",
   "chain_winner": true,
   "authority": "groupbg",
   "group_key": "forest_training_slope",
   "groupbg_asset_id": "6b920ab3-07aa-46e7-8953-c58df18fb015"
  },
  "ref_mode": "재투영 배경+콘티+엔티티 (2택1: 체인 승)",
  "share_plan": {
   "ref_plan": "background"
  }
 },
 "S3sh5::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:39:19.863117+00:00",
  "fingerprint": "b637e9f11813ebf872a30ebf7e35f369f0698ea57d68c82792df6f0c22d1eff4",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S3sh5_sel.png",
  "source_sha256": "84bc73d5f0d1d6c5123791a2dbfa93c01d69eddf8d47e501ca8a0b5ccb82dcbe",
  "file": "S3sh5_cine.png",
  "staged_sha256": "c2df280890c1ca63a9d7a66092d0c18c23394fc5cc980f278f6dc1cb4e37dda5",
  "latency_ms": 13276
 },
 "S3sh8::signage": {
  "fp": "ed61023ae7fba2a9",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S3sh8": {
  "input_fingerprint": "7dae401302d6e65c",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 윌마의 뻗은 검지손가락이 점퍼 벨트 안쪽의 은색 추 표면을 정확히 가리킨 찰나\n\nLOCATION (lock): The exterior forest floor on the wooded ridge, immediately around the laid-out jumper belt. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: 윌마 디어링's extended hand in the middle-left of the frame, foreground, points to silver weight inside the belt; silver weight inside the belt in the middle-right of the frame, midground; flat black plate in the lower-left of the frame, midground.\n- KEY BACKGROUND ELEMENTS: jumper belt interior (open with the flat black plate and silver weights exposed) — The inner face is turned toward the camera, displaying the black plate and overlapping silver weights; used as Forms the bounded instructional field beneath Wilma's entering hand; flat black plate (visible inside the belt) — Its broad face is exposed toward the camera beside the indicated weight; used as Secondary comparison element retained beside the finger's final target; silver weight (exposed and precisely indicated by Wilma's finger) — Its upper surface faces the oblique camera beneath the stopped fingertip; used as Primary instructional target at the end of the crane move.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is rendered in a restrained, desaturated palette with clear tonal separation between the black plate, silver weight, and hand.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 윌마 디어링 — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The captured jumper belt remains open before Tony, exposing its flat black plate and layered heavy silver weights. The phone remains enclosed in the waterproof pouch. 윌마 디어링: She points directly at the belt's weights; her own belt and gun remain coated with gray moss.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 윌마의 뻗은 검지손가락이 점퍼 벨트 안쪽의 은색 추 표면을 정확히 가리킨 찰나\n\nLOCATION (lock): The exterior forest floor on the wooded ridge, immediately around the laid-out jumper belt. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: 윌마 디어링's extended hand in the middle-left of the frame, foreground, points to silver weight inside the belt; silver weight inside the belt in the middle-right of the frame, midground; flat black plate in the lower-left of the frame, midground.\n- KEY BACKGROUND ELEMENTS: jumper belt interior (open with the flat black plate and silver weights exposed) — The inner face is turned toward the camera, displaying the black plate and overlapping silver weights; used as Forms the bounded instructional field beneath Wilma's entering hand; flat black plate (visible inside the belt) — Its broad face is exposed toward the camera beside the indicated weight; used as Secondary comparison element retained beside the finger's final target; silver weight (exposed and precisely indicated by Wilma's finger) — Its upper surface faces the oblique camera beneath the stopped fingertip; used as Primary instructional target at the end of the crane move.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is rendered in a restrained, desaturated palette with clear tonal separation between the black plate, silver weight, and hand.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 윌마 디어링 — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The captured jumper belt remains open before Tony, exposing its flat black plate and layered heavy silver weights. The phone remains enclosed in the waterproof pouch. 윌마 디어링: She points directly at the belt's weights; her own belt and gun remain coated with gray moss.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 윌마의 뻗은 검지손가락이 점퍼 벨트 안쪽의 은색 추 표면을 정확히 가리킨 찰나\n\nLOCATION (lock): The exterior forest floor on the wooded ridge, immediately around the laid-out jumper belt. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: 윌마 디어링's extended hand in the middle-left of the frame, foreground, points to silver weight inside the belt; silver weight inside the belt in the middle-right of the frame, midground; flat black plate in the lower-left of the frame, midground.\n- KEY BACKGROUND ELEMENTS: jumper belt interior (open with the flat black plate and silver weights exposed) — The inner face is turned toward the camera, displaying the black plate and overlapping silver weights; used as Forms the bounded instructional field beneath Wilma's entering hand; flat black plate (visible inside the belt) — Its broad face is exposed toward the camera beside the indicated weight; used as Secondary comparison element retained beside the finger's final target; silver weight (exposed and precisely indicated by Wilma's finger) — Its upper surface faces the oblique camera beneath the stopped fingertip; used as Primary instructional target at the end of the crane move.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is rendered in a restrained, desaturated palette with clear tonal separation between the black plate, silver weight, and hand.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 윌마 디어링 — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The captured jumper belt remains open before Tony, exposing its flat black plate and layered heavy silver weights. The phone remains enclosed in the waterproof pouch. 윌마 디어링: She points directly at the belt's weights; her own belt and gun remain coated with gray moss.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "뻗은 검지손가락이 벨트 안쪽의 은색 추 표면을 정확히 가리키고 있음.",
    "built_space": "숲 바닥 위에 벨트가 펼쳐져 있고, 왼쪽 아래에 평평한 검은색 판, 오른쪽에 은색 추들이 배치됨.",
    "entities": "이전 샷과 일치하는 깨끗한 녹색 소매의 팔, 평평한 검은색 판, 표면이 매끄러운 은색 추들.",
    "hard_violations": [],
    "physics": "손과 팔이 공중에 자연스럽게 지탱되어 있으며, 벨트와 내용물은 지면에 안정적으로 놓여 있음."
   },
   {
    "label": "B",
    "direction": "뻗은 검지손가락이 가장 위에 있는 은색 추를 향하고 있음.",
    "built_space": "숲 바닥 위에 펼쳐진 벨트 안에 두꺼운 검은색 판과 은색 추들이 놓여 있음.",
    "entities": "회색 이끼가 묻어 있는 녹색 소매(이전 샷과 불일치), 두꺼운 블록 형태의 검은색 판, 중앙에 구멍이 뚫린 은색 추.",
    "hard_violations": [],
    "physics": "팔이 자연스럽게 뻗어 손을 지탱하고 있고, 벨트와 부품들은 지면에 고정됨."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "프레이밍과 구도가 프롬프트의 지시를 완벽히 따랐으며, 이전 샷의 녹색 소매 디테일과 질감을 정확히 유지함."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "구도와 동작은 전반적으로 일치하나, 은색 추가 바벨 원판처럼 구멍이 뚫려 있고 이전 샷에 없던 이끼가 소매에 추가됨."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "뻗은 검지손가락이 벨트 안쪽의 은색 추 표면을 정확히 가리키고 있음.",
        "built_space": "숲 바닥 위에 벨트가 펼쳐져 있고, 왼쪽 아래에 평평한 검은색 판, 오른쪽에 은색 추들이 배치됨.",
        "entities": "이전 샷과 일치하는 깨끗한 녹색 소매의 팔, 평평한 검은색 판, 표면이 매끄러운 은색 추들.",
        "hard_violations": [],
        "physics": "손과 팔이 공중에 자연스럽게 지탱되어 있으며, 벨트와 내용물은 지면에 안정적으로 놓여 있음."
       },
       {
        "label": "B",
        "direction": "뻗은 검지손가락이 가장 위에 있는 은색 추를 향하고 있음.",
        "built_space": "숲 바닥 위에 펼쳐진 벨트 안에 두꺼운 검은색 판과 은색 추들이 놓여 있음.",
        "entities": "회색 이끼가 묻어 있는 녹색 소매(이전 샷과 불일치), 두꺼운 블록 형태의 검은색 판, 중앙에 구멍이 뚫린 은색 추.",
        "hard_violations": [],
        "physics": "팔이 자연스럽게 뻗어 손을 지탱하고 있고, 벨트와 부품들은 지면에 고정됨."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "프레이밍과 구도가 프롬프트의 지시를 완벽히 따랐으며, 이전 샷의 녹색 소매 디테일과 질감을 정확히 유지함."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "구도와 동작은 전반적으로 일치하나, 은색 추가 바벨 원판처럼 구멍이 뚫려 있고 이전 샷에 없던 이끼가 소매에 추가됨."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "뻗은 검지손가락이 벨트 안쪽의 은색 추 표면을 정확히 가리키고 있음.",
        "built_space": "숲 바닥 위에 벨트가 펼쳐져 있고, 왼쪽 아래에 평평한 검은색 판, 오른쪽에 은색 추들이 배치됨.",
        "entities": "이전 샷과 일치하는 깨끗한 녹색 소매의 팔, 평평한 검은색 판, 표면이 매끄러운 은색 추들.",
        "hard_violations": [],
        "physics": "손과 팔이 공중에 자연스럽게 지탱되어 있으며, 벨트와 내용물은 지면에 안정적으로 놓여 있음."
       },
       {
        "label": "B",
        "direction": "뻗은 검지손가락이 가장 위에 있는 은색 추를 향하고 있음.",
        "built_space": "숲 바닥 위에 펼쳐진 벨트 안에 두꺼운 검은색 판과 은색 추들이 놓여 있음.",
        "entities": "회색 이끼가 묻어 있는 녹색 소매(이전 샷과 불일치), 두꺼운 블록 형태의 검은색 판, 중앙에 구멍이 뚫린 은색 추.",
        "hard_violations": [],
        "physics": "팔이 자연스럽게 뻗어 손을 지탱하고 있고, 벨트와 부품들은 지면에 고정됨."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "손·은색 추·검은 판의 화면 배치가 지시와 정확히 맞고, 검지가 중간 오른쪽의 은색 추 표면을 직접 짚는 인서트 클로즈업을 가장 충실히 구현했다."
       },
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "검지가 은색 추를 정확히 가리키고 필수 요소도 모두 보이지만, 원형 구멍이 난 추와 다소 넓은 구도가 B보다 점퍼 벨트의 겹친 추 및 세부 인서트 지시에 덜 충실하다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "왼쪽 위에서 들어온 윌마의 손이 오른쪽 아래로 뻗으며, 검지 끝은 중간 오른쪽에 겹쳐 놓인 은색 추의 윗면에 정확히 닿아 있다.",
        "built_space": "숲 바닥 위에 열린 점퍼 벨트 하나가 놓여 있고, 카메라를 향한 내부에는 아래쪽 왼편의 넓은 검은 판 하나와 오른편의 겹친 은색 추 여러 개가 보인다. 손은 왼쪽 전경에서 벨트 내부로 들어오며, 반사면에는 불가능한 인물상이나 중복 구조가 없다.",
        "entities": "보이는 인체는 윌마의 손과 녹색 소매뿐이며 젊은 성인 여성의 손과 잠긴 녹색 복장에 부합한다. 열린 점퍼 벨트, 검은 판, 여러 은색 금속 추, 이끼 낀 숲 바닥이 모두 식별된다. 다만 추는 중앙 구멍이 있는 일반 원판형 운동용 추처럼 보여 B보다 미래형 점퍼 벨트 부품으로서의 형태가 덜 설득력 있다.",
        "hard_violations": [],
        "physics": "손과 검지는 프레임 밖 팔에 자연스럽게 이어지고 손가락 끝이 추 표면에 접촉한다. 은색 추들은 벨트 내부 바닥과 서로의 표면에 받쳐져 있으며, 검은 판과 벨트도 숲 바닥에 안정적으로 놓여 있다. 지지 없이 뜬 물체는 없다."
       },
       {
        "label": "B",
        "direction": "중간 왼쪽에서 들어온 윌마의 검지가 오른쪽 아래 방향으로 곧게 뻗어, 중간 오른쪽에 노출된 은색 추의 윗면을 정확히 짚는다.",
        "built_space": "숲 바닥에 열린 점퍼 벨트 하나가 놓여 있으며 내측 면이 카메라를 향한다. 내부에는 아래쪽 왼편의 넓고 평평한 검은 판 하나, 중간 오른쪽의 지목된 은색 추 하나, 그 뒤로 겹치고 띠에 고정된 은색 추 여러 개가 보인다. 손은 왼쪽 전경에서 이 한정된 내부 설명 영역으로 진입하고, 광택 금속의 반사는 주변 숲과 손에 맞게 자연스럽다.",
        "entities": "유일하게 보이는 인체는 윌마의 손과 녹색 소매이며, 젊은 성인 여성의 손과 캐릭터 참고 이미지의 녹색 복장에 부합한다. 열린 점퍼 벨트, 평평한 검은 판, 겹친 은색 금속 추, 숲 바닥이 명확하며 읽을 수 있는 글자나 불필요한 인물은 없다.",
        "hard_violations": [],
        "physics": "손은 프레임 밖 팔에 연결되어 있고 검지 끝은 은색 추 표면에 실제로 접촉한다. 지목된 추는 벨트 내부 바닥에 놓였고 뒤쪽 추들은 내부 받침과 고정 띠에 의해 지지된다. 검은 판과 열린 벨트는 숲 바닥에 놓여 있어 모든 물체의 지지가 물리적으로 타당하다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "손·은색 추·검은 판의 화면 배치가 지시와 정확히 맞고, 검지가 중간 오른쪽의 은색 추 표면을 직접 짚는 인서트 클로즈업을 가장 충실히 구현했다."
       },
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "검지가 은색 추를 정확히 가리키고 필수 요소도 모두 보이지만, 원형 구멍이 난 추와 다소 넓은 구도가 B보다 점퍼 벨트의 겹친 추 및 세부 인서트 지시에 덜 충실하다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "왼쪽 위에서 들어온 윌마의 손이 오른쪽 아래로 뻗으며, 검지 끝은 중간 오른쪽에 겹쳐 놓인 은색 추의 윗면에 정확히 닿아 있다.",
        "built_space": "숲 바닥 위에 열린 점퍼 벨트 하나가 놓여 있고, 카메라를 향한 내부에는 아래쪽 왼편의 넓은 검은 판 하나와 오른편의 겹친 은색 추 여러 개가 보인다. 손은 왼쪽 전경에서 벨트 내부로 들어오며, 반사면에는 불가능한 인물상이나 중복 구조가 없다.",
        "entities": "보이는 인체는 윌마의 손과 녹색 소매뿐이며 젊은 성인 여성의 손과 잠긴 녹색 복장에 부합한다. 열린 점퍼 벨트, 검은 판, 여러 은색 금속 추, 이끼 낀 숲 바닥이 모두 식별된다. 다만 추는 중앙 구멍이 있는 일반 원판형 운동용 추처럼 보여 B보다 미래형 점퍼 벨트 부품으로서의 형태가 덜 설득력 있다.",
        "hard_violations": [],
        "physics": "손과 검지는 프레임 밖 팔에 자연스럽게 이어지고 손가락 끝이 추 표면에 접촉한다. 은색 추들은 벨트 내부 바닥과 서로의 표면에 받쳐져 있으며, 검은 판과 벨트도 숲 바닥에 안정적으로 놓여 있다. 지지 없이 뜬 물체는 없다."
       },
       {
        "label": "A",
        "direction": "중간 왼쪽에서 들어온 윌마의 검지가 오른쪽 아래 방향으로 곧게 뻗어, 중간 오른쪽에 노출된 은색 추의 윗면을 정확히 짚는다.",
        "built_space": "숲 바닥에 열린 점퍼 벨트 하나가 놓여 있으며 내측 면이 카메라를 향한다. 내부에는 아래쪽 왼편의 넓고 평평한 검은 판 하나, 중간 오른쪽의 지목된 은색 추 하나, 그 뒤로 겹치고 띠에 고정된 은색 추 여러 개가 보인다. 손은 왼쪽 전경에서 이 한정된 내부 설명 영역으로 진입하고, 광택 금속의 반사는 주변 숲과 손에 맞게 자연스럽다.",
        "entities": "유일하게 보이는 인체는 윌마의 손과 녹색 소매이며, 젊은 성인 여성의 손과 캐릭터 참고 이미지의 녹색 복장에 부합한다. 열린 점퍼 벨트, 평평한 검은 판, 겹친 은색 금속 추, 숲 바닥이 명확하며 읽을 수 있는 글자나 불필요한 인물은 없다.",
        "hard_violations": [],
        "physics": "손은 프레임 밖 팔에 연결되어 있고 검지 끝은 은색 추 표면에 실제로 접촉한다. 지목된 추는 벨트 내부 바닥에 놓였고 뒤쪽 추들은 내부 받침과 고정 띠에 의해 지지된다. 검은 판과 열린 벨트는 숲 바닥에 놓여 있어 모든 물체의 지지가 물리적으로 타당하다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.589
   },
   "adjusted": {
    "A": 2.0,
    "B": 1.589
   },
   "violations": {},
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 1589
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "프레이밍과 구도가 프롬프트의 지시를 완벽히 따랐으며, 이전 샷의 녹색 소매 디테일과 질감을 정확히 유지함."
   },
   {
    "label": "B",
    "score": 1589,
    "verdict_ko": "구도와 동작은 전반적으로 일치하나, 은색 추가 바벨 원판처럼 구멍이 뚫려 있고 이전 샷에 없던 이끼가 소매에 추가됨."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The clothing, hair and overall look of 윌마 디어링 — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S3sh5_sel.png",
    "asset_id": "6c9e368a-1dce-4c4d-abb7-6632294d6ab2",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc73-542d-742c-95c7-277d377f6031",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S3sh5"
  }
 },
 "S3sh8::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:40:35.080767+00:00",
  "fingerprint": "48f18901822680564f330216b14ef1397a50d6c393f600bedec7b45f0d09b437",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S3sh8_sel.png",
  "source_sha256": "9f7a964b67bcea7f6a15eab65ca6cb71c52ada3ac00ed8db83ff4554d41dfe8f",
  "file": "S3sh8_cine.png",
  "staged_sha256": "75df2cd201904d06acb85806179adc4bb2b88d6df79ddda977f4464f115f32ca",
  "latency_ms": 12293
 },
 "S4sh5::signage": {
  "fp": "e22ba9f76c221b0d",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S4sh5": {
  "input_fingerprint": "96c48c9f65d4ee1c",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 빽빽한 나뭇가지들 사이에 하반신이 엉켜 거꾸로 처박힌 토니의 옴짝달싹 못하는 모습\n\nLOCATION (lock): The dense exterior branches above a sloping forest floor, where the jumper is wedged upside down. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스), inverted and trapped in the middle-center of the frame, midground; branches trapping Tony's lower body in the upper-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: dense branches (holding Tony's lower body immobilized) — They cross at varied angles around his legs and waist, viewed from beside his inverted torso; used as Physical trap and layered frame around the suspended body; forested slope (visible below the hanging torso) — The slope recedes diagonally beneath Tony; used as Establishes height and the direction of the coming descent.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the sloped forest is kept desaturated with moderate contrast separating Tony's inverted body from the surrounding branches.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nIMMOBILE CHARACTER POSE — CANONICAL (identical wherever this character appears in ANY panel; on any conflict THIS POSE WINS): Tony is wedged upside down among dense tree branches, with his lower body tightly entangled and immobilized while his torso and head hang downward beneath the snag. His exact arm positions are not established.\n\nIMMOBILE BODIES OBEY GRAVITY: a person who is dead or unconscious\nexerts NO muscular effort. Every part of their body — head, torso,\narms, hands, fingers, legs — rests fully on whatever supports it\n(floor, wall, furniture, their own lap) and hangs or slumps with\ngravity. NEVER show any part of an immobile person's body lifted,\nraised, held up in the air, or posed as if presenting something:\nan object in their grip stays clenched in a hand that itself lies\nfallen on a support — the hand does not hold the object up. If the\ncanonical pose leaves a body part unspecified, resolve it as the\nmost gravity-compliant, fully-supported position.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Dense branches surround the elevated point where the failed jump ended. The slope and moss bed below remain undisturbed until the later attempts. 토니(앤서니 로저스): Tony is wedged upside down among dense tree branches, with his lower body tightly entangled and immobilized while his torso and head hang downward beneath the snag, wearing the captured jumper belt with its black plate and silver weights. He still carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 빽빽한 나뭇가지들 사이에 하반신이 엉켜 거꾸로 처박힌 토니의 옴짝달싹 못하는 모습\n\nLOCATION (lock): The dense exterior branches above a sloping forest floor, where the jumper is wedged upside down. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스), inverted and trapped in the middle-center of the frame, midground; branches trapping Tony's lower body in the upper-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: dense branches (holding Tony's lower body immobilized) — They cross at varied angles around his legs and waist, viewed from beside his inverted torso; used as Physical trap and layered frame around the suspended body; forested slope (visible below the hanging torso) — The slope recedes diagonally beneath Tony; used as Establishes height and the direction of the coming descent.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the sloped forest is kept desaturated with moderate contrast separating Tony's inverted body from the surrounding branches.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nIMMOBILE CHARACTER POSE — CANONICAL (identical wherever this character appears in ANY panel; on any conflict THIS POSE WINS): Tony is wedged upside down among dense tree branches, with his lower body tightly entangled and immobilized while his torso and head hang downward beneath the snag. His exact arm positions are not established.\n\nIMMOBILE BODIES OBEY GRAVITY: a person who is dead or unconscious\nexerts NO muscular effort. Every part of their body — head, torso,\narms, hands, fingers, legs — rests fully on whatever supports it\n(floor, wall, furniture, their own lap) and hangs or slumps with\ngravity. NEVER show any part of an immobile person's body lifted,\nraised, held up in the air, or posed as if presenting something:\nan object in their grip stays clenched in a hand that itself lies\nfallen on a support — the hand does not hold the object up. If the\ncanonical pose leaves a body part unspecified, resolve it as the\nmost gravity-compliant, fully-supported position.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Dense branches surround the elevated point where the failed jump ended. The slope and moss bed below remain undisturbed until the later attempts. 토니(앤서니 로저스): Tony is wedged upside down among dense tree branches, with his lower body tightly entangled and immobilized while his torso and head hang downward beneath the snag, wearing the captured jumper belt with its black plate and silver weights. He still carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 빽빽한 나뭇가지들 사이에 하반신이 엉켜 거꾸로 처박힌 토니의 옴짝달싹 못하는 모습\n\nLOCATION (lock): The dense exterior branches above a sloping forest floor, where the jumper is wedged upside down. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스), inverted and trapped in the middle-center of the frame, midground; branches trapping Tony's lower body in the upper-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: dense branches (holding Tony's lower body immobilized) — They cross at varied angles around his legs and waist, viewed from beside his inverted torso; used as Physical trap and layered frame around the suspended body; forested slope (visible below the hanging torso) — The slope recedes diagonally beneath Tony; used as Establishes height and the direction of the coming descent.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the sloped forest is kept desaturated with moderate contrast separating Tony's inverted body from the surrounding branches.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nIMMOBILE CHARACTER POSE — CANONICAL (identical wherever this character appears in ANY panel; on any conflict THIS POSE WINS): Tony is wedged upside down among dense tree branches, with his lower body tightly entangled and immobilized while his torso and head hang downward beneath the snag. His exact arm positions are not established.\n\nIMMOBILE BODIES OBEY GRAVITY: a person who is dead or unconscious\nexerts NO muscular effort. Every part of their body — head, torso,\narms, hands, fingers, legs — rests fully on whatever supports it\n(floor, wall, furniture, their own lap) and hangs or slumps with\ngravity. NEVER show any part of an immobile person's body lifted,\nraised, held up in the air, or posed as if presenting something:\nan object in their grip stays clenched in a hand that itself lies\nfallen on a support — the hand does not hold the object up. If the\ncanonical pose leaves a body part unspecified, resolve it as the\nmost gravity-compliant, fully-supported position.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Dense branches surround the elevated point where the failed jump ended. The slope and moss bed below remain undisturbed until the later attempts. 토니(앤서니 로저스): Tony is wedged upside down among dense tree branches, with his lower body tightly entangled and immobilized while his torso and head hang downward beneath the snag, wearing the captured jumper belt with its black plate and silver weights. He still carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "거꾸로 매달린 토니의 시선이 카메라와 아래쪽 경사면을 향하고 있음.",
    "built_space": "이끼가 낀 경사진 숲 바닥이 대각선으로 펼쳐져 있으며, 화면 상단과 중간에 토니의 하반신을 가두는 나뭇가지들이 빽빽하게 교차하고 있음.",
    "entities": "토니의 얼굴, 헬멧, 전술 장비, 구출용 로프는 참조 이미지와 일치하지만, 프롬프트에 명시된 은색 무게추가 달린 점퍼 벨트를 착용하고 있지 않음.",
    "hard_violations": [],
    "physics": "다리가 나뭇가지들 사이에 단단히 끼워져 몸을 지탱하고 있으며, 몸통과 두 팔은 중력의 영향을 받아 자연스럽게 아래로 늘어져 있음."
   },
   {
    "label": "B",
    "direction": "거꾸로 매달린 토니가 렌즈 방향(아래쪽)을 응시하고 있음.",
    "built_space": "이끼로 덮인 숲의 경사면이 뒤로 물러나듯 묘사되어 높이감을 주며, 두꺼운 나뭇가지들이 토니의 다리를 단단히 감싸고 있음.",
    "entities": "토니의 신체, 얼굴, 의상은 참조와 일치하며, 특히 지시문과 이전 샷에 명시되었던 '은색 추가 달린 검은색 플레이트 형태의 점퍼 벨트'를 허리에 정확히 착용하고 있음.",
    "hard_violations": [],
    "physics": "다리가 굵은 나뭇가지 위로 걸쳐져 체중을 지탱하고 있고, 팔과 여분의 로프는 아무런 힘이 들어가지 않은 채 중력에 의해 똑바로 늘어져 있어 물리법칙에 매우 부합함."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "이전 샷 래퍼런스에서 요구된 은색 추가 달린 검은색 점퍼 벨트를 허리에 착용한 상태를 완벽하게 반영하였으며, 중력에 순응하여 매달린 포즈와 숲의 경사면 묘사가 매우 충실합니다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "지시된 구도와 나뭇가지에 얽혀 거꾸로 매달린 자세는 훌륭하게 구현했으나, 프롬프트에서 필수적으로 요구한 '은색 추가 달린 점퍼 벨트'가 완전히 누락되었습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "거꾸로 매달린 토니의 시선이 카메라와 아래쪽 경사면을 향하고 있음.",
        "built_space": "이끼가 낀 경사진 숲 바닥이 대각선으로 펼쳐져 있으며, 화면 상단과 중간에 토니의 하반신을 가두는 나뭇가지들이 빽빽하게 교차하고 있음.",
        "entities": "토니의 얼굴, 헬멧, 전술 장비, 구출용 로프는 참조 이미지와 일치하지만, 프롬프트에 명시된 은색 무게추가 달린 점퍼 벨트를 착용하고 있지 않음.",
        "hard_violations": [],
        "physics": "다리가 나뭇가지들 사이에 단단히 끼워져 몸을 지탱하고 있으며, 몸통과 두 팔은 중력의 영향을 받아 자연스럽게 아래로 늘어져 있음."
       },
       {
        "label": "B",
        "direction": "거꾸로 매달린 토니가 렌즈 방향(아래쪽)을 응시하고 있음.",
        "built_space": "이끼로 덮인 숲의 경사면이 뒤로 물러나듯 묘사되어 높이감을 주며, 두꺼운 나뭇가지들이 토니의 다리를 단단히 감싸고 있음.",
        "entities": "토니의 신체, 얼굴, 의상은 참조와 일치하며, 특히 지시문과 이전 샷에 명시되었던 '은색 추가 달린 검은색 플레이트 형태의 점퍼 벨트'를 허리에 정확히 착용하고 있음.",
        "hard_violations": [],
        "physics": "다리가 굵은 나뭇가지 위로 걸쳐져 체중을 지탱하고 있고, 팔과 여분의 로프는 아무런 힘이 들어가지 않은 채 중력에 의해 똑바로 늘어져 있어 물리법칙에 매우 부합함."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "이전 샷 래퍼런스에서 요구된 은색 추가 달린 검은색 점퍼 벨트를 허리에 착용한 상태를 완벽하게 반영하였으며, 중력에 순응하여 매달린 포즈와 숲의 경사면 묘사가 매우 충실합니다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "지시된 구도와 나뭇가지에 얽혀 거꾸로 매달린 자세는 훌륭하게 구현했으나, 프롬프트에서 필수적으로 요구한 '은색 추가 달린 점퍼 벨트'가 완전히 누락되었습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "거꾸로 매달린 토니의 시선이 카메라와 아래쪽 경사면을 향하고 있음.",
        "built_space": "이끼가 낀 경사진 숲 바닥이 대각선으로 펼쳐져 있으며, 화면 상단과 중간에 토니의 하반신을 가두는 나뭇가지들이 빽빽하게 교차하고 있음.",
        "entities": "토니의 얼굴, 헬멧, 전술 장비, 구출용 로프는 참조 이미지와 일치하지만, 프롬프트에 명시된 은색 무게추가 달린 점퍼 벨트를 착용하고 있지 않음.",
        "hard_violations": [],
        "physics": "다리가 나뭇가지들 사이에 단단히 끼워져 몸을 지탱하고 있으며, 몸통과 두 팔은 중력의 영향을 받아 자연스럽게 아래로 늘어져 있음."
       },
       {
        "label": "B",
        "direction": "거꾸로 매달린 토니가 렌즈 방향(아래쪽)을 응시하고 있음.",
        "built_space": "이끼로 덮인 숲의 경사면이 뒤로 물러나듯 묘사되어 높이감을 주며, 두꺼운 나뭇가지들이 토니의 다리를 단단히 감싸고 있음.",
        "entities": "토니의 신체, 얼굴, 의상은 참조와 일치하며, 특히 지시문과 이전 샷에 명시되었던 '은색 추가 달린 검은색 플레이트 형태의 점퍼 벨트'를 허리에 정확히 착용하고 있음.",
        "hard_violations": [],
        "physics": "다리가 굵은 나뭇가지 위로 걸쳐져 체중을 지탱하고 있고, 팔과 여분의 로프는 아무런 힘이 들어가지 않은 채 중력에 의해 똑바로 늘어져 있어 물리법칙에 매우 부합함."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "넓은 숲 경사면에서 토니를 중앙에 거꾸로 배치하고, 상단의 가지들이 하반신을 실제로 지지·고정하며 검은 벨트와 은색 웨이트 및 구조 로프까지 충실히 보여 준다."
       },
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "거꾸로 매달린 위치와 가지의 물리적 지지는 정확하지만, 지정된 검은 플레이트와 은색 웨이트 벨트가 뚜렷하지 않아 A보다 소품 연속성이 떨어진다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니는 머리를 아래로 향한 채 프레임 중앙에 거꾸로 매달려 있으며, 얼굴과 시선은 대체로 카메라 쪽을 향한다. 무기나 다른 지향성 물체, 이동 중인 신체는 없다. 몸의 수직 방향은 상단의 붙잡힌 다리에서 하단의 머리와 팔로 이어져 경사면 아래쪽을 명확히 가리킨다.",
        "built_space": "인공 구조물이나 고정 설비가 없는 숲 외부다. 굵고 가는 가지 여러 개가 화면 상단과 중앙에서 다양한 각도로 교차하고, 토니의 허벅지·무릎·허리 주변을 둘러싼다. 토니 아래에는 오른쪽 위에서 왼쪽 아래로 물러나는 이끼 낀 급경사 숲바닥이 보이며, 요구된 높이와 하강 방향이 성립한다.",
        "entities": "등장 인물은 한 명뿐이며, 30대 중반의 미국인 남성 토니로 읽힌다. 헬멧, 어두운 전술복, 장갑과 하네스가 캐릭터 참고 이미지에 부합한다. 허리에는 검은 판과 여러 개의 은색 원형 웨이트가 선명하게 보이고, 갈색 구조 로프도 벨트 옆에 매달려 있다. 추가 인물이나 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "토니의 하반신은 상단의 굵은 교차 가지들 위와 사이에 걸려 있으며, 굽힌 양다리와 허벅지·골반 부근의 접촉이 체중을 지지한다. 몸통과 머리 및 양팔은 그 지점 아래로 중력에 따라 늘어져 있고 손에도 불필요한 근력이 보이지 않는다. 로프는 허리 장비에 연결되거나 걸린 상태로 아래로 늘어져 있어 떠 있지 않는다."
       },
       {
        "label": "B",
        "direction": "토니는 중앙에서 머리를 아래로 향한 채 거꾸로 매달려 있다. 얼굴은 아래쪽과 약간 카메라 쪽을 향하며 눈은 감겼거나 아래를 본다. 무기나 겨냥하는 물체 및 이동 중인 신체는 없고, 몸의 방향은 붙잡힌 다리에서 경사면 아래의 머리와 손으로 이어진다.",
        "built_space": "인공 구조물이나 고정 설비가 없는 숲 외부다. 다수의 굵은 가지가 상단과 중앙을 사선으로 가로지르며 토니의 무릎과 허벅지 양쪽을 끼워 고정한다. 아래에는 오른쪽 위에서 왼쪽 아래로 후퇴하는 이끼 낀 숲 경사면이 넓게 보여 장소와 높이를 잘 설명한다.",
        "entities": "등장 인물은 토니 한 명이며 30대 남성, 헬멧, 어두운 작업복, 장갑, 하네스가 참고 인물과 대체로 맞는다. 갈색 구조 로프 묶음은 오른쪽 허리에 분명히 달려 있다. 다만 지정된 검은 플레이트와 은색 웨이트가 달린 점퍼 벨트는 보이지 않고 일반 하네스와 파우치 위주로 표현됐다. 추가 인물이나 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "양쪽 허벅지와 굽힌 무릎이 여러 굵은 가지의 교차부 위와 사이에 걸려 몸을 확실히 지지한다. 부츠는 위쪽으로 나온 채 가지 뒤에 놓이고, 몸통·머리·팔은 아래로 자연스럽게 늘어진다. 구조 로프는 허리 하네스에 고정된 묶음으로 지지되어 있으며 공중에 뜬 물체는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "넓은 숲 경사면에서 토니를 중앙에 거꾸로 배치하고, 상단의 가지들이 하반신을 실제로 지지·고정하며 검은 벨트와 은색 웨이트 및 구조 로프까지 충실히 보여 준다."
       },
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "거꾸로 매달린 위치와 가지의 물리적 지지는 정확하지만, 지정된 검은 플레이트와 은색 웨이트 벨트가 뚜렷하지 않아 A보다 소품 연속성이 떨어진다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "토니는 머리를 아래로 향한 채 프레임 중앙에 거꾸로 매달려 있으며, 얼굴과 시선은 대체로 카메라 쪽을 향한다. 무기나 다른 지향성 물체, 이동 중인 신체는 없다. 몸의 수직 방향은 상단의 붙잡힌 다리에서 하단의 머리와 팔로 이어져 경사면 아래쪽을 명확히 가리킨다.",
        "built_space": "인공 구조물이나 고정 설비가 없는 숲 외부다. 굵고 가는 가지 여러 개가 화면 상단과 중앙에서 다양한 각도로 교차하고, 토니의 허벅지·무릎·허리 주변을 둘러싼다. 토니 아래에는 오른쪽 위에서 왼쪽 아래로 물러나는 이끼 낀 급경사 숲바닥이 보이며, 요구된 높이와 하강 방향이 성립한다.",
        "entities": "등장 인물은 한 명뿐이며, 30대 중반의 미국인 남성 토니로 읽힌다. 헬멧, 어두운 전술복, 장갑과 하네스가 캐릭터 참고 이미지에 부합한다. 허리에는 검은 판과 여러 개의 은색 원형 웨이트가 선명하게 보이고, 갈색 구조 로프도 벨트 옆에 매달려 있다. 추가 인물이나 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "토니의 하반신은 상단의 굵은 교차 가지들 위와 사이에 걸려 있으며, 굽힌 양다리와 허벅지·골반 부근의 접촉이 체중을 지지한다. 몸통과 머리 및 양팔은 그 지점 아래로 중력에 따라 늘어져 있고 손에도 불필요한 근력이 보이지 않는다. 로프는 허리 장비에 연결되거나 걸린 상태로 아래로 늘어져 있어 떠 있지 않는다."
       },
       {
        "label": "A",
        "direction": "토니는 중앙에서 머리를 아래로 향한 채 거꾸로 매달려 있다. 얼굴은 아래쪽과 약간 카메라 쪽을 향하며 눈은 감겼거나 아래를 본다. 무기나 겨냥하는 물체 및 이동 중인 신체는 없고, 몸의 방향은 붙잡힌 다리에서 경사면 아래의 머리와 손으로 이어진다.",
        "built_space": "인공 구조물이나 고정 설비가 없는 숲 외부다. 다수의 굵은 가지가 상단과 중앙을 사선으로 가로지르며 토니의 무릎과 허벅지 양쪽을 끼워 고정한다. 아래에는 오른쪽 위에서 왼쪽 아래로 후퇴하는 이끼 낀 숲 경사면이 넓게 보여 장소와 높이를 잘 설명한다.",
        "entities": "등장 인물은 토니 한 명이며 30대 남성, 헬멧, 어두운 작업복, 장갑, 하네스가 참고 인물과 대체로 맞는다. 갈색 구조 로프 묶음은 오른쪽 허리에 분명히 달려 있다. 다만 지정된 검은 플레이트와 은색 웨이트가 달린 점퍼 벨트는 보이지 않고 일반 하네스와 파우치 위주로 표현됐다. 추가 인물이나 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "양쪽 허벅지와 굽힌 무릎이 여러 굵은 가지의 교차부 위와 사이에 걸려 몸을 확실히 지지한다. 부츠는 위쪽으로 나온 채 가지 뒤에 놓이고, 몸통·머리·팔은 아래로 자연스럽게 늘어진다. 구조 로프는 허리 하네스에 고정된 묶음으로 지지되어 있으며 공중에 뜬 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.556,
    "B": 2.0
   },
   "adjusted": {
    "A": 1.556,
    "B": 2.0
   },
   "violations": {},
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "B": 2000,
   "A": 1556
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "이전 샷 래퍼런스에서 요구된 은색 추가 달린 검은색 점퍼 벨트를 허리에 착용한 상태를 완벽하게 반영하였으며, 중력에 순응하여 매달린 포즈와 숲의 경사면 묘사가 매우 충실합니다."
   },
   {
    "label": "A",
    "score": 1556,
    "verdict_ko": "지시된 구도와 나뭇가지에 얽혀 거꾸로 매달린 자세는 훌륭하게 구현했으나, 프롬프트에서 필수적으로 요구한 '은색 추가 달린 점퍼 벨트'가 완전히 누락되었습니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S3sh8_sel.png",
    "asset_id": "30f1d460-5819-44a3-8243-3ebcfe58bcd9",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc77-f7b9-74b7-93e7-ec3e89bfa502",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S3sh8"
  }
 },
 "S4sh5::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:41:43.272842+00:00",
  "fingerprint": "b8840a23f260722d942bc8d3e175367f17e95e735cd796adcd15967a1e6eeca0",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S4sh5_sel.png",
  "source_sha256": "4c261aa426e5da90dca2729811bdb20d2c6361063e8897ed06010638f9536f21",
  "file": "S4sh5_cine.png",
  "staged_sha256": "3b3205482366fd3220c19eacc6c3ecd02e060ac6cc8b6d64f99d36785ef4042d",
  "latency_ms": 18139
 },
 "S4sh11::signage": {
  "fp": "e8049832380119f6",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S4sh11": {
  "input_fingerprint": "143a46cdde2b048e",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 은색 추를 토니의 장비 고리에 꽉 맞물려 강하게 밀어 넣고 있는 윌마의 손 클로즈업\n\nLOCATION (lock): An exterior landing spot on the sloping forest floor, beneath the surrounding trees and rocks. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 윌마 디어링's pressing hand in the middle-left of the frame, foreground, reaches for Tony's equipment loop; silver weight entering equipment loop in the middle-center of the frame, foreground; 토니(앤서니 로저스)'s belt area in the middle-right of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: silver weight (being pressed firmly into Tony's equipment loop) — Its locking side is aligned with the loop while its outer surface remains visible to camera; used as Central moving component at the moment of attachment; Tony's equipment loop (receiving the silver weight) — The loop is shown side-on enough to reveal the weight entering and locking into it; used as Mechanical anchor that makes the completed transfer readable; Wilma's belt (one silver weight has been removed from it) — Only a narrow side section remains visible at the opposite edge from Tony's loop; used as Minimal origin reference for the transferred weight.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is desaturated and moderately contrasted to clarify the locking contact between hand, silver weight, and equipment loop.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the dense interlaced branches, mossy forest slope, and filtered daytime light from the reference. Exclude the body caught among the branches and the temporary snagged-twig disturbance; retain only the belt hardware required for this shot.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The training slope and surrounding forest remain unchanged. One silver weight is being transferred between the two jumper belts. 윌마 디어링: She has detached one silver weight from her own moss-smeared belt and is fastening it into another equipment loop. She continues to carry her moss-smeared gun.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마 right now, so 윌마's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 은색 추를 토니의 장비 고리에 꽉 맞물려 강하게 밀어 넣고 있는 윌마의 손 클로즈업\n\nLOCATION (lock): An exterior landing spot on the sloping forest floor, beneath the surrounding trees and rocks. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 윌마 디어링's pressing hand in the middle-left of the frame, foreground, reaches for Tony's equipment loop; silver weight entering equipment loop in the middle-center of the frame, foreground; 토니(앤서니 로저스)'s belt area in the middle-right of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: silver weight (being pressed firmly into Tony's equipment loop) — Its locking side is aligned with the loop while its outer surface remains visible to camera; used as Central moving component at the moment of attachment; Tony's equipment loop (receiving the silver weight) — The loop is shown side-on enough to reveal the weight entering and locking into it; used as Mechanical anchor that makes the completed transfer readable; Wilma's belt (one silver weight has been removed from it) — Only a narrow side section remains visible at the opposite edge from Tony's loop; used as Minimal origin reference for the transferred weight.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is desaturated and moderately contrasted to clarify the locking contact between hand, silver weight, and equipment loop.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the dense interlaced branches, mossy forest slope, and filtered daytime light from the reference. Exclude the body caught among the branches and the temporary snagged-twig disturbance; retain only the belt hardware required for this shot.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The training slope and surrounding forest remain unchanged. One silver weight is being transferred between the two jumper belts. 윌마 디어링: She has detached one silver weight from her own moss-smeared belt and is fastening it into another equipment loop. She continues to carry her moss-smeared gun.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마 right now, so 윌마's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 은색 추를 토니의 장비 고리에 꽉 맞물려 강하게 밀어 넣고 있는 윌마의 손 클로즈업\n\nLOCATION (lock): An exterior landing spot on the sloping forest floor, beneath the surrounding trees and rocks. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 윌마 디어링's pressing hand in the middle-left of the frame, foreground, reaches for Tony's equipment loop; silver weight entering equipment loop in the middle-center of the frame, foreground; 토니(앤서니 로저스)'s belt area in the middle-right of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: silver weight (being pressed firmly into Tony's equipment loop) — Its locking side is aligned with the loop while its outer surface remains visible to camera; used as Central moving component at the moment of attachment; Tony's equipment loop (receiving the silver weight) — The loop is shown side-on enough to reveal the weight entering and locking into it; used as Mechanical anchor that makes the completed transfer readable; Wilma's belt (one silver weight has been removed from it) — Only a narrow side section remains visible at the opposite edge from Tony's loop; used as Minimal origin reference for the transferred weight.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Natural daylight appropriate to the forest setting is desaturated and moderately contrasted to clarify the locking contact between hand, silver weight, and equipment loop.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the dense interlaced branches, mossy forest slope, and filtered daytime light from the reference. Exclude the body caught among the branches and the temporary snagged-twig disturbance; retain only the belt hardware required for this shot.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The training slope and surrounding forest remain unchanged. One silver weight is being transferred between the two jumper belts. 윌마 디어링: She has detached one silver weight from her own moss-smeared belt and is fastening it into another equipment loop. She continues to carry her moss-smeared gun.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마 right now, so 윌마's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "화면 중앙 좌측의 손이 은색 장치를 우측의 벨트 고리로 향해 밀어넣고 있습니다.",
    "built_space": "배경은 이끼가 덮인 숲의 경사면과 나뭇가지들로 구성되어 이전 샷의 환경과 일치합니다.",
    "entities": "오른쪽의 벨트는 위장복 무늬가 있으나 레퍼런스의 둥근 추가 달린 벨트와는 형태가 다릅니다. 동작을 수행하는 손의 소매는 윌마의 단색 녹색 비행복이 아닌 위장복 무늬(토니의 의상)입니다. 화면 좌측 상단에 윌마의 의상과 유사한 단색 녹색 소매의 팔이 의미 없이 잘려 보입니다.",
    "hard_violations": [
     "[gemini-pro] 지정된 인물(윌마)이 아닌 위장복 소매의 인물이 무게추를 조작하고 있음",
     "[gemini-pro] 화면 좌측 상단에 누구의 것인지 알 수 없는 여분의 팔(extra body)이 잘못 렌더링됨",
     "[gpt] 상단 중앙에 윌마의 행동 팔과 별개인 지시되지 않은 두 번째 팔·소매가 추가되어, 허용되지 않은 추가 신체 또는 인물을 만든다."
    ],
    "physics": "위장복 소매의 손이 무게추를 단단히 쥐고 지탱하고 있으나, 좌측 상단의 단색 녹색 팔은 물리적 맥락 없이 허공에 떠 있는 상태입니다."
   },
   {
    "label": "B",
    "direction": "화면 좌측의 손이 둥근 은색 무게추를 우측 벨트에 달린 금속 고리를 향해 정확히 맞춰 밀어넣고 있습니다.",
    "built_space": "배경은 초점 나간 나뭇가지와 이끼 낀 경사면으로 구성되어 이전 샷의 위치와 정확히 일치합니다.",
    "entities": "화면 좌측의 팔은 윌마의 캐릭터 레퍼런스와 일치하는 단색 녹색 비행복 소매와 손가락 끝이 노출된 전술 장갑을 착용하고 있습니다. 좌측 끝에 윌마의 벨트 일부가 보입니다. 우측의 벨트는 이전 샷 레퍼런스의 토니가 착용했던 둥근 은색 추들이 부착된 두꺼운 벨트와 위장복으로 정확히 묘사되었습니다.",
    "hard_violations": [],
    "physics": "장갑을 낀 손이 은색 무게추를 꽉 쥐고 고리에 결합하는 물리적인 접촉과 무게 중심이 매우 자연스럽고 안정적으로 지탱되어 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "무게추를 쥐고 있는 팔이 윌마의 녹색 비행복이 아닌 위장복을 입고 있으며, 화면 좌측 상단에 정체불명의 팔이 하나 더 추가된 치명적인 오류가 있습니다."
       },
       {
        "label": "B",
        "score": 10,
        "verdict_ko": "클로즈업 프레이밍의 지침을 완벽하게 따랐으며, 윌마의 녹색 비행복 소매와 반장갑, 그리고 레퍼런스와 일치하는 토니의 은색 추가 달린 두꺼운 벨트를 정확하게 재현했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "화면 중앙 좌측의 손이 은색 장치를 우측의 벨트 고리로 향해 밀어넣고 있습니다.",
        "built_space": "배경은 이끼가 덮인 숲의 경사면과 나뭇가지들로 구성되어 이전 샷의 환경과 일치합니다.",
        "entities": "오른쪽의 벨트는 위장복 무늬가 있으나 레퍼런스의 둥근 추가 달린 벨트와는 형태가 다릅니다. 동작을 수행하는 손의 소매는 윌마의 단색 녹색 비행복이 아닌 위장복 무늬(토니의 의상)입니다. 화면 좌측 상단에 윌마의 의상과 유사한 단색 녹색 소매의 팔이 의미 없이 잘려 보입니다.",
        "hard_violations": [
         "지정된 인물(윌마)이 아닌 위장복 소매의 인물이 무게추를 조작하고 있음",
         "화면 좌측 상단에 누구의 것인지 알 수 없는 여분의 팔(extra body)이 잘못 렌더링됨"
        ],
        "physics": "위장복 소매의 손이 무게추를 단단히 쥐고 지탱하고 있으나, 좌측 상단의 단색 녹색 팔은 물리적 맥락 없이 허공에 떠 있는 상태입니다."
       },
       {
        "label": "B",
        "direction": "화면 좌측의 손이 둥근 은색 무게추를 우측 벨트에 달린 금속 고리를 향해 정확히 맞춰 밀어넣고 있습니다.",
        "built_space": "배경은 초점 나간 나뭇가지와 이끼 낀 경사면으로 구성되어 이전 샷의 위치와 정확히 일치합니다.",
        "entities": "화면 좌측의 팔은 윌마의 캐릭터 레퍼런스와 일치하는 단색 녹색 비행복 소매와 손가락 끝이 노출된 전술 장갑을 착용하고 있습니다. 좌측 끝에 윌마의 벨트 일부가 보입니다. 우측의 벨트는 이전 샷 레퍼런스의 토니가 착용했던 둥근 은색 추들이 부착된 두꺼운 벨트와 위장복으로 정확히 묘사되었습니다.",
        "hard_violations": [],
        "physics": "장갑을 낀 손이 은색 무게추를 꽉 쥐고 고리에 결합하는 물리적인 접촉과 무게 중심이 매우 자연스럽고 안정적으로 지탱되어 있습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "무게추를 쥐고 있는 팔이 윌마의 녹색 비행복이 아닌 위장복을 입고 있으며, 화면 좌측 상단에 정체불명의 팔이 하나 더 추가된 치명적인 오류가 있습니다."
       },
       {
        "label": "B",
        "score": 10,
        "verdict_ko": "클로즈업 프레이밍의 지침을 완벽하게 따랐으며, 윌마의 녹색 비행복 소매와 반장갑, 그리고 레퍼런스와 일치하는 토니의 은색 추가 달린 두꺼운 벨트를 정확하게 재현했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "화면 중앙 좌측의 손이 은색 장치를 우측의 벨트 고리로 향해 밀어넣고 있습니다.",
        "built_space": "배경은 이끼가 덮인 숲의 경사면과 나뭇가지들로 구성되어 이전 샷의 환경과 일치합니다.",
        "entities": "오른쪽의 벨트는 위장복 무늬가 있으나 레퍼런스의 둥근 추가 달린 벨트와는 형태가 다릅니다. 동작을 수행하는 손의 소매는 윌마의 단색 녹색 비행복이 아닌 위장복 무늬(토니의 의상)입니다. 화면 좌측 상단에 윌마의 의상과 유사한 단색 녹색 소매의 팔이 의미 없이 잘려 보입니다.",
        "hard_violations": [
         "지정된 인물(윌마)이 아닌 위장복 소매의 인물이 무게추를 조작하고 있음",
         "화면 좌측 상단에 누구의 것인지 알 수 없는 여분의 팔(extra body)이 잘못 렌더링됨"
        ],
        "physics": "위장복 소매의 손이 무게추를 단단히 쥐고 지탱하고 있으나, 좌측 상단의 단색 녹색 팔은 물리적 맥락 없이 허공에 떠 있는 상태입니다."
       },
       {
        "label": "B",
        "direction": "화면 좌측의 손이 둥근 은색 무게추를 우측 벨트에 달린 금속 고리를 향해 정확히 맞춰 밀어넣고 있습니다.",
        "built_space": "배경은 초점 나간 나뭇가지와 이끼 낀 경사면으로 구성되어 이전 샷의 위치와 정확히 일치합니다.",
        "entities": "화면 좌측의 팔은 윌마의 캐릭터 레퍼런스와 일치하는 단색 녹색 비행복 소매와 손가락 끝이 노출된 전술 장갑을 착용하고 있습니다. 좌측 끝에 윌마의 벨트 일부가 보입니다. 우측의 벨트는 이전 샷 레퍼런스의 토니가 착용했던 둥근 은색 추들이 부착된 두꺼운 벨트와 위장복으로 정확히 묘사되었습니다.",
        "hard_violations": [],
        "physics": "장갑을 낀 손이 은색 무게추를 꽉 쥐고 고리에 결합하는 물리적인 접촉과 무게 중심이 매우 자연스럽고 안정적으로 지탱되어 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "윌마의 손이 화면 중좌측에서 은색 추를 중앙의 토니 장비 고리로 밀어 넣고, 토니의 벨트가 우측에 놓인 정확한 클로즈업으로 지정된 결합 순간을 가장 충실히 구현했다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "추와 고리의 결합 방향은 맞지만 상단 중앙에 지시되지 않은 두 번째 팔이 추가되었고 윌마 벨트의 최소 기원 표지도 불명확해 사용할 수 없는 프레임이다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마의 장갑 낀 손이 화면 왼쪽에서 오른쪽으로 은색 추를 강하게 밀고 있으며, 추의 잠금 돌기가 화면 중앙의 토니 장비 고리 입구를 정확히 향한다. 추는 고리와 접촉하며 진입하는 순간이고, 목표는 명확히 토니의 고리이다.",
        "built_space": "건축 구조물 내부가 아니라 경사진 숲 바닥의 외부 공간이다. 뒤에는 이전 장소와 부합하는 빽빽하게 얽힌 굵은 가지, 나무줄기, 이끼 낀 사면이 보인다. 전경에는 토니 쪽 수용 고리 1개, 그 벨트에 이미 장착된 원형 은색 추 2개, 이동 중인 추 1개가 보이며, 반대편 왼쪽 가장자리에는 윌마 벨트의 좁은 일부와 버클·장비가 보인다. 반사나 구조적으로 불가능한 배치는 없다.",
        "entities": "중좌측 손과 손목은 젊은 성인 여성에게 자연스러운 크기와 피부, 윌마 참고 이미지에 맞는 녹색 소매와 손가락 노출형 전술 장갑을 지녔다. 중앙 물체는 카메라 쪽에 둥근 금속 외면이 보이는 실물 은색 추이며, 잠금 측면은 토니의 금속 장비 고리에 맞춰져 있다. 우측에는 토니의 벨트 영역만 크롭되어 있고 얼굴이나 별도 인물은 없다. 윌마의 총은 올바른 클로즈업 밖이라 보이지 않아 감점 대상이 아니다.",
        "hard_violations": [],
        "physics": "은색 추는 윌마의 장갑 낀 손가락과 엄지로 확실히 붙잡혀 지지되며, 손에서 오른쪽 고리로 가해지는 압력이 읽힌다. 고리는 토니의 두꺼운 벨트 웨빙에 고정되어 반력을 받을 수 있고, 추의 잠금 돌기와 고리의 접촉도 물리적으로 가능하다. 공중에 무지지 상태로 떠 있는 물체나 신체는 없다."
       },
       {
        "label": "B",
        "direction": "윌마의 장갑 낀 손은 왼쪽에서 오른쪽으로 은색 추를 밀며, 추의 갈고리형 잠금부는 중앙 우측의 토니 장비 고리를 정확히 향해 접촉하고 있다. 목표 자체는 토니 벨트의 고리로 분명하지만, 추의 넓은 외면보다 잠금판이 카메라에 더 크게 노출된다.",
        "built_space": "경사진 숲 사면과 얽힌 가지, 이끼, 주간 여과광은 장소 참고와 대체로 일치한다. 토니 쪽에는 수용 고리 1개, 이동 중인 추 1개, 화면 오른쪽 끝에 기존 은색 추 일부 1개, 여러 파우치와 추가 D링·카라비너가 보인다. 그러나 토니의 상체 장비와 파우치가 넓게 차지하고, 반대편 가장자리에서 보여야 할 윌마 벨트의 좁은 기원 구간은 명확히 확인되지 않는다. 상단 중앙에는 주 행동 팔과 별개인 두 번째 소매·팔이 추가로 내려와 있다.",
        "entities": "주 행동 손은 녹색 소매와 전술 장갑을 착용해 윌마의 일부로 읽히고 은색 추를 잡고 있다. 추와 토니의 금속 장비 고리, 토니의 벨트 영역은 식별된다. 다만 상단 중앙의 갈색 소매 팔은 윌마의 왼쪽에서 이어지는 행동 팔과 별개의 신체 부분으로 보이며, 프롬프트가 허용하지 않은 추가 인물 또는 추가 팔이다. 윌마 자신의 이끼 묻은 벨트에서 추가 제거되었다는 최소 기원 표지도 보이지 않는다.",
        "hard_violations": [
         "상단 중앙에 윌마의 행동 팔과 별개인 지시되지 않은 두 번째 팔·소매가 추가되어, 허용되지 않은 추가 신체 또는 인물을 만든다."
        ],
        "physics": "은색 추는 윌마의 손으로 잡혀 있고 잠금부가 벨트 고리에 닿아 있어 이동 물체 자체의 지지와 결합 동작은 가능하다. 토니의 고리는 벨트 웨빙에 고정되어 있다. 다만 상단 중앙의 추가 팔은 프레임 밖 몸에 이어지는 듯하더라도 이 장면에 배정된 인물이 없으며, 주 행동에 필요한 지지 역할도 하지 않는다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "윌마의 손이 화면 중좌측에서 은색 추를 중앙의 토니 장비 고리로 밀어 넣고, 토니의 벨트가 우측에 놓인 정확한 클로즈업으로 지정된 결합 순간을 가장 충실히 구현했다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "추와 고리의 결합 방향은 맞지만 상단 중앙에 지시되지 않은 두 번째 팔이 추가되었고 윌마 벨트의 최소 기원 표지도 불명확해 사용할 수 없는 프레임이다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "윌마의 장갑 낀 손이 화면 왼쪽에서 오른쪽으로 은색 추를 강하게 밀고 있으며, 추의 잠금 돌기가 화면 중앙의 토니 장비 고리 입구를 정확히 향한다. 추는 고리와 접촉하며 진입하는 순간이고, 목표는 명확히 토니의 고리이다.",
        "built_space": "건축 구조물 내부가 아니라 경사진 숲 바닥의 외부 공간이다. 뒤에는 이전 장소와 부합하는 빽빽하게 얽힌 굵은 가지, 나무줄기, 이끼 낀 사면이 보인다. 전경에는 토니 쪽 수용 고리 1개, 그 벨트에 이미 장착된 원형 은색 추 2개, 이동 중인 추 1개가 보이며, 반대편 왼쪽 가장자리에는 윌마 벨트의 좁은 일부와 버클·장비가 보인다. 반사나 구조적으로 불가능한 배치는 없다.",
        "entities": "중좌측 손과 손목은 젊은 성인 여성에게 자연스러운 크기와 피부, 윌마 참고 이미지에 맞는 녹색 소매와 손가락 노출형 전술 장갑을 지녔다. 중앙 물체는 카메라 쪽에 둥근 금속 외면이 보이는 실물 은색 추이며, 잠금 측면은 토니의 금속 장비 고리에 맞춰져 있다. 우측에는 토니의 벨트 영역만 크롭되어 있고 얼굴이나 별도 인물은 없다. 윌마의 총은 올바른 클로즈업 밖이라 보이지 않아 감점 대상이 아니다.",
        "hard_violations": [],
        "physics": "은색 추는 윌마의 장갑 낀 손가락과 엄지로 확실히 붙잡혀 지지되며, 손에서 오른쪽 고리로 가해지는 압력이 읽힌다. 고리는 토니의 두꺼운 벨트 웨빙에 고정되어 반력을 받을 수 있고, 추의 잠금 돌기와 고리의 접촉도 물리적으로 가능하다. 공중에 무지지 상태로 떠 있는 물체나 신체는 없다."
       },
       {
        "label": "A",
        "direction": "윌마의 장갑 낀 손은 왼쪽에서 오른쪽으로 은색 추를 밀며, 추의 갈고리형 잠금부는 중앙 우측의 토니 장비 고리를 정확히 향해 접촉하고 있다. 목표 자체는 토니 벨트의 고리로 분명하지만, 추의 넓은 외면보다 잠금판이 카메라에 더 크게 노출된다.",
        "built_space": "경사진 숲 사면과 얽힌 가지, 이끼, 주간 여과광은 장소 참고와 대체로 일치한다. 토니 쪽에는 수용 고리 1개, 이동 중인 추 1개, 화면 오른쪽 끝에 기존 은색 추 일부 1개, 여러 파우치와 추가 D링·카라비너가 보인다. 그러나 토니의 상체 장비와 파우치가 넓게 차지하고, 반대편 가장자리에서 보여야 할 윌마 벨트의 좁은 기원 구간은 명확히 확인되지 않는다. 상단 중앙에는 주 행동 팔과 별개인 두 번째 소매·팔이 추가로 내려와 있다.",
        "entities": "주 행동 손은 녹색 소매와 전술 장갑을 착용해 윌마의 일부로 읽히고 은색 추를 잡고 있다. 추와 토니의 금속 장비 고리, 토니의 벨트 영역은 식별된다. 다만 상단 중앙의 갈색 소매 팔은 윌마의 왼쪽에서 이어지는 행동 팔과 별개의 신체 부분으로 보이며, 프롬프트가 허용하지 않은 추가 인물 또는 추가 팔이다. 윌마 자신의 이끼 묻은 벨트에서 추가 제거되었다는 최소 기원 표지도 보이지 않는다.",
        "hard_violations": [
         "상단 중앙에 윌마의 행동 팔과 별개인 지시되지 않은 두 번째 팔·소매가 추가되어, 허용되지 않은 추가 신체 또는 인물을 만든다."
        ],
        "physics": "은색 추는 윌마의 손으로 잡혀 있고 잠금부가 벨트 고리에 닿아 있어 이동 물체 자체의 지지와 결합 동작은 가능하다. 토니의 고리는 벨트 웨빙에 고정되어 있다. 다만 상단 중앙의 추가 팔은 프레임 밖 몸에 이어지는 듯하더라도 이 장면에 배정된 인물이 없으며, 주 행동에 필요한 지지 역할도 하지 않는다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 0.644,
    "B": 2.0
   },
   "adjusted": {
    "A": 0.394,
    "B": 2.0
   },
   "violations": {
    "A": [
     "[gemini-pro] 지정된 인물(윌마)이 아닌 위장복 소매의 인물이 무게추를 조작하고 있음",
     "[gemini-pro] 화면 좌측 상단에 누구의 것인지 알 수 없는 여분의 팔(extra body)이 잘못 렌더링됨",
     "[gpt] 상단 중앙에 윌마의 행동 팔과 별개인 지시되지 않은 두 번째 팔·소매가 추가되어, 허용되지 않은 추가 신체 또는 인물을 만든다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "A": 394,
   "B": 2000
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 394,
    "verdict_ko": "무게추를 쥐고 있는 팔이 윌마의 녹색 비행복이 아닌 위장복을 입고 있으며, 화면 좌측 상단에 정체불명의 팔이 하나 더 추가된 치명적인 오류가 있습니다.  ★위반: [gemini-pro] 지정된 인물(윌마)이 아닌 위장복 소매의 인물이 무게추를 조작하고 있음 / [gemini-pro] 화면 좌측 상단에 누구의 것인지 알 수 없는 여분의 팔(extra body)이 잘못 렌더링됨 / [gpt] 상단 중앙에 윌마의 행동 팔과 별개인 지시되지 않은 두 번째 팔·소매가 추가되어, 허용되지 않은 추가 신체 또는 인물을 만든다."
   },
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "클로즈업 프레이밍의 지침을 완벽하게 따랐으며, 윌마의 녹색 비행복 소매와 반장갑, 그리고 레퍼런스와 일치하는 토니의 은색 추가 달린 두꺼운 벨트를 정확하게 재현했습니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S4sh5_sel.png",
    "asset_id": "f8a7c93b-4b03-4ee9-b761-3f3c27864570",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc7c-98b4-7f61-b822-6679b0c4d974",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S4sh5"
  }
 },
 "S4sh11::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:42:52.650695+00:00",
  "fingerprint": "48625919bb44c33c52dfe8ac96ee88e8ef02edfb98a812fafd81df097b372da3",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S4sh11_sel.png",
  "source_sha256": "4e2c41fa9672a7c461fb52caf22e66adeec114e102cad9babea2e5e445346ff3",
  "file": "S4sh11_cine.png",
  "staged_sha256": "82c76a8c515800ac4f27fec86b79ad5c84ab449d0833d5743329fa701b031d03",
  "latency_ms": 20949
 },
 "S5sh2::signage": {
  "fp": "c68c1ad96b124b21",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S5sh2": {
  "input_fingerprint": "889c60039c9a739b",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): afternoon.\n\nSHOT TEXT (authoritative, Korean): 나뭇잎들 사이로 숲의 흙더미에 반쯤 파묻힌 거대한 콘크리트 곡선 벽면이 드러난 풍경\n\nLOCATION (lock): The exposed exterior face of a ruined stadium, where a vast curved concrete wall emerges from forest soil and foliage. The shot takes place here — the attached LOCATION STRUCTURE PHOTOGRAPH is the single authority for this exact place — its fixed structure and permanent site details are LOCKED to it. No separate location photograph exists for this place. Build everything else strictly from the location text above and the shot text; the layout sketch (when attached) governs framing and placement only, and the shot text governs time of day, lighting and action.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: curved stadium wall in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: foreground leaves (parting around the revealed view); used as Soft foreground framing that completes the crane reveal without obscuring the ruin; curved concrete stadium wall (half-buried in the forest earth mound) — Its broad outer curve faces the camera across the middle distance; used as Primary architectural reveal and scale anchor; forest and earth mound (surrounding and partially covering the stadium wall); used as Environmental scale reference around the exposed concrete curve.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained, desaturated afternoon ambient light maintains moderate contrast across the forest and exposed concrete.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A huge curved concrete wall emerges through the leaves, half-buried in the forest soil. The ruined stadium and its root-torn scoreboard remain embedded in the woods.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): afternoon.\n\nSHOT TEXT (authoritative, Korean): 나뭇잎들 사이로 숲의 흙더미에 반쯤 파묻힌 거대한 콘크리트 곡선 벽면이 드러난 풍경\n\nLOCATION (lock): The exposed exterior face of a ruined stadium, where a vast curved concrete wall emerges from forest soil and foliage. The shot takes place here — the attached LOCATION STRUCTURE PHOTOGRAPH is the single authority for this exact place — its fixed structure and permanent site details are LOCKED to it. No separate location photograph exists for this place. Build everything else strictly from the location text above and the shot text; the layout sketch (when attached) governs framing and placement only, and the shot text governs time of day, lighting and action.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: curved stadium wall in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: foreground leaves (parting around the revealed view); used as Soft foreground framing that completes the crane reveal without obscuring the ruin; curved concrete stadium wall (half-buried in the forest earth mound) — Its broad outer curve faces the camera across the middle distance; used as Primary architectural reveal and scale anchor; forest and earth mound (surrounding and partially covering the stadium wall); used as Environmental scale reference around the exposed concrete curve.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained, desaturated afternoon ambient light maintains moderate contrast across the forest and exposed concrete.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A huge curved concrete wall emerges through the leaves, half-buried in the forest soil. The ruined stadium and its root-torn scoreboard remain embedded in the woods.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): afternoon.\n\nSHOT TEXT (authoritative, Korean): 나뭇잎들 사이로 숲의 흙더미에 반쯤 파묻힌 거대한 콘크리트 곡선 벽면이 드러난 풍경\n\nLOCATION (lock): The exposed exterior face of a ruined stadium, where a vast curved concrete wall emerges from forest soil and foliage. The shot takes place here — the attached LOCATION STRUCTURE PHOTOGRAPH is the single authority for this exact place — its fixed structure and permanent site details are LOCKED to it. No separate location photograph exists for this place. Build everything else strictly from the location text above and the shot text; the layout sketch (when attached) governs framing and placement only, and the shot text governs time of day, lighting and action.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: curved stadium wall in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: foreground leaves (parting around the revealed view); used as Soft foreground framing that completes the crane reveal without obscuring the ruin; curved concrete stadium wall (half-buried in the forest earth mound) — Its broad outer curve faces the camera across the middle distance; used as Primary architectural reveal and scale anchor; forest and earth mound (surrounding and partially covering the stadium wall); used as Environmental scale reference around the exposed concrete curve.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained, desaturated afternoon ambient light maintains moderate contrast across the forest and exposed concrete.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A huge curved concrete wall emerges through the leaves, half-buried in the forest soil. The ruined stadium and its root-torn scoreboard remain embedded in the woods.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라는 전경의 나뭇잎들 사이를 지나 중앙의 곡선형 콘크리트 경기장 벽면과 우측의 스코어보드를 향하고 있음.",
    "built_space": "레퍼런스 이미지와 일치하는 곡선형 경기장 외벽, 상단 보행로, 하단을 지탱하는 기둥들, 그리고 우측의 대형 스코어보드가 명확히 존재함. 기둥 하단은 지시대로 흙더미에 반쯤 파묻혀 있음.",
    "entities": "프레임을 감싸는 전경의 나뭇잎, 흙과 식물에 덮인 거대한 콘크리트 벽, 덩굴이 얽힌 스코어보드(알파벳 텍스트가 식별됨), 배경의 울창한 숲.",
    "hard_violations": [
     "[gpt] 전광판에 영문 제목과 경기 표기가 읽을 수 있게 노출되어, 프레임 어디에도 읽을 수 있는 글자가 없어야 한다는 명시적 조건을 위반했다."
    ],
    "physics": "거대한 콘크리트 구조물이 흙더미 위에 안정적으로 지지되어 있으며, 덩굴과 나무들은 구조물과 지면을 바탕으로 자연스럽게 성장해 있음."
   },
   {
    "label": "B",
    "direction": "카메라는 전경의 나뭇잎들 사이를 지나 중앙의 거대하고 꽉 막힌 콘크리트 장벽을 향하고 있음.",
    "built_space": "레퍼런스 이미지의 경기장(기둥과 개방된 하단부, 관중석, 대형 스코어보드 등)과는 전혀 다른 형태의 댐이나 옹벽과 같은 구조물이 자리잡고 있음.",
    "entities": "전경의 나뭇잎, 거대한 덩어리 형태의 콘크리트 벽, 흙과 숲.",
    "hard_violations": [
     "[gemini-pro] 레퍼런스에 지정된 특정 경기장 구조 대신 완전히 다른 형태의 건축물(invented objects)을 생성함."
    ],
    "physics": "콘크리트 벽면은 지면 위에 고정되어 있으며 식물들이 흙과 벽면에 붙어 안정적으로 지탱됨."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "레퍼런스의 경기장 구조와 스코어보드를 정확히 반영하였고 전경의 나뭇잎을 통한 프레이밍도 훌륭하나, 읽을 수 있는 텍스트를 배제하라는 지시와 달리 스코어보드의 글자가 일부 노출된 점이 아쉽습니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "프레이밍 지시는 따랐으나, 고정된 위치(Location lock)의 필수 구조인 기둥, 관중석, 스코어보드를 모두 누락하고 완전히 다른 형태의 장벽을 생성하여 위치 일관성에서 크게 감점되었습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 전경의 나뭇잎들 사이를 지나 중앙의 곡선형 콘크리트 경기장 벽면과 우측의 스코어보드를 향하고 있음.",
        "built_space": "레퍼런스 이미지와 일치하는 곡선형 경기장 외벽, 상단 보행로, 하단을 지탱하는 기둥들, 그리고 우측의 대형 스코어보드가 명확히 존재함. 기둥 하단은 지시대로 흙더미에 반쯤 파묻혀 있음.",
        "entities": "프레임을 감싸는 전경의 나뭇잎, 흙과 식물에 덮인 거대한 콘크리트 벽, 덩굴이 얽힌 스코어보드(알파벳 텍스트가 식별됨), 배경의 울창한 숲.",
        "hard_violations": [],
        "physics": "거대한 콘크리트 구조물이 흙더미 위에 안정적으로 지지되어 있으며, 덩굴과 나무들은 구조물과 지면을 바탕으로 자연스럽게 성장해 있음."
       },
       {
        "label": "B",
        "direction": "카메라는 전경의 나뭇잎들 사이를 지나 중앙의 거대하고 꽉 막힌 콘크리트 장벽을 향하고 있음.",
        "built_space": "레퍼런스 이미지의 경기장(기둥과 개방된 하단부, 관중석, 대형 스코어보드 등)과는 전혀 다른 형태의 댐이나 옹벽과 같은 구조물이 자리잡고 있음.",
        "entities": "전경의 나뭇잎, 거대한 덩어리 형태의 콘크리트 벽, 흙과 숲.",
        "hard_violations": [
         "레퍼런스에 지정된 특정 경기장 구조 대신 완전히 다른 형태의 건축물(invented objects)을 생성함."
        ],
        "physics": "콘크리트 벽면은 지면 위에 고정되어 있으며 식물들이 흙과 벽면에 붙어 안정적으로 지탱됨."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "레퍼런스의 경기장 구조와 스코어보드를 정확히 반영하였고 전경의 나뭇잎을 통한 프레이밍도 훌륭하나, 읽을 수 있는 텍스트를 배제하라는 지시와 달리 스코어보드의 글자가 일부 노출된 점이 아쉽습니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "프레이밍 지시는 따랐으나, 고정된 위치(Location lock)의 필수 구조인 기둥, 관중석, 스코어보드를 모두 누락하고 완전히 다른 형태의 장벽을 생성하여 위치 일관성에서 크게 감점되었습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 전경의 나뭇잎들 사이를 지나 중앙의 곡선형 콘크리트 경기장 벽면과 우측의 스코어보드를 향하고 있음.",
        "built_space": "레퍼런스 이미지와 일치하는 곡선형 경기장 외벽, 상단 보행로, 하단을 지탱하는 기둥들, 그리고 우측의 대형 스코어보드가 명확히 존재함. 기둥 하단은 지시대로 흙더미에 반쯤 파묻혀 있음.",
        "entities": "프레임을 감싸는 전경의 나뭇잎, 흙과 식물에 덮인 거대한 콘크리트 벽, 덩굴이 얽힌 스코어보드(알파벳 텍스트가 식별됨), 배경의 울창한 숲.",
        "hard_violations": [],
        "physics": "거대한 콘크리트 구조물이 흙더미 위에 안정적으로 지지되어 있으며, 덩굴과 나무들은 구조물과 지면을 바탕으로 자연스럽게 성장해 있음."
       },
       {
        "label": "B",
        "direction": "카메라는 전경의 나뭇잎들 사이를 지나 중앙의 거대하고 꽉 막힌 콘크리트 장벽을 향하고 있음.",
        "built_space": "레퍼런스 이미지의 경기장(기둥과 개방된 하단부, 관중석, 대형 스코어보드 등)과는 전혀 다른 형태의 댐이나 옹벽과 같은 구조물이 자리잡고 있음.",
        "entities": "전경의 나뭇잎, 거대한 덩어리 형태의 콘크리트 벽, 흙과 숲.",
        "hard_violations": [
         "레퍼런스에 지정된 특정 경기장 구조 대신 완전히 다른 형태의 건축물(invented objects)을 생성함."
        ],
        "physics": "콘크리트 벽면은 지면 위에 고정되어 있으며 식물들이 흙과 벽면에 붙어 안정적으로 지탱됨."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 5,
        "verdict_ko": "나뭇잎으로 둘러싸인 와이드 숏과 반쯤 매몰된 콘크리트 벽은 맞지만, 경기장의 넓은 곡률·연속 개구부·뿌리에 얽힌 전광판이 사라져 장소 재현도가 크게 떨어진다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "곡선 경기장 벽, 흙더미, 숲, 전광판까지 장소와 구도는 가장 정확하지만 전광판의 영문이 읽혀 ‘읽을 수 있는 글자 금지’를 직접 위반하므로 탈락한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "사람, 시선, 무기, 지향성 소품이나 이동하는 물체가 없어 겨냥 방향과 대상은 없다.",
        "built_space": "중앙 중경에 하나의 거대한 콘크리트 구조물이 있고 하부에 검은 개구부 하나가 보인다. 흙더미가 구조물의 양쪽과 하부를 덮으며 전경·상단·좌우의 나뭇잎이 시야를 갈라 놓는다. 다만 벽은 기준 사진의 넓고 연속적인 경기장 곡면보다는 각진 요새나 절벽형 건물처럼 보이고, 반복되는 기둥·개구부·난간 및 전광판이 보이지 않는다.",
        "entities": "숲, 흙더미, 낡은 콘크리트 벽은 존재하고 사람은 없다. 그러나 구조물은 기준 경기장의 고유한 곡선과 개방형 관람석 외벽을 충분히 재현하지 못하며, 명시된 뿌리에 뜯긴 전광판도 누락됐다. 읽을 수 있는 글자는 보이지 않는다.",
        "hard_violations": [],
        "physics": "콘크리트 구조물은 지면과 매몰된 기초에 지지되고 흙더미와 식생도 지면 위에 놓여 있다. 공중에 떠 있거나 손 없이 떠 있는 물체는 없고 물리적으로 불가능한 배치도 보이지 않는다."
       },
       {
        "label": "B",
        "direction": "사람, 시선, 무기 또는 이동 물체가 없어 겨냥 방향과 대상은 없다.",
        "built_space": "중앙 중경에 하나의 대형 곡선 경기장 외벽이 있고, 상부 난간 하나가 곡선을 따라 이어지며 하부에는 여러 콘크리트 기둥과 개구부가 반복된다. 오른쪽에는 두 지지대에 선 전광판 하나가 있고 뿌리와 덩굴이 얽혀 있다. 흙더미가 외벽 하부를 부분적으로 매몰하며 전경과 상단의 잎이 폐허를 부드럽게 액자처럼 감싼다. 기준 사진의 경기장 구조와 배치는 대체로 일치한다.",
        "entities": "곡선 콘크리트 경기장 외벽, 숲, 흙더미, 뿌리에 얽힌 전광판이 모두 보이고 사람은 없다. 다만 전광판 상단의 영문과 내부 표기가 일부 읽힐 정도로 선명해 글자 비가독성 조건을 충족하지 못한다.",
        "hard_violations": [
         "전광판에 영문 제목과 경기 표기가 읽을 수 있게 노출되어, 프레임 어디에도 읽을 수 있는 글자가 없어야 한다는 명시적 조건을 위반했다."
        ],
        "physics": "경기장 벽과 기둥은 지면 및 매몰된 기초에 지지되고, 전광판은 두 개의 기둥에 세워져 있으며 뿌리와 덩굴도 구조물에 붙어 내려온다. 흙과 식생은 지면에 놓여 있고 지지 없이 떠 있는 물체는 없다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 5,
        "verdict_ko": "나뭇잎으로 둘러싸인 와이드 숏과 반쯤 매몰된 콘크리트 벽은 맞지만, 경기장의 넓은 곡률·연속 개구부·뿌리에 얽힌 전광판이 사라져 장소 재현도가 크게 떨어진다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "곡선 경기장 벽, 흙더미, 숲, 전광판까지 장소와 구도는 가장 정확하지만 전광판의 영문이 읽혀 ‘읽을 수 있는 글자 금지’를 직접 위반하므로 탈락한다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "사람, 시선, 무기, 지향성 소품이나 이동하는 물체가 없어 겨냥 방향과 대상은 없다.",
        "built_space": "중앙 중경에 하나의 거대한 콘크리트 구조물이 있고 하부에 검은 개구부 하나가 보인다. 흙더미가 구조물의 양쪽과 하부를 덮으며 전경·상단·좌우의 나뭇잎이 시야를 갈라 놓는다. 다만 벽은 기준 사진의 넓고 연속적인 경기장 곡면보다는 각진 요새나 절벽형 건물처럼 보이고, 반복되는 기둥·개구부·난간 및 전광판이 보이지 않는다.",
        "entities": "숲, 흙더미, 낡은 콘크리트 벽은 존재하고 사람은 없다. 그러나 구조물은 기준 경기장의 고유한 곡선과 개방형 관람석 외벽을 충분히 재현하지 못하며, 명시된 뿌리에 뜯긴 전광판도 누락됐다. 읽을 수 있는 글자는 보이지 않는다.",
        "hard_violations": [],
        "physics": "콘크리트 구조물은 지면과 매몰된 기초에 지지되고 흙더미와 식생도 지면 위에 놓여 있다. 공중에 떠 있거나 손 없이 떠 있는 물체는 없고 물리적으로 불가능한 배치도 보이지 않는다."
       },
       {
        "label": "A",
        "direction": "사람, 시선, 무기 또는 이동 물체가 없어 겨냥 방향과 대상은 없다.",
        "built_space": "중앙 중경에 하나의 대형 곡선 경기장 외벽이 있고, 상부 난간 하나가 곡선을 따라 이어지며 하부에는 여러 콘크리트 기둥과 개구부가 반복된다. 오른쪽에는 두 지지대에 선 전광판 하나가 있고 뿌리와 덩굴이 얽혀 있다. 흙더미가 외벽 하부를 부분적으로 매몰하며 전경과 상단의 잎이 폐허를 부드럽게 액자처럼 감싼다. 기준 사진의 경기장 구조와 배치는 대체로 일치한다.",
        "entities": "곡선 콘크리트 경기장 외벽, 숲, 흙더미, 뿌리에 얽힌 전광판이 모두 보이고 사람은 없다. 다만 전광판 상단의 영문과 내부 표기가 일부 읽힐 정도로 선명해 글자 비가독성 조건을 충족하지 못한다.",
        "hard_violations": [
         "전광판에 영문 제목과 경기 표기가 읽을 수 있게 노출되어, 프레임 어디에도 읽을 수 있는 글자가 없어야 한다는 명시적 조건을 위반했다."
        ],
        "physics": "경기장 벽과 기둥은 지면 및 매몰된 기초에 지지되고, 전광판은 두 개의 기둥에 세워져 있으며 뿌리와 덩굴도 구조물에 붙어 내려온다. 흙과 식생은 지면에 놓여 있고 지지 없이 떠 있는 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.8,
    "B": 1.25
   },
   "adjusted": {
    "A": 1.55,
    "B": 1.0
   },
   "violations": {
    "B": [
     "[gemini-pro] 레퍼런스에 지정된 특정 경기장 구조 대신 완전히 다른 형태의 건축물(invented objects)을 생성함."
    ],
    "A": [
     "[gpt] 전광판에 영문 제목과 경기 표기가 읽을 수 있게 노출되어, 프레임 어디에도 읽을 수 있는 글자가 없어야 한다는 명시적 조건을 위반했다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "agreed": false
  },
  "totals": {
   "A": 1550,
   "B": 1000
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1550,
    "verdict_ko": "레퍼런스의 경기장 구조와 스코어보드를 정확히 반영하였고 전경의 나뭇잎을 통한 프레이밍도 훌륭하나, 읽을 수 있는 텍스트를 배제하라는 지시와 달리 스코어보드의 글자가 일부 노출된 점이 아쉽습니다.  ★위반: [gpt] 전광판에 영문 제목과 경기 표기가 읽을 수 있게 노출되어, 프레임 어디에도 읽을 수 있는 글자가 없어야 한다는 명시적 조건을 위반했다."
   },
   {
    "label": "B",
    "score": 1000,
    "verdict_ko": "프레이밍 지시는 따랐으나, 고정된 위치(Location lock)의 필수 구조인 기둥, 관중석, 스코어보드를 모두 누락하고 완전히 다른 형태의 장벽을 생성하여 위치 일관성에서 크게 감점되었습니다.  ★위반: [gemini-pro] 레퍼런스에 지정된 특정 경기장 구조 대신 완전히 다른 형태의 건축물(invented objects)을 생성함."
   }
  ],
  "refs": [
   {
    "label": "LOCATION STRUCTURE PHOTOGRAPH — the confirmed photograph of this exact place and its fixed structure: it is the SINGLE authority for the location, the structure's shape, proportions, materials, colors, openings and every permanent site detail. Never copy its camera framing, time of day or lighting — the shot text is the authority for those.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/background_chain/seed_bg_ruined_stadium_sel.png",
    "asset_id": "94a93d7b-1211-466c-b57f-a34526130968",
    "role": "location_seed_bg"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc81-1b6c-7397-bf71-4dbe060c5f93",
  "ref_mode": "seed-bg만 (배경 전용)",
  "share_plan": {
   "ref_plan": "background"
  },
  "lane_policy": "ab_select_bypass:bg_only"
 },
 "S5sh2::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:44:08.014750+00:00",
  "fingerprint": "4be064558667edd90f534c249ce3e40f12ef134f4278a7f4e4b32554c471b7fd",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S5sh2_sel.png",
  "source_sha256": "baef553f603c3bbbbdf185e982a09303baefc92b12a15e021a3ca8a5f58206f8",
  "file": "S5sh2_cine.png",
  "staged_sha256": "0bfff0c8fa503a88ffe2dbcd79cb0375db81b5c1c90d7b204ff117cf56e45fa3",
  "latency_ms": 14130
 },
 "S5sh5::signage": {
  "fp": "b68e1fbdbceaff6e",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S5sh5": {
  "input_fingerprint": "2e9e4a53f9c4f729",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): afternoon.\n\nSHOT TEXT (authoritative, Korean): 입술을 굳게 다문 채 눈동자를 고정하고 굳어 있는 토니의 얼굴\n\nLOCATION (lock): The overgrown exterior spectator stands of the ruined stadium, among cracked seats invaded by the forest. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-right of the frame, foreground, looks toward Wilma off-screen.\n- KEY BACKGROUND ELEMENTS: cracked stadium seat (cracked) — Only its angled upper edge is visible behind Tony; used as Peripheral spatial reminder of the ruined stands and Tony’s loss.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated afternoon ambient light is shaped into restrained low-to-moderate-key contrast across Tony’s fixed eyes and compressed mouth.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the root-split concrete ruins, packed forest soil, encroaching foliage, and subdued afternoon light from the reference. Exclude restored stadium fixtures, active spectators, modern event equipment, or any clean undamaged surfaces.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The baseball stadium remains half-buried in the forest, with its scoreboard torn by roots and the faint text “...ANTON RAILRIDERS” still visible. A cracked spectator chair remains by Tony's hand. 토니(앤서니 로저스): He stands rigid after hearing the blunt statement, wearing the jumper belt with its added silver weight and carrying his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): afternoon.\n\nSHOT TEXT (authoritative, Korean): 입술을 굳게 다문 채 눈동자를 고정하고 굳어 있는 토니의 얼굴\n\nLOCATION (lock): The overgrown exterior spectator stands of the ruined stadium, among cracked seats invaded by the forest. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-right of the frame, foreground, looks toward Wilma off-screen.\n- KEY BACKGROUND ELEMENTS: cracked stadium seat (cracked) — Only its angled upper edge is visible behind Tony; used as Peripheral spatial reminder of the ruined stands and Tony’s loss.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated afternoon ambient light is shaped into restrained low-to-moderate-key contrast across Tony’s fixed eyes and compressed mouth.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the root-split concrete ruins, packed forest soil, encroaching foliage, and subdued afternoon light from the reference. Exclude restored stadium fixtures, active spectators, modern event equipment, or any clean undamaged surfaces.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The baseball stadium remains half-buried in the forest, with its scoreboard torn by roots and the faint text “...ANTON RAILRIDERS” still visible. A cracked spectator chair remains by Tony's hand. 토니(앤서니 로저스): He stands rigid after hearing the blunt statement, wearing the jumper belt with its added silver weight and carrying his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): afternoon.\n\nSHOT TEXT (authoritative, Korean): 입술을 굳게 다문 채 눈동자를 고정하고 굳어 있는 토니의 얼굴\n\nLOCATION (lock): The overgrown exterior spectator stands of the ruined stadium, among cracked seats invaded by the forest. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-right of the frame, foreground, looks toward Wilma off-screen.\n- KEY BACKGROUND ELEMENTS: cracked stadium seat (cracked) — Only its angled upper edge is visible behind Tony; used as Peripheral spatial reminder of the ruined stands and Tony’s loss.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated afternoon ambient light is shaped into restrained low-to-moderate-key contrast across Tony’s fixed eyes and compressed mouth.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the root-split concrete ruins, packed forest soil, encroaching foliage, and subdued afternoon light from the reference. Exclude restored stadium fixtures, active spectators, modern event equipment, or any clean undamaged surfaces.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The baseball stadium remains half-buried in the forest, with its scoreboard torn by roots and the faint text “...ANTON RAILRIDERS” still visible. A cracked spectator chair remains by Tony's hand. 토니(앤서니 로저스): He stands rigid after hearing the blunt statement, wearing the jumper belt with its added silver weight and carrying his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "토니의 시선이 화면 왼쪽 전면에 있는 여성의 머리를 향하고 있음.",
    "built_space": "콘크리트 계단과 금이 간 관중석 의자 하나가 토니의 등 뒤에 배치되어 있음.",
    "entities": "토니(참조 이미지의 외모와 일치)와 프롬프트에 허용되지 않은 여성의 뒷모습(어깨와 머리)이 보임.",
    "hard_violations": [
     "[gemini-pro] 프롬프트에 명시되지 않은 추가 인물(화면 밖 윌마)이 프레임 내에 등장함",
     "[gpt] 화면 밖에 있어야 하는 윌마로 보이는 여성을 전경에 추가해, 허용된 유일한 가시 인물인 토니 외의 인물을 등장시켰다."
    ],
    "physics": "두 인물 모두 바닥에 서서 안정적으로 지지받고 있음."
   },
   {
    "label": "B",
    "direction": "토니의 시선이 화면 밖 왼쪽을 고정하여 응시하고 있음.",
    "built_space": "식물로 뒤덮인 낡은 경기장 스탠드와 여러 개의 금이 간 관중석 의자가 뒤편에 배치되어 있음.",
    "entities": "로프와 은색 무게추가 달린 벨트를 착용한 토니(참조 이미지와 일치)가 단독으로 등장함.",
    "hard_violations": [],
    "physics": "토니는 지면에 서서 몸의 하중을 안정적으로 지지하고 있음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "화면 밖에 있어야 할 윌마를 화면 안 전면에 등장시킴으로써, 명시된 인물 외에는 추가하지 말라는 지침을 심각하게 위반했습니다."
       },
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "단일 인물 규칙과 소품, 굳어있는 표정 연기를 잘 구현했으나, 요구된 클로즈업 숏보다 프레이밍이 넓게 잡혀 배경 요소가 더 많이 노출된 점이 아쉽습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선이 화면 왼쪽 전면에 있는 여성의 머리를 향하고 있음.",
        "built_space": "콘크리트 계단과 금이 간 관중석 의자 하나가 토니의 등 뒤에 배치되어 있음.",
        "entities": "토니(참조 이미지의 외모와 일치)와 프롬프트에 허용되지 않은 여성의 뒷모습(어깨와 머리)이 보임.",
        "hard_violations": [
         "프롬프트에 명시되지 않은 추가 인물(화면 밖 윌마)이 프레임 내에 등장함"
        ],
        "physics": "두 인물 모두 바닥에 서서 안정적으로 지지받고 있음."
       },
       {
        "label": "B",
        "direction": "토니의 시선이 화면 밖 왼쪽을 고정하여 응시하고 있음.",
        "built_space": "식물로 뒤덮인 낡은 경기장 스탠드와 여러 개의 금이 간 관중석 의자가 뒤편에 배치되어 있음.",
        "entities": "로프와 은색 무게추가 달린 벨트를 착용한 토니(참조 이미지와 일치)가 단독으로 등장함.",
        "hard_violations": [],
        "physics": "토니는 지면에 서서 몸의 하중을 안정적으로 지지하고 있음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "화면 밖에 있어야 할 윌마를 화면 안 전면에 등장시킴으로써, 명시된 인물 외에는 추가하지 말라는 지침을 심각하게 위반했습니다."
       },
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "단일 인물 규칙과 소품, 굳어있는 표정 연기를 잘 구현했으나, 요구된 클로즈업 숏보다 프레이밍이 넓게 잡혀 배경 요소가 더 많이 노출된 점이 아쉽습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선이 화면 왼쪽 전면에 있는 여성의 머리를 향하고 있음.",
        "built_space": "콘크리트 계단과 금이 간 관중석 의자 하나가 토니의 등 뒤에 배치되어 있음.",
        "entities": "토니(참조 이미지의 외모와 일치)와 프롬프트에 허용되지 않은 여성의 뒷모습(어깨와 머리)이 보임.",
        "hard_violations": [
         "프롬프트에 명시되지 않은 추가 인물(화면 밖 윌마)이 프레임 내에 등장함"
        ],
        "physics": "두 인물 모두 바닥에 서서 안정적으로 지지받고 있음."
       },
       {
        "label": "B",
        "direction": "토니의 시선이 화면 밖 왼쪽을 고정하여 응시하고 있음.",
        "built_space": "식물로 뒤덮인 낡은 경기장 스탠드와 여러 개의 금이 간 관중석 의자가 뒤편에 배치되어 있음.",
        "entities": "로프와 은색 무게추가 달린 벨트를 착용한 토니(참조 이미지와 일치)가 단독으로 등장함.",
        "hard_violations": [],
        "physics": "토니는 지면에 서서 몸의 하중을 안정적으로 지지하고 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "토니만 중간 오른쪽에 두고 화면 밖 왼쪽을 응시하는 배치는 맞지만, 얼굴보다 상반신을 넓게 담았고 입술이 벌어져 있으며 좌석의 윗모서리만 보여야 한다는 배경 조건도 어겼다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "더 밀착된 얼굴과 굳게 다문 입술은 좋지만, 화면 밖에 있어야 할 윌마를 전경에 실제 인물로 넣은 치명적 위반 때문에 사용할 수 없다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 얼굴과 눈은 화면 왼쪽, 카메라 축에서 약간 벗어난 화면 밖 대상을 향한다. 따라서 화면 밖 윌마를 바라보는 방향으로 읽히며 무기나 다른 지향성 물체는 없다.",
        "built_space": "토니는 숲에 잠식된 낡은 관중석 통로에 서 있다. 뒤쪽 오른편에 금이 간 좌석이 여러 개 보이고, 콘크리트 난간·계단·뿌리와 이끼도 확인된다. 그러나 요구된 단 하나의 좌석 ‘기울어진 윗모서리’만 주변적으로 보이는 구성이 아니라 좌석의 등받이와 구조가 크게 드러난다.",
        "entities": "보이는 사람은 미국인 30대 남성 토니 한 명뿐이며 얼굴·머리·체격은 캐릭터 참고와 대체로 부합한다. 구조용 로프는 어깨에 걸려 있고 은색 추로 읽히는 금속 장비도 가슴에 있다. 다만 참고 의상의 헬멧은 없고, 표정은 눈이 고정된 듯 보이지만 입술이 굳게 다물리지 않고 살짝 벌어져 있다.",
        "hard_violations": [],
        "physics": "토니는 직립한 상반신으로 보이며 다리와 발은 프레이밍 밖이라 지면 접촉은 확인되지 않지만 부유하는 정황은 없다. 로프는 어깨에 걸쳐져 중력에 따라 아래로 늘어지고, 금속 장비는 하네스에 고정되어 있어 모두 지지 상태가 자연스럽다."
       },
       {
        "label": "B",
        "direction": "토니의 눈은 화면 왼쪽 전경에 보이는 여성의 얼굴을 정확히 향한다. 시선 표적 자체는 윌마로 읽히지만, 지시된 ‘화면 밖 윌마’가 아니라 화면 안 인물을 직접 바라본다.",
        "built_space": "토니는 균열 난 콘크리트 단과 숲에 잠식된 관중석 사이에 있다. 바로 뒤에 금이 크게 간 좌석 등받이 전체가 보이며 주변에도 여러 좌석이 있다. 요구된 좌석의 기울어진 윗모서리만 주변적으로 보이는 배경 구성이 아니다.",
        "entities": "토니는 미국인 30대 남성으로 얼굴·머리·체격이 참고와 대체로 맞고, 구조용 로프가 오른쪽 어깨에 걸려 있다. 입술은 압축되어 있고 눈도 여성에게 고정되어 있다. 그러나 왼쪽 전경에 여성의 머리·얼굴·어깨가 추가로 보이며, 이는 토니만 등장해야 하고 윌마는 화면 밖이어야 한다는 명시를 위반한다. 토니의 참고 의상에 있는 헬멧도 없다.",
        "hard_violations": [
         "화면 밖에 있어야 하는 윌마로 보이는 여성을 전경에 추가해, 허용된 유일한 가시 인물인 토니 외의 인물을 등장시켰다."
        ],
        "physics": "토니와 전경 여성 모두 하체가 프레임 밖이어서 직접적인 접지점은 보이지 않지만 정상적인 직립 또는 착석 자세로 읽히며 공중에 뜬 정황은 없다. 로프는 토니의 어깨가 지지하고 아래로 자연스럽게 늘어진다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "토니만 중간 오른쪽에 두고 화면 밖 왼쪽을 응시하는 배치는 맞지만, 얼굴보다 상반신을 넓게 담았고 입술이 벌어져 있으며 좌석의 윗모서리만 보여야 한다는 배경 조건도 어겼다."
       },
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "더 밀착된 얼굴과 굳게 다문 입술은 좋지만, 화면 밖에 있어야 할 윌마를 전경에 실제 인물로 넣은 치명적 위반 때문에 사용할 수 없다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "토니의 얼굴과 눈은 화면 왼쪽, 카메라 축에서 약간 벗어난 화면 밖 대상을 향한다. 따라서 화면 밖 윌마를 바라보는 방향으로 읽히며 무기나 다른 지향성 물체는 없다.",
        "built_space": "토니는 숲에 잠식된 낡은 관중석 통로에 서 있다. 뒤쪽 오른편에 금이 간 좌석이 여러 개 보이고, 콘크리트 난간·계단·뿌리와 이끼도 확인된다. 그러나 요구된 단 하나의 좌석 ‘기울어진 윗모서리’만 주변적으로 보이는 구성이 아니라 좌석의 등받이와 구조가 크게 드러난다.",
        "entities": "보이는 사람은 미국인 30대 남성 토니 한 명뿐이며 얼굴·머리·체격은 캐릭터 참고와 대체로 부합한다. 구조용 로프는 어깨에 걸려 있고 은색 추로 읽히는 금속 장비도 가슴에 있다. 다만 참고 의상의 헬멧은 없고, 표정은 눈이 고정된 듯 보이지만 입술이 굳게 다물리지 않고 살짝 벌어져 있다.",
        "hard_violations": [],
        "physics": "토니는 직립한 상반신으로 보이며 다리와 발은 프레이밍 밖이라 지면 접촉은 확인되지 않지만 부유하는 정황은 없다. 로프는 어깨에 걸쳐져 중력에 따라 아래로 늘어지고, 금속 장비는 하네스에 고정되어 있어 모두 지지 상태가 자연스럽다."
       },
       {
        "label": "A",
        "direction": "토니의 눈은 화면 왼쪽 전경에 보이는 여성의 얼굴을 정확히 향한다. 시선 표적 자체는 윌마로 읽히지만, 지시된 ‘화면 밖 윌마’가 아니라 화면 안 인물을 직접 바라본다.",
        "built_space": "토니는 균열 난 콘크리트 단과 숲에 잠식된 관중석 사이에 있다. 바로 뒤에 금이 크게 간 좌석 등받이 전체가 보이며 주변에도 여러 좌석이 있다. 요구된 좌석의 기울어진 윗모서리만 주변적으로 보이는 배경 구성이 아니다.",
        "entities": "토니는 미국인 30대 남성으로 얼굴·머리·체격이 참고와 대체로 맞고, 구조용 로프가 오른쪽 어깨에 걸려 있다. 입술은 압축되어 있고 눈도 여성에게 고정되어 있다. 그러나 왼쪽 전경에 여성의 머리·얼굴·어깨가 추가로 보이며, 이는 토니만 등장해야 하고 윌마는 화면 밖이어야 한다는 명시를 위반한다. 토니의 참고 의상에 있는 헬멧도 없다.",
        "hard_violations": [
         "화면 밖에 있어야 하는 윌마로 보이는 여성을 전경에 추가해, 허용된 유일한 가시 인물인 토니 외의 인물을 등장시켰다."
        ],
        "physics": "토니와 전경 여성 모두 하체가 프레임 밖이어서 직접적인 접지점은 보이지 않지만 정상적인 직립 또는 착석 자세로 읽히며 공중에 뜬 정황은 없다. 로프는 토니의 어깨가 지지하고 아래로 자연스럽게 늘어진다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 0.583,
    "B": 2.0
   },
   "adjusted": {
    "A": 0.333,
    "B": 2.0
   },
   "violations": {
    "A": [
     "[gemini-pro] 프롬프트에 명시되지 않은 추가 인물(화면 밖 윌마)이 프레임 내에 등장함",
     "[gpt] 화면 밖에 있어야 하는 윌마로 보이는 여성을 전경에 추가해, 허용된 유일한 가시 인물인 토니 외의 인물을 등장시켰다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "A": 333,
   "B": 2000
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 333,
    "verdict_ko": "화면 밖에 있어야 할 윌마를 화면 안 전면에 등장시킴으로써, 명시된 인물 외에는 추가하지 말라는 지침을 심각하게 위반했습니다.  ★위반: [gemini-pro] 프롬프트에 명시되지 않은 추가 인물(화면 밖 윌마)이 프레임 내에 등장함 / [gpt] 화면 밖에 있어야 하는 윌마로 보이는 여성을 전경에 추가해, 허용된 유일한 가시 인물인 토니 외의 인물을 등장시켰다."
   },
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "단일 인물 규칙과 소품, 굳어있는 표정 연기를 잘 구현했으나, 요구된 클로즈업 숏보다 프레이밍이 넓게 잡혀 배경 요소가 더 많이 노출된 점이 아쉽습니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S5sh2_sel.png",
    "asset_id": "8f75d7f9-bd39-480a-ad0e-08f12a9f86b6",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc85-646c-7669-9539-5f5c4363cc7b",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S5sh2"
  }
 },
 "S5sh5::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:45:04.824920+00:00",
  "fingerprint": "56babaebc6d63da9ec716924b80baaafddb0360de1f5afe4967a181f4e8ddd3d",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S5sh5_sel.png",
  "source_sha256": "4abc589c49b1da3cf85edd1739528d528a5eaff39bffa653f0d8b99d43563ee3",
  "file": "S5sh5_cine.png",
  "staged_sha256": "79cb135ef334b3655477c62a0f655d3c3de05ed6c52a3c68d18cc8149e3e8256",
  "latency_ms": 14578
 },
 "S5sh10::signage": {
  "fp": "a0db556df75fc7e7",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S5sh10": {
  "input_fingerprint": "9a05d7734cc443b3",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): afternoon.\n\nSHOT TEXT (authoritative, Korean): 숲의 깊은 어둠 속 어딘가를 향해 시선을 고정한 채 바짝 몸을 낮춘 두 사람의 긴장한 자세\n\nLOCATION (lock): An exterior patch beneath a tree beside the ruined stadium seating, facing into the dark surrounding forest. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward deep forest; 윌마 디어링 in the lower-left of the frame, midground, looks toward deep forest; deep forest in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: deep forest (dark beyond the crouched figures); used as Threat-facing negative space receiving both characters’ sightlines; stadium edge (ruined and partially surrounded by the forest) — A partial side edge recedes behind the crouched pair; used as Secondary depth marker behind the reaction.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Afternoon ambient light falls into tense low-key contrast toward the deep darkness of the forest they are watching.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 토니(앤서니 로저스) — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the half-buried concrete stadium, invasive roots, dense surrounding woods, and dim afternoon palette from the reference. Exclude any visible pursuer, moving machine, fresh artificial lights, or active stadium equipment; the metallic threat remains unseen.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The ruined stadium remains half-buried among the trees. One old seat has been pulled free and set beneath a tree. 윌마 디어링: She is alert in the stadium area, wearing her moss-smeared belt with one silver weight removed and carrying her moss-smeared gun. 토니(앤서니 로저스): He is alert in the stadium area, wearing the jumper belt with the added silver weight and carrying his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): afternoon.\n\nSHOT TEXT (authoritative, Korean): 숲의 깊은 어둠 속 어딘가를 향해 시선을 고정한 채 바짝 몸을 낮춘 두 사람의 긴장한 자세\n\nLOCATION (lock): An exterior patch beneath a tree beside the ruined stadium seating, facing into the dark surrounding forest. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward deep forest; 윌마 디어링 in the lower-left of the frame, midground, looks toward deep forest; deep forest in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: deep forest (dark beyond the crouched figures); used as Threat-facing negative space receiving both characters’ sightlines; stadium edge (ruined and partially surrounded by the forest) — A partial side edge recedes behind the crouched pair; used as Secondary depth marker behind the reaction.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Afternoon ambient light falls into tense low-key contrast toward the deep darkness of the forest they are watching.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 토니(앤서니 로저스) — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the half-buried concrete stadium, invasive roots, dense surrounding woods, and dim afternoon palette from the reference. Exclude any visible pursuer, moving machine, fresh artificial lights, or active stadium equipment; the metallic threat remains unseen.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The ruined stadium remains half-buried among the trees. One old seat has been pulled free and set beneath a tree. 윌마 디어링: She is alert in the stadium area, wearing her moss-smeared belt with one silver weight removed and carrying her moss-smeared gun. 토니(앤서니 로저스): He is alert in the stadium area, wearing the jumper belt with the added silver weight and carrying his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): afternoon.\n\nSHOT TEXT (authoritative, Korean): 숲의 깊은 어둠 속 어딘가를 향해 시선을 고정한 채 바짝 몸을 낮춘 두 사람의 긴장한 자세\n\nLOCATION (lock): An exterior patch beneath a tree beside the ruined stadium seating, facing into the dark surrounding forest. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward deep forest; 윌마 디어링 in the lower-left of the frame, midground, looks toward deep forest; deep forest in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: deep forest (dark beyond the crouched figures); used as Threat-facing negative space receiving both characters’ sightlines; stadium edge (ruined and partially surrounded by the forest) — A partial side edge recedes behind the crouched pair; used as Secondary depth marker behind the reaction.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Afternoon ambient light falls into tense low-key contrast toward the deep darkness of the forest they are watching.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 토니(앤서니 로저스) — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the half-buried concrete stadium, invasive roots, dense surrounding woods, and dim afternoon palette from the reference. Exclude any visible pursuer, moving machine, fresh artificial lights, or active stadium equipment; the metallic threat remains unseen.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The ruined stadium remains half-buried among the trees. One old seat has been pulled free and set beneath a tree. 윌마 디어링: She is alert in the stadium area, wearing her moss-smeared belt with one silver weight removed and carrying her moss-smeared gun. 토니(앤서니 로저스): He is alert in the stadium area, wearing the jumper belt with the added silver weight and carrying his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "토니와 윌마는 화면 왼쪽에서 몸을 바짝 낮춘 채, 오른쪽의 깊은 숲(위협이 도사리는 어둠)을 향해 시선을 고정하고 있습니다. 윌마가 든 총의 총구 역시 숲을 향하고 있습니다.",
    "built_space": "폐허가 된 경기장의 가장자리입니다. 화면 앞쪽에 이끼 낀 콘크리트 장벽이 있고, 인물들 우측 뒤편으로 경기장 좌석들이 배경을 향해 물러서듯 배치되어 있습니다. 프롬프트에 명시된 '나무 아래 분리되어 놓인 낡은 좌석'은 보이지 않습니다.",
    "entities": "토니는 참고 이미지의 인물과 일치하며 구명 밧줄을 메고 있으나 추가된 은색 추는 명확히 보이지 않습니다. 윌마 역시 참고 이미지와 일치하는 외모에 녹색 점프수트를 입고 있으며, 지시대로 이끼가 잔뜩 묻은 총을 들고 있습니다.",
    "hard_violations": [],
    "physics": "두 인물 모두 지면에 발을 딛고 덩굴과 장벽에 의지해 안정적으로 몸을 낮추고 있습니다. 윌마의 왼손은 장벽을 자연스럽게 짚고 있으며 모든 자세가 지탱되어 있습니다."
   },
   {
    "label": "B",
    "direction": "두 인물은 화면 왼쪽에서 몸을 낮추고 오른쪽의 깊고 어두운 숲을 응시하고 있습니다. 윌마는 총을 숲 방향으로 겨누고 있습니다.",
    "built_space": "화면 왼쪽 뒤로 경기장 좌석이 늘어서 있고, 나무 옆 바닥에는 프롬프트가 지시한 분리된 낡은 경기장 좌석이 놓여 있습니다.",
    "entities": "토니는 밧줄을 메고 있으나 은색 추가 보이지 않으며, 윌마는 녹색 점프수트를 입었으나 그녀가 든 총은 지시와 달리 이끼가 전혀 없이 깨끗합니다.",
    "hard_violations": [
     "[gemini-pro] 해부학적으로 불가능한 신체 구조 (토니의 오른쪽 다리가 상체 위치와 무관하게 윌마의 뒤편으로 비정상적으로 길게 뻗어 있음)"
    ],
    "physics": "윌마는 땅에 발을 딛고 총을 든 채 안정적으로 몸을 웅크리고 있으나, 토니의 경우 상체는 윌마 바로 뒤에 있음에도 불구하고 오른쪽 다리가 해부학적 한계를 넘어 뒤로 길게 뻗어 있어 물리적으로 불가능한 자세를 취하고 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "나무 아래 분리된 좌석 소품이 누락되었으나, 지정된 구도와 조명 분위기를 정확히 구현했으며 물리적으로 자연스러운 자세를 보여줍니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "분리된 좌석 소품은 포함되었으나, 토니의 오른쪽 다리가 해부학적으로 불가능하게 연장되어 심각한 물리적 오류를 발생시켰습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니와 윌마는 화면 왼쪽에서 몸을 바짝 낮춘 채, 오른쪽의 깊은 숲(위협이 도사리는 어둠)을 향해 시선을 고정하고 있습니다. 윌마가 든 총의 총구 역시 숲을 향하고 있습니다.",
        "built_space": "폐허가 된 경기장의 가장자리입니다. 화면 앞쪽에 이끼 낀 콘크리트 장벽이 있고, 인물들 우측 뒤편으로 경기장 좌석들이 배경을 향해 물러서듯 배치되어 있습니다. 프롬프트에 명시된 '나무 아래 분리되어 놓인 낡은 좌석'은 보이지 않습니다.",
        "entities": "토니는 참고 이미지의 인물과 일치하며 구명 밧줄을 메고 있으나 추가된 은색 추는 명확히 보이지 않습니다. 윌마 역시 참고 이미지와 일치하는 외모에 녹색 점프수트를 입고 있으며, 지시대로 이끼가 잔뜩 묻은 총을 들고 있습니다.",
        "hard_violations": [],
        "physics": "두 인물 모두 지면에 발을 딛고 덩굴과 장벽에 의지해 안정적으로 몸을 낮추고 있습니다. 윌마의 왼손은 장벽을 자연스럽게 짚고 있으며 모든 자세가 지탱되어 있습니다."
       },
       {
        "label": "B",
        "direction": "두 인물은 화면 왼쪽에서 몸을 낮추고 오른쪽의 깊고 어두운 숲을 응시하고 있습니다. 윌마는 총을 숲 방향으로 겨누고 있습니다.",
        "built_space": "화면 왼쪽 뒤로 경기장 좌석이 늘어서 있고, 나무 옆 바닥에는 프롬프트가 지시한 분리된 낡은 경기장 좌석이 놓여 있습니다.",
        "entities": "토니는 밧줄을 메고 있으나 은색 추가 보이지 않으며, 윌마는 녹색 점프수트를 입었으나 그녀가 든 총은 지시와 달리 이끼가 전혀 없이 깨끗합니다.",
        "hard_violations": [
         "해부학적으로 불가능한 신체 구조 (토니의 오른쪽 다리가 상체 위치와 무관하게 윌마의 뒤편으로 비정상적으로 길게 뻗어 있음)"
        ],
        "physics": "윌마는 땅에 발을 딛고 총을 든 채 안정적으로 몸을 웅크리고 있으나, 토니의 경우 상체는 윌마 바로 뒤에 있음에도 불구하고 오른쪽 다리가 해부학적 한계를 넘어 뒤로 길게 뻗어 있어 물리적으로 불가능한 자세를 취하고 있습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "나무 아래 분리된 좌석 소품이 누락되었으나, 지정된 구도와 조명 분위기를 정확히 구현했으며 물리적으로 자연스러운 자세를 보여줍니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "분리된 좌석 소품은 포함되었으나, 토니의 오른쪽 다리가 해부학적으로 불가능하게 연장되어 심각한 물리적 오류를 발생시켰습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "토니와 윌마는 화면 왼쪽에서 몸을 바짝 낮춘 채, 오른쪽의 깊은 숲(위협이 도사리는 어둠)을 향해 시선을 고정하고 있습니다. 윌마가 든 총의 총구 역시 숲을 향하고 있습니다.",
        "built_space": "폐허가 된 경기장의 가장자리입니다. 화면 앞쪽에 이끼 낀 콘크리트 장벽이 있고, 인물들 우측 뒤편으로 경기장 좌석들이 배경을 향해 물러서듯 배치되어 있습니다. 프롬프트에 명시된 '나무 아래 분리되어 놓인 낡은 좌석'은 보이지 않습니다.",
        "entities": "토니는 참고 이미지의 인물과 일치하며 구명 밧줄을 메고 있으나 추가된 은색 추는 명확히 보이지 않습니다. 윌마 역시 참고 이미지와 일치하는 외모에 녹색 점프수트를 입고 있으며, 지시대로 이끼가 잔뜩 묻은 총을 들고 있습니다.",
        "hard_violations": [],
        "physics": "두 인물 모두 지면에 발을 딛고 덩굴과 장벽에 의지해 안정적으로 몸을 낮추고 있습니다. 윌마의 왼손은 장벽을 자연스럽게 짚고 있으며 모든 자세가 지탱되어 있습니다."
       },
       {
        "label": "B",
        "direction": "두 인물은 화면 왼쪽에서 몸을 낮추고 오른쪽의 깊고 어두운 숲을 응시하고 있습니다. 윌마는 총을 숲 방향으로 겨누고 있습니다.",
        "built_space": "화면 왼쪽 뒤로 경기장 좌석이 늘어서 있고, 나무 옆 바닥에는 프롬프트가 지시한 분리된 낡은 경기장 좌석이 놓여 있습니다.",
        "entities": "토니는 밧줄을 메고 있으나 은색 추가 보이지 않으며, 윌마는 녹색 점프수트를 입었으나 그녀가 든 총은 지시와 달리 이끼가 전혀 없이 깨끗합니다.",
        "hard_violations": [
         "해부학적으로 불가능한 신체 구조 (토니의 오른쪽 다리가 상체 위치와 무관하게 윌마의 뒤편으로 비정상적으로 길게 뻗어 있음)"
        ],
        "physics": "윌마는 땅에 발을 딛고 총을 든 채 안정적으로 몸을 웅크리고 있으나, 토니의 경우 상체는 윌마 바로 뒤에 있음에도 불구하고 오른쪽 다리가 해부학적 한계를 넘어 뒤로 길게 뻗어 있어 물리적으로 불가능한 자세를 취하고 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "두 인물이 지정된 왼쪽 위치에서 몸을 낮추고 중간 오른쪽의 깊고 어두운 숲을 함께 응시하며, 뒤로 물러나는 폐허 관람석 가장자리와 분리된 낡은 좌석까지 가장 충실하게 구현했다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "인물·의상·낮은 자세는 대체로 맞지만 두 시선이 깊은 숲보다 화면 오른쪽 관람석 쪽으로 향하고, 위협을 받는 숲의 음영 공간도 중간 오른쪽이 아닌 중앙에 놓여 핵심 연출이 어긋난다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니와 윌마는 모두 몸과 얼굴을 화면 오른쪽으로 향하고 있으며, 두 시선은 중간 오른쪽 배경의 짙고 어두운 숲속에 닿는다. 윌마의 총구도 같은 방향의 숲을 향하되 안전하게 아래쪽으로 낮춰져 있다.",
        "built_space": "이끼 낀 콘크리트 관람석 가장자리와 낡은 고정 좌석 여러 개가 두 사람 뒤에서 후방으로 물러난다. 나무 아래 통로에는 고정 열에서 빠져나온 것으로 보이는 낡은 좌석 한 개가 따로 놓여 있다. 토니는 나무 옆 중간 왼쪽, 윌마는 그 앞의 아래 왼쪽에 자리하며 구조물과 충돌하지 않는다. 반사면이나 광학적으로 불가능한 반사는 없다.",
        "entities": "토니는 미국인 30대 남성으로 보이며 기준과 가까운 얼굴·짧은 갈색 머리·어두운 작업복을 갖췄고, 어깨에 구조용 밧줄과 은색 장비 무게를 지닌다. 윌마는 미국인 20대 후반 여성으로 보이며 기준과 가까운 금발 묶음머리·녹색 점프수트·장갑·장비 벨트를 착용하고 총을 들고 있다. 추가 인물이나 보이는 위협 기계는 없으며, 장소는 뿌리와 이끼가 침범한 반매몰 콘크리트 경기장이다.",
        "hard_violations": [],
        "physics": "두 사람 모두 실제 지면과 뿌리에 체중을 싣고 낮게 웅크린다. 토니는 굽힌 다리와 뒤로 뻗은 다리로 지지되고, 윌마는 한쪽 무릎과 반대쪽 발로 지지된다. 윌마의 총은 양손이 손잡이와 앞부분을 확실히 붙들고 있고, 토니의 밧줄은 어깨와 몸에 걸려 있어 뜨는 물체가 없다."
       },
       {
        "label": "B",
        "direction": "토니와 윌마는 모두 화면 오른쪽을 바라보지만, 보이는 시선의 도착점은 깊은 숲보다 오른쪽 위의 낡은 관람석과 경기장 가장자리 쪽에 가깝다. 어두운 숲은 두 사람 뒤쪽 중앙에 있어 시선이 받는 중간 오른쪽 위협 공간으로 기능하지 않는다. 윌마의 총구는 오른쪽 아래로 낮춰져 있다.",
        "built_space": "이끼 낀 콘크리트 벽과 침입한 굵은 뿌리가 통로 양쪽에 있고, 낡은 고정 관람석 여러 개는 화면 오른쪽 위에 모여 있다. 토니는 중간 왼쪽의 뿌리 위, 윌마는 아래 왼쪽의 통로 가장자리에 웅크리고 있어 접촉은 가능하다. 그러나 분리해 나무 아래 둔 낡은 좌석 한 개는 보이지 않고, 경기장 가장자리가 인물 뒤의 보조 깊이 표지라기보다 화면 오른쪽 전경을 크게 차지한다. 불가능한 반사는 없다.",
        "entities": "토니는 기준과 가까운 30대 남성 얼굴·짧은 갈색 머리·어두운 작업복과 구조용 밧줄을 갖췄다. 윌마는 기준과 가까운 20대 후반 여성 얼굴·묶은 금발·녹색 점프수트·장비 벨트를 갖추고 총을 든다. 추가 인물이나 기계 위협은 없고, 반매몰 콘크리트 경기장·뿌리·숲이라는 장소 요소는 대체로 일치한다. 토니의 추가 은색 무게는 A보다 식별이 어렵다.",
        "hard_violations": [],
        "physics": "토니는 한쪽 무릎과 반대쪽 발을 뿌리 덮인 지면에 대고 있으며, 윌마도 굽힌 다리와 발로 체중을 받친다. 윌마는 총을 양손으로 잡고 있고 토니의 밧줄은 어깨에 걸려 있다. 모든 몸과 휴대품에 지지점이 있으며 부유하거나 물리적으로 불가능한 동작은 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "두 인물이 지정된 왼쪽 위치에서 몸을 낮추고 중간 오른쪽의 깊고 어두운 숲을 함께 응시하며, 뒤로 물러나는 폐허 관람석 가장자리와 분리된 낡은 좌석까지 가장 충실하게 구현했다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "인물·의상·낮은 자세는 대체로 맞지만 두 시선이 깊은 숲보다 화면 오른쪽 관람석 쪽으로 향하고, 위협을 받는 숲의 음영 공간도 중간 오른쪽이 아닌 중앙에 놓여 핵심 연출이 어긋난다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "토니와 윌마는 모두 몸과 얼굴을 화면 오른쪽으로 향하고 있으며, 두 시선은 중간 오른쪽 배경의 짙고 어두운 숲속에 닿는다. 윌마의 총구도 같은 방향의 숲을 향하되 안전하게 아래쪽으로 낮춰져 있다.",
        "built_space": "이끼 낀 콘크리트 관람석 가장자리와 낡은 고정 좌석 여러 개가 두 사람 뒤에서 후방으로 물러난다. 나무 아래 통로에는 고정 열에서 빠져나온 것으로 보이는 낡은 좌석 한 개가 따로 놓여 있다. 토니는 나무 옆 중간 왼쪽, 윌마는 그 앞의 아래 왼쪽에 자리하며 구조물과 충돌하지 않는다. 반사면이나 광학적으로 불가능한 반사는 없다.",
        "entities": "토니는 미국인 30대 남성으로 보이며 기준과 가까운 얼굴·짧은 갈색 머리·어두운 작업복을 갖췄고, 어깨에 구조용 밧줄과 은색 장비 무게를 지닌다. 윌마는 미국인 20대 후반 여성으로 보이며 기준과 가까운 금발 묶음머리·녹색 점프수트·장갑·장비 벨트를 착용하고 총을 들고 있다. 추가 인물이나 보이는 위협 기계는 없으며, 장소는 뿌리와 이끼가 침범한 반매몰 콘크리트 경기장이다.",
        "hard_violations": [],
        "physics": "두 사람 모두 실제 지면과 뿌리에 체중을 싣고 낮게 웅크린다. 토니는 굽힌 다리와 뒤로 뻗은 다리로 지지되고, 윌마는 한쪽 무릎과 반대쪽 발로 지지된다. 윌마의 총은 양손이 손잡이와 앞부분을 확실히 붙들고 있고, 토니의 밧줄은 어깨와 몸에 걸려 있어 뜨는 물체가 없다."
       },
       {
        "label": "A",
        "direction": "토니와 윌마는 모두 화면 오른쪽을 바라보지만, 보이는 시선의 도착점은 깊은 숲보다 오른쪽 위의 낡은 관람석과 경기장 가장자리 쪽에 가깝다. 어두운 숲은 두 사람 뒤쪽 중앙에 있어 시선이 받는 중간 오른쪽 위협 공간으로 기능하지 않는다. 윌마의 총구는 오른쪽 아래로 낮춰져 있다.",
        "built_space": "이끼 낀 콘크리트 벽과 침입한 굵은 뿌리가 통로 양쪽에 있고, 낡은 고정 관람석 여러 개는 화면 오른쪽 위에 모여 있다. 토니는 중간 왼쪽의 뿌리 위, 윌마는 아래 왼쪽의 통로 가장자리에 웅크리고 있어 접촉은 가능하다. 그러나 분리해 나무 아래 둔 낡은 좌석 한 개는 보이지 않고, 경기장 가장자리가 인물 뒤의 보조 깊이 표지라기보다 화면 오른쪽 전경을 크게 차지한다. 불가능한 반사는 없다.",
        "entities": "토니는 기준과 가까운 30대 남성 얼굴·짧은 갈색 머리·어두운 작업복과 구조용 밧줄을 갖췄다. 윌마는 기준과 가까운 20대 후반 여성 얼굴·묶은 금발·녹색 점프수트·장비 벨트를 갖추고 총을 든다. 추가 인물이나 기계 위협은 없고, 반매몰 콘크리트 경기장·뿌리·숲이라는 장소 요소는 대체로 일치한다. 토니의 추가 은색 무게는 A보다 식별이 어렵다.",
        "hard_violations": [],
        "physics": "토니는 한쪽 무릎과 반대쪽 발을 뿌리 덮인 지면에 대고 있으며, 윌마도 굽힌 다리와 발로 체중을 받친다. 윌마는 총을 양손으로 잡고 있고 토니의 밧줄은 어깨에 걸려 있다. 모든 몸과 휴대품에 지지점이 있으며 부유하거나 물리적으로 불가능한 동작은 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.667,
    "B": 1.25
   },
   "adjusted": {
    "A": 1.667,
    "B": 1.0
   },
   "violations": {
    "B": [
     "[gemini-pro] 해부학적으로 불가능한 신체 구조 (토니의 오른쪽 다리가 상체 위치와 무관하게 윌마의 뒤편으로 비정상적으로 길게 뻗어 있음)"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "agreed": false
  },
  "totals": {
   "A": 1667,
   "B": 1000
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1667,
    "verdict_ko": "나무 아래 분리된 좌석 소품이 누락되었으나, 지정된 구도와 조명 분위기를 정확히 구현했으며 물리적으로 자연스러운 자세를 보여줍니다."
   },
   {
    "label": "B",
    "score": 1000,
    "verdict_ko": "분리된 좌석 소품은 포함되었으나, 토니의 오른쪽 다리가 해부학적으로 불가능하게 연장되어 심각한 물리적 오류를 발생시켰습니다.  ★위반: [gemini-pro] 해부학적으로 불가능한 신체 구조 (토니의 오른쪽 다리가 상체 위치와 무관하게 윌마의 뒤편으로 비정상적으로 길게 뻗어 있음)"
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The clothing, hair and overall look of 토니(앤서니 로저스) — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S5sh5_sel.png",
    "asset_id": "3f82aaa7-f7a8-4940-b2c5-249ce12e18fb",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc88-f8aa-7897-9c85-6a04ea87639d",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S5sh5"
  }
 },
 "S5sh10::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:46:44.621250+00:00",
  "fingerprint": "93e851000439417135d7b299a75c2d6e8edf2185478ab8a080eeb855f28857a0",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S5sh10_sel.png",
  "source_sha256": "f9ea1835f15d5b3e2e892028bf045bb98e150a5829f232e802078d98c0a437a4",
  "file": "S5sh10_cine.png",
  "staged_sha256": "2ace2f937bb73e72c8dac3ff7cf2cf68f0110c6b90fc24fd42f9ce5473ea790e",
  "latency_ms": 15488
 },
 "S6sh4::signage": {
  "fp": "6d8cb6cfb1f43580",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "era_assess::1517515e26fa2d79": {
  "subjects": [],
  "subject_text": "산림 협곡 접근로와 절벽 사이 상공\n울창한 숲길이 양쪽 절벽 사이의 깊은 협곡으로 이어지는 산악 지형. 해질녘 빛 아래 바위와 수목이 층층이 겹친다.",
  "identity": "canonical",
  "scope_id": "L09",
  "scope_role": "location_exterior",
  "scope_sha": "e8761361effa538a"
 },
 "S6sh4::bgfirst_bg": {
  "input_fingerprint": "cee533880ecdaeb9",
  "prompt": "Turn the attached storyboard sketch into a photorealistic film still of\nits LOCATION ONLY, keeping the figures exactly as they are.\n\nKEEP EXACTLY: the camera framing, the horizon, and where every element\nsits in the frame. Each mannequin figure stays a plain grey featureless\nmannequin standing in the very same spot, at the same size, in the same\npose, turned the same way — do not turn them into people, do not move,\nrotate, mirror or re-pose them, do not add or remove figures.\n\nBUILD PHOTOREALISTICALLY: everything that is not a figure — ground,\nsurfacing, structures, vegetation, sky, water, distance. The bare lines\nof the sketch are a layout guide; replace them with the real materials,\ndepth and lighting of the place described below.\n\nSHOT TEXT this background must serve (Korean): 불을 뿜는 탄자가 지나가는 가운데, 이를 피해 양옆 허공으로 몸을 비틀어 막 날아오른 윌마와 토니의 도약 찰나\n\nLOCATION (lock): Open exterior airspace above the rocky canyon approach, between the forested ground and the cliff sides.\n\nTIME OF DAY (lock): sunset.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-left of the frame, midground, moves toward left side of the firing line; 토니(앤서니 로저스) in the middle-right of the frame, midground, moves toward right side of the firing line; projectile crossing gap in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: flaming projectile (in flight between Wilma and Tony) — Its forward axis crosses obliquely through the central space between the two bodies; used as Central moving separator and immediate threat cue; canyon approach rocks (in the projectile’s path); used as Ground and depth reference beneath the airborne split; canyon approach (open between the opposing jumps); used as Wide environmental context that makes the opposite trajectories readable.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated sunset ambient light is interrupted by the explicitly fiery projectile, creating a brief hard accent within restrained action contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHINGS AT THIS PLACE:\nThe canyon approach occupies the near side of the site and terminates at the near edge of an open canyon chasm. A tree beside the canyonward portion of the approach provides a fixed vertical landmark near that edge. The open canyon airspace extends between the approach side and the opposite cliff. Across the chasm, the top of the opposite cliff forms the far-side ground toward which the canyon-spanning trajectory descends. A forest directly adjoins the landward side of that cliff-top ground, allowing movement from the exposed rim into the trees.\n- approach: canyon approach route\n- tree: canyon-edge tree beside the approach\n- canyon: open canyon chasm\n- cliff: opposite cliff and its rim\n- forest: forest adjoining the opposite cliff rim\n\nWORLD FACTS (creator-confirmed — always true):\n- Region (real-world reference): 미국 펜실베이니아의 산림·산악권\n- Era: 25세기 초반의 먼 미래(2419년), 21세기 문명의 유산이 폐허와 구형 장비로 잔존하는 시대\n- 현재 사용되는 미래 기술과 오래되어 훼손·매몰된 구문명 유산을 명확히 구분한다. 구문명 요소는 보존된 현대 공간처럼 만들지 말고 장기간의 방치와 자연 침식이 축적된 외형으로 표현한다.\n- 사용자는 추진 비행이나 공중 정지 대신 중력의 영향을 받는 포물선 궤도로 이동시킨다. 검은 중력 제어판과 탈착식 은색 추를 벨트의 핵심 구성으로 유지하고, 추의 수와 장착 상태에 따라 도약 높이와 하강성이 달라 보이게 한다.\n- 탐색 장치는 작은 발광점이 아니라 렌즈와 금속성 구조를 지닌 물리적 센서로 표현한다. 소거 효과는 폭발이나 화염보다 경계가 비정상적으로 깨끗한 결손과 압력 교란으로 나타내며, 그래픽한 잔해 표현은 피한다.\n- 구시대 스마트폰은 익숙한 현대형 물건으로 유지하되 미래형 단말기로 재설계하지 않는다. 차폐 상태에서는 방수 주머니와 금속성 결속을 분명히 보여주고, 음성 인공지능은 별도의 인간형 실체 없이 기기에서 발생하는 인터페이스로 처리한다.\n- 위장된 표면은 닫혀 있을 때 주변 지형과 연속되어야 하며, 개방 시에만 인공적인 경계와 내부 구조를 드러낸다. 통신 정보와 탐지 표식은 물리 공간에 떠 있는 장식이 아니라 활성화된 판면이나 지도 인터페이스에 종속시킨다.\n\nTHINGS THAT LIVE AT THIS PLACE (what they are, not where they go): the\nlist above names what permanently belongs to this site. Where each one\nsits in this frame is decided by the attached sketch alone — do not add\nor reposition anything on the strength of the list, build only what the\nsketch already has lines for, and do not invent facilities that are not\nlisted.\n\nNo readable writing anywhere: surfaces that would carry writing may be\npresent, but stage any wording out of legibility — an oblique angle,\ndistance, shallow focus. No captions, watermarks or overlay text,\nand none of the sketch's diagram symbols, lines or markers anywhere in\nthe image.",
  "effective_prompt": "Turn the attached storyboard sketch into a photorealistic film still of\nits LOCATION ONLY, keeping the figures exactly as they are.\n\nKEEP EXACTLY: the camera framing, the horizon, and where every element\nsits in the frame. Each mannequin figure stays a plain grey featureless\nmannequin standing in the very same spot, at the same size, in the same\npose, turned the same way — do not turn them into people, do not move,\nrotate, mirror or re-pose them, do not add or remove figures.\n\nBUILD PHOTOREALISTICALLY: everything that is not a figure — ground,\nsurfacing, structures, vegetation, sky, water, distance. The bare lines\nof the sketch are a layout guide; replace them with the real materials,\ndepth and lighting of the place described below.\n\nSHOT TEXT this background must serve (Korean): 불을 뿜는 탄자가 지나가는 가운데, 이를 피해 양옆 허공으로 몸을 비틀어 막 날아오른 윌마와 토니의 도약 찰나\n\nLOCATION (lock): Open exterior airspace above the rocky canyon approach, between the forested ground and the cliff sides.\n\nTIME OF DAY (lock): sunset.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-left of the frame, midground, moves toward left side of the firing line; 토니(앤서니 로저스) in the middle-right of the frame, midground, moves toward right side of the firing line; projectile crossing gap in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: flaming projectile (in flight between Wilma and Tony) — Its forward axis crosses obliquely through the central space between the two bodies; used as Central moving separator and immediate threat cue; canyon approach rocks (in the projectile’s path); used as Ground and depth reference beneath the airborne split; canyon approach (open between the opposing jumps); used as Wide environmental context that makes the opposite trajectories readable.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated sunset ambient light is interrupted by the explicitly fiery projectile, creating a brief hard accent within restrained action contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHINGS AT THIS PLACE:\nThe canyon approach occupies the near side of the site and terminates at the near edge of an open canyon chasm. A tree beside the canyonward portion of the approach provides a fixed vertical landmark near that edge. The open canyon airspace extends between the approach side and the opposite cliff. Across the chasm, the top of the opposite cliff forms the far-side ground toward which the canyon-spanning trajectory descends. A forest directly adjoins the landward side of that cliff-top ground, allowing movement from the exposed rim into the trees.\n- approach: canyon approach route\n- tree: canyon-edge tree beside the approach\n- canyon: open canyon chasm\n- cliff: opposite cliff and its rim\n- forest: forest adjoining the opposite cliff rim\n\nWORLD FACTS (creator-confirmed — always true):\n- Region (real-world reference): 미국 펜실베이니아의 산림·산악권\n- Era: 25세기 초반의 먼 미래(2419년), 21세기 문명의 유산이 폐허와 구형 장비로 잔존하는 시대\n- 현재 사용되는 미래 기술과 오래되어 훼손·매몰된 구문명 유산을 명확히 구분한다. 구문명 요소는 보존된 현대 공간처럼 만들지 말고 장기간의 방치와 자연 침식이 축적된 외형으로 표현한다.\n- 사용자는 추진 비행이나 공중 정지 대신 중력의 영향을 받는 포물선 궤도로 이동시킨다. 검은 중력 제어판과 탈착식 은색 추를 벨트의 핵심 구성으로 유지하고, 추의 수와 장착 상태에 따라 도약 높이와 하강성이 달라 보이게 한다.\n- 탐색 장치는 작은 발광점이 아니라 렌즈와 금속성 구조를 지닌 물리적 센서로 표현한다. 소거 효과는 폭발이나 화염보다 경계가 비정상적으로 깨끗한 결손과 압력 교란으로 나타내며, 그래픽한 잔해 표현은 피한다.\n- 구시대 스마트폰은 익숙한 현대형 물건으로 유지하되 미래형 단말기로 재설계하지 않는다. 차폐 상태에서는 방수 주머니와 금속성 결속을 분명히 보여주고, 음성 인공지능은 별도의 인간형 실체 없이 기기에서 발생하는 인터페이스로 처리한다.\n- 위장된 표면은 닫혀 있을 때 주변 지형과 연속되어야 하며, 개방 시에만 인공적인 경계와 내부 구조를 드러낸다. 통신 정보와 탐지 표식은 물리 공간에 떠 있는 장식이 아니라 활성화된 판면이나 지도 인터페이스에 종속시킨다.\n\nTHINGS THAT LIVE AT THIS PLACE (what they are, not where they go): the\nlist above names what permanently belongs to this site. Where each one\nsits in this frame is decided by the attached sketch alone — do not add\nor reposition anything on the strength of the list, build only what the\nsketch already has lines for, and do not invent facilities that are not\nlisted.\n\nNo readable writing anywhere: surfaces that would carry writing may be\npresent, but stage any wording out of legibility — an oblique angle,\ndistance, shallow focus. No captions, watermarks or overlay text,\nand none of the sketch's diagram symbols, lines or markers anywhere in\nthe image.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh4__bgfirst_bg.png",
  "asset_id": "64bb41b9-15fb-41ee-959e-2f8866bd7565",
  "input_asset_ids": [
   "6d78d4b3-e79d-4663-b353-21120ee172a0"
  ]
 },
 "S6sh4": {
  "input_fingerprint": "b6113a0e1e159ef7",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): sunset.\n\nSHOT TEXT (authoritative, Korean): 불을 뿜는 탄자가 지나가는 가운데, 이를 피해 양옆 허공으로 몸을 비틀어 막 날아오른 윌마와 토니의 도약 찰나\n\nLOCATION (lock): Open exterior airspace above the rocky canyon approach, between the forested ground and the cliff sides. The shot takes place here — the attached STORYBOARD SKETCH fixes the staging, camera and figure placement of this exact place. No location photograph is attached — build the location itself strictly from the location text above and the shot text, inventing nothing beyond them.\n\nFRAMING SCALE (follow exactly — this alone decides how much of the frame the subject fills; the storyboard sketch supplies where the figures stand and which way they face, never this scale):\n- FRAMING SCALE: wide shot\nMatch this shot size exactly. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated sunset ambient light is interrupted by the explicitly fiery projectile, creating a brief hard accent within restrained action contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Three jumper landing grooves mark the ground, one of them fresh. A rocket projectile passes between the opposing jumps and blasts the intervening rock. 윌마 디어링: She is launching sideways to evade the shot, wearing her moss-smeared belt with one silver weight removed and carrying her moss-smeared gun. 토니(앤서니 로저스): He is launching in the opposite direction, wearing the jumper belt before either silver weight falls away. He carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Replace every grey or outlined mannequin figure in the FIRST attached\nimage with the real people the CHARACTER REFERENCE images show, and\noutput a photorealistic film still. Each mannequin becomes the character\nwhose POSE CANON and the shot text match that figure's pose and\nposition; never repeat one person across two mannequins. If no CHARACTER\nREFERENCE image is attached for a figure, still replace it with a\nplausible real person consistent with the text. Every mannequin becomes\na person — erasing a figure, or leaving one out, is not a replacement,\nand the number of figures never changes.\n\nKEEP EXACTLY: the background of the first image, the camera framing,\nand each mannequin's position, size, pose and the direction it is\nturned. A person must be exactly where their mannequin was, at the same\nscale, in the same pose, facing the same way — never mirrored, never\nre-staged, never straightened up, stood upright or otherwise re-posed.\nNo mannequin, grey figure or sketch line may remain anywhere in the\noutput.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): sunset.\n\nSHOT TEXT (authoritative, Korean): 불을 뿜는 탄자가 지나가는 가운데, 이를 피해 양옆 허공으로 몸을 비틀어 막 날아오른 윌마와 토니의 도약 찰나\n\nLOCATION (lock): Open exterior airspace above the rocky canyon approach, between the forested ground and the cliff sides. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-left of the frame, midground, moves toward left side of the firing line; 토니(앤서니 로저스) in the middle-right of the frame, midground, moves toward right side of the firing line; projectile crossing gap in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: flaming projectile (in flight between Wilma and Tony) — Its forward axis crosses obliquely through the central space between the two bodies; used as Central moving separator and immediate threat cue; canyon approach rocks (in the projectile’s path); used as Ground and depth reference beneath the airborne split; canyon approach (open between the opposing jumps); used as Wide environmental context that makes the opposite trajectories readable.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated sunset ambient light is interrupted by the explicitly fiery projectile, creating a brief hard accent within restrained action contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Three jumper landing grooves mark the ground, one of them fresh. A rocket projectile passes between the opposing jumps and blasts the intervening rock. 윌마 디어링: She is launching sideways to evade the shot, wearing her moss-smeared belt with one silver weight removed and carrying her moss-smeared gun. 토니(앤서니 로저스): He is launching in the opposite direction, wearing the jumper belt before either silver weight falls away. He carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Replace every grey or outlined mannequin figure in the FIRST attached\nimage with the real people the CHARACTER REFERENCE images show, and\noutput a photorealistic film still. Each mannequin becomes the character\nwhose POSE CANON and the shot text match that figure's pose and\nposition; never repeat one person across two mannequins. If no CHARACTER\nREFERENCE image is attached for a figure, still replace it with a\nplausible real person consistent with the text. Every mannequin becomes\na person — erasing a figure, or leaving one out, is not a replacement,\nand the number of figures never changes.\n\nKEEP EXACTLY: the background of the first image, the camera framing,\nand each mannequin's position, size, pose and the direction it is\nturned. A person must be exactly where their mannequin was, at the same\nscale, in the same pose, facing the same way — never mirrored, never\nre-staged, never straightened up, stood upright or otherwise re-posed.\nNo mannequin, grey figure or sketch line may remain anywhere in the\noutput.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): sunset.\n\nSHOT TEXT (authoritative, Korean): 불을 뿜는 탄자가 지나가는 가운데, 이를 피해 양옆 허공으로 몸을 비틀어 막 날아오른 윌마와 토니의 도약 찰나\n\nLOCATION (lock): Open exterior airspace above the rocky canyon approach, between the forested ground and the cliff sides. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-left of the frame, midground, moves toward left side of the firing line; 토니(앤서니 로저스) in the middle-right of the frame, midground, moves toward right side of the firing line; projectile crossing gap in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: flaming projectile (in flight between Wilma and Tony) — Its forward axis crosses obliquely through the central space between the two bodies; used as Central moving separator and immediate threat cue; canyon approach rocks (in the projectile’s path); used as Ground and depth reference beneath the airborne split; canyon approach (open between the opposing jumps); used as Wide environmental context that makes the opposite trajectories readable.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated sunset ambient light is interrupted by the explicitly fiery projectile, creating a brief hard accent within restrained action contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Three jumper landing grooves mark the ground, one of them fresh. A rocket projectile passes between the opposing jumps and blasts the intervening rock. 윌마 디어링: She is launching sideways to evade the shot, wearing her moss-smeared belt with one silver weight removed and carrying her moss-smeared gun. 토니(앤서니 로저스): He is launching in the opposite direction, wearing the jumper belt before either silver weight falls away. He carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh4__bgfirst_bg.png",
     "asset_id": "64bb41b9-15fb-41ee-959e-2f8866bd7565",
     "role": "bgfirst_bg"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    },
    {
     "label": "PROP REFERENCE — 로켓 피스톨과 폭발 탄자: the exact object appearing in this shot; match its look, material and wear exactly.",
     "path": "<bytes:1032965>",
     "asset_id": "f47d954a-99d1-49af-9d61-f4fed0d5e81f",
     "role": "prop_ref"
    }
   ],
   "B": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh4__bgfirst_bg.png",
     "asset_id": "64bb41b9-15fb-41ee-959e-2f8866bd7565",
     "role": "bgfirst_bg"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    },
    {
     "label": "PROP REFERENCE — 로켓 피스톨과 폭발 탄자: the exact object appearing in this shot; match its look, material and wear exactly.",
     "path": "<bytes:1032965>",
     "asset_id": "f47d954a-99d1-49af-9d61-f4fed0d5e81f",
     "role": "prop_ref"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "로켓탄은 우측 상단을 향해 날아가며, 윌마는 왼쪽으로, 토니는 오른쪽으로 몸을 비틀어 회피하는 올바른 방향성을 보임.",
    "built_space": "숲을 배경으로 한 바위 계곡. 왼쪽 절벽 바닥에 3개의 길쭉한 착지 홈이 패여 있고, 그 옆에 은색 무게추 1개가 놓여 있음.",
    "entities": "윌마는 참조 이미지의 얼굴과 녹색 점프슈트를 갖추고 총을 들고 있음. 토니는 참조 이미지의 얼굴, 헬멧, 전술복을 정확히 착용하고 왼손에 로프를 들고 있음.",
    "hard_violations": [
     "[gemini-pro] 물리적 지지 및 추진력 부재: 두 인물 모두 도약의 반동과 힘 없이, 단순히 달리는 자세를 뒤로 회전시킨 채 허공에 떠 있음 (Physics 위반)"
    ],
    "physics": "두 인물 모두 절벽에서 밀어내는 도약의 추진력(launch)을 보여주지 못하고, 참조용 마네킹의 달리는 자세를 그대로 뒤로 눕혀 허공에 띄워 놓아 물리적 지지(nothing supports it)가 부재함."
   },
   {
    "label": "B",
    "direction": "로켓탄은 우측 상단으로 날아감. 윌마는 왼쪽을 향하나, 토니 역시 왼쪽(로켓과 윌마가 있는 계곡 안쪽)을 향해 도약하고 있어 회피 방향이 틀림.",
    "built_space": "바위 계곡 구조. 왼쪽 절벽 바닥에 3개의 착지 홈과 은색 무게추 1개가 놓여 있음.",
    "entities": "윌마는 은색 무게추가 1개 남은 벨트를 차고 총을 쥠. 토니는 은색 무게추 2개가 달린 벨트와 로프를 가졌으나, 필수 요소인 헬멧을 착용하지 않은 맨머리임.",
    "hard_violations": [
     "[gemini-pro] 물리적 지지 부재: 도약의 추진력 없이 단순히 앞으로 달리는 자세로 허공에 떠 있음 (Physics 위반)",
     "[gemini-pro] 잘못된 스테이징: 토니가 회피 방향(오른쪽)이 아닌 로켓 쪽(왼쪽)으로 뛰어들어 액션 지시와 모순됨"
    ],
    "physics": "두 인물 모두 허공에서 앞으로 달려나가는 폼을 취하고 있음. 특히 토니는 뒤로 회피해야 함에도 절벽 바깥쪽으로 몸을 던지는 형태라 물리적 동선과 상황이 전혀 맞지 않음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "스토리보드의 구도와 인물의 복장(토니의 헬멧 포함)을 충실히 반영했으나, 인물들의 도약 자세가 마네킹의 달리는 폼을 단순히 뒤로 기울인 형태라 물리적 아쉬움이 있습니다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "토니가 헬멧을 착용하지 않았으며, 양옆으로 회피해야 할 토니가 반대 방향이 아닌 계곡 안쪽(왼쪽)으로 도약하여 액션 지시를 심각하게 위반했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "로켓탄은 우측 상단을 향해 날아가며, 윌마는 왼쪽으로, 토니는 오른쪽으로 몸을 비틀어 회피하는 올바른 방향성을 보임.",
        "built_space": "숲을 배경으로 한 바위 계곡. 왼쪽 절벽 바닥에 3개의 길쭉한 착지 홈이 패여 있고, 그 옆에 은색 무게추 1개가 놓여 있음.",
        "entities": "윌마는 참조 이미지의 얼굴과 녹색 점프슈트를 갖추고 총을 들고 있음. 토니는 참조 이미지의 얼굴, 헬멧, 전술복을 정확히 착용하고 왼손에 로프를 들고 있음.",
        "hard_violations": [
         "물리적 지지 및 추진력 부재: 두 인물 모두 도약의 반동과 힘 없이, 단순히 달리는 자세를 뒤로 회전시킨 채 허공에 떠 있음 (Physics 위반)"
        ],
        "physics": "두 인물 모두 절벽에서 밀어내는 도약의 추진력(launch)을 보여주지 못하고, 참조용 마네킹의 달리는 자세를 그대로 뒤로 눕혀 허공에 띄워 놓아 물리적 지지(nothing supports it)가 부재함."
       },
       {
        "label": "B",
        "direction": "로켓탄은 우측 상단으로 날아감. 윌마는 왼쪽을 향하나, 토니 역시 왼쪽(로켓과 윌마가 있는 계곡 안쪽)을 향해 도약하고 있어 회피 방향이 틀림.",
        "built_space": "바위 계곡 구조. 왼쪽 절벽 바닥에 3개의 착지 홈과 은색 무게추 1개가 놓여 있음.",
        "entities": "윌마는 은색 무게추가 1개 남은 벨트를 차고 총을 쥠. 토니는 은색 무게추 2개가 달린 벨트와 로프를 가졌으나, 필수 요소인 헬멧을 착용하지 않은 맨머리임.",
        "hard_violations": [
         "물리적 지지 부재: 도약의 추진력 없이 단순히 앞으로 달리는 자세로 허공에 떠 있음 (Physics 위반)",
         "잘못된 스테이징: 토니가 회피 방향(오른쪽)이 아닌 로켓 쪽(왼쪽)으로 뛰어들어 액션 지시와 모순됨"
        ],
        "physics": "두 인물 모두 허공에서 앞으로 달려나가는 폼을 취하고 있음. 특히 토니는 뒤로 회피해야 함에도 절벽 바깥쪽으로 몸을 던지는 형태라 물리적 동선과 상황이 전혀 맞지 않음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "스토리보드의 구도와 인물의 복장(토니의 헬멧 포함)을 충실히 반영했으나, 인물들의 도약 자세가 마네킹의 달리는 폼을 단순히 뒤로 기울인 형태라 물리적 아쉬움이 있습니다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "토니가 헬멧을 착용하지 않았으며, 양옆으로 회피해야 할 토니가 반대 방향이 아닌 계곡 안쪽(왼쪽)으로 도약하여 액션 지시를 심각하게 위반했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "로켓탄은 우측 상단을 향해 날아가며, 윌마는 왼쪽으로, 토니는 오른쪽으로 몸을 비틀어 회피하는 올바른 방향성을 보임.",
        "built_space": "숲을 배경으로 한 바위 계곡. 왼쪽 절벽 바닥에 3개의 길쭉한 착지 홈이 패여 있고, 그 옆에 은색 무게추 1개가 놓여 있음.",
        "entities": "윌마는 참조 이미지의 얼굴과 녹색 점프슈트를 갖추고 총을 들고 있음. 토니는 참조 이미지의 얼굴, 헬멧, 전술복을 정확히 착용하고 왼손에 로프를 들고 있음.",
        "hard_violations": [
         "물리적 지지 및 추진력 부재: 두 인물 모두 도약의 반동과 힘 없이, 단순히 달리는 자세를 뒤로 회전시킨 채 허공에 떠 있음 (Physics 위반)"
        ],
        "physics": "두 인물 모두 절벽에서 밀어내는 도약의 추진력(launch)을 보여주지 못하고, 참조용 마네킹의 달리는 자세를 그대로 뒤로 눕혀 허공에 띄워 놓아 물리적 지지(nothing supports it)가 부재함."
       },
       {
        "label": "B",
        "direction": "로켓탄은 우측 상단으로 날아감. 윌마는 왼쪽을 향하나, 토니 역시 왼쪽(로켓과 윌마가 있는 계곡 안쪽)을 향해 도약하고 있어 회피 방향이 틀림.",
        "built_space": "바위 계곡 구조. 왼쪽 절벽 바닥에 3개의 착지 홈과 은색 무게추 1개가 놓여 있음.",
        "entities": "윌마는 은색 무게추가 1개 남은 벨트를 차고 총을 쥠. 토니는 은색 무게추 2개가 달린 벨트와 로프를 가졌으나, 필수 요소인 헬멧을 착용하지 않은 맨머리임.",
        "hard_violations": [
         "물리적 지지 부재: 도약의 추진력 없이 단순히 앞으로 달리는 자세로 허공에 떠 있음 (Physics 위반)",
         "잘못된 스테이징: 토니가 회피 방향(오른쪽)이 아닌 로켓 쪽(왼쪽)으로 뛰어들어 액션 지시와 모순됨"
        ],
        "physics": "두 인물 모두 허공에서 앞으로 달려나가는 폼을 취하고 있음. 특히 토니는 뒤로 회피해야 함에도 절벽 바깥쪽으로 몸을 던지는 형태라 물리적 동선과 상황이 전혀 맞지 않음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "와이드 숏과 협곡·노을·두 인물·불붙은 탄자의 배치는 충실하고 토니의 헬멧·전술복·구조 로프도 A보다 정확하지만, 탄자가 바위를 폭파하지 않고 허공을 향하며 지면 홈도 세 개가 명확하지 않다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "도약 배치와 협곡 와이드 숏은 잘 맞지만 토니의 헬멧과 주요 전술 장비가 빠져 인물 레퍼런스 충실도가 B보다 낮고, 탄자 역시 개입한 바위를 타격하지 않는다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "탄자는 꼬리 화염을 왼쪽 아래에 두고 오른쪽 위로 진행하며, 토니 바로 왼쪽의 빈 공중을 향한다. 탄두가 맞거나 폭파하는 바위는 보이지 않는다. 윌마는 오른쪽의 탄자와 토니 쪽을 보며 총구도 오른쪽 위 공중으로 향하고, 토니는 왼쪽의 탄자와 윌마 쪽을 본다. 두 사람의 몸은 서로 반대 방향의 운동 자세를 취하지만 결과적으로 협곡 중앙 쪽으로 접근하는 인상이 강하다.",
        "built_space": "야외 협곡이며 좌우에 암반 선반 두 곳, 그 사이 깊은 틈, 맞은편 수직 절벽과 상부 침엽수림, 왼쪽 나무 한 그루가 보인다. 윌마는 왼쪽 선반 가장자리 위, 토니는 오른쪽 선반 가장자리에 배치되어 스토리보드의 좌우 구도를 따른다. 왼쪽 지면에는 긴 홈이 두 줄 정도만 뚜렷하여 요구된 세 개의 착지 홈과 그중 하나의 신선함이 명확히 판독되지 않는다.",
        "entities": "등장 인물은 윌마와 토니로 읽히는 성인 미국인 여성 한 명과 남성 한 명뿐이다. 윌마는 녹색 점프슈트와 벨트, 손에 든 총을 갖췄고 지면에는 떨어진 은색 추로 읽히는 물체가 있다. 토니는 어두운 작업복과 구조 로프를 지녔지만 캐릭터 레퍼런스의 헬멧과 가슴 전술 장비가 빠져 동일 의상 재현이 약하다. 불붙은 금속 로켓 탄자는 존재하며 읽을 수 있는 글자나 추가 인물은 없다.",
        "hard_violations": [],
        "physics": "윌마는 왼쪽 선반에서 막 박차고 오른쪽으로 떠오른 달리기형 도약 자세이며 발은 지면에서 떨어졌지만 바로 뒤에 발사 지점인 선반이 있어 도약의 출발이 성립한다. 토니는 오른쪽 선반 끝의 한 발로 마지막 추진 접촉을 유지하고 몸을 왼쪽으로 기울여 발사 순간이 성립한다. 윌마의 총과 토니의 로프는 각각 손에 잡혀 있다. 탄자는 후방 화염 추진으로 비행이 설명되지만 바위에 충돌하거나 폭발을 일으키는 물리적 접점은 없다."
       },
       {
        "label": "B",
        "direction": "탄자는 후방 화염을 왼쪽 아래로 뿜으며 오른쪽 위, 즉 토니의 왼쪽 아래에 있는 빈 공중으로 진행한다. 요구된 개입 바위에는 조준되거나 충돌하지 않는다. 윌마의 시선과 총구는 오른쪽의 탄자·토니 쪽이고, 토니는 왼쪽의 탄자와 윌마를 본다. 두 사람은 서로 반대되는 좌우 도약 자세이나 양쪽 가장자리에서 협곡 중앙 쪽으로 향하는 인상이다.",
        "built_space": "좌우 암반 선반 두 곳과 중앙의 깊은 협곡, 맞은편 암벽, 능선 위 침엽수림, 왼쪽 전경 나무 한 그루로 이루어진 외부 공중 공간이다. 윌마는 왼쪽 선반 위, 토니는 오른쪽 가장자리에서 도약하여 스토리보드 배치와 카메라 축을 따른다. 왼쪽 바닥의 착지 홈은 두 줄 정도만 분명하고 요구된 세 줄 및 신선한 한 줄의 구별은 불명확하다.",
        "entities": "정확히 두 사람만 있으며 왼쪽은 20대 후반 여성 윌마, 오른쪽은 30대 중반 남성 토니로 읽힌다. 윌마는 녹색 점프슈트, 벨트와 손에 잡은 총을 갖췄고 바닥에는 제거된 은색 추로 해석 가능한 물체가 있다. 토니는 레퍼런스에 가까운 헬멧, 어두운 전술복, 가슴 장비와 구조 로프를 갖추며 A보다 인물·의상 일치도가 높다. 두 개의 벨트 추로 보이는 금속 부품도 남아 있다. 금속 로켓 탄자와 화염은 보이고 추가 인물이나 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "윌마는 왼쪽 선반을 막 박차고 떠오른 자세로, 뒤쪽 선반이 발사 지점이며 굽힌 다리와 뻗은 다리가 추진 동작을 설명한다. 토니는 오른쪽 발끝이 선반 가장자리에 닿은 상태에서 몸을 왼쪽으로 던져 막 이륙하는 순간이라 지지와 추진이 보인다. 총은 윌마의 손에, 로프는 토니의 오른손과 장비에 확실히 연결되어 떠 있지 않는다. 탄자의 비행은 꼬리 화염으로 설명되지만 바위 폭파 동작은 나타나지 않는다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "와이드 숏과 협곡·노을·두 인물·불붙은 탄자의 배치는 충실하고 토니의 헬멧·전술복·구조 로프도 A보다 정확하지만, 탄자가 바위를 폭파하지 않고 허공을 향하며 지면 홈도 세 개가 명확하지 않다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "도약 배치와 협곡 와이드 숏은 잘 맞지만 토니의 헬멧과 주요 전술 장비가 빠져 인물 레퍼런스 충실도가 B보다 낮고, 탄자 역시 개입한 바위를 타격하지 않는다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "탄자는 꼬리 화염을 왼쪽 아래에 두고 오른쪽 위로 진행하며, 토니 바로 왼쪽의 빈 공중을 향한다. 탄두가 맞거나 폭파하는 바위는 보이지 않는다. 윌마는 오른쪽의 탄자와 토니 쪽을 보며 총구도 오른쪽 위 공중으로 향하고, 토니는 왼쪽의 탄자와 윌마 쪽을 본다. 두 사람의 몸은 서로 반대 방향의 운동 자세를 취하지만 결과적으로 협곡 중앙 쪽으로 접근하는 인상이 강하다.",
        "built_space": "야외 협곡이며 좌우에 암반 선반 두 곳, 그 사이 깊은 틈, 맞은편 수직 절벽과 상부 침엽수림, 왼쪽 나무 한 그루가 보인다. 윌마는 왼쪽 선반 가장자리 위, 토니는 오른쪽 선반 가장자리에 배치되어 스토리보드의 좌우 구도를 따른다. 왼쪽 지면에는 긴 홈이 두 줄 정도만 뚜렷하여 요구된 세 개의 착지 홈과 그중 하나의 신선함이 명확히 판독되지 않는다.",
        "entities": "등장 인물은 윌마와 토니로 읽히는 성인 미국인 여성 한 명과 남성 한 명뿐이다. 윌마는 녹색 점프슈트와 벨트, 손에 든 총을 갖췄고 지면에는 떨어진 은색 추로 읽히는 물체가 있다. 토니는 어두운 작업복과 구조 로프를 지녔지만 캐릭터 레퍼런스의 헬멧과 가슴 전술 장비가 빠져 동일 의상 재현이 약하다. 불붙은 금속 로켓 탄자는 존재하며 읽을 수 있는 글자나 추가 인물은 없다.",
        "hard_violations": [],
        "physics": "윌마는 왼쪽 선반에서 막 박차고 오른쪽으로 떠오른 달리기형 도약 자세이며 발은 지면에서 떨어졌지만 바로 뒤에 발사 지점인 선반이 있어 도약의 출발이 성립한다. 토니는 오른쪽 선반 끝의 한 발로 마지막 추진 접촉을 유지하고 몸을 왼쪽으로 기울여 발사 순간이 성립한다. 윌마의 총과 토니의 로프는 각각 손에 잡혀 있다. 탄자는 후방 화염 추진으로 비행이 설명되지만 바위에 충돌하거나 폭발을 일으키는 물리적 접점은 없다."
       },
       {
        "label": "A",
        "direction": "탄자는 후방 화염을 왼쪽 아래로 뿜으며 오른쪽 위, 즉 토니의 왼쪽 아래에 있는 빈 공중으로 진행한다. 요구된 개입 바위에는 조준되거나 충돌하지 않는다. 윌마의 시선과 총구는 오른쪽의 탄자·토니 쪽이고, 토니는 왼쪽의 탄자와 윌마를 본다. 두 사람은 서로 반대되는 좌우 도약 자세이나 양쪽 가장자리에서 협곡 중앙 쪽으로 향하는 인상이다.",
        "built_space": "좌우 암반 선반 두 곳과 중앙의 깊은 협곡, 맞은편 암벽, 능선 위 침엽수림, 왼쪽 전경 나무 한 그루로 이루어진 외부 공중 공간이다. 윌마는 왼쪽 선반 위, 토니는 오른쪽 가장자리에서 도약하여 스토리보드 배치와 카메라 축을 따른다. 왼쪽 바닥의 착지 홈은 두 줄 정도만 분명하고 요구된 세 줄 및 신선한 한 줄의 구별은 불명확하다.",
        "entities": "정확히 두 사람만 있으며 왼쪽은 20대 후반 여성 윌마, 오른쪽은 30대 중반 남성 토니로 읽힌다. 윌마는 녹색 점프슈트, 벨트와 손에 잡은 총을 갖췄고 바닥에는 제거된 은색 추로 해석 가능한 물체가 있다. 토니는 레퍼런스에 가까운 헬멧, 어두운 전술복, 가슴 장비와 구조 로프를 갖추며 A보다 인물·의상 일치도가 높다. 두 개의 벨트 추로 보이는 금속 부품도 남아 있다. 금속 로켓 탄자와 화염은 보이고 추가 인물이나 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "윌마는 왼쪽 선반을 막 박차고 떠오른 자세로, 뒤쪽 선반이 발사 지점이며 굽힌 다리와 뻗은 다리가 추진 동작을 설명한다. 토니는 오른쪽 발끝이 선반 가장자리에 닿은 상태에서 몸을 왼쪽으로 던져 막 이륙하는 순간이라 지지와 추진이 보인다. 총은 윌마의 손에, 로프는 토니의 오른손과 장비에 확실히 연결되어 떠 있지 않는다. 탄자의 비행은 꼬리 화염으로 설명되지만 바위 폭파 동작은 나타나지 않는다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.357
   },
   "adjusted": {
    "A": 1.75,
    "B": 1.107
   },
   "violations": {
    "A": [
     "[gemini-pro] 물리적 지지 및 추진력 부재: 두 인물 모두 도약의 반동과 힘 없이, 단순히 달리는 자세를 뒤로 회전시킨 채 허공에 떠 있음 (Physics 위반)"
    ],
    "B": [
     "[gemini-pro] 물리적 지지 부재: 도약의 추진력 없이 단순히 앞으로 달리는 자세로 허공에 떠 있음 (Physics 위반)",
     "[gemini-pro] 잘못된 스테이징: 토니가 회피 방향(오른쪽)이 아닌 로켓 쪽(왼쪽)으로 뛰어들어 액션 지시와 모순됨"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 1750,
   "B": 1107
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1750,
    "verdict_ko": "스토리보드의 구도와 인물의 복장(토니의 헬멧 포함)을 충실히 반영했으나, 인물들의 도약 자세가 마네킹의 달리는 폼을 단순히 뒤로 기울인 형태라 물리적 아쉬움이 있습니다.  ★위반: [gemini-pro] 물리적 지지 및 추진력 부재: 두 인물 모두 도약의 반동과 힘 없이, 단순히 달리는 자세를 뒤로 회전시킨 채 허공에 떠 있음 (Physics 위반)"
   },
   {
    "label": "B",
    "score": 1107,
    "verdict_ko": "토니가 헬멧을 착용하지 않았으며, 양옆으로 회피해야 할 토니가 반대 방향이 아닌 계곡 안쪽(왼쪽)으로 도약하여 액션 지시를 심각하게 위반했습니다.  ★위반: [gemini-pro] 물리적 지지 부재: 도약의 추진력 없이 단순히 앞으로 달리는 자세로 허공에 떠 있음 (Physics 위반) / [gemini-pro] 잘못된 스테이징: 토니가 회피 방향(오른쪽)이 아닌 로켓 쪽(왼쪽)으로 뛰어들어 액션 지시와 모순됨"
   }
  ],
  "refs": [
   {
    "label": "SHOT BACKGROUND",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh4__bgfirst_bg.png",
    "asset_id": "64bb41b9-15fb-41ee-959e-2f8866bd7565",
    "role": "bgfirst_bg"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   },
   {
    "label": "PROP REFERENCE — 로켓 피스톨과 폭발 탄자: the exact object appearing in this shot; match its look, material and wear exactly.",
    "path": "<bytes:1032965>",
    "asset_id": "f47d954a-99d1-49af-9d61-f4fed0d5e81f",
    "role": "prop_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc8f-439d-7473-8c48-dd898f407b3a",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh4__bgfirst_bg.png",
   "bg_asset_id": "64bb41b9-15fb-41ee-959e-2f8866bd7565",
   "bg_record_key": "S6sh4::bgfirst_bg",
   "chain_winner": true,
   "authority": "lane_conti_only"
  },
  "ref_mode": "lane(map_marker): 스케치+엔티티",
  "share_plan": {
   "ref_plan": "background"
  }
 },
 "S6sh4::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:50:23.391956+00:00",
  "fingerprint": "fb8a227ec811f6ff5edf00c47dc6ddec927befb8ac271659876e4c33c782f74b",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S6sh4_sel.png",
  "source_sha256": "34cbba69afaecd99871e2fb8413df76012cec12e418b1473c6dfc749fd223d85",
  "file": "S6sh4_cine.png",
  "staged_sha256": "0c480fefdd0886daca82d969d42260a194b4282b5fef39947ca2b2ec295df7c4",
  "latency_ms": 12785
 },
 "S6sh13::signage": {
  "fp": "a0527f09370c13b0",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S6sh13::bgfirst_bg": {
  "input_fingerprint": "6d63d185b2edb26f",
  "prompt": "Turn the attached storyboard sketch into a photorealistic film still of\nits LOCATION ONLY, keeping the figures exactly as they are.\n\nKEEP EXACTLY: the camera framing, the horizon, and where every element\nsits in the frame. Each mannequin figure stays a plain grey featureless\nmannequin standing in the very same spot, at the same size, in the same\npose, turned the same way — do not turn them into people, do not move,\nrotate, mirror or re-pose them, do not add or remove figures.\n\nBUILD PHOTOREALISTICALLY: everything that is not a figure — ground,\nsurfacing, structures, vegetation, sky, water, distance. The bare lines\nof the sketch are a layout guide; replace them with the real materials,\ndepth and lighting of the place described below.\n\nSHOT TEXT this background must serve (Korean): 서로를 단단히 붙잡은 두 사람의 몸이 협곡 반대편 아래쪽을 향해 사선으로 기울어진 채 허공에 떠 있는 순간\n\nLOCATION (lock): Open exterior airspace over the canyon, descending diagonally toward the opposite rim.\n\nTIME OF DAY (lock): sunset.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, midground, moves toward lower opposite side of canyon; 토니(앤서니 로저스) in the middle-right of the frame, midground, moves toward lower opposite side of canyon; opposite canyon side in the lower-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: opposite side of the canyon (approaching along their descending trajectory) — Its upper edge runs diagonally beneath the pair’s direction of travel; used as Destination and scale reference for the redirected descent; canyon gap (open beneath the airborne pair); used as Negative depth that clarifies their suspended height and downward motion.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained, desaturated sunset light preserves the canyon depth with tense moderate contrast around the descending figures.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHINGS AT THIS PLACE:\nThe canyon approach occupies the near side of the site and terminates at the near edge of an open canyon chasm. A tree beside the canyonward portion of the approach provides a fixed vertical landmark near that edge. The open canyon airspace extends between the approach side and the opposite cliff. Across the chasm, the top of the opposite cliff forms the far-side ground toward which the canyon-spanning trajectory descends. A forest directly adjoins the landward side of that cliff-top ground, allowing movement from the exposed rim into the trees.\n- approach: canyon approach route\n- tree: canyon-edge tree beside the approach\n- canyon: open canyon chasm\n- cliff: opposite cliff and its rim\n- forest: forest adjoining the opposite cliff rim\n\nWORLD FACTS (creator-confirmed — always true):\n- Region (real-world reference): 미국 펜실베이니아의 산림·산악권\n- Era: 25세기 초반의 먼 미래(2419년), 21세기 문명의 유산이 폐허와 구형 장비로 잔존하는 시대\n- 현재 사용되는 미래 기술과 오래되어 훼손·매몰된 구문명 유산을 명확히 구분한다. 구문명 요소는 보존된 현대 공간처럼 만들지 말고 장기간의 방치와 자연 침식이 축적된 외형으로 표현한다.\n- 사용자는 추진 비행이나 공중 정지 대신 중력의 영향을 받는 포물선 궤도로 이동시킨다. 검은 중력 제어판과 탈착식 은색 추를 벨트의 핵심 구성으로 유지하고, 추의 수와 장착 상태에 따라 도약 높이와 하강성이 달라 보이게 한다.\n- 탐색 장치는 작은 발광점이 아니라 렌즈와 금속성 구조를 지닌 물리적 센서로 표현한다. 소거 효과는 폭발이나 화염보다 경계가 비정상적으로 깨끗한 결손과 압력 교란으로 나타내며, 그래픽한 잔해 표현은 피한다.\n- 구시대 스마트폰은 익숙한 현대형 물건으로 유지하되 미래형 단말기로 재설계하지 않는다. 차폐 상태에서는 방수 주머니와 금속성 결속을 분명히 보여주고, 음성 인공지능은 별도의 인간형 실체 없이 기기에서 발생하는 인터페이스로 처리한다.\n- 위장된 표면은 닫혀 있을 때 주변 지형과 연속되어야 하며, 개방 시에만 인공적인 경계와 내부 구조를 드러낸다. 통신 정보와 탐지 표식은 물리 공간에 떠 있는 장식이 아니라 활성화된 판면이나 지도 인터페이스에 종속시킨다.\n\nTHINGS THAT LIVE AT THIS PLACE (what they are, not where they go): the\nlist above names what permanently belongs to this site. Where each one\nsits in this frame is decided by the attached sketch alone — do not add\nor reposition anything on the strength of the list, build only what the\nsketch already has lines for, and do not invent facilities that are not\nlisted.\n\nNo readable writing anywhere: surfaces that would carry writing may be\npresent, but stage any wording out of legibility — an oblique angle,\ndistance, shallow focus. No captions, watermarks or overlay text,\nand none of the sketch's diagram symbols, lines or markers anywhere in\nthe image.",
  "effective_prompt": "Turn the attached storyboard sketch into a photorealistic film still of\nits LOCATION ONLY, keeping the figures exactly as they are.\n\nKEEP EXACTLY: the camera framing, the horizon, and where every element\nsits in the frame. Each mannequin figure stays a plain grey featureless\nmannequin standing in the very same spot, at the same size, in the same\npose, turned the same way — do not turn them into people, do not move,\nrotate, mirror or re-pose them, do not add or remove figures.\n\nBUILD PHOTOREALISTICALLY: everything that is not a figure — ground,\nsurfacing, structures, vegetation, sky, water, distance. The bare lines\nof the sketch are a layout guide; replace them with the real materials,\ndepth and lighting of the place described below.\n\nSHOT TEXT this background must serve (Korean): 서로를 단단히 붙잡은 두 사람의 몸이 협곡 반대편 아래쪽을 향해 사선으로 기울어진 채 허공에 떠 있는 순간\n\nLOCATION (lock): Open exterior airspace over the canyon, descending diagonally toward the opposite rim.\n\nTIME OF DAY (lock): sunset.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, midground, moves toward lower opposite side of canyon; 토니(앤서니 로저스) in the middle-right of the frame, midground, moves toward lower opposite side of canyon; opposite canyon side in the lower-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: opposite side of the canyon (approaching along their descending trajectory) — Its upper edge runs diagonally beneath the pair’s direction of travel; used as Destination and scale reference for the redirected descent; canyon gap (open beneath the airborne pair); used as Negative depth that clarifies their suspended height and downward motion.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained, desaturated sunset light preserves the canyon depth with tense moderate contrast around the descending figures.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHINGS AT THIS PLACE:\nThe canyon approach occupies the near side of the site and terminates at the near edge of an open canyon chasm. A tree beside the canyonward portion of the approach provides a fixed vertical landmark near that edge. The open canyon airspace extends between the approach side and the opposite cliff. Across the chasm, the top of the opposite cliff forms the far-side ground toward which the canyon-spanning trajectory descends. A forest directly adjoins the landward side of that cliff-top ground, allowing movement from the exposed rim into the trees.\n- approach: canyon approach route\n- tree: canyon-edge tree beside the approach\n- canyon: open canyon chasm\n- cliff: opposite cliff and its rim\n- forest: forest adjoining the opposite cliff rim\n\nWORLD FACTS (creator-confirmed — always true):\n- Region (real-world reference): 미국 펜실베이니아의 산림·산악권\n- Era: 25세기 초반의 먼 미래(2419년), 21세기 문명의 유산이 폐허와 구형 장비로 잔존하는 시대\n- 현재 사용되는 미래 기술과 오래되어 훼손·매몰된 구문명 유산을 명확히 구분한다. 구문명 요소는 보존된 현대 공간처럼 만들지 말고 장기간의 방치와 자연 침식이 축적된 외형으로 표현한다.\n- 사용자는 추진 비행이나 공중 정지 대신 중력의 영향을 받는 포물선 궤도로 이동시킨다. 검은 중력 제어판과 탈착식 은색 추를 벨트의 핵심 구성으로 유지하고, 추의 수와 장착 상태에 따라 도약 높이와 하강성이 달라 보이게 한다.\n- 탐색 장치는 작은 발광점이 아니라 렌즈와 금속성 구조를 지닌 물리적 센서로 표현한다. 소거 효과는 폭발이나 화염보다 경계가 비정상적으로 깨끗한 결손과 압력 교란으로 나타내며, 그래픽한 잔해 표현은 피한다.\n- 구시대 스마트폰은 익숙한 현대형 물건으로 유지하되 미래형 단말기로 재설계하지 않는다. 차폐 상태에서는 방수 주머니와 금속성 결속을 분명히 보여주고, 음성 인공지능은 별도의 인간형 실체 없이 기기에서 발생하는 인터페이스로 처리한다.\n- 위장된 표면은 닫혀 있을 때 주변 지형과 연속되어야 하며, 개방 시에만 인공적인 경계와 내부 구조를 드러낸다. 통신 정보와 탐지 표식은 물리 공간에 떠 있는 장식이 아니라 활성화된 판면이나 지도 인터페이스에 종속시킨다.\n\nTHINGS THAT LIVE AT THIS PLACE (what they are, not where they go): the\nlist above names what permanently belongs to this site. Where each one\nsits in this frame is decided by the attached sketch alone — do not add\nor reposition anything on the strength of the list, build only what the\nsketch already has lines for, and do not invent facilities that are not\nlisted.\n\nNo readable writing anywhere: surfaces that would carry writing may be\npresent, but stage any wording out of legibility — an oblique angle,\ndistance, shallow focus. No captions, watermarks or overlay text,\nand none of the sketch's diagram symbols, lines or markers anywhere in\nthe image.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh13__bgfirst_bg.png",
  "asset_id": "9d3c25dc-789d-4bce-b6d2-d2035cf3c5ba",
  "input_asset_ids": [
   "ec96045f-fe90-4b9a-8c99-96b2df75f661"
  ]
 },
 "S6sh13": {
  "input_fingerprint": "fcddf589170ecd85",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): sunset.\n\nSHOT TEXT (authoritative, Korean): 서로를 단단히 붙잡은 두 사람의 몸이 협곡 반대편 아래쪽을 향해 사선으로 기울어진 채 허공에 떠 있는 순간\n\nLOCATION (lock): Open exterior airspace over the canyon, descending diagonally toward the opposite rim. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nFRAMING SCALE (follow exactly — this alone decides how much of the frame the subject fills; the storyboard sketch supplies where the figures stand and which way they face, never this scale):\n- FRAMING SCALE: wide shot\nMatch this shot size exactly. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained, desaturated sunset light preserves the canyon depth with tense moderate contrast around the descending figures.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 윌마 디어링, 토니(앤서니 로저스) — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the forested canyon, exposed rock faces, open air between the cliffs, and fading daylight from the reference. Exclude the earlier rocket projectile, its flame, and the fresh rock explosion; preserve only the undisturbed canyon materials required here.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The struck rock remains blasted apart. Two silver weights have fallen free from Tony's belt toward the gorge. 윌마 디어링: She is airborne on a descending diagonal toward the far side, gripping a jumper harness. Her moss-smeared belt remains missing the previously transferred weight, and she carries her moss-smeared gun. 토니(앤서니 로저스): He is airborne on the same descending diagonal, suspended by his harness after losing two silver weights. He still carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Replace every grey or outlined mannequin figure in the FIRST attached\nimage with the real people the CHARACTER REFERENCE images show, and\noutput a photorealistic film still. Each mannequin becomes the character\nwhose POSE CANON and the shot text match that figure's pose and\nposition; never repeat one person across two mannequins. If no CHARACTER\nREFERENCE image is attached for a figure, still replace it with a\nplausible real person consistent with the text. Every mannequin becomes\na person — erasing a figure, or leaving one out, is not a replacement,\nand the number of figures never changes.\n\nKEEP EXACTLY: the background of the first image, the camera framing,\nand each mannequin's position, size, pose and the direction it is\nturned. A person must be exactly where their mannequin was, at the same\nscale, in the same pose, facing the same way — never mirrored, never\nre-staged, never straightened up, stood upright or otherwise re-posed.\nNo mannequin, grey figure or sketch line may remain anywhere in the\noutput.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): sunset.\n\nSHOT TEXT (authoritative, Korean): 서로를 단단히 붙잡은 두 사람의 몸이 협곡 반대편 아래쪽을 향해 사선으로 기울어진 채 허공에 떠 있는 순간\n\nLOCATION (lock): Open exterior airspace over the canyon, descending diagonally toward the opposite rim. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, midground, moves toward lower opposite side of canyon; 토니(앤서니 로저스) in the middle-right of the frame, midground, moves toward lower opposite side of canyon; opposite canyon side in the lower-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: opposite side of the canyon (approaching along their descending trajectory) — Its upper edge runs diagonally beneath the pair’s direction of travel; used as Destination and scale reference for the redirected descent; canyon gap (open beneath the airborne pair); used as Negative depth that clarifies their suspended height and downward motion.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained, desaturated sunset light preserves the canyon depth with tense moderate contrast around the descending figures.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The struck rock remains blasted apart. Two silver weights have fallen free from Tony's belt toward the gorge. 윌마 디어링: She is airborne on a descending diagonal toward the far side, gripping a jumper harness. Her moss-smeared belt remains missing the previously transferred weight, and she carries her moss-smeared gun. 토니(앤서니 로저스): He is airborne on the same descending diagonal, suspended by his harness after losing two silver weights. He still carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Replace every grey or outlined mannequin figure in the FIRST attached\nimage with the real people the CHARACTER REFERENCE images show, and\noutput a photorealistic film still. Each mannequin becomes the character\nwhose POSE CANON and the shot text match that figure's pose and\nposition; never repeat one person across two mannequins. If no CHARACTER\nREFERENCE image is attached for a figure, still replace it with a\nplausible real person consistent with the text. Every mannequin becomes\na person — erasing a figure, or leaving one out, is not a replacement,\nand the number of figures never changes.\n\nKEEP EXACTLY: the background of the first image, the camera framing,\nand each mannequin's position, size, pose and the direction it is\nturned. A person must be exactly where their mannequin was, at the same\nscale, in the same pose, facing the same way — never mirrored, never\nre-staged, never straightened up, stood upright or otherwise re-posed.\nNo mannequin, grey figure or sketch line may remain anywhere in the\noutput.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): sunset.\n\nSHOT TEXT (authoritative, Korean): 서로를 단단히 붙잡은 두 사람의 몸이 협곡 반대편 아래쪽을 향해 사선으로 기울어진 채 허공에 떠 있는 순간\n\nLOCATION (lock): Open exterior airspace over the canyon, descending diagonally toward the opposite rim. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, midground, moves toward lower opposite side of canyon; 토니(앤서니 로저스) in the middle-right of the frame, midground, moves toward lower opposite side of canyon; opposite canyon side in the lower-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: opposite side of the canyon (approaching along their descending trajectory) — Its upper edge runs diagonally beneath the pair’s direction of travel; used as Destination and scale reference for the redirected descent; canyon gap (open beneath the airborne pair); used as Negative depth that clarifies their suspended height and downward motion.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained, desaturated sunset light preserves the canyon depth with tense moderate contrast around the descending figures.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The struck rock remains blasted apart. Two silver weights have fallen free from Tony's belt toward the gorge. 윌마 디어링: She is airborne on a descending diagonal toward the far side, gripping a jumper harness. Her moss-smeared belt remains missing the previously transferred weight, and she carries her moss-smeared gun. 토니(앤서니 로저스): He is airborne on the same descending diagonal, suspended by his harness after losing two silver weights. He still carries his rescue rope.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh13__bgfirst_bg.png",
     "asset_id": "9d3c25dc-789d-4bce-b6d2-d2035cf3c5ba",
     "role": "bgfirst_bg"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ],
   "B": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh13__bgfirst_bg.png",
     "asset_id": "9d3c25dc-789d-4bce-b6d2-d2035cf3c5ba",
     "role": "bgfirst_bg"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "initial_roll_all_fail": true,
  "readings": [
   {
    "label": "A",
    "direction": "두 사람 모두 협곡 반대편 아래쪽을 향하고 있으며, 윌마는 총구를 앞쪽으로 겨누고 있습니다.",
    "built_space": "협곡 상공의 공중 공간입니다. 절벽, 나무, 빛의 방향 등 배경 요소가 레퍼런스 이미지와 동일하게 구현되었습니다.",
    "entities": "윌마는 이끼가 묻은 총을 들고 있으며 헬멧을 쓰지 않았습니다. 토니는 헬멧을 착용하지 않았고 하네스를 입고 있습니다. 은색 추는 토니에게 줄로 매달린 2개와 아래로 떨어지는 1개가 보입니다.",
    "hard_violations": [
     "[gemini-pro] 지지대나 위로 연결된 로프 없이 허공에 떠 있는 인물들",
     "[gemini-pro] 손이나 지지대 없이 허공에 떠 있는 은색 추",
     "[gpt] 토니의 하네스를 실제로 매다는 팽팽한 선이나 고정점이 전혀 없고 도약의 출발 동작도 없어, 두 공중 인물을 지지하는 것이 없다.",
     "[gpt] 지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
    ],
    "physics": "두 사람 모두 도약이나 착지의 흔적, 혹은 위에서 끌어당기는 로프 등 몸을 지지하는 요소가 전혀 없이 허공에 떠 있습니다. 떨어지는 은색 추 역시 아무도 잡고 있지 않은 상태로 허공에 떠 있습니다."
   },
   {
    "label": "B",
    "direction": "두 사람 모두 협곡 반대편 아래쪽을 향해 있으며, 윌마는 총구를 앞쪽 목표를 향해 겨누고 있습니다.",
    "built_space": "협곡 상공의 공중 공간입니다. 레퍼런스의 절벽 지형과 숲 등 배경 요소가 정확한 위치에 렌더링되었습니다.",
    "entities": "윌마는 헬멧 없이 이끼가 묻은 총을 들고 있습니다. 토니는 레퍼런스에 있는 헬멧과 하네스를 착용했습니다. 은색 추는 줄에 매달린 2개와 아래에 홀로 떠 있는 1개가 있습니다.",
    "hard_violations": [
     "[gemini-pro] 아무런 지지대 없이 허공에 떠 있는 인물들",
     "[gemini-pro] 손이나 연결 고리 없이 허공에 떠 있는 은색 추",
     "[gpt] 토니의 하네스에 연결된 팽팽한 지지선이나 고정점이 없고 두 사람의 도약 출발도 표현되지 않아, 두 몸을 공중에 유지하는 물리적 지지가 없다.",
     "[gpt] 지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
    ],
    "physics": "두 인물 모두 위로 연결된 케이블이나 도약/착지 동작 등 몸을 지탱하는 요소 없이 공중에 떠 있습니다. 아래의 은색 추 역시 지지대나 잡고 있는 손 없이 허공에 떠 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "토니의 헬멧 착용과 레퍼런스의 자세를 잘 유지했으나, 인물과 사물이 지지대 없이 허공에 떠 있어 물리 법칙(Physics)을 위반한 심각한 오류가 있습니다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "인물과 사물이 지지대 없이 허공에 떠 있는 물리 법칙 위반이 있으며, 토니의 헬멧이 누락되었고 두 사람의 자세가 레퍼런스와 다르게 변형되었습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "두 사람 모두 협곡 반대편 아래쪽을 향하고 있으며, 윌마는 총구를 앞쪽으로 겨누고 있습니다.",
        "built_space": "협곡 상공의 공중 공간입니다. 절벽, 나무, 빛의 방향 등 배경 요소가 레퍼런스 이미지와 동일하게 구현되었습니다.",
        "entities": "윌마는 이끼가 묻은 총을 들고 있으며 헬멧을 쓰지 않았습니다. 토니는 헬멧을 착용하지 않았고 하네스를 입고 있습니다. 은색 추는 토니에게 줄로 매달린 2개와 아래로 떨어지는 1개가 보입니다.",
        "hard_violations": [
         "지지대나 위로 연결된 로프 없이 허공에 떠 있는 인물들",
         "손이나 지지대 없이 허공에 떠 있는 은색 추"
        ],
        "physics": "두 사람 모두 도약이나 착지의 흔적, 혹은 위에서 끌어당기는 로프 등 몸을 지지하는 요소가 전혀 없이 허공에 떠 있습니다. 떨어지는 은색 추 역시 아무도 잡고 있지 않은 상태로 허공에 떠 있습니다."
       },
       {
        "label": "B",
        "direction": "두 사람 모두 협곡 반대편 아래쪽을 향해 있으며, 윌마는 총구를 앞쪽 목표를 향해 겨누고 있습니다.",
        "built_space": "협곡 상공의 공중 공간입니다. 레퍼런스의 절벽 지형과 숲 등 배경 요소가 정확한 위치에 렌더링되었습니다.",
        "entities": "윌마는 헬멧 없이 이끼가 묻은 총을 들고 있습니다. 토니는 레퍼런스에 있는 헬멧과 하네스를 착용했습니다. 은색 추는 줄에 매달린 2개와 아래에 홀로 떠 있는 1개가 있습니다.",
        "hard_violations": [
         "아무런 지지대 없이 허공에 떠 있는 인물들",
         "손이나 연결 고리 없이 허공에 떠 있는 은색 추"
        ],
        "physics": "두 인물 모두 위로 연결된 케이블이나 도약/착지 동작 등 몸을 지탱하는 요소 없이 공중에 떠 있습니다. 아래의 은색 추 역시 지지대나 잡고 있는 손 없이 허공에 떠 있습니다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "토니의 헬멧 착용과 레퍼런스의 자세를 잘 유지했으나, 인물과 사물이 지지대 없이 허공에 떠 있어 물리 법칙(Physics)을 위반한 심각한 오류가 있습니다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "인물과 사물이 지지대 없이 허공에 떠 있는 물리 법칙 위반이 있으며, 토니의 헬멧이 누락되었고 두 사람의 자세가 레퍼런스와 다르게 변형되었습니다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "A",
        "direction": "두 사람 모두 협곡 반대편 아래쪽을 향하고 있으며, 윌마는 총구를 앞쪽으로 겨누고 있습니다.",
        "built_space": "협곡 상공의 공중 공간입니다. 절벽, 나무, 빛의 방향 등 배경 요소가 레퍼런스 이미지와 동일하게 구현되었습니다.",
        "entities": "윌마는 이끼가 묻은 총을 들고 있으며 헬멧을 쓰지 않았습니다. 토니는 헬멧을 착용하지 않았고 하네스를 입고 있습니다. 은색 추는 토니에게 줄로 매달린 2개와 아래로 떨어지는 1개가 보입니다.",
        "hard_violations": [
         "지지대나 위로 연결된 로프 없이 허공에 떠 있는 인물들",
         "손이나 지지대 없이 허공에 떠 있는 은색 추"
        ],
        "physics": "두 사람 모두 도약이나 착지의 흔적, 혹은 위에서 끌어당기는 로프 등 몸을 지지하는 요소가 전혀 없이 허공에 떠 있습니다. 떨어지는 은색 추 역시 아무도 잡고 있지 않은 상태로 허공에 떠 있습니다."
       },
       {
        "label": "B",
        "direction": "두 사람 모두 협곡 반대편 아래쪽을 향해 있으며, 윌마는 총구를 앞쪽 목표를 향해 겨누고 있습니다.",
        "built_space": "협곡 상공의 공중 공간입니다. 레퍼런스의 절벽 지형과 숲 등 배경 요소가 정확한 위치에 렌더링되었습니다.",
        "entities": "윌마는 헬멧 없이 이끼가 묻은 총을 들고 있습니다. 토니는 레퍼런스에 있는 헬멧과 하네스를 착용했습니다. 은색 추는 줄에 매달린 2개와 아래에 홀로 떠 있는 1개가 있습니다.",
        "hard_violations": [
         "아무런 지지대 없이 허공에 떠 있는 인물들",
         "손이나 연결 고리 없이 허공에 떠 있는 은색 추"
        ],
        "physics": "두 인물 모두 위로 연결된 케이블이나 도약/착지 동작 등 몸을 지탱하는 요소 없이 공중에 떠 있습니다. 아래의 은색 추 역시 지지대나 잡고 있는 손 없이 허공에 떠 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "넓은 화면에서 두 사람이 손을 분명히 맞잡고 아래 왼쪽으로 함께 기울어진 순간은 A보다 정확하지만, 지지선 없는 공중 체류와 은색 추 1개 추가가 치명적이며 토니의 잠긴 헬멧도 빠졌다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "협곡·일몰·복장은 잘 이어지지만 서로를 단단히 붙잡기보다 어깨에 손을 얹은 관계로 보여 핵심 동작이 약하고, 지지 없는 공중 체류와 세 번째 은색 추가 치명적이다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "두 몸의 머리가 왼쪽, 발이 오른쪽에 놓여 아래 왼쪽의 협곡 공간과 전경 절벽 쪽으로 사선 이동하는 모습이다. 윌마의 총구도 아래 왼쪽 협곡과 왼쪽 암벽을 향하며 특정 인물을 겨누지는 않는다. 토니는 윌마 쪽을 내려다보고 윌마는 이동 방향인 왼쪽을 본다.",
        "built_space": "인공 구조물이나 고정 설비는 없다. 왼쪽 전경의 바위 절벽과 소나무, 중앙의 깊은 협곡, 맞은편 수직 암벽과 숲이 덮인 능선이 참고 장소와 같은 배치로 보이며 일몰도 유지된다. 두 사람은 어느 표면에도 닿지 않은 채 협곡 중앙 공중에 있다.",
        "entities": "미국인 성인 남녀 두 명만 보인다. 토니는 30대 남성의 얼굴·검은 작업복·장갑·헬멧·하네스와 구조용 로프를 갖춰 참고 복장과 대체로 맞고, 윌마는 20대 후반 여성의 묶은 금발·녹색 작업복·벨트·권총을 갖췄다. 다만 은색 원통형 추는 토니 아래 줄에 매달린 2개와 더 아래 자유 낙하 중인 1개로 총 3개가 보여, 지정된 2개보다 하나 많다. 로켓·폭발·제3자는 없다.",
        "hard_violations": [
         "토니의 하네스에 연결된 팽팽한 지지선이나 고정점이 없고 두 사람의 도약 출발도 표현되지 않아, 두 몸을 공중에 유지하는 물리적 지지가 없다.",
         "지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
        ],
        "physics": "두 사람은 다리를 뒤로 흘리며 아래 왼쪽으로 이동하는 자세이고 토니의 손은 윌마의 어깨에 닿아 있으며 윌마의 반대팔도 토니 쪽에 접촉한다. 그러나 이는 서로의 몸을 지지할 뿐 두 사람 전체를 공중에 매다는 지지가 아니다. 하네스의 줄은 아래로 느슨하게 늘어져 고정점에 연결되지 않았고, 보이는 절벽에서 밀어낸 발동작도 없다. 두 개의 추는 줄에 매달려 있지만 세 번째 추는 낙하 중이다."
       },
       {
        "label": "B",
        "direction": "두 몸은 머리가 왼쪽, 발이 오른쪽에 놓인 채 아래 왼쪽의 협곡과 전경 절벽 방향으로 함께 기울어져 있다. 윌마는 이동 방향을 보면서 총구를 수평에 가깝게 왼쪽 소나무와 왼쪽 암벽 쪽으로 겨눈다. 토니는 윌마와 두 사람이 잡은 손 쪽을 바라본다.",
        "built_space": "인공 구조물이나 고정 설비는 없다. 왼쪽 전경 바위와 소나무, 중앙의 열린 협곡, 맞은편 암벽과 숲 능선이 참고 장소와 일치하고 절제된 일몰광도 이어진다. 두 인물은 협곡 중앙 공중에 있으며 주변 암벽과의 실제 거리감도 넓은 화면에 맞는다.",
        "entities": "미국인 성인 남녀 두 명만 있으며 얼굴과 체격은 토니와 윌마로 읽힌다. 윌마는 묶은 금발, 녹색 작업복, 벨트와 권총을 갖췄고 토니는 검은 작업복, 하네스와 감긴 구조용 로프를 지녔다. 다만 참고에서 잠긴 토니의 헬멧이 없어 머리와 복장 연속성이 어긋난다. 은색 추는 줄 끝의 2개와 오른쪽 아래에서 떨어지는 1개로 총 3개가 보여 하나가 추가됐다. 제3자나 로켓·폭발은 없다.",
        "hard_violations": [
         "토니의 하네스를 실제로 매다는 팽팽한 선이나 고정점이 전혀 없고 도약의 출발 동작도 없어, 두 공중 인물을 지지하는 것이 없다.",
         "지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
        ],
        "physics": "윌마와 토니는 한 손을 분명히 맞잡아 상호 접촉이 확실하고 다리가 뒤로 흐르는 자세도 같은 아래 왼쪽 궤적을 암시한다. 하지만 손잡기는 두 사람을 서로 연결할 뿐 전체 몸을 떠받치지 못한다. 토니의 하네스에서 내려오는 선들은 모두 느슨하며 고정점이 없고, 발이 밀고 나온 표면이나 추진 장치도 보이지 않는다. 줄 끝의 두 추는 줄이 지지하지만 세 번째 추는 자유 낙하한다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "넓은 화면에서 두 사람이 손을 분명히 맞잡고 아래 왼쪽으로 함께 기울어진 순간은 A보다 정확하지만, 지지선 없는 공중 체류와 은색 추 1개 추가가 치명적이며 토니의 잠긴 헬멧도 빠졌다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "협곡·일몰·복장은 잘 이어지지만 서로를 단단히 붙잡기보다 어깨에 손을 얹은 관계로 보여 핵심 동작이 약하고, 지지 없는 공중 체류와 세 번째 은색 추가 치명적이다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "두 몸의 머리가 왼쪽, 발이 오른쪽에 놓여 아래 왼쪽의 협곡 공간과 전경 절벽 쪽으로 사선 이동하는 모습이다. 윌마의 총구도 아래 왼쪽 협곡과 왼쪽 암벽을 향하며 특정 인물을 겨누지는 않는다. 토니는 윌마 쪽을 내려다보고 윌마는 이동 방향인 왼쪽을 본다.",
        "built_space": "인공 구조물이나 고정 설비는 없다. 왼쪽 전경의 바위 절벽과 소나무, 중앙의 깊은 협곡, 맞은편 수직 암벽과 숲이 덮인 능선이 참고 장소와 같은 배치로 보이며 일몰도 유지된다. 두 사람은 어느 표면에도 닿지 않은 채 협곡 중앙 공중에 있다.",
        "entities": "미국인 성인 남녀 두 명만 보인다. 토니는 30대 남성의 얼굴·검은 작업복·장갑·헬멧·하네스와 구조용 로프를 갖춰 참고 복장과 대체로 맞고, 윌마는 20대 후반 여성의 묶은 금발·녹색 작업복·벨트·권총을 갖췄다. 다만 은색 원통형 추는 토니 아래 줄에 매달린 2개와 더 아래 자유 낙하 중인 1개로 총 3개가 보여, 지정된 2개보다 하나 많다. 로켓·폭발·제3자는 없다.",
        "hard_violations": [
         "토니의 하네스에 연결된 팽팽한 지지선이나 고정점이 없고 두 사람의 도약 출발도 표현되지 않아, 두 몸을 공중에 유지하는 물리적 지지가 없다.",
         "지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
        ],
        "physics": "두 사람은 다리를 뒤로 흘리며 아래 왼쪽으로 이동하는 자세이고 토니의 손은 윌마의 어깨에 닿아 있으며 윌마의 반대팔도 토니 쪽에 접촉한다. 그러나 이는 서로의 몸을 지지할 뿐 두 사람 전체를 공중에 매다는 지지가 아니다. 하네스의 줄은 아래로 느슨하게 늘어져 고정점에 연결되지 않았고, 보이는 절벽에서 밀어낸 발동작도 없다. 두 개의 추는 줄에 매달려 있지만 세 번째 추는 낙하 중이다."
       },
       {
        "label": "A",
        "direction": "두 몸은 머리가 왼쪽, 발이 오른쪽에 놓인 채 아래 왼쪽의 협곡과 전경 절벽 방향으로 함께 기울어져 있다. 윌마는 이동 방향을 보면서 총구를 수평에 가깝게 왼쪽 소나무와 왼쪽 암벽 쪽으로 겨눈다. 토니는 윌마와 두 사람이 잡은 손 쪽을 바라본다.",
        "built_space": "인공 구조물이나 고정 설비는 없다. 왼쪽 전경 바위와 소나무, 중앙의 열린 협곡, 맞은편 암벽과 숲 능선이 참고 장소와 일치하고 절제된 일몰광도 이어진다. 두 인물은 협곡 중앙 공중에 있으며 주변 암벽과의 실제 거리감도 넓은 화면에 맞는다.",
        "entities": "미국인 성인 남녀 두 명만 있으며 얼굴과 체격은 토니와 윌마로 읽힌다. 윌마는 묶은 금발, 녹색 작업복, 벨트와 권총을 갖췄고 토니는 검은 작업복, 하네스와 감긴 구조용 로프를 지녔다. 다만 참고에서 잠긴 토니의 헬멧이 없어 머리와 복장 연속성이 어긋난다. 은색 추는 줄 끝의 2개와 오른쪽 아래에서 떨어지는 1개로 총 3개가 보여 하나가 추가됐다. 제3자나 로켓·폭발은 없다.",
        "hard_violations": [
         "토니의 하네스를 실제로 매다는 팽팽한 선이나 고정점이 전혀 없고 도약의 출발 동작도 없어, 두 공중 인물을 지지하는 것이 없다.",
         "지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
        ],
        "physics": "윌마와 토니는 한 손을 분명히 맞잡아 상호 접촉이 확실하고 다리가 뒤로 흐르는 자세도 같은 아래 왼쪽 궤적을 암시한다. 하지만 손잡기는 두 사람을 서로 연결할 뿐 전체 몸을 떠받치지 못한다. 토니의 하네스에서 내려오는 선들은 모두 느슨하며 고정점이 없고, 발이 밀고 나온 표면이나 추진 장치도 보이지 않는다. 줄 끝의 두 추는 줄이 지지하지만 세 번째 추는 자유 낙하한다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.75,
    "B": 1.75
   },
   "adjusted": {
    "A": 1.5,
    "B": 1.5
   },
   "violations": {
    "A": [
     "[gemini-pro] 지지대나 위로 연결된 로프 없이 허공에 떠 있는 인물들",
     "[gemini-pro] 손이나 지지대 없이 허공에 떠 있는 은색 추",
     "[gpt] 토니의 하네스를 실제로 매다는 팽팽한 선이나 고정점이 전혀 없고 도약의 출발 동작도 없어, 두 공중 인물을 지지하는 것이 없다.",
     "[gpt] 지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
    ],
    "B": [
     "[gemini-pro] 아무런 지지대 없이 허공에 떠 있는 인물들",
     "[gemini-pro] 손이나 연결 고리 없이 허공에 떠 있는 은색 추",
     "[gpt] 토니의 하네스에 연결된 팽팽한 지지선이나 고정점이 없고 두 사람의 도약 출발도 표현되지 않아, 두 몸을 공중에 유지하는 물리적 지지가 없다.",
     "[gpt] 지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "agreed": false
  },
  "totals": {
   "B": 1500,
   "A": 1500
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 1500,
    "verdict_ko": "토니의 헬멧 착용과 레퍼런스의 자세를 잘 유지했으나, 인물과 사물이 지지대 없이 허공에 떠 있어 물리 법칙(Physics)을 위반한 심각한 오류가 있습니다.  ★위반: [gemini-pro] 아무런 지지대 없이 허공에 떠 있는 인물들 / [gemini-pro] 손이나 연결 고리 없이 허공에 떠 있는 은색 추 / [gpt] 토니의 하네스에 연결된 팽팽한 지지선이나 고정점이 없고 두 사람의 도약 출발도 표현되지 않아, 두 몸을 공중에 유지하는 물리적 지지가 없다. / [gpt] 지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
   },
   {
    "label": "A",
    "score": 1500,
    "verdict_ko": "인물과 사물이 지지대 없이 허공에 떠 있는 물리 법칙 위반이 있으며, 토니의 헬멧이 누락되었고 두 사람의 자세가 레퍼런스와 다르게 변형되었습니다.  ★위반: [gemini-pro] 지지대나 위로 연결된 로프 없이 허공에 떠 있는 인물들 / [gemini-pro] 손이나 지지대 없이 허공에 떠 있는 은색 추 / [gpt] 토니의 하네스를 실제로 매다는 팽팽한 선이나 고정점이 전혀 없고 도약의 출발 동작도 없어, 두 공중 인물을 지지하는 것이 없다. / [gpt] 지정된 두 개 외에 세 번째 은색 추가 추가되어 물체 수가 중복되었다."
   }
  ],
  "refs": [
   {
    "label": "SHOT BACKGROUND",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh13__bgfirst_bg.png",
    "asset_id": "9d3c25dc-789d-4bce-b6d2-d2035cf3c5ba",
    "role": "bgfirst_bg"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "needs_reshoot": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bc9c-c500-74b7-b960-da40e965dedf",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh13__bgfirst_bg.png",
   "bg_asset_id": "9d3c25dc-789d-4bce-b6d2-d2035cf3c5ba",
   "bg_record_key": "S6sh13::bgfirst_bg",
   "chain_winner": true,
   "authority": "lane_conti_only"
  },
  "ref_mode": "lane(map_marker): 스케치+prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S6sh4"
  }
 },
 "S6sh13::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:54:14.270391+00:00",
  "fingerprint": "1c8b705fd6a2b0b00c1b9d3af0c5275fe7560fb0ab5708c3236062f67bc6442d",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S6sh13_sel.png",
  "source_sha256": "8ebf2cabacbb5fc51762225deb5eae4cff3b96c2f10a9cb00f45077aa09cc6f4",
  "file": "S6sh13_cine.png",
  "staged_sha256": "446b1ce7dde74294328d43f37c261bfe359306a41b0430f2311b67fb2172f3b1",
  "latency_ms": 12208
 },
 "S6sh18::signage": {
  "fp": "d410374220d31288",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S6sh18": {
  "input_fingerprint": "71c53e2f9278a16d",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): sunset.\n\nSHOT TEXT (authoritative, Korean): 팽팽해진 줄의 장력으로 인해 네모난 송신기가 흑인 남성 추격자의 어깨에서 뜯겨져 허공에 떠 있는 찰나\n\nLOCATION (lock): The exposed exterior cliff edge above the canyon, where the torn-off transmitter is suspended over the drop. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- KEY BACKGROUND ELEMENTS: square signal transmitter (airborne immediately after being torn from the shoulder) — Its detached side faces back toward the pursuer while its outer face turns toward the camera; used as Primary mechanical detail suspended beside the pursuer’s shoulder; hooked rescue rope (taut under tension) — It crosses diagonally from outside the frame to the transmitter hook; used as Graphic force line explaining why the transmitter has detached; canyon approach (visible only as limited background context); used as Scale reference behind the tightly framed action.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated sunset ambient light holds the violent separation in tense moderate contrast without adding an unmotivated source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A taut rescue rope stretches across the gorge, its hook caught on the rectangular signal transmitter. The transmitter has been torn free and is suspended momentarily before falling into the gorge.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 흑인 남성 추격자 (흑인 남성, 성인 연령대, 얼굴형 미상, 자연 범위의 머리색, 머리 길이와 스타일 미상) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): sunset.\n\nSHOT TEXT (authoritative, Korean): 팽팽해진 줄의 장력으로 인해 네모난 송신기가 흑인 남성 추격자의 어깨에서 뜯겨져 허공에 떠 있는 찰나\n\nLOCATION (lock): The exposed exterior cliff edge above the canyon, where the torn-off transmitter is suspended over the drop. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- KEY BACKGROUND ELEMENTS: square signal transmitter (airborne immediately after being torn from the shoulder) — Its detached side faces back toward the pursuer while its outer face turns toward the camera; used as Primary mechanical detail suspended beside the pursuer’s shoulder; hooked rescue rope (taut under tension) — It crosses diagonally from outside the frame to the transmitter hook; used as Graphic force line explaining why the transmitter has detached; canyon approach (visible only as limited background context); used as Scale reference behind the tightly framed action.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated sunset ambient light holds the violent separation in tense moderate contrast without adding an unmotivated source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A taut rescue rope stretches across the gorge, its hook caught on the rectangular signal transmitter. The transmitter has been torn free and is suspended momentarily before falling into the gorge.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 흑인 남성 추격자 (흑인 남성, 성인 연령대, 얼굴형 미상, 자연 범위의 머리색, 머리 길이와 스타일 미상) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): sunset.\n\nSHOT TEXT (authoritative, Korean): 팽팽해진 줄의 장력으로 인해 네모난 송신기가 흑인 남성 추격자의 어깨에서 뜯겨져 허공에 떠 있는 찰나\n\nLOCATION (lock): The exposed exterior cliff edge above the canyon, where the torn-off transmitter is suspended over the drop. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- KEY BACKGROUND ELEMENTS: square signal transmitter (airborne immediately after being torn from the shoulder) — Its detached side faces back toward the pursuer while its outer face turns toward the camera; used as Primary mechanical detail suspended beside the pursuer’s shoulder; hooked rescue rope (taut under tension) — It crosses diagonally from outside the frame to the transmitter hook; used as Graphic force line explaining why the transmitter has detached; canyon approach (visible only as limited background context); used as Scale reference behind the tightly framed action.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Desaturated sunset ambient light holds the violent separation in tense moderate contrast without adding an unmotivated source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A taut rescue rope stretches across the gorge, its hook caught on the rectangular signal transmitter. The transmitter has been torn free and is suspended momentarily before falling into the gorge.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 흑인 남성 추격자 (흑인 남성, 성인 연령대, 얼굴형 미상, 자연 범위의 머리색, 머리 길이와 스타일 미상) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "팽팽한 줄이 화면 우측 상단으로 향하며 송신기를 당기고 있음. 카메라는 남성의 어깨 너머로 협곡을 내려다봄.",
    "built_space": "참조 이미지와 일치하는 깊은 협곡 위 절벽 가장자리. 일몰의 빛이 협곡 전체에 부드럽게 깔려 있음.",
    "entities": "흑인 남성 추격자(뒷모습/측면), 네모난 송신기(갈라진 틈과 회로 노출), 구조용 밧줄과 갈고리 모두 프롬프트와 일치함.",
    "hard_violations": [],
    "physics": "줄의 장력에 의해 송신기가 허공에 떠 있으며, 송신기와 어깨 양쪽 모두에 뜯겨진 천 조각과 기계 장치가 남아있어 물리적 분리 과정이 완벽히 설명됨."
   },
   {
    "label": "B",
    "direction": "줄이 화면 좌측 상단으로 뻗어 있으며, 남성은 전방/하단을 응시하고 있음.",
    "built_space": "협곡을 배경으로 한 절벽 가장자리의 좁은 길. 참조 이미지의 환경과 부합함.",
    "entities": "흑인 남성 추격자(얼굴 노출), 네모난 송신기(끊어진 전선 노출), 구조용 밧줄과 갈고리.",
    "hard_violations": [
     "[gemini-pro] 물리적으로 불가능한 연출: 송신기가 어깨에서 뜯겨져 나갔다고 명시되어 있으나, 남성의 어깨 쪽 의상에는 아무런 찢어짐이나 부착 흔적이 없이 멀쩡함."
    ],
    "physics": "송신기가 줄에 매달려 공중에 떠 있으나, 떨어져 나온 원래 위치(어깨)에 아무런 물리적 흔적이 없어 뜯겨져 나간 액션이 성립하지 않음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "어깨에서 뜯겨져 나간 송신기의 물리적 상태와 팽팽한 줄의 장력을 완벽한 클로즈업 샷으로 훌륭하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "송신기가 공중에 떠 있으나, 남성의 어깨 옷차림이 전혀 찢어지거나 손상되지 않아 뜯겨져 나갔다는 물리적 인과관계가 성립하지 않습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "팽팽한 줄이 화면 우측 상단으로 향하며 송신기를 당기고 있음. 카메라는 남성의 어깨 너머로 협곡을 내려다봄.",
        "built_space": "참조 이미지와 일치하는 깊은 협곡 위 절벽 가장자리. 일몰의 빛이 협곡 전체에 부드럽게 깔려 있음.",
        "entities": "흑인 남성 추격자(뒷모습/측면), 네모난 송신기(갈라진 틈과 회로 노출), 구조용 밧줄과 갈고리 모두 프롬프트와 일치함.",
        "hard_violations": [],
        "physics": "줄의 장력에 의해 송신기가 허공에 떠 있으며, 송신기와 어깨 양쪽 모두에 뜯겨진 천 조각과 기계 장치가 남아있어 물리적 분리 과정이 완벽히 설명됨."
       },
       {
        "label": "B",
        "direction": "줄이 화면 좌측 상단으로 뻗어 있으며, 남성은 전방/하단을 응시하고 있음.",
        "built_space": "협곡을 배경으로 한 절벽 가장자리의 좁은 길. 참조 이미지의 환경과 부합함.",
        "entities": "흑인 남성 추격자(얼굴 노출), 네모난 송신기(끊어진 전선 노출), 구조용 밧줄과 갈고리.",
        "hard_violations": [
         "물리적으로 불가능한 연출: 송신기가 어깨에서 뜯겨져 나갔다고 명시되어 있으나, 남성의 어깨 쪽 의상에는 아무런 찢어짐이나 부착 흔적이 없이 멀쩡함."
        ],
        "physics": "송신기가 줄에 매달려 공중에 떠 있으나, 떨어져 나온 원래 위치(어깨)에 아무런 물리적 흔적이 없어 뜯겨져 나간 액션이 성립하지 않음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "어깨에서 뜯겨져 나간 송신기의 물리적 상태와 팽팽한 줄의 장력을 완벽한 클로즈업 샷으로 훌륭하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "송신기가 공중에 떠 있으나, 남성의 어깨 옷차림이 전혀 찢어지거나 손상되지 않아 뜯겨져 나갔다는 물리적 인과관계가 성립하지 않습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "팽팽한 줄이 화면 우측 상단으로 향하며 송신기를 당기고 있음. 카메라는 남성의 어깨 너머로 협곡을 내려다봄.",
        "built_space": "참조 이미지와 일치하는 깊은 협곡 위 절벽 가장자리. 일몰의 빛이 협곡 전체에 부드럽게 깔려 있음.",
        "entities": "흑인 남성 추격자(뒷모습/측면), 네모난 송신기(갈라진 틈과 회로 노출), 구조용 밧줄과 갈고리 모두 프롬프트와 일치함.",
        "hard_violations": [],
        "physics": "줄의 장력에 의해 송신기가 허공에 떠 있으며, 송신기와 어깨 양쪽 모두에 뜯겨진 천 조각과 기계 장치가 남아있어 물리적 분리 과정이 완벽히 설명됨."
       },
       {
        "label": "B",
        "direction": "줄이 화면 좌측 상단으로 뻗어 있으며, 남성은 전방/하단을 응시하고 있음.",
        "built_space": "협곡을 배경으로 한 절벽 가장자리의 좁은 길. 참조 이미지의 환경과 부합함.",
        "entities": "흑인 남성 추격자(얼굴 노출), 네모난 송신기(끊어진 전선 노출), 구조용 밧줄과 갈고리.",
        "hard_violations": [
         "물리적으로 불가능한 연출: 송신기가 어깨에서 뜯겨져 나갔다고 명시되어 있으나, 남성의 어깨 쪽 의상에는 아무런 찢어짐이나 부착 흔적이 없이 멀쩡함."
        ],
        "physics": "송신기가 줄에 매달려 공중에 떠 있으나, 떨어져 나온 원래 위치(어깨)에 아무런 물리적 흔적이 없어 뜯겨져 나간 액션이 성립하지 않음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "송신기의 외측 면이 카메라를 향하고 뜯긴 면은 추격자 쪽을 향하며, 팽팽한 대각선 로프가 어깨에서 분리시키는 순간을 정확한 클로즈업으로 보여준다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "로프 장력과 어깨에서 뜯겨 나온 순간은 명확하지만, 추격자 뒤쪽 카메라에서 송신기의 회로 면이 보이므로 ‘외측 면은 카메라, 뜯긴 면은 추격자’라는 방향 관계가 맞지 않는다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "추격자의 시선은 바로 왼쪽 어깨 옆에 떠 있는 송신기에 닿는다. 로프는 화면 왼쪽 위 바깥에서 송신기 고리까지 대각선으로 팽팽하게 이어져 송신기를 어깨에서 왼쪽 위로 잡아당긴다. 송신기의 밝은 외측 면은 카메라를 향하고, 노출된 전선과 뜯긴 측면은 추격자의 어깨 쪽을 향한다.",
        "built_space": "인공 구조물 내부가 아니라 노출된 절벽 가장자리다. 왼쪽에 바위 절벽 상단과 좁은 접근 지면이 보이고, 뒤에는 깊은 협곡과 맞은편 수직 암벽이 제한된 배경으로 배치된다. 추격자는 절벽 가장자리에서 상반신만 클로즈업되며, 고정 설비나 불가능한 반사는 없다.",
        "entities": "성인 흑인 남성 추격자 한 명만 보이며 인종·성별·연령대가 지시와 맞는다. 네모난 금속 송신기 한 개가 어깨 바로 옆에 분리되어 있고, 고리 달린 구조와 해진 전선이 보인다. 구조 로프 한 줄이 송신기 고리에 연결되어 있다. 협곡과 암벽의 재질 및 황혼 색조도 참조 장소와 부합하며 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "송신기는 손이나 어깨에 붙어 있지 않지만, 고리에 걸린 팽팽한 로프가 명확히 지지하고 왼쪽 위로 당긴다. 뜯긴 전선과 어깨 바로 옆 위치가 방금 분리된 관성을 설명한다. 남성의 하체와 발은 프레임 밖이지만 몸통은 정상적인 직립·기울기 상태로 이어져 공중에 떠 있는 것으로 보이지 않는다."
       },
       {
        "label": "B",
        "direction": "로프는 화면 오른쪽 위 바깥에서 송신기 고리까지 팽팽하게 이어져 송신기를 추격자의 왼쪽 어깨에서 오른쪽 위로 당긴다. 힘의 방향과 분리 대상은 명확하다. 다만 카메라가 추격자의 뒤쪽에 있는데 카메라와 추격자 양쪽에서 같은 회로 노출 면을 보게 되어, 뜯긴 면은 추격자 쪽이고 외측 면은 카메라 쪽이어야 한다는 지정 방향을 충족하지 못한다. 추격자의 눈은 보이지 않아 시선의 도착점은 확인할 수 없다.",
        "built_space": "노출된 절벽 가장자리에서 내려다본 협곡이 배경이며, 아래 계곡 바닥과 맞은편 암벽이 제한적으로 보인다. 추격자는 절벽 쪽을 향한 채 어깨와 뒤통수 위주로 클로즈업된다. 인공 구조물이나 고정 설비는 없고 불가능한 반사도 없다.",
        "entities": "성인 흑인 남성 추격자 한 명, 어깨에서 분리된 네모난 송신기 한 개, 송신기의 어깨 장착 잔여부, 고리에 걸린 구조 로프 한 줄이 보인다. 사람의 인종·성별·연령대는 맞고 협곡 장소도 참조와 일치한다. 송신기의 카메라 쪽 면은 외장 면보다 내부 회로 면처럼 보여 지정된 면 배치가 어긋나며, 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "공중의 송신기는 오른쪽 위로 이어진 팽팽한 로프와 금속 고리가 지지한다. 어깨에 남은 찢어진 장착부와 송신기의 해진 섬유가 실제 분리 지점을 설명하므로 부유하지 않는다. 추격자의 하체는 프레임 밖이지만 상체가 절벽 가장자리 쪽으로 기울어진 정상적인 자세여서 지지 불능으로 보이지 않는다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "송신기의 외측 면이 카메라를 향하고 뜯긴 면은 추격자 쪽을 향하며, 팽팽한 대각선 로프가 어깨에서 분리시키는 순간을 정확한 클로즈업으로 보여준다."
       },
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "로프 장력과 어깨에서 뜯겨 나온 순간은 명확하지만, 추격자 뒤쪽 카메라에서 송신기의 회로 면이 보이므로 ‘외측 면은 카메라, 뜯긴 면은 추격자’라는 방향 관계가 맞지 않는다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "추격자의 시선은 바로 왼쪽 어깨 옆에 떠 있는 송신기에 닿는다. 로프는 화면 왼쪽 위 바깥에서 송신기 고리까지 대각선으로 팽팽하게 이어져 송신기를 어깨에서 왼쪽 위로 잡아당긴다. 송신기의 밝은 외측 면은 카메라를 향하고, 노출된 전선과 뜯긴 측면은 추격자의 어깨 쪽을 향한다.",
        "built_space": "인공 구조물 내부가 아니라 노출된 절벽 가장자리다. 왼쪽에 바위 절벽 상단과 좁은 접근 지면이 보이고, 뒤에는 깊은 협곡과 맞은편 수직 암벽이 제한된 배경으로 배치된다. 추격자는 절벽 가장자리에서 상반신만 클로즈업되며, 고정 설비나 불가능한 반사는 없다.",
        "entities": "성인 흑인 남성 추격자 한 명만 보이며 인종·성별·연령대가 지시와 맞는다. 네모난 금속 송신기 한 개가 어깨 바로 옆에 분리되어 있고, 고리 달린 구조와 해진 전선이 보인다. 구조 로프 한 줄이 송신기 고리에 연결되어 있다. 협곡과 암벽의 재질 및 황혼 색조도 참조 장소와 부합하며 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "송신기는 손이나 어깨에 붙어 있지 않지만, 고리에 걸린 팽팽한 로프가 명확히 지지하고 왼쪽 위로 당긴다. 뜯긴 전선과 어깨 바로 옆 위치가 방금 분리된 관성을 설명한다. 남성의 하체와 발은 프레임 밖이지만 몸통은 정상적인 직립·기울기 상태로 이어져 공중에 떠 있는 것으로 보이지 않는다."
       },
       {
        "label": "A",
        "direction": "로프는 화면 오른쪽 위 바깥에서 송신기 고리까지 팽팽하게 이어져 송신기를 추격자의 왼쪽 어깨에서 오른쪽 위로 당긴다. 힘의 방향과 분리 대상은 명확하다. 다만 카메라가 추격자의 뒤쪽에 있는데 카메라와 추격자 양쪽에서 같은 회로 노출 면을 보게 되어, 뜯긴 면은 추격자 쪽이고 외측 면은 카메라 쪽이어야 한다는 지정 방향을 충족하지 못한다. 추격자의 눈은 보이지 않아 시선의 도착점은 확인할 수 없다.",
        "built_space": "노출된 절벽 가장자리에서 내려다본 협곡이 배경이며, 아래 계곡 바닥과 맞은편 암벽이 제한적으로 보인다. 추격자는 절벽 쪽을 향한 채 어깨와 뒤통수 위주로 클로즈업된다. 인공 구조물이나 고정 설비는 없고 불가능한 반사도 없다.",
        "entities": "성인 흑인 남성 추격자 한 명, 어깨에서 분리된 네모난 송신기 한 개, 송신기의 어깨 장착 잔여부, 고리에 걸린 구조 로프 한 줄이 보인다. 사람의 인종·성별·연령대는 맞고 협곡 장소도 참조와 일치한다. 송신기의 카메라 쪽 면은 외장 면보다 내부 회로 면처럼 보여 지정된 면 배치가 어긋나며, 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "공중의 송신기는 오른쪽 위로 이어진 팽팽한 로프와 금속 고리가 지지한다. 어깨에 남은 찢어진 장착부와 송신기의 해진 섬유가 실제 분리 지점을 설명하므로 부유하지 않는다. 추격자의 하체는 프레임 밖이지만 상체가 절벽 가장자리 쪽으로 기울어진 정상적인 자세여서 지지 불능으로 보이지 않는다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.778,
    "B": 1.4
   },
   "adjusted": {
    "A": 1.778,
    "B": 1.15
   },
   "violations": {
    "B": [
     "[gemini-pro] 물리적으로 불가능한 연출: 송신기가 어깨에서 뜯겨져 나갔다고 명시되어 있으나, 남성의 어깨 쪽 의상에는 아무런 찢어짐이나 부착 흔적이 없이 멀쩡함."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "agreed": false
  },
  "totals": {
   "A": 1778,
   "B": 1150
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1778,
    "verdict_ko": "어깨에서 뜯겨져 나간 송신기의 물리적 상태와 팽팽한 줄의 장력을 완벽한 클로즈업 샷으로 훌륭하게 구현했습니다."
   },
   {
    "label": "B",
    "score": 1150,
    "verdict_ko": "송신기가 공중에 떠 있으나, 남성의 어깨 옷차림이 전혀 찢어지거나 손상되지 않아 뜯겨져 나갔다는 물리적 인과관계가 성립하지 않습니다.  ★위반: [gemini-pro] 물리적으로 불가능한 연출: 송신기가 어깨에서 뜯겨져 나갔다고 명시되어 있으나, 남성의 어깨 쪽 의상에는 아무런 찢어짐이나 부착 흔적이 없이 멀쩡함."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S6sh13_sel.png",
    "asset_id": "049ca1fe-b2bd-413a-a2af-5f1bab8c67a7",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bcab-2a24-7228-a518-2b977c2da165",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S6sh13"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S6sh18::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:55:12.992689+00:00",
  "fingerprint": "0fc560526a4597babea97cc196e8a8514a65a940366c89e022b7d7c1f96ab3ba",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S6sh18_sel.png",
  "source_sha256": "7ad992a5becc2f42bfe0227744ceaec2d4f47f649277d82402b4f36312c00be0",
  "file": "S6sh18_cine.png",
  "staged_sha256": "05524d3d5d25d06632f597bac986358d248adf6730212503fc74a8e5a8b5b7de",
  "latency_ms": 13154
 },
 "S7sh4::signage": {
  "fp": "564894326659a4ea",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "era_assess::a678850a4b8b52be": {
  "subjects": [],
  "subject_text": "바위벽으로 위장된 산악 접촉소 외부 입구\n자연 암벽처럼 이어진 은폐형 산악 입구. 닫히면 평범한 바위 표면이며, 황혼의 숲과 산맥에 둘러싸여 있다.",
  "identity": "canonical",
  "scope_id": "L10",
  "scope_role": "location_exterior",
  "scope_sha": "99ecca21ab579e45"
 },
 "S7sh4::bgfirst_bg": {
  "input_fingerprint": "3ecf152f24925f28",
  "prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 완전히 열린 바위벽 안쪽 어둠 속에서 장총을 든 백인 남성이 밖을 향해 총구를 겨눈 전신\n\nLOCATION (lock): Just inside the open concealed rock doorway of a small underground contact room, facing outward toward the mountain exterior. The interior remains dark against the twilight entrance.\n\nTIME OF DAY (lock): twilight.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: opened rock wall entrance (fully open) — The separated rock face exposes its inner side around the cellar opening; used as Architectural frame around the armed full-body reveal; small cellar interior (dark and revealed behind the opened wall); used as Background field isolating the armed figure; long rifle (shouldered and aimed outward toward Wilma) — The barrel points obliquely past the camera position rather than directly into the lens; used as Threat line leading from the interior toward the entrance.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Twilight exterior ambience separates the armed figure from the explicitly dark cellar interior in restrained low-key contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.\n\nThe THIRD attached image (STRUCTURE LOOK) is the identity source of the fixed structure at this location: its faces, openings, levels, materials and signage are truth. Where it conflicts with the LOCATION PHOTOGRAPH about the structure itself, the STRUCTURE LOOK wins; the photograph still governs the surroundings, time of day and lighting.",
  "effective_prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 완전히 열린 바위벽 안쪽 어둠 속에서 장총을 든 백인 남성이 밖을 향해 총구를 겨눈 전신\n\nLOCATION (lock): Just inside the open concealed rock doorway of a small underground contact room, facing outward toward the mountain exterior. The interior remains dark against the twilight entrance.\n\nTIME OF DAY (lock): twilight.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: opened rock wall entrance (fully open) — The separated rock face exposes its inner side around the cellar opening; used as Architectural frame around the armed full-body reveal; small cellar interior (dark and revealed behind the opened wall); used as Background field isolating the armed figure; long rifle (shouldered and aimed outward toward Wilma) — The barrel points obliquely past the camera position rather than directly into the lens; used as Threat line leading from the interior toward the entrance.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Twilight exterior ambience separates the armed figure from the explicitly dark cellar interior in restrained low-key contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.\n\nThe THIRD attached image (STRUCTURE LOOK) is the identity source of the fixed structure at this location: its faces, openings, levels, materials and signage are truth. Where it conflicts with the LOCATION PHOTOGRAPH about the structure itself, the STRUCTURE LOOK wins; the photograph still governs the surroundings, time of day and lighting.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S7sh4__bgfirst_bg.png",
  "asset_id": "9b24e4cd-4d4f-468b-989e-a586203ee9c2",
  "input_asset_ids": [
   "998c505f-f6c3-4cc5-a1bc-ab444444c757",
   "267f2645-3473-4888-9e13-cc2b20962e1d",
   "95a8376a-4dea-407c-8b9a-dcae414da6d8"
  ]
 },
 "S7sh4": {
  "input_fingerprint": "94d81ff0e13ca3f8",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): twilight.\n\nSHOT TEXT (authoritative, Korean): 완전히 열린 바위벽 안쪽 어둠 속에서 장총을 든 백인 남성이 밖을 향해 총구를 겨눈 전신\n\nLOCATION (lock): Just inside the open concealed rock doorway of a small underground contact room, facing outward toward the mountain exterior. The interior remains dark against the twilight entrance. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nSTRUCTURE LOOK AUTHORITY: the attached STRUCTURE LOOK photograph is the identity of the fixed structure at this location — wherever that structure appears in the frame, its shape, proportions, openings, materials and colors are LOCKED to it. The LOCATION PHOTOGRAPH remains the authority for this shot's sub-space, surroundings, time of day and lighting. If the two conflict on the structure itself, the STRUCTURE LOOK photo wins; for everything else, the LOCATION PHOTOGRAPH wins.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: opened rock wall entrance (fully open) — The separated rock face exposes its inner side around the cellar opening; used as Architectural frame around the armed full-body reveal; small cellar interior (dark and revealed behind the opened wall); used as Background field isolating the armed figure; long rifle (shouldered and aimed outward toward Wilma) — The barrel points obliquely past the camera position rather than directly into the lens; used as Threat line leading from the interior toward the entrance.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Twilight exterior ambience separates the armed figure from the explicitly dark cellar interior in restrained low-key contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The ordinary-looking rock wall is fully open, revealing a small dark underground room. The round communication panel and ceiling map inside are not yet illuminated. 백인 남성: He stands inside the opened contact station aiming a long gun outward.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 백인 남성 right now, so 백인 남성's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 백인 남성: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 백인 남성 (백인 남성, 성인 연령대, 얼굴형 미상, 자연 범위의 머리색, 머리 길이와 스타일 미상) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Stage the shot. The FIRST attached image (SHOT BACKGROUND) is the finished empty background of this shot — keep it EXACTLY: its camera, perspective, architecture, lighting and every fixed feature stay untouched. The SECOND attached image (LAYOUT SKETCH) tells you ONLY where the people go: each sketched person's position, screen size, pose and the gaze/motion arrows. Ignore the sketch's background lines. The CHARACTER REFERENCE photographs show the real people.\nPlace the real people into the background at exactly the sketched positions, sizes and poses, following the arrow directions. No sketch lines or arrows may remain.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): twilight.\n\nSHOT TEXT (authoritative, Korean): 완전히 열린 바위벽 안쪽 어둠 속에서 장총을 든 백인 남성이 밖을 향해 총구를 겨눈 전신\n\nLOCATION (lock): Just inside the open concealed rock doorway of a small underground contact room, facing outward toward the mountain exterior. The interior remains dark against the twilight entrance. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: opened rock wall entrance (fully open) — The separated rock face exposes its inner side around the cellar opening; used as Architectural frame around the armed full-body reveal; small cellar interior (dark and revealed behind the opened wall); used as Background field isolating the armed figure; long rifle (shouldered and aimed outward toward Wilma) — The barrel points obliquely past the camera position rather than directly into the lens; used as Threat line leading from the interior toward the entrance.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Twilight exterior ambience separates the armed figure from the explicitly dark cellar interior in restrained low-key contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The ordinary-looking rock wall is fully open, revealing a small dark underground room. The round communication panel and ceiling map inside are not yet illuminated. 백인 남성: He stands inside the opened contact station aiming a long gun outward.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 백인 남성 right now, so 백인 남성's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 백인 남성: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 백인 남성 (백인 남성, 성인 연령대, 얼굴형 미상, 자연 범위의 머리색, 머리 길이와 스타일 미상) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): twilight.\n\nSHOT TEXT (authoritative, Korean): 완전히 열린 바위벽 안쪽 어둠 속에서 장총을 든 백인 남성이 밖을 향해 총구를 겨눈 전신\n\nLOCATION (lock): Just inside the open concealed rock doorway of a small underground contact room, facing outward toward the mountain exterior. The interior remains dark against the twilight entrance. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nSTRUCTURE LOOK AUTHORITY: the attached STRUCTURE LOOK photograph is the identity of the fixed structure at this location — wherever that structure appears in the frame, its shape, proportions, openings, materials and colors are LOCKED to it. The LOCATION PHOTOGRAPH remains the authority for this shot's sub-space, surroundings, time of day and lighting. If the two conflict on the structure itself, the STRUCTURE LOOK photo wins; for everything else, the LOCATION PHOTOGRAPH wins.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: opened rock wall entrance (fully open) — The separated rock face exposes its inner side around the cellar opening; used as Architectural frame around the armed full-body reveal; small cellar interior (dark and revealed behind the opened wall); used as Background field isolating the armed figure; long rifle (shouldered and aimed outward toward Wilma) — The barrel points obliquely past the camera position rather than directly into the lens; used as Threat line leading from the interior toward the entrance.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Twilight exterior ambience separates the armed figure from the explicitly dark cellar interior in restrained low-key contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The ordinary-looking rock wall is fully open, revealing a small dark underground room. The round communication panel and ceiling map inside are not yet illuminated. 백인 남성: He stands inside the opened contact station aiming a long gun outward.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 백인 남성 right now, so 백인 남성's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 백인 남성: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 백인 남성 (백인 남성, 성인 연령대, 얼굴형 미상, 자연 범위의 머리색, 머리 길이와 스타일 미상) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S7sh4__bgfirst_bg.png",
     "asset_id": "9b24e4cd-4d4f-468b-989e-a586203ee9c2",
     "role": "bgfirst_bg"
    },
    {
     "label": "LAYOUT SKETCH (people placement only)",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/conti/conti_S7sh4.png",
     "asset_id": "998c505f-f6c3-4cc5-a1bc-ab444444c757",
     "role": "conti_light"
    }
   ],
   "B": [
    {
     "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its spatial layout, surroundings, fixed features, time of day and lighting mood are spatial truth; stage the moment inside this place. If a STRUCTURE LOOK photograph is also attached, that photo wins for the fixed structure itself — this photograph wins for everything around it. Never copy its camera framing.",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/episodes/9fc33eb1-970b-4e5e-8b98-53732d321d63/images/background_chain/L10B01.png",
     "asset_id": "267f2645-3473-4888-9e13-cc2b20962e1d",
     "role": "location_plate"
    },
    {
     "label": "STRUCTURE LOOK — the confirmed photograph of the fixed structure at this location: wherever the structure appears in the frame, its shape, proportions, materials, colors and openings are LOCKED to this photo. Never copy its camera framing, time of day or lighting — the shot text and the LOCATION PHOTOGRAPH are the authorities for those.",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/background_chain/seed_bg_mountain_outpost_sel.png",
     "asset_id": "95a8376a-4dea-407c-8b9a-dcae414da6d8",
     "role": "structure_seed_look"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "judge_flip_skipped": "judge_fn_owns_order",
  "readings": [
   {
    "label": "A",
    "direction": "남성이 들고 있는 장총의 총구는 카메라의 오른쪽을 향해 사선으로 밖을 겨누고 있습니다.",
    "built_space": "참조 이미지와 동일한 두꺼운 암석 문과 철제 프레임이 있는 지하 벙커 입구가 묘사되었고, 남성은 그 내부 바닥에 안정적으로 위치해 있습니다.",
    "entities": "백인 남성, 전신, 장총 모두 프롬프트의 요구사항과 정확히 일치합니다.",
    "hard_violations": [],
    "physics": "남성은 두 발로 바닥을 확실히 지탱하고 있으며, 두 손으로 장총의 무게를 자연스럽게 받치고 견착한 자세를 취하고 있습니다."
   },
   {
    "label": "B",
    "direction": "남성이 무기를 카메라의 왼쪽을 향해 사선으로 겨누고 있으나 총구의 형태가 명확하지 않습니다.",
    "built_space": "참조 이미지의 암석 문과 내부 공간이 잘 묘사되어 있으며, 인물은 문틀 안쪽에 서 있습니다.",
    "entities": "백인 남성과 전신 구도는 일치하나, 손에 들고 있는 무기가 '장총(long rifle)'이 아닌 총열이 잘려나간 기형적인 물체로 묘사되었습니다.",
    "hard_violations": [
     "[gemini-pro] 물리적으로 불가능하거나 심하게 왜곡된 물체(총열이 잘려나간 기형적인 형태의 총기)"
    ],
    "physics": "인물이 바닥을 딛고 서 있는 자세와 파지법은 존재하지만, 들고 있는 물체의 앞부분이 공기 중으로 사라진 듯 비정상적인 형태를 띠고 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "프롬프트가 요구한 '전신' 구도와 '장총을 겨누는 자세'를 참조된 공간 안에서 매우 사실적이고 정확하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "공간과 인물의 배치는 준수하나, 인물이 들고 있는 장총의 총열이 중간에 잘린 형태로 렌더링되어 치명적인 객체 왜곡이 발생했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "남성이 들고 있는 장총의 총구는 카메라의 오른쪽을 향해 사선으로 밖을 겨누고 있습니다.",
        "built_space": "참조 이미지와 동일한 두꺼운 암석 문과 철제 프레임이 있는 지하 벙커 입구가 묘사되었고, 남성은 그 내부 바닥에 안정적으로 위치해 있습니다.",
        "entities": "백인 남성, 전신, 장총 모두 프롬프트의 요구사항과 정확히 일치합니다.",
        "hard_violations": [],
        "physics": "남성은 두 발로 바닥을 확실히 지탱하고 있으며, 두 손으로 장총의 무게를 자연스럽게 받치고 견착한 자세를 취하고 있습니다."
       },
       {
        "label": "B",
        "direction": "남성이 무기를 카메라의 왼쪽을 향해 사선으로 겨누고 있으나 총구의 형태가 명확하지 않습니다.",
        "built_space": "참조 이미지의 암석 문과 내부 공간이 잘 묘사되어 있으며, 인물은 문틀 안쪽에 서 있습니다.",
        "entities": "백인 남성과 전신 구도는 일치하나, 손에 들고 있는 무기가 '장총(long rifle)'이 아닌 총열이 잘려나간 기형적인 물체로 묘사되었습니다.",
        "hard_violations": [
         "물리적으로 불가능하거나 심하게 왜곡된 물체(총열이 잘려나간 기형적인 형태의 총기)"
        ],
        "physics": "인물이 바닥을 딛고 서 있는 자세와 파지법은 존재하지만, 들고 있는 물체의 앞부분이 공기 중으로 사라진 듯 비정상적인 형태를 띠고 있습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "프롬프트가 요구한 '전신' 구도와 '장총을 겨누는 자세'를 참조된 공간 안에서 매우 사실적이고 정확하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "공간과 인물의 배치는 준수하나, 인물이 들고 있는 장총의 총열이 중간에 잘린 형태로 렌더링되어 치명적인 객체 왜곡이 발생했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "남성이 들고 있는 장총의 총구는 카메라의 오른쪽을 향해 사선으로 밖을 겨누고 있습니다.",
        "built_space": "참조 이미지와 동일한 두꺼운 암석 문과 철제 프레임이 있는 지하 벙커 입구가 묘사되었고, 남성은 그 내부 바닥에 안정적으로 위치해 있습니다.",
        "entities": "백인 남성, 전신, 장총 모두 프롬프트의 요구사항과 정확히 일치합니다.",
        "hard_violations": [],
        "physics": "남성은 두 발로 바닥을 확실히 지탱하고 있으며, 두 손으로 장총의 무게를 자연스럽게 받치고 견착한 자세를 취하고 있습니다."
       },
       {
        "label": "B",
        "direction": "남성이 무기를 카메라의 왼쪽을 향해 사선으로 겨누고 있으나 총구의 형태가 명확하지 않습니다.",
        "built_space": "참조 이미지의 암석 문과 내부 공간이 잘 묘사되어 있으며, 인물은 문틀 안쪽에 서 있습니다.",
        "entities": "백인 남성과 전신 구도는 일치하나, 손에 들고 있는 무기가 '장총(long rifle)'이 아닌 총열이 잘려나간 기형적인 물체로 묘사되었습니다.",
        "hard_violations": [
         "물리적으로 불가능하거나 심하게 왜곡된 물체(총열이 잘려나간 기형적인 형태의 총기)"
        ],
        "physics": "인물이 바닥을 딛고 서 있는 자세와 파지법은 존재하지만, 들고 있는 물체의 앞부분이 공기 중으로 사라진 듯 비정상적인 형태를 띠고 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "열린 암벽 출입구 안쪽의 어두운 방에 선 백인 남성의 전신과, 카메라를 비껴 외부의 윌마 쪽으로 향하는 장총의 위협선이 권위적 쇼트 지시를 가장 정확히 구현한다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "와이드 전신과 열린 출입구는 맞지만 총구가 외부의 윌마가 있을 방향보다 화면 오른쪽 암벽 쪽을 향해 핵심 조준 관계가 어긋나고, 남성도 방 안쪽보다 문턱에 치우쳐 있다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "남성의 시선과 장총 총구가 모두 화면 왼쪽의 산 외부를 향한다. 총구는 렌즈를 정면으로 겨누지 않고 카메라 위치의 왼쪽을 비껴가며, 화면에는 보이지 않는 외부의 윌마를 겨누는 방향으로 읽힌다.",
        "built_space": "암벽에 숨겨진 출입구 하나가 완전히 열려 있고, 분리된 두꺼운 암석 문짝이 화면 왼쪽에 펼쳐져 있다. 금속 문틀 너머에는 작은 지하실 하나가 어둡게 드러나며 남성은 문턱보다 안쪽 바닥에 서 있다. 내부의 원형 패널은 희미한 비점등 상태이고 천장 지도도 점등되지 않았다. 반사나 중복된 고정 설비는 보이지 않는다.",
        "entities": "등장 인물은 성인 백인 남성 한 명뿐이며 다른 사람은 없다. 남성은 자연스러운 짧은 갈색 머리와 야외용 회갈색 복장을 하고 있으며, 두 손으로 실제 장총을 잡고 있다. 장총은 개머리판·총열·방아쇠 부위가 식별되고 읽을 수 있는 글자나 로고는 없다.",
        "hard_violations": [],
        "physics": "남성의 두 발이 실내 바닥에 닿아 체중을 지지하며, 약간 벌린 안정된 사격 자세다. 개머리판은 어깨에 붙고 한 손은 방아쇠 쪽 손잡이, 다른 손은 총열 아래를 받쳐 장총의 무게와 조준을 물리적으로 지지한다."
       },
       {
        "label": "B",
        "direction": "남성의 시선은 장총을 따라 화면 오른쪽을 향하고 총구도 오른쪽 문설주와 인접 암벽 방향으로 뻗는다. 카메라를 비껴가기는 하지만 열린 산 외부나 화면 왼쪽 바깥의 윌마를 겨누기보다 출입구 오른쪽 암벽 쪽을 겨누는 것으로 읽힌다.",
        "built_space": "암벽 출입구 하나와 화면 왼쪽으로 완전히 열린 두꺼운 암석 문짝 하나가 보인다. 금속 문틀 뒤에 작은 지하실이 있으며 기둥형 설비와 어두운 천장, 후면 벽이 보인다. 남성은 깊은 실내가 아니라 문턱 바로 안쪽에 서 있고 오른발은 하단 문틀 가까이에 있다. 원형 패널은 어둡고 별도 조명은 켜지지 않았으며 불가능한 반사나 설비 중복은 없다.",
        "entities": "성인 백인 남성 한 명과 장총 한 정이 보이며 추가 인물은 없다. 남성은 짧은 갈색 머리와 어두운 재킷·바지를 착용했다. 장총은 충분히 긴 총열을 지닌 실물 소총으로 보이고 두 손이 모두 총에 닿아 있다. 읽을 수 있는 글자, 자막, 로고는 없다.",
        "hard_violations": [],
        "physics": "남성은 두 발을 문턱 안쪽 바닥에 넓게 딛고 있어 몸이 지지된다. 개머리판이 어깨에 닿고 오른손이 방아쇠 부근을 잡으며 왼손이 앞부분을 받쳐 총도 지지된다. 부유하거나 지지되지 않은 신체·물체는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "열린 암벽 출입구 안쪽의 어두운 방에 선 백인 남성의 전신과, 카메라를 비껴 외부의 윌마 쪽으로 향하는 장총의 위협선이 권위적 쇼트 지시를 가장 정확히 구현한다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "와이드 전신과 열린 출입구는 맞지만 총구가 외부의 윌마가 있을 방향보다 화면 오른쪽 암벽 쪽을 향해 핵심 조준 관계가 어긋나고, 남성도 방 안쪽보다 문턱에 치우쳐 있다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "남성의 시선과 장총 총구가 모두 화면 왼쪽의 산 외부를 향한다. 총구는 렌즈를 정면으로 겨누지 않고 카메라 위치의 왼쪽을 비껴가며, 화면에는 보이지 않는 외부의 윌마를 겨누는 방향으로 읽힌다.",
        "built_space": "암벽에 숨겨진 출입구 하나가 완전히 열려 있고, 분리된 두꺼운 암석 문짝이 화면 왼쪽에 펼쳐져 있다. 금속 문틀 너머에는 작은 지하실 하나가 어둡게 드러나며 남성은 문턱보다 안쪽 바닥에 서 있다. 내부의 원형 패널은 희미한 비점등 상태이고 천장 지도도 점등되지 않았다. 반사나 중복된 고정 설비는 보이지 않는다.",
        "entities": "등장 인물은 성인 백인 남성 한 명뿐이며 다른 사람은 없다. 남성은 자연스러운 짧은 갈색 머리와 야외용 회갈색 복장을 하고 있으며, 두 손으로 실제 장총을 잡고 있다. 장총은 개머리판·총열·방아쇠 부위가 식별되고 읽을 수 있는 글자나 로고는 없다.",
        "hard_violations": [],
        "physics": "남성의 두 발이 실내 바닥에 닿아 체중을 지지하며, 약간 벌린 안정된 사격 자세다. 개머리판은 어깨에 붙고 한 손은 방아쇠 쪽 손잡이, 다른 손은 총열 아래를 받쳐 장총의 무게와 조준을 물리적으로 지지한다."
       },
       {
        "label": "A",
        "direction": "남성의 시선은 장총을 따라 화면 오른쪽을 향하고 총구도 오른쪽 문설주와 인접 암벽 방향으로 뻗는다. 카메라를 비껴가기는 하지만 열린 산 외부나 화면 왼쪽 바깥의 윌마를 겨누기보다 출입구 오른쪽 암벽 쪽을 겨누는 것으로 읽힌다.",
        "built_space": "암벽 출입구 하나와 화면 왼쪽으로 완전히 열린 두꺼운 암석 문짝 하나가 보인다. 금속 문틀 뒤에 작은 지하실이 있으며 기둥형 설비와 어두운 천장, 후면 벽이 보인다. 남성은 깊은 실내가 아니라 문턱 바로 안쪽에 서 있고 오른발은 하단 문틀 가까이에 있다. 원형 패널은 어둡고 별도 조명은 켜지지 않았으며 불가능한 반사나 설비 중복은 없다.",
        "entities": "성인 백인 남성 한 명과 장총 한 정이 보이며 추가 인물은 없다. 남성은 짧은 갈색 머리와 어두운 재킷·바지를 착용했다. 장총은 충분히 긴 총열을 지닌 실물 소총으로 보이고 두 손이 모두 총에 닿아 있다. 읽을 수 있는 글자, 자막, 로고는 없다.",
        "hard_violations": [],
        "physics": "남성은 두 발을 문턱 안쪽 바닥에 넓게 딛고 있어 몸이 지지된다. 개머리판이 어깨에 닿고 오른손이 방아쇠 부근을 잡으며 왼손이 앞부분을 받쳐 총도 지지된다. 부유하거나 지지되지 않은 신체·물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.667,
    "B": 1.444
   },
   "adjusted": {
    "A": 1.667,
    "B": 1.194
   },
   "violations": {
    "B": [
     "[gemini-pro] 물리적으로 불가능하거나 심하게 왜곡된 물체(총열이 잘려나간 기형적인 형태의 총기)"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "agreed": false
  },
  "totals": {
   "A": 1667,
   "B": 1194
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1667,
    "verdict_ko": "프롬프트가 요구한 '전신' 구도와 '장총을 겨누는 자세'를 참조된 공간 안에서 매우 사실적이고 정확하게 구현했습니다."
   },
   {
    "label": "B",
    "score": 1194,
    "verdict_ko": "공간과 인물의 배치는 준수하나, 인물이 들고 있는 장총의 총열이 중간에 잘린 형태로 렌더링되어 치명적인 객체 왜곡이 발생했습니다.  ★위반: [gemini-pro] 물리적으로 불가능하거나 심하게 왜곡된 물체(총열이 잘려나간 기형적인 형태의 총기)"
   }
  ],
  "refs": [
   {
    "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its spatial layout, surroundings, fixed features, time of day and lighting mood are spatial truth; stage the moment inside this place. If a STRUCTURE LOOK photograph is also attached, that photo wins for the fixed structure itself — this photograph wins for everything around it. Never copy its camera framing.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/episodes/9fc33eb1-970b-4e5e-8b98-53732d321d63/images/background_chain/L10B01.png",
    "asset_id": "267f2645-3473-4888-9e13-cc2b20962e1d",
    "role": "location_plate"
   },
   {
    "label": "STRUCTURE LOOK — the confirmed photograph of the fixed structure at this location: wherever the structure appears in the frame, its shape, proportions, materials, colors and openings are LOCKED to this photo. Never copy its camera framing, time of day or lighting — the shot text and the LOCATION PHOTOGRAPH are the authorities for those.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/background_chain/seed_bg_mountain_outpost_sel.png",
    "asset_id": "95a8376a-4dea-407c-8b9a-dcae414da6d8",
    "role": "structure_seed_look"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bcae-e484-7bd3-9703-7a2212676dd9",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S7sh4__bgfirst_bg.png",
   "bg_asset_id": "9b24e4cd-4d4f-468b-989e-a586203ee9c2",
   "bg_record_key": "S7sh4::bgfirst_bg",
   "chain_winner": true,
   "authority": "plate",
   "seed_attached": true
  },
  "ref_mode": "재투영 배경+콘티+엔티티 (2택1: 체인 승)",
  "share_plan": {
   "ref_plan": "background"
  },
  "lane_policy": "ab_select_ready"
 },
 "S7sh4::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:57:43.654555+00:00",
  "fingerprint": "a6e44cd9964df8fe395a6db78d881980299aa1b54a46a49ff0ce03f5ae249808",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S7sh4_sel.png",
  "source_sha256": "b31277a58e7b6c6305fd1dc9c8547e92aef27b67954e1ca3155d6c2a1050d96e",
  "file": "S7sh4_cine.png",
  "staged_sha256": "68ac4d43905917a6ec663fe558e333a14139584124721de57819d466e7d1b2dc",
  "latency_ms": 12578
 },
 "S7sh15::signage": {
  "fp": "63f4ce09f06147e9",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S7sh15": {
  "input_fingerprint": "55acba2b4396551a",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): twilight.\n\nSHOT TEXT (authoritative, Korean): 천장 지도 표면의 앞쪽 산맥 위치에 붉은 점들이 반원형의 빽빽한 포위망 형태로 켜진 모습\n\nLOCATION (lock): Inside the small underground contact room, directly beneath its ceiling-mounted map as red indicators illuminate the mountain positions. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: forward mountain sector on ceiling map in the middle-center of the frame, background; semicircle of red points in the upper-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: ceiling map (active with red points illuminated over the forward mountain positions) — The map face is visible from below, showing the forward mountains enclosed by a dense semicircle of red points; used as Primary information surface and encirclement reveal; surrounding cellar ceiling (visible around the mapped sector) — Its underside faces the upward-looking camera; used as Perimeter scale reference that keeps the map physically attached to the room.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The activated red points punctuate the cellar’s restrained low-key ambience while the overall palette remains desaturated.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The rock wall remains open. The round communication panel is lit with “UNAUTHORIZED LEGACY AUTHORITY DETECTED,” while dense red points form a semicircular encirclement around the forward mountains on the ceiling map.\n\nPEOPLE: the SHOT TEXT alone decides who is visible in this shot. People known to appear somewhere in this scene: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 백인 남성 (백인 남성, 성인 연령대, 얼굴형 미상, 자연 범위의 머리색, 머리 길이와 스타일 미상). That list is scene-level, not a cast list for this frame — it may name someone this shot does not show, and it may omit someone this shot does show. If the shot text names a person who is not on the list, draw that person exactly as the shot text describes them; the list does not override the shot text. Never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): twilight.\n\nSHOT TEXT (authoritative, Korean): 천장 지도 표면의 앞쪽 산맥 위치에 붉은 점들이 반원형의 빽빽한 포위망 형태로 켜진 모습\n\nLOCATION (lock): Inside the small underground contact room, directly beneath its ceiling-mounted map as red indicators illuminate the mountain positions. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: forward mountain sector on ceiling map in the middle-center of the frame, background; semicircle of red points in the upper-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: ceiling map (active with red points illuminated over the forward mountain positions) — The map face is visible from below, showing the forward mountains enclosed by a dense semicircle of red points; used as Primary information surface and encirclement reveal; surrounding cellar ceiling (visible around the mapped sector) — Its underside faces the upward-looking camera; used as Perimeter scale reference that keeps the map physically attached to the room.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The activated red points punctuate the cellar’s restrained low-key ambience while the overall palette remains desaturated.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The rock wall remains open. The round communication panel is lit with “UNAUTHORIZED LEGACY AUTHORITY DETECTED,” while dense red points form a semicircular encirclement around the forward mountains on the ceiling map.\n\nPEOPLE: the SHOT TEXT alone decides who is visible in this shot. People known to appear somewhere in this scene: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 백인 남성 (백인 남성, 성인 연령대, 얼굴형 미상, 자연 범위의 머리색, 머리 길이와 스타일 미상). That list is scene-level, not a cast list for this frame — it may name someone this shot does not show, and it may omit someone this shot does show. If the shot text names a person who is not on the list, draw that person exactly as the shot text describes them; the list does not override the shot text. Never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): twilight.\n\nSHOT TEXT (authoritative, Korean): 천장 지도 표면의 앞쪽 산맥 위치에 붉은 점들이 반원형의 빽빽한 포위망 형태로 켜진 모습\n\nLOCATION (lock): Inside the small underground contact room, directly beneath its ceiling-mounted map as red indicators illuminate the mountain positions. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: forward mountain sector on ceiling map in the middle-center of the frame, background; semicircle of red points in the upper-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: ceiling map (active with red points illuminated over the forward mountain positions) — The map face is visible from below, showing the forward mountains enclosed by a dense semicircle of red points; used as Primary information surface and encirclement reveal; surrounding cellar ceiling (visible around the mapped sector) — Its underside faces the upward-looking camera; used as Perimeter scale reference that keeps the map physically attached to the room.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The activated red points punctuate the cellar’s restrained low-key ambience while the overall palette remains desaturated.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The rock wall remains open. The round communication panel is lit with “UNAUTHORIZED LEGACY AUTHORITY DETECTED,” while dense red points form a semicircular encirclement around the forward mountains on the ceiling map.\n\nPEOPLE: the SHOT TEXT alone decides who is visible in this shot. People known to appear somewhere in this scene: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 백인 남성 (백인 남성, 성인 연령대, 얼굴형 미상, 자연 범위의 머리색, 머리 길이와 스타일 미상). That list is scene-level, not a cast list for this frame — it may name someone this shot does not show, and it may omit someone this shot does show. If the shot text names a person who is not on the list, draw that person exactly as the shot text describes them; the list does not override the shot text. Never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라는 수직 위를 향해 천장에 장착된 지도를 똑바로 올려다보고 있음.",
    "built_space": "거친 콘크리트 질감의 천장에 사각형 프레임으로 둘러싸인 지도가 단단히 고정되어 있음. 주변부 천장이 프레임 가장자리에 보임.",
    "entities": "표면에 산맥이 그려진 지도와, 그 산맥 앞쪽을 둥글게 감싸는 매우 빽빽하게 밀집된 반원형의 붉은 발광점들.",
    "hard_violations": [],
    "physics": "지도는 천장 구조물에 물리적으로 결합되어 있으며, 발광하는 붉은 점들은 지도 표면에 안착되어 자연스럽게 빛을 냄."
   },
   {
    "label": "B",
    "direction": "카메라는 위를 향해 천장의 지도를 비스듬히 올려다보고 있음.",
    "built_space": "천장에 부착된 입체적인 부조 형태의 지도가 있으며, 가장자리 주변으로 평평한 천장 표면이 보임.",
    "entities": "입체적으로 튀어나온 산맥 지형과, 이를 둘러싼 두 줄의 비교적 간격이 넓은 반원형 붉은 점들.",
    "hard_violations": [],
    "physics": "부조 지도는 천장에 안정적으로 고정되어 있고, 붉은 조명은 지형의 표면을 따라 발광하고 있음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "지시된 프레이밍을 정확히 따랐으며, '빽빽한 포위망 형태'라는 텍스트의 핵심 묘사를 매우 조밀하고 선명한 붉은 점들로 완벽하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "카메라 앵글과 기본 구성은 지시사항을 따랐으나, 붉은 점들의 배열이 듬성듬성하여 '빽빽한 포위망'이라는 텍스트의 세부 묘사를 충실히 반영하지 못했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 수직 위를 향해 천장에 장착된 지도를 똑바로 올려다보고 있음.",
        "built_space": "거친 콘크리트 질감의 천장에 사각형 프레임으로 둘러싸인 지도가 단단히 고정되어 있음. 주변부 천장이 프레임 가장자리에 보임.",
        "entities": "표면에 산맥이 그려진 지도와, 그 산맥 앞쪽을 둥글게 감싸는 매우 빽빽하게 밀집된 반원형의 붉은 발광점들.",
        "hard_violations": [],
        "physics": "지도는 천장 구조물에 물리적으로 결합되어 있으며, 발광하는 붉은 점들은 지도 표면에 안착되어 자연스럽게 빛을 냄."
       },
       {
        "label": "B",
        "direction": "카메라는 위를 향해 천장의 지도를 비스듬히 올려다보고 있음.",
        "built_space": "천장에 부착된 입체적인 부조 형태의 지도가 있으며, 가장자리 주변으로 평평한 천장 표면이 보임.",
        "entities": "입체적으로 튀어나온 산맥 지형과, 이를 둘러싼 두 줄의 비교적 간격이 넓은 반원형 붉은 점들.",
        "hard_violations": [],
        "physics": "부조 지도는 천장에 안정적으로 고정되어 있고, 붉은 조명은 지형의 표면을 따라 발광하고 있음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "지시된 프레이밍을 정확히 따랐으며, '빽빽한 포위망 형태'라는 텍스트의 핵심 묘사를 매우 조밀하고 선명한 붉은 점들로 완벽하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "카메라 앵글과 기본 구성은 지시사항을 따랐으나, 붉은 점들의 배열이 듬성듬성하여 '빽빽한 포위망'이라는 텍스트의 세부 묘사를 충실히 반영하지 못했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 수직 위를 향해 천장에 장착된 지도를 똑바로 올려다보고 있음.",
        "built_space": "거친 콘크리트 질감의 천장에 사각형 프레임으로 둘러싸인 지도가 단단히 고정되어 있음. 주변부 천장이 프레임 가장자리에 보임.",
        "entities": "표면에 산맥이 그려진 지도와, 그 산맥 앞쪽을 둥글게 감싸는 매우 빽빽하게 밀집된 반원형의 붉은 발광점들.",
        "hard_violations": [],
        "physics": "지도는 천장 구조물에 물리적으로 결합되어 있으며, 발광하는 붉은 점들은 지도 표면에 안착되어 자연스럽게 빛을 냄."
       },
       {
        "label": "B",
        "direction": "카메라는 위를 향해 천장의 지도를 비스듬히 올려다보고 있음.",
        "built_space": "천장에 부착된 입체적인 부조 형태의 지도가 있으며, 가장자리 주변으로 평평한 천장 표면이 보임.",
        "entities": "입체적으로 튀어나온 산맥 지형과, 이를 둘러싼 두 줄의 비교적 간격이 넓은 반원형 붉은 점들.",
        "hard_violations": [],
        "physics": "부조 지도는 천장에 안정적으로 고정되어 있고, 붉은 조명은 지형의 표면을 따라 발광하고 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "천장 지도의 전방 산악 지대를 중간에 두고 빽빽한 붉은 점들이 상부 중심에서 포위 반원을 이루어, 핵심 인서트 구도와 ‘조밀한 포위망’ 묘사를 가장 충실히 구현했다."
       },
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "산맥을 감싼 반원 방향과 인서트 구도는 정확하지만 붉은 점의 수와 밀도가 낮아, 프롬프트가 강조한 빽빽한 포위망의 위압감은 B보다 약하다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "시선, 무기, 이동체는 없다. 두 줄가량의 붉은 표시점들이 화면 중앙 산맥을 목표로 삼아 왼쪽 아래에서 정상부를 거쳐 오른쪽 아래까지 반원형으로 둘러싼다.",
        "built_space": "천장에 붙은 지도 면 1개와 그 둘레의 지하실 천장 일부가 보인다. 지도는 화면 대부분을 차지하는 인서트 클로즈업이며, 가장자리와 주변 천장이 함께 보여 천장 부착물이라는 공간 관계가 성립한다. 사람이나 반사는 없다.",
        "entities": "천장 지도 1개, 전방 산맥 부조, 붉은 표시점들이 모두 확인된다. 붉은 점은 개별 광원처럼 표면에 붙어 있으나 요구된 ‘빽빽한’ 수준보다는 성기다. 사람은 없으며, 인물을 명시하지 않은 샷 텍스트에 부합한다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "지도는 테두리와 천장 면에 고정되어 있어 지지 관계가 자연스럽고, 붉은 점들도 지도 표면에 매립된 표시등으로 보인다. 공중에 뜨거나 지지되지 않은 물체는 없다."
       },
       {
        "label": "B",
        "direction": "시선, 무기, 이동체는 없다. 다중 열의 붉은 표시점들이 중앙 산악 지대를 목표로 하여 위쪽과 양옆을 조밀하게 감싸며 포위 방향이 명확하다. 양 끝이 아래로 내려와 순수한 180도 반원보다 다소 깊은 말굽형에 가깝다.",
        "built_space": "금속 테두리가 있는 천장 지도 1개와 그 주위의 낡은 콘크리트 천장이 보인다. 지도 아래에서 올려다본 구도가 분명하고, 지도 테두리와 주변 천장이 물리적 부착 및 크기 기준을 제공한다. 사람이나 불가능한 반사는 없다.",
        "entities": "천장 지도 1개, 중앙의 산악 지형, 매우 많은 개별 붉은 표시점이 확인된다. 표시점들은 산악 지대를 둘러싼 조밀한 반원형 포위망이라는 핵심 정보를 강하게 전달한다. 사람은 없으며 샷 텍스트에 맞고, 읽을 수 있는 글자도 없다.",
        "hard_violations": [],
        "physics": "지도는 금속 프레임을 통해 천장에 고정되어 있고 표시등은 지도 면에 매립된 광원으로 보인다. 아무 지지 없이 떠 있는 물체나 비현실적인 운동은 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "천장 지도의 전방 산악 지대를 중간에 두고 빽빽한 붉은 점들이 상부 중심에서 포위 반원을 이루어, 핵심 인서트 구도와 ‘조밀한 포위망’ 묘사를 가장 충실히 구현했다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "산맥을 감싼 반원 방향과 인서트 구도는 정확하지만 붉은 점의 수와 밀도가 낮아, 프롬프트가 강조한 빽빽한 포위망의 위압감은 B보다 약하다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "시선, 무기, 이동체는 없다. 두 줄가량의 붉은 표시점들이 화면 중앙 산맥을 목표로 삼아 왼쪽 아래에서 정상부를 거쳐 오른쪽 아래까지 반원형으로 둘러싼다.",
        "built_space": "천장에 붙은 지도 면 1개와 그 둘레의 지하실 천장 일부가 보인다. 지도는 화면 대부분을 차지하는 인서트 클로즈업이며, 가장자리와 주변 천장이 함께 보여 천장 부착물이라는 공간 관계가 성립한다. 사람이나 반사는 없다.",
        "entities": "천장 지도 1개, 전방 산맥 부조, 붉은 표시점들이 모두 확인된다. 붉은 점은 개별 광원처럼 표면에 붙어 있으나 요구된 ‘빽빽한’ 수준보다는 성기다. 사람은 없으며, 인물을 명시하지 않은 샷 텍스트에 부합한다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "지도는 테두리와 천장 면에 고정되어 있어 지지 관계가 자연스럽고, 붉은 점들도 지도 표면에 매립된 표시등으로 보인다. 공중에 뜨거나 지지되지 않은 물체는 없다."
       },
       {
        "label": "A",
        "direction": "시선, 무기, 이동체는 없다. 다중 열의 붉은 표시점들이 중앙 산악 지대를 목표로 하여 위쪽과 양옆을 조밀하게 감싸며 포위 방향이 명확하다. 양 끝이 아래로 내려와 순수한 180도 반원보다 다소 깊은 말굽형에 가깝다.",
        "built_space": "금속 테두리가 있는 천장 지도 1개와 그 주위의 낡은 콘크리트 천장이 보인다. 지도 아래에서 올려다본 구도가 분명하고, 지도 테두리와 주변 천장이 물리적 부착 및 크기 기준을 제공한다. 사람이나 불가능한 반사는 없다.",
        "entities": "천장 지도 1개, 중앙의 산악 지형, 매우 많은 개별 붉은 표시점이 확인된다. 표시점들은 산악 지대를 둘러싼 조밀한 반원형 포위망이라는 핵심 정보를 강하게 전달한다. 사람은 없으며 샷 텍스트에 맞고, 읽을 수 있는 글자도 없다.",
        "hard_violations": [],
        "physics": "지도는 금속 프레임을 통해 천장에 고정되어 있고 표시등은 지도 면에 매립된 광원으로 보인다. 아무 지지 없이 떠 있는 물체나 비현실적인 운동은 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.478
   },
   "adjusted": {
    "A": 2.0,
    "B": 1.478
   },
   "violations": {},
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 1478
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "지시된 프레이밍을 정확히 따랐으며, '빽빽한 포위망 형태'라는 텍스트의 핵심 묘사를 매우 조밀하고 선명한 붉은 점들로 완벽하게 구현했습니다."
   },
   {
    "label": "B",
    "score": 1478,
    "verdict_ko": "카메라 앵글과 기본 구성은 지시사항을 따랐으나, 붉은 점들의 배열이 듬성듬성하여 '빽빽한 포위망'이라는 텍스트의 세부 묘사를 충실히 반영하지 못했습니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features, lighting mood and each person's clothing are LOCKED to this photo; never copy its camera framing. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S7sh4_sel.png",
    "asset_id": "538af073-51a8-4dd2-83dc-5d2b5fddf502",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bcb8-461a-776d-a222-25b7980d1808",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S7sh4"
  },
  "lane_policy": "ab_select_bypass:bg_only:share_plan_prev_bgonly"
 },
 "S7sh15::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T07:58:32.276192+00:00",
  "fingerprint": "9a149abe5f3462059f39969ab0801319fa3ee6971b7592e66a2d883fec367db8",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S7sh15_sel.png",
  "source_sha256": "2fa60f13e1e5476d181f835999b66e96dd6cab29e5fde7363cfe4a6fe5ad973b",
  "file": "S7sh15_cine.png",
  "staged_sha256": "9a2bd56ec10c5d6072ff32aff330034bd7e7f8cf4d5993c978e466bb87723280",
  "latency_ms": 15219
 },
 "S7sh20::signage": {
  "fp": "ea236f4bcdbfe56a",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S7sh20": {
  "input_fingerprint": "4e45835bfce5c52d",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): twilight.\n\nSHOT TEXT (authoritative, Korean): 방수 주머니 안의 휴대폰을 내려다보며 눈을 크게 뜬 토니의 긴장된 얼굴\n\nLOCATION (lock): Inside the small underground contact room near its active wall communication panel, with the open rock entrance admitting twilight from outside. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-center of the frame, foreground, looks toward phone in waterproof pouch.\n- KEY BACKGROUND ELEMENTS: waterproof pouch (containing the phone) — Its pouch side is angled toward the camera at the lower frame edge; used as Lower-edge gaze anchor linking Tony’s face to the vibration source; phone (off after vibrating once) — A small portion of the unlit screen face is visible inside the pouch, angled upward toward Tony; used as Secondary focal evidence beneath Tony’s lowered gaze.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained illumination from the active communication systems and ceiling map supports low-key contrast across Tony’s tense face without introducing another source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The rock wall remains open, the communication panel remains active, and the ceiling map retains its dense red encirclement. The switched-off phone vibrates once inside the waterproof pouch. 토니(앤서니 로저스): He carries the waterproof pouch and reacts to the single vibration from the switched-off phone inside it. His jumper belt still lacks the two silver weights lost over the gorge.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 토니 right now, so 토니's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 토니: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): twilight.\n\nSHOT TEXT (authoritative, Korean): 방수 주머니 안의 휴대폰을 내려다보며 눈을 크게 뜬 토니의 긴장된 얼굴\n\nLOCATION (lock): Inside the small underground contact room near its active wall communication panel, with the open rock entrance admitting twilight from outside. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-center of the frame, foreground, looks toward phone in waterproof pouch.\n- KEY BACKGROUND ELEMENTS: waterproof pouch (containing the phone) — Its pouch side is angled toward the camera at the lower frame edge; used as Lower-edge gaze anchor linking Tony’s face to the vibration source; phone (off after vibrating once) — A small portion of the unlit screen face is visible inside the pouch, angled upward toward Tony; used as Secondary focal evidence beneath Tony’s lowered gaze.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained illumination from the active communication systems and ceiling map supports low-key contrast across Tony’s tense face without introducing another source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The rock wall remains open, the communication panel remains active, and the ceiling map retains its dense red encirclement. The switched-off phone vibrates once inside the waterproof pouch. 토니(앤서니 로저스): He carries the waterproof pouch and reacts to the single vibration from the switched-off phone inside it. His jumper belt still lacks the two silver weights lost over the gorge.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 토니 right now, so 토니's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 토니: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — far-future Pennsylvania, United States; all people are American and English-speaking unless stated. TIME OF DAY (lock): twilight.\n\nSHOT TEXT (authoritative, Korean): 방수 주머니 안의 휴대폰을 내려다보며 눈을 크게 뜬 토니의 긴장된 얼굴\n\nLOCATION (lock): Inside the small underground contact room near its active wall communication panel, with the open rock entrance admitting twilight from outside. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-center of the frame, foreground, looks toward phone in waterproof pouch.\n- KEY BACKGROUND ELEMENTS: waterproof pouch (containing the phone) — Its pouch side is angled toward the camera at the lower frame edge; used as Lower-edge gaze anchor linking Tony’s face to the vibration source; phone (off after vibrating once) — A small portion of the unlit screen face is visible inside the pouch, angled upward toward Tony; used as Secondary focal evidence beneath Tony’s lowered gaze.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained illumination from the active communication systems and ceiling map supports low-key contrast across Tony’s tense face without introducing another source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The rock wall remains open, the communication panel remains active, and the ceiling map retains its dense red encirclement. The switched-off phone vibrates once inside the waterproof pouch. 토니(앤서니 로저스): He carries the waterproof pouch and reacts to the single vibration from the switched-off phone inside it. His jumper belt still lacks the two silver weights lost over the gorge.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 토니 right now, so 토니's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 토니: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "토니의 시선은 두 손으로 든 방수 주머니 안의 휴대폰을 향하고 있으며, 화면 대신 휴대폰의 후면(카메라 렌즈)이 카메라 쪽을 향함.",
    "built_space": "암석으로 된 벽면과 두 개의 통로가 보이며, 천장에는 레퍼런스의 두꺼운 LED 점선 반원 대신 붉은색 네온 동심원 형태의 조명이 배치되어 공간적 일치도가 떨어짐.",
    "entities": "토니의 얼굴, 전술복, 장갑은 레퍼런스와 잘 일치하나 헬멧은 벗은 상태임. 휴대폰이 들어 있는 두꺼운 방수 주머니의 형태가 사실적으로 묘사됨.",
    "hard_violations": [
     "[gpt] 단일 고정 공간 요소여야 할 열린 암반 입구를 좌우 두 개의 별도 개구부처럼 중복 구성했다."
    ],
    "physics": "양손이 방수 주머니의 양쪽 가장자리를 쥐고 안정적으로 들고 있음."
   },
   {
    "label": "B",
    "direction": "토니의 시선이 손에 든 얇은 방수 주머니 속 휴대폰을 향하고 있으며, 화면 대신 휴대폰의 후면이 카메라를 향해 있음.",
    "built_space": "레퍼런스와 일치하는 노출 콘크리트 벽면과 천장 형태를 갖춤. 천장에는 붉은색 LED 점들로 이루어진 반원형 맵이 정확히 묘사되었고, 왼쪽에 빛나는 통신 패널, 오른쪽에 외부 빛이 들어오는 바위 입구가 배치됨.",
    "entities": "토니의 얼굴과 전술복은 일치하나 장갑을 끼지 않은 맨손으로 묘사됨. 휴대폰을 감싼 방수 주머니는 다소 얇은 투명 지퍼백 형태로 표현됨.",
    "hard_violations": [],
    "physics": "오른손이 휴대폰 뒷면을 전체적으로 감싸 쥐며 중력에 맞게 안정적으로 지지하고 있음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "레퍼런스의 실내 공간(천장 LED 맵 포함)과 차분한 조명 분위기를 지시대로 잘 구현했으나, 레퍼런스에 있는 장갑을 누락했고 카메라를 향해야 할 휴대폰 화면 대신 뒷면이 묘사된 점이 아쉽습니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "장갑 착용 상태는 훌륭하나, 천장 맵의 형태가 레퍼런스와 다르고 무엇보다 추가 조명을 금지한 지시를 어기고 아래에서 강한 주황색 빛을 비추어 크게 감점되었습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선은 두 손으로 든 방수 주머니 안의 휴대폰을 향하고 있으며, 화면 대신 휴대폰의 후면(카메라 렌즈)이 카메라 쪽을 향함.",
        "built_space": "암석으로 된 벽면과 두 개의 통로가 보이며, 천장에는 레퍼런스의 두꺼운 LED 점선 반원 대신 붉은색 네온 동심원 형태의 조명이 배치되어 공간적 일치도가 떨어짐.",
        "entities": "토니의 얼굴, 전술복, 장갑은 레퍼런스와 잘 일치하나 헬멧은 벗은 상태임. 휴대폰이 들어 있는 두꺼운 방수 주머니의 형태가 사실적으로 묘사됨.",
        "hard_violations": [],
        "physics": "양손이 방수 주머니의 양쪽 가장자리를 쥐고 안정적으로 들고 있음."
       },
       {
        "label": "B",
        "direction": "토니의 시선이 손에 든 얇은 방수 주머니 속 휴대폰을 향하고 있으며, 화면 대신 휴대폰의 후면이 카메라를 향해 있음.",
        "built_space": "레퍼런스와 일치하는 노출 콘크리트 벽면과 천장 형태를 갖춤. 천장에는 붉은색 LED 점들로 이루어진 반원형 맵이 정확히 묘사되었고, 왼쪽에 빛나는 통신 패널, 오른쪽에 외부 빛이 들어오는 바위 입구가 배치됨.",
        "entities": "토니의 얼굴과 전술복은 일치하나 장갑을 끼지 않은 맨손으로 묘사됨. 휴대폰을 감싼 방수 주머니는 다소 얇은 투명 지퍼백 형태로 표현됨.",
        "hard_violations": [],
        "physics": "오른손이 휴대폰 뒷면을 전체적으로 감싸 쥐며 중력에 맞게 안정적으로 지지하고 있음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "레퍼런스의 실내 공간(천장 LED 맵 포함)과 차분한 조명 분위기를 지시대로 잘 구현했으나, 레퍼런스에 있는 장갑을 누락했고 카메라를 향해야 할 휴대폰 화면 대신 뒷면이 묘사된 점이 아쉽습니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "장갑 착용 상태는 훌륭하나, 천장 맵의 형태가 레퍼런스와 다르고 무엇보다 추가 조명을 금지한 지시를 어기고 아래에서 강한 주황색 빛을 비추어 크게 감점되었습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선은 두 손으로 든 방수 주머니 안의 휴대폰을 향하고 있으며, 화면 대신 휴대폰의 후면(카메라 렌즈)이 카메라 쪽을 향함.",
        "built_space": "암석으로 된 벽면과 두 개의 통로가 보이며, 천장에는 레퍼런스의 두꺼운 LED 점선 반원 대신 붉은색 네온 동심원 형태의 조명이 배치되어 공간적 일치도가 떨어짐.",
        "entities": "토니의 얼굴, 전술복, 장갑은 레퍼런스와 잘 일치하나 헬멧은 벗은 상태임. 휴대폰이 들어 있는 두꺼운 방수 주머니의 형태가 사실적으로 묘사됨.",
        "hard_violations": [],
        "physics": "양손이 방수 주머니의 양쪽 가장자리를 쥐고 안정적으로 들고 있음."
       },
       {
        "label": "B",
        "direction": "토니의 시선이 손에 든 얇은 방수 주머니 속 휴대폰을 향하고 있으며, 화면 대신 휴대폰의 후면이 카메라를 향해 있음.",
        "built_space": "레퍼런스와 일치하는 노출 콘크리트 벽면과 천장 형태를 갖춤. 천장에는 붉은색 LED 점들로 이루어진 반원형 맵이 정확히 묘사되었고, 왼쪽에 빛나는 통신 패널, 오른쪽에 외부 빛이 들어오는 바위 입구가 배치됨.",
        "entities": "토니의 얼굴과 전술복은 일치하나 장갑을 끼지 않은 맨손으로 묘사됨. 휴대폰을 감싼 방수 주머니는 다소 얇은 투명 지퍼백 형태로 표현됨.",
        "hard_violations": [],
        "physics": "오른손이 휴대폰 뒷면을 전체적으로 감싸 쥐며 중력에 맞게 안정적으로 지지하고 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "얼굴 중심 클로즈업, 휴대폰을 향한 긴장된 시선, 단일 입구와 활성 패널은 충실하지만 화면 면이 보이지 않고 고정 의상인 헬멧과 장갑이 빠졌다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "시선과 양손 파지는 정확하고 장갑도 이어지지만 휴대폰 뒷면만 보이며, 단일이어야 할 암반 입구가 두 곳으로 중복되어 공간 잠금을 위반한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 두 눈은 아래쪽 중앙에서 자신이 든 방수 주머니 속 휴대폰을 정확히 향한다. 무기나 다른 지향 물체는 없다. 휴대폰의 화면이 아니라 카메라 모듈이 있는 뒷면이 카메라를 향한다.",
        "built_space": "지하 암반 접촉실 내부로 보이며, 천장에는 붉게 활성화된 지도 설비 1개, 왼쪽 벽에는 활성 통신 패널 설비 1군, 오른쪽에는 황혼빛을 들이는 열린 암반 입구 1개가 보인다. 토니는 방 중앙 전경에 서 있어 설비와의 배치가 자연스럽고, 불가능한 반사는 없다.",
        "entities": "토니는 미국인 30대 중반 남성으로 보이고 얼굴·머리·체격과 낡은 어두운 작업복은 인물 참고와 대체로 맞는다. 다만 참고에서 지속되어야 할 헬멧과 장갑이 없으며 맨손이다. 투명한 방수 주머니와 그 안의 꺼진 휴대폰은 존재하지만, 요구된 꺼진 화면 면의 작은 부분은 보이지 않고 휴대폰 뒷면만 보인다. 천장 지도, 활성 통신 패널, 열린 암반 입구는 확인된다.",
        "hard_violations": [],
        "physics": "토니의 맨손이 방수 주머니 전면과 가장자리를 확실히 움켜쥐어 지지하며, 휴대폰은 주머니 내부에 수납되어 지지된다. 몸은 수직으로 서 있는 자연스러운 자세이고 공중에 뜬 물체나 신체는 없다."
       },
       {
        "label": "B",
        "direction": "토니의 두 눈은 아래쪽 중앙에서 양손으로 든 방수 주머니 속 휴대폰을 정확히 향한다. 휴대폰 화면은 토니 쪽을 향한 것으로 보이고 카메라에는 렌즈가 있는 뒷면만 보여, 명시된 화면 면의 일부는 확인되지 않는다.",
        "built_space": "천장에는 붉게 활성화된 지도 설비 1개가 있고, 오른쪽 가장자리에는 활성 통신 패널 1개가 보인다. 그러나 토니 뒤로 황혼이 보이는 왼쪽 암반 개구부와 어두운 오른쪽 암반 통로가 각각 보여 단일 열린 암반 입구가 두 곳처럼 구성됐다. 토니는 방 중앙 전경에 서 있으며 불가능한 반사는 없다.",
        "entities": "토니는 미국인 30대 중반 남성으로 보이며 얼굴·머리·체격과 어두운 작업복은 참고 인물에 대체로 부합한다. 참고 의상의 장갑은 양손에 유지됐지만 헬멧은 없다. 방수 주머니와 꺼진 휴대폰은 분명하나 휴대폰 뒷면 전체가 크게 보이고 요구된 unlit screen face의 작은 부분은 보이지 않는다. 천장 지도와 활성 패널은 존재한다.",
        "hard_violations": [
         "단일 고정 공간 요소여야 할 열린 암반 입구를 좌우 두 개의 별도 개구부처럼 중복 구성했다."
        ],
        "physics": "양쪽 장갑 낀 손이 방수 주머니의 좌우를 단단히 잡고 있고 휴대폰은 주머니 안에서 지지된다. 토니의 몸은 수직으로 서 있으며 떠 있거나 지지되지 않은 물체는 없다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "얼굴 중심 클로즈업, 휴대폰을 향한 긴장된 시선, 단일 입구와 활성 패널은 충실하지만 화면 면이 보이지 않고 고정 의상인 헬멧과 장갑이 빠졌다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "시선과 양손 파지는 정확하고 장갑도 이어지지만 휴대폰 뒷면만 보이며, 단일이어야 할 암반 입구가 두 곳으로 중복되어 공간 잠금을 위반한다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "토니의 두 눈은 아래쪽 중앙에서 자신이 든 방수 주머니 속 휴대폰을 정확히 향한다. 무기나 다른 지향 물체는 없다. 휴대폰의 화면이 아니라 카메라 모듈이 있는 뒷면이 카메라를 향한다.",
        "built_space": "지하 암반 접촉실 내부로 보이며, 천장에는 붉게 활성화된 지도 설비 1개, 왼쪽 벽에는 활성 통신 패널 설비 1군, 오른쪽에는 황혼빛을 들이는 열린 암반 입구 1개가 보인다. 토니는 방 중앙 전경에 서 있어 설비와의 배치가 자연스럽고, 불가능한 반사는 없다.",
        "entities": "토니는 미국인 30대 중반 남성으로 보이고 얼굴·머리·체격과 낡은 어두운 작업복은 인물 참고와 대체로 맞는다. 다만 참고에서 지속되어야 할 헬멧과 장갑이 없으며 맨손이다. 투명한 방수 주머니와 그 안의 꺼진 휴대폰은 존재하지만, 요구된 꺼진 화면 면의 작은 부분은 보이지 않고 휴대폰 뒷면만 보인다. 천장 지도, 활성 통신 패널, 열린 암반 입구는 확인된다.",
        "hard_violations": [],
        "physics": "토니의 맨손이 방수 주머니 전면과 가장자리를 확실히 움켜쥐어 지지하며, 휴대폰은 주머니 내부에 수납되어 지지된다. 몸은 수직으로 서 있는 자연스러운 자세이고 공중에 뜬 물체나 신체는 없다."
       },
       {
        "label": "A",
        "direction": "토니의 두 눈은 아래쪽 중앙에서 양손으로 든 방수 주머니 속 휴대폰을 정확히 향한다. 휴대폰 화면은 토니 쪽을 향한 것으로 보이고 카메라에는 렌즈가 있는 뒷면만 보여, 명시된 화면 면의 일부는 확인되지 않는다.",
        "built_space": "천장에는 붉게 활성화된 지도 설비 1개가 있고, 오른쪽 가장자리에는 활성 통신 패널 1개가 보인다. 그러나 토니 뒤로 황혼이 보이는 왼쪽 암반 개구부와 어두운 오른쪽 암반 통로가 각각 보여 단일 열린 암반 입구가 두 곳처럼 구성됐다. 토니는 방 중앙 전경에 서 있으며 불가능한 반사는 없다.",
        "entities": "토니는 미국인 30대 중반 남성으로 보이며 얼굴·머리·체격과 어두운 작업복은 참고 인물에 대체로 부합한다. 참고 의상의 장갑은 양손에 유지됐지만 헬멧은 없다. 방수 주머니와 꺼진 휴대폰은 분명하나 휴대폰 뒷면 전체가 크게 보이고 요구된 unlit screen face의 작은 부분은 보이지 않는다. 천장 지도와 활성 패널은 존재한다.",
        "hard_violations": [
         "단일 고정 공간 요소여야 할 열린 암반 입구를 좌우 두 개의 별도 개구부처럼 중복 구성했다."
        ],
        "physics": "양쪽 장갑 낀 손이 방수 주머니의 좌우를 단단히 잡고 있고 휴대폰은 주머니 안에서 지지된다. 토니의 몸은 수직으로 서 있으며 떠 있거나 지지되지 않은 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.143,
    "B": 2.0
   },
   "adjusted": {
    "A": 0.893,
    "B": 2.0
   },
   "violations": {
    "A": [
     "[gpt] 단일 고정 공간 요소여야 할 열린 암반 입구를 좌우 두 개의 별도 개구부처럼 중복 구성했다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "B": 2000,
   "A": 893
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "레퍼런스의 실내 공간(천장 LED 맵 포함)과 차분한 조명 분위기를 지시대로 잘 구현했으나, 레퍼런스에 있는 장갑을 누락했고 카메라를 향해야 할 휴대폰 화면 대신 뒷면이 묘사된 점이 아쉽습니다."
   },
   {
    "label": "A",
    "score": 893,
    "verdict_ko": "장갑 착용 상태는 훌륭하나, 천장 맵의 형태가 레퍼런스와 다르고 무엇보다 추가 조명을 금지한 지시를 어기고 아래에서 강한 주황색 빛을 비추어 크게 감점되었습니다.  ★위반: [gpt] 단일 고정 공간 요소여야 할 열린 암반 입구를 좌우 두 개의 별도 개구부처럼 중복 구성했다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features, lighting mood and each person's clothing are LOCKED to this photo; never copy its camera framing. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/9fc33eb1-970b-4e5e-8b98-53732d321d63/scene/recipe/S7sh15_sel.png",
    "asset_id": "e0979e6e-9755-4f09-87b8-c7e5aaa7f313",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9bcbb-7a33-7c60-a443-3e44b453e2b7",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S7sh15"
  }
 },
 "S7sh20::cine": {
  "applied": true,
  "attempted_at": "2026-09-05T08:01:02.017855+00:00",
  "fingerprint": "41c5e91b2006247ef0325c5db17f13e548e85017e72d9ff8a0dc01c8cd9618c2",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S7sh20_sel.png",
  "source_sha256": "97c01a7a5a3322d1a8db9e3f09fc6eaa19579f1e191036b9a1c1ba24a574d808",
  "file": "S7sh20_cine.png",
  "staged_sha256": "04d7096e644dba7b458ebf964ce66abcba650c787cd0130c712b88af6ce0b27a",
  "latency_ms": 13368
 }
}