{
 "S1sh8::signage": {
  "fp": "4dc15ce3071e1375",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "era_assess::872293120ab2d887": {
  "subjects": [],
  "subject_text": "아케론 딥 볼트 12 폐광 갱도 내부\n좁고 어두운 폐광 갱도. 콘크리트와 금속 방폭문, 천장 조명, 센서가 배치되고 석탄층 사이에 검은 유리질 광맥이 드러난다.",
  "identity": "canonical",
  "scope_id": "L01",
  "scope_role": "location_interior",
  "scope_sha": "1208e6c705acca73"
 },
 "S1sh8::bgfirst_bg": {
  "input_fingerprint": "7878419728ecccd4",
  "prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 검은 광맥을 향해 손을 뻗은 채 옅은 미소를 띤 '토니(앤서니 로저스)'의 상체.\n\nLOCATION (lock): Inside the far end of a narrow abandoned mine tunnel, directly before a black glasslike vein embedded between coal layers. The darkness is cut by a single helmet lamp.\n\nTIME OF DAY (lock): night.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: medium shot\n- KEY BACKGROUND ELEMENTS: black glass-like vein (embedded between coal layers and absorbing light); used as Placed beside Tony's outstretched hand as the visual endpoint of his gesture; coal layers (surrounding the black vein); used as Provide a dark spatial boundary behind the reaching hand.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The single helmet light cuts through the dark tunnel in restrained low-key contrast while the black vein appears to swallow the illumination reaching it.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "effective_prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 검은 광맥을 향해 손을 뻗은 채 옅은 미소를 띤 '토니(앤서니 로저스)'의 상체.\n\nLOCATION (lock): Inside the far end of a narrow abandoned mine tunnel, directly before a black glasslike vein embedded between coal layers. The darkness is cut by a single helmet lamp.\n\nTIME OF DAY (lock): night.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: medium shot\n- KEY BACKGROUND ELEMENTS: black glass-like vein (embedded between coal layers and absorbing light); used as Placed beside Tony's outstretched hand as the visual endpoint of his gesture; coal layers (surrounding the black vein); used as Provide a dark spatial boundary behind the reaching hand.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The single helmet light cuts through the dark tunnel in restrained low-key contrast while the black vein appears to swallow the illumination reaching it.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S1sh8__bgfirst_bg.png",
  "asset_id": "9530c7f7-b6ae-4a9b-8576-5073250951cf",
  "input_asset_ids": [
   "0ef26097-5093-4790-b90d-6d45bf8f9e3c",
   "136764b8-0c9b-4d51-b99c-c74624fb8b19"
  ]
 },
 "S1sh8": {
  "input_fingerprint": "eba1e4a59e529b6c",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 검은 광맥을 향해 손을 뻗은 채 옅은 미소를 띤 '토니(앤서니 로저스)'의 상체.\n\nLOCATION (lock): Inside the far end of a narrow abandoned mine tunnel, directly before a black glasslike vein embedded between coal layers. The darkness is cut by a single helmet lamp. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: medium shot\n- KEY BACKGROUND ELEMENTS: black glass-like vein (embedded between coal layers and absorbing light); used as Placed beside Tony's outstretched hand as the visual endpoint of his gesture; coal layers (surrounding the black vein); used as Provide a dark spatial boundary behind the reaching hand.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The single helmet light cuts through the dark tunnel in restrained low-key contrast while the black vein appears to swallow the illumination reaching it.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The black glasslike vein lies between the coal seams, absorbing light with a pulse-like effect. Three wall sensors blink red simultaneously, while MIDGE flies ahead in the narrow tunnel. 토니(앤서니 로저스): He is in the narrow tunnel with one hand extended toward the black vein and a faint smile on his face. He wears a rescue rope or harness at his waist and a sensor module on his chest.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Stage the shot. The FIRST attached image (SHOT BACKGROUND) is the finished empty background of this shot — keep it EXACTLY: its camera, perspective, architecture, lighting and every fixed feature stay untouched. The SECOND attached image (LAYOUT SKETCH) tells you ONLY where the people go: each sketched person's position, screen size, pose and the gaze/motion arrows. Ignore the sketch's background lines. The CHARACTER REFERENCE photographs show the real people.\nPlace the real people into the background at exactly the sketched positions, sizes and poses, following the arrow directions. No sketch lines or arrows may remain.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 검은 광맥을 향해 손을 뻗은 채 옅은 미소를 띤 '토니(앤서니 로저스)'의 상체.\n\nLOCATION (lock): Inside the far end of a narrow abandoned mine tunnel, directly before a black glasslike vein embedded between coal layers. The darkness is cut by a single helmet lamp. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: medium shot\n- KEY BACKGROUND ELEMENTS: black glass-like vein (embedded between coal layers and absorbing light); used as Placed beside Tony's outstretched hand as the visual endpoint of his gesture; coal layers (surrounding the black vein); used as Provide a dark spatial boundary behind the reaching hand.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The single helmet light cuts through the dark tunnel in restrained low-key contrast while the black vein appears to swallow the illumination reaching it.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The black glasslike vein lies between the coal seams, absorbing light with a pulse-like effect. Three wall sensors blink red simultaneously, while MIDGE flies ahead in the narrow tunnel. 토니(앤서니 로저스): He is in the narrow tunnel with one hand extended toward the black vein and a faint smile on his face. He wears a rescue rope or harness at his waist and a sensor module on his chest.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 검은 광맥을 향해 손을 뻗은 채 옅은 미소를 띤 '토니(앤서니 로저스)'의 상체.\n\nLOCATION (lock): Inside the far end of a narrow abandoned mine tunnel, directly before a black glasslike vein embedded between coal layers. The darkness is cut by a single helmet lamp. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: medium shot\n- KEY BACKGROUND ELEMENTS: black glass-like vein (embedded between coal layers and absorbing light); used as Placed beside Tony's outstretched hand as the visual endpoint of his gesture; coal layers (surrounding the black vein); used as Provide a dark spatial boundary behind the reaching hand.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The single helmet light cuts through the dark tunnel in restrained low-key contrast while the black vein appears to swallow the illumination reaching it.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The black glasslike vein lies between the coal seams, absorbing light with a pulse-like effect. Three wall sensors blink red simultaneously, while MIDGE flies ahead in the narrow tunnel. 토니(앤서니 로저스): He is in the narrow tunnel with one hand extended toward the black vein and a faint smile on his face. He wears a rescue rope or harness at his waist and a sensor module on his chest.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S1sh8__bgfirst_bg.png",
     "asset_id": "9530c7f7-b6ae-4a9b-8576-5073250951cf",
     "role": "bgfirst_bg"
    },
    {
     "label": "LAYOUT SKETCH (people placement only)",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/conti/conti_S1sh8.png",
     "asset_id": "0ef26097-5093-4790-b90d-6d45bf8f9e3c",
     "role": "conti_light"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    }
   ],
   "B": [
    {
     "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/episodes/c804efc3-0697-4c22-98b4-6992a70c2b20/images/background_chain/L01B03.png",
     "asset_id": "136764b8-0c9b-4d51-b99c-c74624fb8b19",
     "role": "location_plate"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "judge_flip_skipped": "judge_fn_owns_order",
  "readings": [
   {
    "label": "A",
    "direction": "토니의 시선과 뻗은 왼손은 모두 오른쪽 벽면에 위치한 검은 광맥을 정확히 향하고 있습니다.",
    "built_space": "로케이션 사진과 정확히 일치하는 폐광 터널 내부입니다. 왼쪽에 거대한 철제 구조물(문)이 있고 바닥에는 선로가 깔려 있습니다. 우측 벽면의 광맥 위로 붉은 빛이 켜진 3개의 기계식 센서가 나란히 부착되어 있습니다.",
    "entities": "토니(레퍼런스와 일치하는 얼굴, 헬멧, 흉부 센서, 허리 하네스 및 로프 착용). 우측 벽의 검은 광맥. 3개의 벽면 센서. 프롬프트에 명시된 드론(MIDGE)은 화면에 존재하지 않습니다.",
    "hard_violations": [
     "[gpt] 오른쪽 벽에 원통형 센서 장치가 3개씩 두 벌, 총 6개로 보이며 명시된 3개짜리 고정 설비를 중복시켰다."
    ],
    "physics": "자연스럽게 서서 팔을 뻗은 안정적인 포즈이며, 보이지 않는 바닥에 의해 체중이 정상적으로 지지되고 있습니다."
   },
   {
    "label": "B",
    "direction": "토니의 시선과 뻗은 오른손은 왼쪽 벽면에 있는 검은 광맥을 향하고 있으며, 손끝이 광맥에 닿아 있습니다.",
    "built_space": "암석으로 이루어진 광산 터널이나, 레퍼런스 이미지에 있던 좌측 철문이나 바닥 선로 등 구체적인 건축 요소가 보이지 않습니다. 터널 안쪽 배경 벽에 3개의 붉은 불빛이 켜져 있습니다.",
    "entities": "토니(레퍼런스와 일치하는 얼굴, 헬멧, 흉부 센서 착용. 프레임이 가슴선에서 잘려 허리 로프는 보이지 않음). 좌측 벽의 검은 광맥. 배경 벽의 붉은 점 3개(센서 형태 불분명). 터널 안쪽에서 비행 중인 소형 드론(MIDGE).",
    "hard_violations": [],
    "physics": "안정적인 스탠딩 자세로 서 있으며, 배경의 드론은 회전익을 통해 공중에 정상적으로 떠 있는 비행 상태를 보여줍니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "지정된 로케이션 레퍼런스의 터널 구조(왼쪽 철문, 바닥 선로 등)를 완벽히 재현했으며 인물의 포즈와 장비(허리 하네스, 벽면 센서) 디테일이 뛰어나지만, 드론(MIDGE)이 누락된 점이 아쉽습니다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "공중에 떠 있는 드론(MIDGE)은 묘사되었으나 로케이션 레퍼런스의 핵심 건축물(대형 철문, 선로)이 완전히 생략되었으며, 프레이밍이 좁아 명시된 허리 로프가 보이지 않고 센서의 형태도 단순화되었습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선과 뻗은 왼손은 모두 오른쪽 벽면에 위치한 검은 광맥을 정확히 향하고 있습니다.",
        "built_space": "로케이션 사진과 정확히 일치하는 폐광 터널 내부입니다. 왼쪽에 거대한 철제 구조물(문)이 있고 바닥에는 선로가 깔려 있습니다. 우측 벽면의 광맥 위로 붉은 빛이 켜진 3개의 기계식 센서가 나란히 부착되어 있습니다.",
        "entities": "토니(레퍼런스와 일치하는 얼굴, 헬멧, 흉부 센서, 허리 하네스 및 로프 착용). 우측 벽의 검은 광맥. 3개의 벽면 센서. 프롬프트에 명시된 드론(MIDGE)은 화면에 존재하지 않습니다.",
        "hard_violations": [],
        "physics": "자연스럽게 서서 팔을 뻗은 안정적인 포즈이며, 보이지 않는 바닥에 의해 체중이 정상적으로 지지되고 있습니다."
       },
       {
        "label": "B",
        "direction": "토니의 시선과 뻗은 오른손은 왼쪽 벽면에 있는 검은 광맥을 향하고 있으며, 손끝이 광맥에 닿아 있습니다.",
        "built_space": "암석으로 이루어진 광산 터널이나, 레퍼런스 이미지에 있던 좌측 철문이나 바닥 선로 등 구체적인 건축 요소가 보이지 않습니다. 터널 안쪽 배경 벽에 3개의 붉은 불빛이 켜져 있습니다.",
        "entities": "토니(레퍼런스와 일치하는 얼굴, 헬멧, 흉부 센서 착용. 프레임이 가슴선에서 잘려 허리 로프는 보이지 않음). 좌측 벽의 검은 광맥. 배경 벽의 붉은 점 3개(센서 형태 불분명). 터널 안쪽에서 비행 중인 소형 드론(MIDGE).",
        "hard_violations": [],
        "physics": "안정적인 스탠딩 자세로 서 있으며, 배경의 드론은 회전익을 통해 공중에 정상적으로 떠 있는 비행 상태를 보여줍니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "지정된 로케이션 레퍼런스의 터널 구조(왼쪽 철문, 바닥 선로 등)를 완벽히 재현했으며 인물의 포즈와 장비(허리 하네스, 벽면 센서) 디테일이 뛰어나지만, 드론(MIDGE)이 누락된 점이 아쉽습니다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "공중에 떠 있는 드론(MIDGE)은 묘사되었으나 로케이션 레퍼런스의 핵심 건축물(대형 철문, 선로)이 완전히 생략되었으며, 프레이밍이 좁아 명시된 허리 로프가 보이지 않고 센서의 형태도 단순화되었습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선과 뻗은 왼손은 모두 오른쪽 벽면에 위치한 검은 광맥을 정확히 향하고 있습니다.",
        "built_space": "로케이션 사진과 정확히 일치하는 폐광 터널 내부입니다. 왼쪽에 거대한 철제 구조물(문)이 있고 바닥에는 선로가 깔려 있습니다. 우측 벽면의 광맥 위로 붉은 빛이 켜진 3개의 기계식 센서가 나란히 부착되어 있습니다.",
        "entities": "토니(레퍼런스와 일치하는 얼굴, 헬멧, 흉부 센서, 허리 하네스 및 로프 착용). 우측 벽의 검은 광맥. 3개의 벽면 센서. 프롬프트에 명시된 드론(MIDGE)은 화면에 존재하지 않습니다.",
        "hard_violations": [],
        "physics": "자연스럽게 서서 팔을 뻗은 안정적인 포즈이며, 보이지 않는 바닥에 의해 체중이 정상적으로 지지되고 있습니다."
       },
       {
        "label": "B",
        "direction": "토니의 시선과 뻗은 오른손은 왼쪽 벽면에 있는 검은 광맥을 향하고 있으며, 손끝이 광맥에 닿아 있습니다.",
        "built_space": "암석으로 이루어진 광산 터널이나, 레퍼런스 이미지에 있던 좌측 철문이나 바닥 선로 등 구체적인 건축 요소가 보이지 않습니다. 터널 안쪽 배경 벽에 3개의 붉은 불빛이 켜져 있습니다.",
        "entities": "토니(레퍼런스와 일치하는 얼굴, 헬멧, 흉부 센서 착용. 프레임이 가슴선에서 잘려 허리 로프는 보이지 않음). 좌측 벽의 검은 광맥. 배경 벽의 붉은 점 3개(센서 형태 불분명). 터널 안쪽에서 비행 중인 소형 드론(MIDGE).",
        "hard_violations": [],
        "physics": "안정적인 스탠딩 자세로 서 있으며, 배경의 드론은 회전익을 통해 공중에 정상적으로 떠 있는 비행 상태를 보여줍니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "상체 중심의 미디엄 숏에서 토니가 검은 유리질 광맥을 정확히 바라보며 손을 뻗고 옅게 미소 짓고 있어 핵심 프레이밍과 순간을 가장 충실히 구현했다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "손과 시선은 광맥을 향하지만 지나치게 넓은 허벅지 크롭, 여러 백색 터널등, 두 벌로 보이는 벽 센서 때문에 단일 헬멧등의 상체 미디엄 숏이라는 핵심 조건을 어겼다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선과 얼굴은 화면 왼쪽의 검은 유리질 광맥을 향한다. 뻗은 오른팔과 장갑 낀 손가락도 같은 광맥으로 이어지며, 손끝이 광맥 바로 앞에 닿아 제스처의 시각적 종착점이 정확하다. 중앙 뒤편의 MIDGE는 터널 안쪽에서 앞서 비행하는 방향으로 보인다.",
        "built_space": "거친 석탄층으로 둘러싸인 좁은 갱도 끝부분이 보이며, 검은 유리질 광맥은 왼쪽 벽의 석탄층 사이에 박혀 있다. 가까운 미디엄 숏이라 기준 사진의 레일·강재 지지틀·대형 금속 설비는 프레임 밖이다. 벽에는 동시에 빛나는 붉은 센서 표시 3개가 보이고, 주된 백색 광원은 토니의 헬멧등 하나다. 토니는 벽 바로 앞에 서서 팔을 뻗을 수 있는 자연스러운 거리에 있다.",
        "entities": "토니 한 명만 등장하며, 30대 중반 미국인 남성으로 보이는 얼굴·짧은 자연색 머리·체격과 광부 구조복이 인물 참고와 잘 맞는다. 헬멧, 장갑, 가슴 센서 모듈과 하네스 스트랩이 보인다. 검고 유리처럼 반사되는 광맥과 이를 둘러싼 석탄층, 붉은 센서 3개, 소형 비행체 MIDGE가 모두 식별된다. 표정은 분명한 옅은 미소이며 읽을 수 있는 문구는 없다.",
        "hard_violations": [],
        "physics": "토니의 몸은 프레임 아래로 이어져 서 있는 상태로 읽히며, 뻗은 손은 어깨와 팔에 정상적으로 연결되어 있다. 손은 광맥에 거의 닿거나 가볍게 접촉하고 있어 공중에 떠 있는 물체가 아니다. MIDGE는 소형 로터형 비행체로 보이므로 자체 추진 비행이라는 지지가 성립하며, 다른 물체도 비현실적으로 떠 있지 않다."
       },
       {
        "label": "B",
        "direction": "토니의 시선은 화면 오른쪽 벽의 검은 광맥과 자신의 손을 향한다. 오른팔과 펼친 손도 광맥 쪽으로 곧게 뻗어 손바닥이 광맥 바로 앞에 놓이므로 제스처의 목표는 맞는다. MIDGE도 손과 광맥 사이에서 갱도 앞쪽으로 비행하는 것으로 보인다.",
        "built_space": "기준 장소의 왼쪽 대형 금속 설비, 바닥 레일, 반복되는 강재 지지틀과 오른쪽 암벽이 넓게 보인다. 토니는 레일 옆에 서서 오른쪽 벽까지 팔을 뻗고 있다. 오른쪽 벽에는 붉게 켜진 원통형 센서 3개가 있지만 그 위쪽에도 별도의 원통형 장치 3개가 한 벌 더 보여 센서 설비가 중복된다. 또한 헬멧등 외에 천장과 갱도 뒤쪽의 백색 고정등 여러 개가 켜져 있어 ‘단일 헬멧등만이 어둠을 가른다’는 조명 구조와 맞지 않는다.",
        "entities": "토니 한 명만 보이며 성별·연령대·얼굴·체격과 작업복은 참고 인물에 대체로 부합한다. 헬멧, 장갑, 가슴 센서 모듈, 허리 구조 로프와 하네스가 확인된다. 오른쪽에는 석탄층 사이의 검은 유리질 광맥이 있고, 소형 MIDGE와 붉은 센서 3개도 있다. 다만 추가 원통형 센서 세트와 여러 활성 터널등이 함께 나타난다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [
         "오른쪽 벽에 원통형 센서 장치가 3개씩 두 벌, 총 6개로 보이며 명시된 3개짜리 고정 설비를 중복시켰다."
        ],
        "physics": "토니는 바닥에 선 자세로 체중이 하체에 실리고 팔은 어깨에서 자연스럽게 뻗어 있다. 손은 광맥 바로 앞에 있으나 접촉 직전이라도 팔이 확실히 지지한다. MIDGE는 로터가 있는 소형 비행체로 보여 자체 추진으로 떠 있는 것이 가능하다. 부유하거나 지지되지 않은 인체·휴대 물체는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "상체 중심의 미디엄 숏에서 토니가 검은 유리질 광맥을 정확히 바라보며 손을 뻗고 옅게 미소 짓고 있어 핵심 프레이밍과 순간을 가장 충실히 구현했다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "손과 시선은 광맥을 향하지만 지나치게 넓은 허벅지 크롭, 여러 백색 터널등, 두 벌로 보이는 벽 센서 때문에 단일 헬멧등의 상체 미디엄 숏이라는 핵심 조건을 어겼다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "토니의 시선과 얼굴은 화면 왼쪽의 검은 유리질 광맥을 향한다. 뻗은 오른팔과 장갑 낀 손가락도 같은 광맥으로 이어지며, 손끝이 광맥 바로 앞에 닿아 제스처의 시각적 종착점이 정확하다. 중앙 뒤편의 MIDGE는 터널 안쪽에서 앞서 비행하는 방향으로 보인다.",
        "built_space": "거친 석탄층으로 둘러싸인 좁은 갱도 끝부분이 보이며, 검은 유리질 광맥은 왼쪽 벽의 석탄층 사이에 박혀 있다. 가까운 미디엄 숏이라 기준 사진의 레일·강재 지지틀·대형 금속 설비는 프레임 밖이다. 벽에는 동시에 빛나는 붉은 센서 표시 3개가 보이고, 주된 백색 광원은 토니의 헬멧등 하나다. 토니는 벽 바로 앞에 서서 팔을 뻗을 수 있는 자연스러운 거리에 있다.",
        "entities": "토니 한 명만 등장하며, 30대 중반 미국인 남성으로 보이는 얼굴·짧은 자연색 머리·체격과 광부 구조복이 인물 참고와 잘 맞는다. 헬멧, 장갑, 가슴 센서 모듈과 하네스 스트랩이 보인다. 검고 유리처럼 반사되는 광맥과 이를 둘러싼 석탄층, 붉은 센서 3개, 소형 비행체 MIDGE가 모두 식별된다. 표정은 분명한 옅은 미소이며 읽을 수 있는 문구는 없다.",
        "hard_violations": [],
        "physics": "토니의 몸은 프레임 아래로 이어져 서 있는 상태로 읽히며, 뻗은 손은 어깨와 팔에 정상적으로 연결되어 있다. 손은 광맥에 거의 닿거나 가볍게 접촉하고 있어 공중에 떠 있는 물체가 아니다. MIDGE는 소형 로터형 비행체로 보이므로 자체 추진 비행이라는 지지가 성립하며, 다른 물체도 비현실적으로 떠 있지 않다."
       },
       {
        "label": "A",
        "direction": "토니의 시선은 화면 오른쪽 벽의 검은 광맥과 자신의 손을 향한다. 오른팔과 펼친 손도 광맥 쪽으로 곧게 뻗어 손바닥이 광맥 바로 앞에 놓이므로 제스처의 목표는 맞는다. MIDGE도 손과 광맥 사이에서 갱도 앞쪽으로 비행하는 것으로 보인다.",
        "built_space": "기준 장소의 왼쪽 대형 금속 설비, 바닥 레일, 반복되는 강재 지지틀과 오른쪽 암벽이 넓게 보인다. 토니는 레일 옆에 서서 오른쪽 벽까지 팔을 뻗고 있다. 오른쪽 벽에는 붉게 켜진 원통형 센서 3개가 있지만 그 위쪽에도 별도의 원통형 장치 3개가 한 벌 더 보여 센서 설비가 중복된다. 또한 헬멧등 외에 천장과 갱도 뒤쪽의 백색 고정등 여러 개가 켜져 있어 ‘단일 헬멧등만이 어둠을 가른다’는 조명 구조와 맞지 않는다.",
        "entities": "토니 한 명만 보이며 성별·연령대·얼굴·체격과 작업복은 참고 인물에 대체로 부합한다. 헬멧, 장갑, 가슴 센서 모듈, 허리 구조 로프와 하네스가 확인된다. 오른쪽에는 석탄층 사이의 검은 유리질 광맥이 있고, 소형 MIDGE와 붉은 센서 3개도 있다. 다만 추가 원통형 센서 세트와 여러 활성 터널등이 함께 나타난다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [
         "오른쪽 벽에 원통형 센서 장치가 3개씩 두 벌, 총 6개로 보이며 명시된 3개짜리 고정 설비를 중복시켰다."
        ],
        "physics": "토니는 바닥에 선 자세로 체중이 하체에 실리고 팔은 어깨에서 자연스럽게 뻗어 있다. 손은 광맥 바로 앞에 있으나 접촉 직전이라도 팔이 확실히 지지한다. MIDGE는 로터가 있는 소형 비행체로 보여 자체 추진으로 떠 있는 것이 가능하다. 부유하거나 지지되지 않은 인체·휴대 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.444,
    "B": 1.75
   },
   "adjusted": {
    "A": 1.194,
    "B": 1.75
   },
   "violations": {
    "A": [
     "[gpt] 오른쪽 벽에 원통형 센서 장치가 3개씩 두 벌, 총 6개로 보이며 명시된 3개짜리 고정 설비를 중복시켰다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "agreed": false
  },
  "totals": {
   "A": 1194,
   "B": 1750
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1194,
    "verdict_ko": "지정된 로케이션 레퍼런스의 터널 구조(왼쪽 철문, 바닥 선로 등)를 완벽히 재현했으며 인물의 포즈와 장비(허리 하네스, 벽면 센서) 디테일이 뛰어나지만, 드론(MIDGE)이 누락된 점이 아쉽습니다.  ★위반: [gpt] 오른쪽 벽에 원통형 센서 장치가 3개씩 두 벌, 총 6개로 보이며 명시된 3개짜리 고정 설비를 중복시켰다."
   },
   {
    "label": "B",
    "score": 1750,
    "verdict_ko": "공중에 떠 있는 드론(MIDGE)은 묘사되었으나 로케이션 레퍼런스의 핵심 건축물(대형 철문, 선로)이 완전히 생략되었으며, 프레이밍이 좁아 명시된 허리 로프가 보이지 않고 센서의 형태도 단순화되었습니다."
   }
  ],
  "refs": [
   {
    "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/episodes/c804efc3-0697-4c22-98b4-6992a70c2b20/images/background_chain/L01B03.png",
    "asset_id": "136764b8-0c9b-4d51-b99c-c74624fb8b19",
    "role": "location_plate"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b269-bd27-7c28-bcd7-05ec3abc9f91",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S1sh8__bgfirst_bg.png",
   "bg_asset_id": "9530c7f7-b6ae-4a9b-8576-5073250951cf",
   "bg_record_key": "S1sh8::bgfirst_bg",
   "chain_winner": false,
   "authority": "plate"
  },
  "ref_mode": "플레이트+엔티티 (2택1: 무콘티 승)",
  "share_plan": {
   "ref_plan": "background"
  }
 },
 "S1sh8::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:17:08.146414+00:00",
  "fingerprint": "89847444433e9e071d39aa2848a69cb8fcdc39a4a343b3455a07f482806f68e1",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S1sh8_sel.png",
  "source_sha256": "2cc208d4a9465521cc2b073c93b8a627624c9e226554fcbe50e8ed3acb6c4bb8",
  "file": "S1sh8_cine.png",
  "staged_sha256": "60b2a74f070cc39f320b90fc10fa83782915e6fee4389cd42da37867f540a077",
  "latency_ms": 15083
 },
 "S1sh9::signage": {
  "fp": "af62c3f63bfc21c1",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S1sh9": {
  "input_fingerprint": "e2920097ad990cb6",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 갱도 천장에서 바닥을 향해 무거운 육중함을 뽐내며 내려오는 궤적의 한 위치에서 포착된 철제 방폭문.\n\nLOCATION (lock): Inside the rear section of the abandoned mine tunnel, at the threshold beneath a descending steel blast door. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: descending blast door in the middle-center of the frame, midground, moves toward tunnel floor; tunnel floor beneath doorway in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: iron blast door (partway through its downward descent) — Its broad Tony-side face and descending lower edge are visible from below; used as Primary threat and vertical compositional axis; rear doorway (being progressively sealed by the descending door) — The opening is viewed from Tony's side and steeply from below; used as Defines the narrowing gap and full descent path; tunnel floor (below the still-open portion of the doorway); used as Lower anchor toward which the door is moving.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Low-key tunnel illumination preserves the mineral-black and muted industrial surfaces without softening the door's severe silhouette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The heavy steel blast door is descending behind the tunnel position. The black vein remains exposed, the three wall sensors remain red, and MIDGE is still in the tunnel ahead.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 갱도 천장에서 바닥을 향해 무거운 육중함을 뽐내며 내려오는 궤적의 한 위치에서 포착된 철제 방폭문.\n\nLOCATION (lock): Inside the rear section of the abandoned mine tunnel, at the threshold beneath a descending steel blast door. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: descending blast door in the middle-center of the frame, midground, moves toward tunnel floor; tunnel floor beneath doorway in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: iron blast door (partway through its downward descent) — Its broad Tony-side face and descending lower edge are visible from below; used as Primary threat and vertical compositional axis; rear doorway (being progressively sealed by the descending door) — The opening is viewed from Tony's side and steeply from below; used as Defines the narrowing gap and full descent path; tunnel floor (below the still-open portion of the doorway); used as Lower anchor toward which the door is moving.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Low-key tunnel illumination preserves the mineral-black and muted industrial surfaces without softening the door's severe silhouette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The heavy steel blast door is descending behind the tunnel position. The black vein remains exposed, the three wall sensors remain red, and MIDGE is still in the tunnel ahead.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 갱도 천장에서 바닥을 향해 무거운 육중함을 뽐내며 내려오는 궤적의 한 위치에서 포착된 철제 방폭문.\n\nLOCATION (lock): Inside the rear section of the abandoned mine tunnel, at the threshold beneath a descending steel blast door. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: descending blast door in the middle-center of the frame, midground, moves toward tunnel floor; tunnel floor beneath doorway in the lower-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: iron blast door (partway through its downward descent) — Its broad Tony-side face and descending lower edge are visible from below; used as Primary threat and vertical compositional axis; rear doorway (being progressively sealed by the descending door) — The opening is viewed from Tony's side and steeply from below; used as Defines the narrowing gap and full descent path; tunnel floor (below the still-open portion of the doorway); used as Lower anchor toward which the door is moving.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Low-key tunnel illumination preserves the mineral-black and muted industrial surfaces without softening the door's severe silhouette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The heavy steel blast door is descending behind the tunnel position. The black vein remains exposed, the three wall sensors remain red, and MIDGE is still in the tunnel ahead.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라 시선은 터널 안쪽을 정면으로 향하며, 방폭문은 위에서 아래를 향해 수직으로 하강하고 있음.",
    "built_space": "거친 암석으로 이루어진 터널 내부. 왼쪽 벽면에는 검고 윤기 나는 광맥이 보임. 터널 중앙(미드그라운드)에 철제 방폭문 구조물이 설치되어 있고 문이 절반쯤 내려온 상태임. 방폭문 뒤쪽 터널의 우측 벽면을 따라 2개의 붉은 센서등이 배치되어 있음.",
    "entities": "무거운 철제 방폭문, 터널 바닥, 왼쪽 벽의 검은 광맥이 존재함. 프롬프트가 요구한 3개의 붉은 센서 중 2개만 우측 벽에 보이며, 드론(MIDGE)은 화면 내에 존재하지 않음. 사람은 없음.",
    "hard_violations": [],
    "physics": "방폭문은 상단 프레임 구조물에 의해 지탱되며 정상적으로 하강하고 있음. 중력을 거스르거나 허공에 뜬 물체는 없음."
   },
   {
    "label": "B",
    "direction": "카메라는 터널 안쪽을 향하고 있으나, 화면 전체가 크게 기울어진 더치 앵글(Dutch angle)로 촬영됨.",
    "built_space": "터널 내부에 거대한 방폭문이 화면의 전경을 꽉 채우며 배치되어 있음. 터널 좌우 벽면 모두에 붉은 센서가 달려 있으며, 특히 우측 벽에는 레퍼런스 공간에 존재하지 않았던 긴 흰색 형광등 형태의 조명들이 부착되어 있음.",
    "entities": "기계적 디테일이 과장된 거대한 방폭문, 터널 바닥이 보임. 붉은 센서등이 양쪽 벽에 다수 존재하여 수량이 맞지 않음. 드론(MIDGE)은 존재하지 않음. 사람은 없음.",
    "hard_violations": [
     "[gemini-pro] 발명된 사물/고정물 추가: 록(lock)된 장소 레퍼런스에 없는 현대적인 긴 튜브 형태의 흰색 조명 기구가 우측 벽에 설치됨.",
     "[gemini-pro] 프레이밍 및 피사체 크기 위반: 배경 요소여야 할 방폭문이 화면을 가득 채우는 전경(foreground) 요소로 과도하게 확대되어, 프롬프트의 '배경 요소를 전경으로 확대하지 말 것' 지시를 위반함.",
     "[gpt] 고정 설비인 적색 벽 센서가 정확히 세 개가 아니라 최소 네 개로 중복되어 보인다."
    ],
    "physics": "크고 무거운 금속 문이 양옆의 궤도와 피스톤에 의해 지탱되고 있음. 화면이 크게 기울어져 있으나 구조물 자체의 지지는 확인됨."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "요구된 와이드 샷 프레이밍과 터널의 기본 구조를 훌륭하게 구현했으나, 명시된 드론이 보이지 않고 센서의 위치가 변경된 점은 아쉽습니다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "방폭문이 화면을 압도할 정도로 과장되어 프레이밍 지시를 위반했으며, 공간 레퍼런스에 없는 새로운 조명 기구를 임의로 추가했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "카메라 시선은 터널 안쪽을 정면으로 향하며, 방폭문은 위에서 아래를 향해 수직으로 하강하고 있음.",
        "built_space": "거친 암석으로 이루어진 터널 내부. 왼쪽 벽면에는 검고 윤기 나는 광맥이 보임. 터널 중앙(미드그라운드)에 철제 방폭문 구조물이 설치되어 있고 문이 절반쯤 내려온 상태임. 방폭문 뒤쪽 터널의 우측 벽면을 따라 2개의 붉은 센서등이 배치되어 있음.",
        "entities": "무거운 철제 방폭문, 터널 바닥, 왼쪽 벽의 검은 광맥이 존재함. 프롬프트가 요구한 3개의 붉은 센서 중 2개만 우측 벽에 보이며, 드론(MIDGE)은 화면 내에 존재하지 않음. 사람은 없음.",
        "hard_violations": [],
        "physics": "방폭문은 상단 프레임 구조물에 의해 지탱되며 정상적으로 하강하고 있음. 중력을 거스르거나 허공에 뜬 물체는 없음."
       },
       {
        "label": "B",
        "direction": "카메라는 터널 안쪽을 향하고 있으나, 화면 전체가 크게 기울어진 더치 앵글(Dutch angle)로 촬영됨.",
        "built_space": "터널 내부에 거대한 방폭문이 화면의 전경을 꽉 채우며 배치되어 있음. 터널 좌우 벽면 모두에 붉은 센서가 달려 있으며, 특히 우측 벽에는 레퍼런스 공간에 존재하지 않았던 긴 흰색 형광등 형태의 조명들이 부착되어 있음.",
        "entities": "기계적 디테일이 과장된 거대한 방폭문, 터널 바닥이 보임. 붉은 센서등이 양쪽 벽에 다수 존재하여 수량이 맞지 않음. 드론(MIDGE)은 존재하지 않음. 사람은 없음.",
        "hard_violations": [
         "발명된 사물/고정물 추가: 록(lock)된 장소 레퍼런스에 없는 현대적인 긴 튜브 형태의 흰색 조명 기구가 우측 벽에 설치됨.",
         "프레이밍 및 피사체 크기 위반: 배경 요소여야 할 방폭문이 화면을 가득 채우는 전경(foreground) 요소로 과도하게 확대되어, 프롬프트의 '배경 요소를 전경으로 확대하지 말 것' 지시를 위반함."
        ],
        "physics": "크고 무거운 금속 문이 양옆의 궤도와 피스톤에 의해 지탱되고 있음. 화면이 크게 기울어져 있으나 구조물 자체의 지지는 확인됨."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "요구된 와이드 샷 프레이밍과 터널의 기본 구조를 훌륭하게 구현했으나, 명시된 드론이 보이지 않고 센서의 위치가 변경된 점은 아쉽습니다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "방폭문이 화면을 압도할 정도로 과장되어 프레이밍 지시를 위반했으며, 공간 레퍼런스에 없는 새로운 조명 기구를 임의로 추가했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "카메라 시선은 터널 안쪽을 정면으로 향하며, 방폭문은 위에서 아래를 향해 수직으로 하강하고 있음.",
        "built_space": "거친 암석으로 이루어진 터널 내부. 왼쪽 벽면에는 검고 윤기 나는 광맥이 보임. 터널 중앙(미드그라운드)에 철제 방폭문 구조물이 설치되어 있고 문이 절반쯤 내려온 상태임. 방폭문 뒤쪽 터널의 우측 벽면을 따라 2개의 붉은 센서등이 배치되어 있음.",
        "entities": "무거운 철제 방폭문, 터널 바닥, 왼쪽 벽의 검은 광맥이 존재함. 프롬프트가 요구한 3개의 붉은 센서 중 2개만 우측 벽에 보이며, 드론(MIDGE)은 화면 내에 존재하지 않음. 사람은 없음.",
        "hard_violations": [],
        "physics": "방폭문은 상단 프레임 구조물에 의해 지탱되며 정상적으로 하강하고 있음. 중력을 거스르거나 허공에 뜬 물체는 없음."
       },
       {
        "label": "B",
        "direction": "카메라는 터널 안쪽을 향하고 있으나, 화면 전체가 크게 기울어진 더치 앵글(Dutch angle)로 촬영됨.",
        "built_space": "터널 내부에 거대한 방폭문이 화면의 전경을 꽉 채우며 배치되어 있음. 터널 좌우 벽면 모두에 붉은 센서가 달려 있으며, 특히 우측 벽에는 레퍼런스 공간에 존재하지 않았던 긴 흰색 형광등 형태의 조명들이 부착되어 있음.",
        "entities": "기계적 디테일이 과장된 거대한 방폭문, 터널 바닥이 보임. 붉은 센서등이 양쪽 벽에 다수 존재하여 수량이 맞지 않음. 드론(MIDGE)은 존재하지 않음. 사람은 없음.",
        "hard_violations": [
         "발명된 사물/고정물 추가: 록(lock)된 장소 레퍼런스에 없는 현대적인 긴 튜브 형태의 흰색 조명 기구가 우측 벽에 설치됨.",
         "프레이밍 및 피사체 크기 위반: 배경 요소여야 할 방폭문이 화면을 가득 채우는 전경(foreground) 요소로 과도하게 확대되어, 프롬프트의 '배경 요소를 전경으로 확대하지 말 것' 지시를 위반함."
        ],
        "physics": "크고 무거운 금속 문이 양옆의 궤도와 피스톤에 의해 지탱되고 있음. 화면이 크게 기울어져 있으나 구조물 자체의 지지는 확인됨."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "중앙의 철제 방폭문이 레일에 지지된 채 바닥으로 내려오며 통로의 틈을 좁히고, 검은 광맥과 정확히 세 개의 적색 센서도 유지되어 핵심 무대 지시를 가장 충실히 구현했다."
       },
       {
        "label": "A",
        "score": 5,
        "verdict_ko": "육중한 문의 넓은 면과 낮은 시점은 매우 잘 맞지만 적색 센서가 네 개 이상 보여 고정 설비 수를 위반하며, 문의 과도한 기계 장치도 이전 장소와의 연속성을 약화한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "시선·무기·여행하는 인물은 없다. 화면 중앙의 문의 수평 하단 모서리는 아래쪽 터널 바닥을 향하고 있어 하강 방향과 목표는 맞는다.",
        "built_space": "암석 갱도 중앙에 대형 리벳식 철문 하나가 있고, 양옆 레일·롤러·유압 장치가 문을 감싼다. 문 아래에는 아직 열린 낮은 틈과 그 너머의 갱도 바닥이 보인다. 카메라는 바닥 가까이에서 위를 올려다보며 문의 넓은 면과 하단을 본다. 적색 센서는 왼쪽 벽에 두 개, 오른쪽 벽에 최소 두 개가 보여 총 네 개 이상이며, 고정된 세 개라는 조건과 맞지 않는다.",
        "entities": "사람·얼굴·신체는 없다. 중앙 물체는 마모된 리벳식 철제 방폭문으로 읽히고, 젖은 검은 광물성 벽면과 갱도 바닥도 보인다. 검은 광맥은 노출되어 있으나 적색 센서 수가 세 개를 초과한다. 읽을 수 있는 글자나 로고는 없다.",
        "hard_violations": [
         "고정 설비인 적색 벽 센서가 정확히 세 개가 아니라 최소 네 개로 중복되어 보인다."
        ],
        "physics": "문은 양쪽의 두꺼운 수직 레일, 다수의 롤러와 유압 실린더에 물려 지지된다. 하단은 바닥 위에 떠 있지만 이는 하강 중인 문을 상부·측면 기구가 받치는 상태로 물리적으로 설명되며, 이동 경로도 바닥으로 이어진다."
       },
       {
        "label": "B",
        "direction": "시선·무기·여행하는 인물은 없다. 중앙 문의 톱니형 하단은 바로 아래의 갱도 바닥을 향하며, 하강하면 통로를 막는 방향으로 정확히 이동한다.",
        "built_space": "암석 갱도 문턱에 직사각형 철제 프레임 하나가 있고 그 안에 철제 방폭문 하나가 일부 내려와 있다. 문 아래의 열린 틈을 통해 후방 갱도와 바닥이 하부 중앙에 이어진다. 오른쪽 후방 벽에는 적색 센서가 정확히 세 개 보이며 검은 광맥성 표면도 노출되어 있다. 인물이나 불가능한 반사는 없다. 다만 시점은 A보다 덜 가파르고 문의 넓은 면이 다소 작게 보인다.",
        "entities": "사람·얼굴·신체는 없다. 중앙 구조물은 마모된 강철 방폭문과 문틀로 읽히며, 암석 터널, 젖은 검은 광맥, 세 개의 적색 벽 센서가 모두 확인된다. 읽을 수 있는 문구·로고·오버레이는 없다.",
        "hard_violations": [],
        "physics": "문은 상부 구조와 양쪽 철제 문틀·수직 레일에 결합되어 지지된다. 바닥 위에 떠 있는 하단은 하강 도중의 한 위치로 자연스럽고, 아래에는 닫힐 바닥과 명확한 이동 여유가 있다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "중앙의 철제 방폭문이 레일에 지지된 채 바닥으로 내려오며 통로의 틈을 좁히고, 검은 광맥과 정확히 세 개의 적색 센서도 유지되어 핵심 무대 지시를 가장 충실히 구현했다."
       },
       {
        "label": "B",
        "score": 5,
        "verdict_ko": "육중한 문의 넓은 면과 낮은 시점은 매우 잘 맞지만 적색 센서가 네 개 이상 보여 고정 설비 수를 위반하며, 문의 과도한 기계 장치도 이전 장소와의 연속성을 약화한다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "시선·무기·여행하는 인물은 없다. 화면 중앙의 문의 수평 하단 모서리는 아래쪽 터널 바닥을 향하고 있어 하강 방향과 목표는 맞는다.",
        "built_space": "암석 갱도 중앙에 대형 리벳식 철문 하나가 있고, 양옆 레일·롤러·유압 장치가 문을 감싼다. 문 아래에는 아직 열린 낮은 틈과 그 너머의 갱도 바닥이 보인다. 카메라는 바닥 가까이에서 위를 올려다보며 문의 넓은 면과 하단을 본다. 적색 센서는 왼쪽 벽에 두 개, 오른쪽 벽에 최소 두 개가 보여 총 네 개 이상이며, 고정된 세 개라는 조건과 맞지 않는다.",
        "entities": "사람·얼굴·신체는 없다. 중앙 물체는 마모된 리벳식 철제 방폭문으로 읽히고, 젖은 검은 광물성 벽면과 갱도 바닥도 보인다. 검은 광맥은 노출되어 있으나 적색 센서 수가 세 개를 초과한다. 읽을 수 있는 글자나 로고는 없다.",
        "hard_violations": [
         "고정 설비인 적색 벽 센서가 정확히 세 개가 아니라 최소 네 개로 중복되어 보인다."
        ],
        "physics": "문은 양쪽의 두꺼운 수직 레일, 다수의 롤러와 유압 실린더에 물려 지지된다. 하단은 바닥 위에 떠 있지만 이는 하강 중인 문을 상부·측면 기구가 받치는 상태로 물리적으로 설명되며, 이동 경로도 바닥으로 이어진다."
       },
       {
        "label": "A",
        "direction": "시선·무기·여행하는 인물은 없다. 중앙 문의 톱니형 하단은 바로 아래의 갱도 바닥을 향하며, 하강하면 통로를 막는 방향으로 정확히 이동한다.",
        "built_space": "암석 갱도 문턱에 직사각형 철제 프레임 하나가 있고 그 안에 철제 방폭문 하나가 일부 내려와 있다. 문 아래의 열린 틈을 통해 후방 갱도와 바닥이 하부 중앙에 이어진다. 오른쪽 후방 벽에는 적색 센서가 정확히 세 개 보이며 검은 광맥성 표면도 노출되어 있다. 인물이나 불가능한 반사는 없다. 다만 시점은 A보다 덜 가파르고 문의 넓은 면이 다소 작게 보인다.",
        "entities": "사람·얼굴·신체는 없다. 중앙 구조물은 마모된 강철 방폭문과 문틀로 읽히며, 암석 터널, 젖은 검은 광맥, 세 개의 적색 벽 센서가 모두 확인된다. 읽을 수 있는 문구·로고·오버레이는 없다.",
        "hard_violations": [],
        "physics": "문은 상부 구조와 양쪽 철제 문틀·수직 레일에 결합되어 지지된다. 바닥 위에 떠 있는 하단은 하강 도중의 한 위치로 자연스럽고, 아래에는 닫힐 바닥과 명확한 이동 여유가 있다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.054
   },
   "adjusted": {
    "A": 2.0,
    "B": 0.804
   },
   "violations": {
    "B": [
     "[gemini-pro] 발명된 사물/고정물 추가: 록(lock)된 장소 레퍼런스에 없는 현대적인 긴 튜브 형태의 흰색 조명 기구가 우측 벽에 설치됨.",
     "[gemini-pro] 프레이밍 및 피사체 크기 위반: 배경 요소여야 할 방폭문이 화면을 가득 채우는 전경(foreground) 요소로 과도하게 확대되어, 프롬프트의 '배경 요소를 전경으로 확대하지 말 것' 지시를 위반함.",
     "[gpt] 고정 설비인 적색 벽 센서가 정확히 세 개가 아니라 최소 네 개로 중복되어 보인다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 804
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "요구된 와이드 샷 프레이밍과 터널의 기본 구조를 훌륭하게 구현했으나, 명시된 드론이 보이지 않고 센서의 위치가 변경된 점은 아쉽습니다."
   },
   {
    "label": "B",
    "score": 804,
    "verdict_ko": "방폭문이 화면을 압도할 정도로 과장되어 프레이밍 지시를 위반했으며, 공간 레퍼런스에 없는 새로운 조명 기구를 임의로 추가했습니다.  ★위반: [gemini-pro] 발명된 사물/고정물 추가: 록(lock)된 장소 레퍼런스에 없는 현대적인 긴 튜브 형태의 흰색 조명 기구가 우측 벽에 설치됨. / [gemini-pro] 프레이밍 및 피사체 크기 위반: 배경 요소여야 할 방폭문이 화면을 가득 채우는 전경(foreground) 요소로 과도하게 확대되어, 프롬프트의 '배경 요소를 전경으로 확대하지 말 것' 지시를 위반함. / [gpt] 고정 설비인 적색 벽 센서가 정확히 세 개가 아니라 최소 네 개로 중복되어 보인다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S1sh8_sel.png",
    "asset_id": "f40e6858-b7d1-4a0e-84e9-97b6033653e1",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b275-358d-7434-8d92-d2bb4d727f18",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S1sh8"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S1sh9::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:18:24.061817+00:00",
  "fingerprint": "89bc66455ff35503114b171700329f37f80e95adf450e2e9bbfd66aff7509a6a",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S1sh9_sel.png",
  "source_sha256": "2c6b286a0a4727a1e4e97a69a27ce1460183ca911c1d7707063da1a8733bed7d",
  "file": "S1sh9_cine.png",
  "staged_sha256": "c42696d5488f0ed7786c9eef2812d0f4ceae41df31df0461bb7d1a969049b4f2",
  "latency_ms": 15466
 },
 "S2sh12::signage": {
  "fp": "8df8059faf2cd23a",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S2sh12": {
  "input_fingerprint": "b11781b21766ba0b",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 굉음과 함께 갱도 천장에서 쏟아져 내리는 돌더미와 흙먼지.\n\nLOCATION (lock): Inside the abandoned mine tunnel beneath a collapsing roof, where rocks and soil are pouring into the passage amid thick dust. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: falling rock mass in the middle-center of the frame, midground, moves toward tunnel floor below frame; ruptured tunnel ceiling in the upper-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: tunnel ceiling (breaking apart during the collapse); used as Upper-frame origin of the falling debris; falling rocks (pouring downward from the broken ceiling); used as Primary moving mass crossing the central frame; falling dust (descending with the rocks); used as Separates the debris into foreground, midground, and background layers.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Failing tunnel illumination renders the collapse in low-key, rugged contrast as the ceiling lights begin going out in sequence.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The blast door is fully sealed with MANUAL SEAL CONFIRMED displayed on the smartphone. Rock and dust pour from the collapsing tunnel as the ceiling lights go out one by one; MIDGE has struck the wall and fallen to the floor.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 굉음과 함께 갱도 천장에서 쏟아져 내리는 돌더미와 흙먼지.\n\nLOCATION (lock): Inside the abandoned mine tunnel beneath a collapsing roof, where rocks and soil are pouring into the passage amid thick dust. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: falling rock mass in the middle-center of the frame, midground, moves toward tunnel floor below frame; ruptured tunnel ceiling in the upper-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: tunnel ceiling (breaking apart during the collapse); used as Upper-frame origin of the falling debris; falling rocks (pouring downward from the broken ceiling); used as Primary moving mass crossing the central frame; falling dust (descending with the rocks); used as Separates the debris into foreground, midground, and background layers.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Failing tunnel illumination renders the collapse in low-key, rugged contrast as the ceiling lights begin going out in sequence.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The blast door is fully sealed with MANUAL SEAL CONFIRMED displayed on the smartphone. Rock and dust pour from the collapsing tunnel as the ceiling lights go out one by one; MIDGE has struck the wall and fallen to the floor.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 굉음과 함께 갱도 천장에서 쏟아져 내리는 돌더미와 흙먼지.\n\nLOCATION (lock): Inside the abandoned mine tunnel beneath a collapsing roof, where rocks and soil are pouring into the passage amid thick dust. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: falling rock mass in the middle-center of the frame, midground, moves toward tunnel floor below frame; ruptured tunnel ceiling in the upper-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: tunnel ceiling (breaking apart during the collapse); used as Upper-frame origin of the falling debris; falling rocks (pouring downward from the broken ceiling); used as Primary moving mass crossing the central frame; falling dust (descending with the rocks); used as Separates the debris into foreground, midground, and background layers.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Failing tunnel illumination renders the collapse in low-key, rugged contrast as the ceiling lights begin going out in sequence.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The blast door is fully sealed with MANUAL SEAL CONFIRMED displayed on the smartphone. Rock and dust pour from the collapsing tunnel as the ceiling lights go out one by one; MIDGE has struck the wall and fallen to the floor.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "바위와 흙먼지가 천장 구멍에서 갱도 바닥을 향해 수직으로 쏟아져 내림.",
    "built_space": "암석으로 이루어진 갱도 내부. 양측면에 수직 철골 기둥은 있으나, 레퍼런스의 거대한 철제 방폭문 구조가 사라지고 그 자리에 밝은 빛이 새어 들어오는 큰 구멍이 생김.",
    "entities": "쏟아지는 바위 덩어리들과 짙은 흙먼지. 프롬프트에 언급된 스마트폰이나 사람의 흔적은 없음.",
    "hard_violations": [],
    "physics": "중력에 의해 허공에서 바닥으로 떨어지고 있는 돌과 먼지 입자들."
   },
   {
    "label": "B",
    "direction": "천장에서 바닥으로 파편과 먼지가 쏟아지고 있으며, 우측 하단의 스마트폰 스크린은 위를 향함.",
    "built_space": "어두운 갱도 내부. 레퍼런스의 방폭문(수직 패널과 하단부)이 먼지 너머로 보이며, 상단을 가로지르는 철골 빔이 끊어지고 꺾여 붕괴(ruptured tunnel ceiling)를 정확히 보여줌.",
    "entities": "무너져 내리는 바위와 철골 파편, 두꺼운 먼지. 우측 하단에 초록색 화면이 켜진 스마트폰을 쥐고 바닥에 쓰러진 팔.",
    "hard_violations": [
     "[gpt] 장소와 붕괴만 보여야 하고 사람을 발명하지 말라는 명시적 지시와 달리, 우하단에 스마트폰을 든 사람의 팔과 손이 추가되었다.",
     "[gpt] 이 순간의 숏 텍스트가 스마트폰을 들거나 조작하는 행동을 연출하지 않는데도 스마트폰 소품을 전경에 발명했다."
    ],
    "physics": "충격으로 튕겨져 나와 허공을 가르며 떨어지는 바위들. 바닥에 기대어 스마트폰의 무게를 지탱하고 있는 손."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "붕괴되는 천장과 배경의 방폭문 구조를 충실히 구현했으며, 바닥에 쓰러져 스마트폰을 쥔 팔을 화면에 포함하여 프롬프트의 상황(Carried state)을 규칙에 어긋남 없이 완벽하게 연출했습니다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "천장의 붕괴는 묘사되었으나 레퍼런스의 핵심인 중앙 방폭문 형태가 사라졌으며, 프롬프트에 명시된 스마트폰 및 쓰러진 인물의 디테일이 완전히 누락되었습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "바위와 흙먼지가 천장 구멍에서 갱도 바닥을 향해 수직으로 쏟아져 내림.",
        "built_space": "암석으로 이루어진 갱도 내부. 양측면에 수직 철골 기둥은 있으나, 레퍼런스의 거대한 철제 방폭문 구조가 사라지고 그 자리에 밝은 빛이 새어 들어오는 큰 구멍이 생김.",
        "entities": "쏟아지는 바위 덩어리들과 짙은 흙먼지. 프롬프트에 언급된 스마트폰이나 사람의 흔적은 없음.",
        "hard_violations": [],
        "physics": "중력에 의해 허공에서 바닥으로 떨어지고 있는 돌과 먼지 입자들."
       },
       {
        "label": "B",
        "direction": "천장에서 바닥으로 파편과 먼지가 쏟아지고 있으며, 우측 하단의 스마트폰 스크린은 위를 향함.",
        "built_space": "어두운 갱도 내부. 레퍼런스의 방폭문(수직 패널과 하단부)이 먼지 너머로 보이며, 상단을 가로지르는 철골 빔이 끊어지고 꺾여 붕괴(ruptured tunnel ceiling)를 정확히 보여줌.",
        "entities": "무너져 내리는 바위와 철골 파편, 두꺼운 먼지. 우측 하단에 초록색 화면이 켜진 스마트폰을 쥐고 바닥에 쓰러진 팔.",
        "hard_violations": [],
        "physics": "충격으로 튕겨져 나와 허공을 가르며 떨어지는 바위들. 바닥에 기대어 스마트폰의 무게를 지탱하고 있는 손."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "붕괴되는 천장과 배경의 방폭문 구조를 충실히 구현했으며, 바닥에 쓰러져 스마트폰을 쥔 팔을 화면에 포함하여 프롬프트의 상황(Carried state)을 규칙에 어긋남 없이 완벽하게 연출했습니다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "천장의 붕괴는 묘사되었으나 레퍼런스의 핵심인 중앙 방폭문 형태가 사라졌으며, 프롬프트에 명시된 스마트폰 및 쓰러진 인물의 디테일이 완전히 누락되었습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "바위와 흙먼지가 천장 구멍에서 갱도 바닥을 향해 수직으로 쏟아져 내림.",
        "built_space": "암석으로 이루어진 갱도 내부. 양측면에 수직 철골 기둥은 있으나, 레퍼런스의 거대한 철제 방폭문 구조가 사라지고 그 자리에 밝은 빛이 새어 들어오는 큰 구멍이 생김.",
        "entities": "쏟아지는 바위 덩어리들과 짙은 흙먼지. 프롬프트에 언급된 스마트폰이나 사람의 흔적은 없음.",
        "hard_violations": [],
        "physics": "중력에 의해 허공에서 바닥으로 떨어지고 있는 돌과 먼지 입자들."
       },
       {
        "label": "B",
        "direction": "천장에서 바닥으로 파편과 먼지가 쏟아지고 있으며, 우측 하단의 스마트폰 스크린은 위를 향함.",
        "built_space": "어두운 갱도 내부. 레퍼런스의 방폭문(수직 패널과 하단부)이 먼지 너머로 보이며, 상단을 가로지르는 철골 빔이 끊어지고 꺾여 붕괴(ruptured tunnel ceiling)를 정확히 보여줌.",
        "entities": "무너져 내리는 바위와 철골 파편, 두꺼운 먼지. 우측 하단에 초록색 화면이 켜진 스마트폰을 쥐고 바닥에 쓰러진 팔.",
        "hard_violations": [],
        "physics": "충격으로 튕겨져 나와 허공을 가르며 떨어지는 바위들. 바닥에 기대어 스마트폰의 무게를 지탱하고 있는 손."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "상부 중앙의 파열된 천장에서 중앙의 암석과 흙먼지가 바닥 쪽으로 쏟아지는 와이드 숏을 정확히 구현하며, 인물·문자·불필요한 소품도 없다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "붕괴 방향과 장소는 대체로 맞지만, 장소만 보여야 하는 숏에 스마트폰을 든 팔을 새로 넣은 것이 명백한 인물·소품 발명이라 실격성 위반이다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "암석과 흙먼지는 상부 중앙 부근에서 발생해 화면 중앙을 지나 터널 바닥과 하단 프레임 방향으로 내려온다. 우하단 스마트폰은 화면 면이 카메라와 위쪽을 향하며, 프레임 밖 소유자의 눈을 정확히 향하는지는 확인하기 어렵다.",
        "built_space": "젖은 암벽 터널 안에 좌우 수직 철골 지지대 한 쌍, 상부의 파손된 가로 구조물, 상단 중앙 조명 하나가 보인다. 중앙 뒤에는 기존 방폭문 계통의 금속 구조가 일부 남아 있으나 먼지와 낙석에 가려 밀폐 상태는 확인되지 않는다. 우하단에는 사람이 있어야 할 위치에 팔과 스마트폰이 들어와 있다.",
        "entities": "파열되는 갱도 천장, 실제 암석처럼 보이는 낙석, 흙먼지, 젖은 암벽과 철골 구조는 모두 요청된 대상에 부합한다. 그러나 프롬프트가 이 숏에서 배제한 사람의 팔과 스마트폰이 추가되었다. 화면의 녹색 인터페이스는 보이지만 판독 가능한 문구는 없다.",
        "hard_violations": [
         "장소와 붕괴만 보여야 하고 사람을 발명하지 말라는 명시적 지시와 달리, 우하단에 스마트폰을 든 사람의 팔과 손이 추가되었다.",
         "이 순간의 숏 텍스트가 스마트폰을 들거나 조작하는 행동을 연출하지 않는데도 스마트폰 소품을 전경에 발명했다."
        ],
        "physics": "낙석은 파손된 상부 구조에서 분리되어 중력 방향으로 떨어지고 일부는 바닥에 도달해 쌓여 있어 발생점과 착지점이 성립한다. 먼지도 낙석과 함께 하강한다. 스마트폰은 손이 실제로 쥐고 있고 팔은 프레임 밖 몸으로 이어져 물리적으로 떠 있지는 않는다."
       },
       {
        "label": "B",
        "direction": "암석과 토사는 상부 중앙의 뚜렷한 천장 파열부에서 시작해 화면 중앙을 수직으로 가로지르며 터널 바닥과 하단 프레임 방향으로 떨어진다. 낙하 목표는 바로 아래 통로 바닥이며, 일부 큰 돌은 이미 그 바닥에 도달했다.",
        "built_space": "젖은 암벽 터널 안에 좌우 수직 철골 지지대 한 쌍과 양쪽에서 중앙 쪽으로 이어지다 파손된 상부 철골들이 보인다. 상부 중앙 천장이 크게 뚫려 있고, 중앙 통로 바닥에는 낙석이 쌓인다. 방폭문 자체의 완전 밀폐 상태는 먼지와 잔해에 가려 확인하기 어렵지만, 중복된 문이나 불가능한 반사는 없다.",
        "entities": "요청된 파열된 천장, 중앙 낙석 덩어리, 층을 이루는 흙먼지, 젖고 낡은 암벽 및 철골 터널이 모두 실물 재질로 표현되었다. 사람·얼굴·신체 일부·스마트폰·판독 가능한 문자는 보이지 않는다.",
        "hard_violations": [],
        "physics": "모든 공중 암석은 상부 중앙의 붕괴 지점에서 떨어지는 중이며 중력 방향의 궤적을 보인다. 아래에는 충돌하고 쌓일 터널 바닥과 기존 낙석이 명확히 있어 발생점과 착지점이 모두 성립한다. 먼지는 낙석과 함께 하강하고 바닥 가까이 퍼진다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "상부 중앙의 파열된 천장에서 중앙의 암석과 흙먼지가 바닥 쪽으로 쏟아지는 와이드 숏을 정확히 구현하며, 인물·문자·불필요한 소품도 없다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "붕괴 방향과 장소는 대체로 맞지만, 장소만 보여야 하는 숏에 스마트폰을 든 팔을 새로 넣은 것이 명백한 인물·소품 발명이라 실격성 위반이다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "암석과 흙먼지는 상부 중앙 부근에서 발생해 화면 중앙을 지나 터널 바닥과 하단 프레임 방향으로 내려온다. 우하단 스마트폰은 화면 면이 카메라와 위쪽을 향하며, 프레임 밖 소유자의 눈을 정확히 향하는지는 확인하기 어렵다.",
        "built_space": "젖은 암벽 터널 안에 좌우 수직 철골 지지대 한 쌍, 상부의 파손된 가로 구조물, 상단 중앙 조명 하나가 보인다. 중앙 뒤에는 기존 방폭문 계통의 금속 구조가 일부 남아 있으나 먼지와 낙석에 가려 밀폐 상태는 확인되지 않는다. 우하단에는 사람이 있어야 할 위치에 팔과 스마트폰이 들어와 있다.",
        "entities": "파열되는 갱도 천장, 실제 암석처럼 보이는 낙석, 흙먼지, 젖은 암벽과 철골 구조는 모두 요청된 대상에 부합한다. 그러나 프롬프트가 이 숏에서 배제한 사람의 팔과 스마트폰이 추가되었다. 화면의 녹색 인터페이스는 보이지만 판독 가능한 문구는 없다.",
        "hard_violations": [
         "장소와 붕괴만 보여야 하고 사람을 발명하지 말라는 명시적 지시와 달리, 우하단에 스마트폰을 든 사람의 팔과 손이 추가되었다.",
         "이 순간의 숏 텍스트가 스마트폰을 들거나 조작하는 행동을 연출하지 않는데도 스마트폰 소품을 전경에 발명했다."
        ],
        "physics": "낙석은 파손된 상부 구조에서 분리되어 중력 방향으로 떨어지고 일부는 바닥에 도달해 쌓여 있어 발생점과 착지점이 성립한다. 먼지도 낙석과 함께 하강한다. 스마트폰은 손이 실제로 쥐고 있고 팔은 프레임 밖 몸으로 이어져 물리적으로 떠 있지는 않는다."
       },
       {
        "label": "A",
        "direction": "암석과 토사는 상부 중앙의 뚜렷한 천장 파열부에서 시작해 화면 중앙을 수직으로 가로지르며 터널 바닥과 하단 프레임 방향으로 떨어진다. 낙하 목표는 바로 아래 통로 바닥이며, 일부 큰 돌은 이미 그 바닥에 도달했다.",
        "built_space": "젖은 암벽 터널 안에 좌우 수직 철골 지지대 한 쌍과 양쪽에서 중앙 쪽으로 이어지다 파손된 상부 철골들이 보인다. 상부 중앙 천장이 크게 뚫려 있고, 중앙 통로 바닥에는 낙석이 쌓인다. 방폭문 자체의 완전 밀폐 상태는 먼지와 잔해에 가려 확인하기 어렵지만, 중복된 문이나 불가능한 반사는 없다.",
        "entities": "요청된 파열된 천장, 중앙 낙석 덩어리, 층을 이루는 흙먼지, 젖고 낡은 암벽 및 철골 터널이 모두 실물 재질로 표현되었다. 사람·얼굴·신체 일부·스마트폰·판독 가능한 문자는 보이지 않는다.",
        "hard_violations": [],
        "physics": "모든 공중 암석은 상부 중앙의 붕괴 지점에서 떨어지는 중이며 중력 방향의 궤적을 보인다. 아래에는 충돌하고 쌓일 터널 바닥과 기존 낙석이 명확히 있어 발생점과 착지점이 모두 성립한다. 먼지는 낙석과 함께 하강하고 바닥 가까이 퍼진다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.667,
    "B": 1.333
   },
   "adjusted": {
    "A": 1.667,
    "B": 1.083
   },
   "violations": {
    "B": [
     "[gpt] 장소와 붕괴만 보여야 하고 사람을 발명하지 말라는 명시적 지시와 달리, 우하단에 스마트폰을 든 사람의 팔과 손이 추가되었다.",
     "[gpt] 이 순간의 숏 텍스트가 스마트폰을 들거나 조작하는 행동을 연출하지 않는데도 스마트폰 소품을 전경에 발명했다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "agreed": false
  },
  "totals": {
   "B": 1083,
   "A": 1667
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 1083,
    "verdict_ko": "붕괴되는 천장과 배경의 방폭문 구조를 충실히 구현했으며, 바닥에 쓰러져 스마트폰을 쥔 팔을 화면에 포함하여 프롬프트의 상황(Carried state)을 규칙에 어긋남 없이 완벽하게 연출했습니다.  ★위반: [gpt] 장소와 붕괴만 보여야 하고 사람을 발명하지 말라는 명시적 지시와 달리, 우하단에 스마트폰을 든 사람의 팔과 손이 추가되었다. / [gpt] 이 순간의 숏 텍스트가 스마트폰을 들거나 조작하는 행동을 연출하지 않는데도 스마트폰 소품을 전경에 발명했다."
   },
   {
    "label": "A",
    "score": 1667,
    "verdict_ko": "천장의 붕괴는 묘사되었으나 레퍼런스의 핵심인 중앙 방폭문 형태가 사라졌으며, 프롬프트에 명시된 스마트폰 및 쓰러진 인물의 디테일이 완전히 누락되었습니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S1sh9_sel.png",
    "asset_id": "c46e09ef-645c-4a03-83ce-953c1e3ac054",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b279-fa5c-70eb-bf60-ef2a1a198bf2",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S1sh9"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S2sh12::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:19:35.552320+00:00",
  "fingerprint": "732d6010eac94496974d0f3455d9ae813fe155d0adc7c3c304de3382d9ec603e",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S2sh12_sel.png",
  "source_sha256": "e09a5fc5f4502817de70f3cc246a67a586fa2424009c693589798e4ef687bab3",
  "file": "S2sh12_cine.png",
  "staged_sha256": "3ddad6a12f5b5b021e9e7fd6b883ee5f6ef9157744a71d1a12bcb171d9f6352c",
  "latency_ms": 15804
 },
 "S2sh21::signage": {
  "fp": "a738cad1b60dfaf9",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S2sh21": {
  "input_fingerprint": "c72147a423571ee2",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 허공에 완전히 정지한 채 떠 있는 흙먼지 입자들.\n\nLOCATION (lock): Inside the rubble-choked section of the collapsed mine tunnel near the black vein. All fixed lights are out, leaving only the smartphone as illumination. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- KEY BACKGROUND ELEMENTS: suspended dust field (completely motionless in midair); used as Primary focus and layered foreground passage for the next camera move; black vein (its surface showing a ripple); used as Side anchor locating the camera at knee height within the tunnel; tunnel floor (beneath the suspended dust); used as Provides a stable reference that makes the particles' arrested motion legible.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The smartphone remains the only stated illumination, giving the suspended field localized low-key visibility without introducing another light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The sealed tunnel is filled with collapse rubble, and MIDGE lies on the floor. All ceiling lights are out, the smartphone is the only light, metallic gas creeps along the floor, and dust particles are suspended motionless beside the rippling black vein.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 허공에 완전히 정지한 채 떠 있는 흙먼지 입자들.\n\nLOCATION (lock): Inside the rubble-choked section of the collapsed mine tunnel near the black vein. All fixed lights are out, leaving only the smartphone as illumination. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- KEY BACKGROUND ELEMENTS: suspended dust field (completely motionless in midair); used as Primary focus and layered foreground passage for the next camera move; black vein (its surface showing a ripple); used as Side anchor locating the camera at knee height within the tunnel; tunnel floor (beneath the suspended dust); used as Provides a stable reference that makes the particles' arrested motion legible.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The smartphone remains the only stated illumination, giving the suspended field localized low-key visibility without introducing another light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The sealed tunnel is filled with collapse rubble, and MIDGE lies on the floor. All ceiling lights are out, the smartphone is the only light, metallic gas creeps along the floor, and dust particles are suspended motionless beside the rippling black vein.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 허공에 완전히 정지한 채 떠 있는 흙먼지 입자들.\n\nLOCATION (lock): Inside the rubble-choked section of the collapsed mine tunnel near the black vein. All fixed lights are out, leaving only the smartphone as illumination. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- KEY BACKGROUND ELEMENTS: suspended dust field (completely motionless in midair); used as Primary focus and layered foreground passage for the next camera move; black vein (its surface showing a ripple); used as Side anchor locating the camera at knee height within the tunnel; tunnel floor (beneath the suspended dust); used as Provides a stable reference that makes the particles' arrested motion legible.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The smartphone remains the only stated illumination, giving the suspended field localized low-key visibility without introducing another light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The sealed tunnel is filled with collapse rubble, and MIDGE lies on the floor. All ceiling lights are out, the smartphone is the only light, metallic gas creeps along the floor, and dust particles are suspended motionless beside the rippling black vein.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "화면 왼쪽에서 손에 들린 스마트폰 플래시 빛이 오른쪽 허공에 떠 있는 흙먼지와 우측의 검은 광맥을 향해 곧게 비추고 있다.",
    "built_space": "어둡고 무너진 갱도 내부. 바닥에는 돌무더기가 있고 오른쪽에는 빛을 반사하는 젖은 질감의 검은 광맥 벽이 위치해 있다. 시점이 바닥과 가까운 무릎 높이로 적절히 설정되었다.",
    "entities": "스마트폰을 단단히 쥔 손, 빛에 반사되는 수많은 흙먼지 입자들, 검은 광맥 벽, 바닥의 잔해들이 프롬프트와 일치하게 나타나며, 다른 인물의 모습은 배제되었다.",
    "hard_violations": [
     "[gpt] 허용된 손이나 팔을 넘어 왼쪽 가장자리에 사람의 머리와 옆얼굴 윤곽이 보여 ‘사람 없음, 얼굴 없음’ 조건을 위반한다."
    ],
    "physics": "흙먼지 입자들이 프롬프트의 요구대로 물리 법칙(중력)을 무시한 채 허공에 완전히 정지된 상태로 떠 있으며, 스마트폰은 손에 의해 안정적으로 지탱되고 있다."
   },
   {
    "label": "B",
    "direction": "천장의 무너진 틈 사이로 크고 작은 바위와 흙먼지가 아래를 향해 쏟아져 내리고 있다.",
    "built_space": "넓게 잡힌 갱도 내부. 좌우로 철제 지지대 빔이 세워져 있으며, 천장이 무너져 내린 구조가 레퍼런스 이미지의 공간과 똑같이 배치되어 있다.",
    "entities": "쏟아지는 바위 파편들과 짙은 먼지구름, 철제 빔 등은 있으나, 프롬프트에서 필수적으로 요구한 '스마트폰'과 '정지된 먼지 입자'는 존재하지 않는다.",
    "hard_violations": [
     "[gemini-pro] 클로즈업(close-up) 프레이밍 지시를 어기고 레퍼런스와 동일한 와이드 샷으로 연출됨",
     "[gemini-pro] 유일한 광원으로 지시된 스마트폰이 등장하지 않음",
     "[gpt] 모든 고정등이 꺼져 있어야 하는데 화면 상단의 고정 천장등처럼 보이는 광원이 켜져 있어 스마트폰 단독 조명 조건을 위반한다.",
     "[gpt] 여러 개의 큰 돌조각이 공중에 떠 있으나 이를 붙들거나 정지시킬 지지물이 없고, 요구된 부유 대상인 미세 흙먼지를 넘어선 물체가 무중력처럼 제시된다."
    ],
    "physics": "바위와 흙먼지가 정지해 있지 않고 낙하하는 운동 상태를 보여주고 있다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "스마트폰 조명, 허공에 정지한 흙먼지 입자, 클로즈업 프레이밍 및 시점 등 프롬프트의 구체적인 지침들을 완벽하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "요구된 클로즈업 카메라 구도와 스마트폰 조명 조건을 완전히 무시하고 이전 샷(레퍼런스)의 와이드 구도를 그대로 복제하여 심각한 위반에 해당합니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "화면 왼쪽에서 손에 들린 스마트폰 플래시 빛이 오른쪽 허공에 떠 있는 흙먼지와 우측의 검은 광맥을 향해 곧게 비추고 있다.",
        "built_space": "어둡고 무너진 갱도 내부. 바닥에는 돌무더기가 있고 오른쪽에는 빛을 반사하는 젖은 질감의 검은 광맥 벽이 위치해 있다. 시점이 바닥과 가까운 무릎 높이로 적절히 설정되었다.",
        "entities": "스마트폰을 단단히 쥔 손, 빛에 반사되는 수많은 흙먼지 입자들, 검은 광맥 벽, 바닥의 잔해들이 프롬프트와 일치하게 나타나며, 다른 인물의 모습은 배제되었다.",
        "hard_violations": [],
        "physics": "흙먼지 입자들이 프롬프트의 요구대로 물리 법칙(중력)을 무시한 채 허공에 완전히 정지된 상태로 떠 있으며, 스마트폰은 손에 의해 안정적으로 지탱되고 있다."
       },
       {
        "label": "B",
        "direction": "천장의 무너진 틈 사이로 크고 작은 바위와 흙먼지가 아래를 향해 쏟아져 내리고 있다.",
        "built_space": "넓게 잡힌 갱도 내부. 좌우로 철제 지지대 빔이 세워져 있으며, 천장이 무너져 내린 구조가 레퍼런스 이미지의 공간과 똑같이 배치되어 있다.",
        "entities": "쏟아지는 바위 파편들과 짙은 먼지구름, 철제 빔 등은 있으나, 프롬프트에서 필수적으로 요구한 '스마트폰'과 '정지된 먼지 입자'는 존재하지 않는다.",
        "hard_violations": [
         "클로즈업(close-up) 프레이밍 지시를 어기고 레퍼런스와 동일한 와이드 샷으로 연출됨",
         "유일한 광원으로 지시된 스마트폰이 등장하지 않음"
        ],
        "physics": "바위와 흙먼지가 정지해 있지 않고 낙하하는 운동 상태를 보여주고 있다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "스마트폰 조명, 허공에 정지한 흙먼지 입자, 클로즈업 프레이밍 및 시점 등 프롬프트의 구체적인 지침들을 완벽하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "요구된 클로즈업 카메라 구도와 스마트폰 조명 조건을 완전히 무시하고 이전 샷(레퍼런스)의 와이드 구도를 그대로 복제하여 심각한 위반에 해당합니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "화면 왼쪽에서 손에 들린 스마트폰 플래시 빛이 오른쪽 허공에 떠 있는 흙먼지와 우측의 검은 광맥을 향해 곧게 비추고 있다.",
        "built_space": "어둡고 무너진 갱도 내부. 바닥에는 돌무더기가 있고 오른쪽에는 빛을 반사하는 젖은 질감의 검은 광맥 벽이 위치해 있다. 시점이 바닥과 가까운 무릎 높이로 적절히 설정되었다.",
        "entities": "스마트폰을 단단히 쥔 손, 빛에 반사되는 수많은 흙먼지 입자들, 검은 광맥 벽, 바닥의 잔해들이 프롬프트와 일치하게 나타나며, 다른 인물의 모습은 배제되었다.",
        "hard_violations": [],
        "physics": "흙먼지 입자들이 프롬프트의 요구대로 물리 법칙(중력)을 무시한 채 허공에 완전히 정지된 상태로 떠 있으며, 스마트폰은 손에 의해 안정적으로 지탱되고 있다."
       },
       {
        "label": "B",
        "direction": "천장의 무너진 틈 사이로 크고 작은 바위와 흙먼지가 아래를 향해 쏟아져 내리고 있다.",
        "built_space": "넓게 잡힌 갱도 내부. 좌우로 철제 지지대 빔이 세워져 있으며, 천장이 무너져 내린 구조가 레퍼런스 이미지의 공간과 똑같이 배치되어 있다.",
        "entities": "쏟아지는 바위 파편들과 짙은 먼지구름, 철제 빔 등은 있으나, 프롬프트에서 필수적으로 요구한 '스마트폰'과 '정지된 먼지 입자'는 존재하지 않는다.",
        "hard_violations": [
         "클로즈업(close-up) 프레이밍 지시를 어기고 레퍼런스와 동일한 와이드 샷으로 연출됨",
         "유일한 광원으로 지시된 스마트폰이 등장하지 않음"
        ],
        "physics": "바위와 흙먼지가 정지해 있지 않고 낙하하는 운동 상태를 보여주고 있다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "이전 숏과 유사한 넓은 붕괴 장면을 반복해 먼지 입자의 클로즈업을 놓쳤고, 켜진 천장 광원과 공중의 큰 파편도 ‘스마트폰만의 조명’ 및 정지한 흙먼지라는 핵심 순간에 어긋난다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "무릎 높이의 클로즈업에서 스마트폰 빛이 정지한 먼지층과 오른쪽 검은 광맥을 비추고 바닥을 함께 보여 핵심 구도는 가장 정확하지만, 허용 범위를 넘은 사람의 얼굴 윤곽이 나타난다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "시선이나 무기는 없다. 먼지뿐 아니라 여러 돌조각이 중앙에서 아래쪽 바닥을 향해 떨어지는 붕괴 장면처럼 보인다. 빛은 스마트폰이 아니라 화면 상단의 밝은 천장 광원에서 아래로 향하며 먼지장을 비춘다.",
        "built_space": "붕괴한 광산 터널을 정면으로 넓게 보여 준다. 좌우에 각각 수직 철제 지지대와 사선 보강재가 보이고, 중앙에는 무너진 개구부, 아래에는 큰 암석과 잔해가 쌓인 바닥, 오른쪽에는 젖고 검은 암벽이 있다. 상단 중앙에는 켜진 고정 조명처럼 보이는 밝은 광원이 하나 있고 뒤쪽에도 작은 밝은 점이 보여, 모든 고정등이 꺼졌다는 공간 상태와 맞지 않는다.",
        "entities": "사람과 문자는 없다. 붕괴 잔해, 터널 바닥, 먼지, 오른쪽의 검은 광맥 계열 암벽은 보이지만 광맥 표면의 파문은 분명하지 않다. 먼지보다 큰 암석 파편이 다수 공중에 있어 요구된 ‘정지한 흙먼지 입자’의 순수한 주제가 흐려진다. 스마트폰은 화면에 없으며 그 빛으로 식별되는 조명도 아니다.",
        "hard_violations": [
         "모든 고정등이 꺼져 있어야 하는데 화면 상단의 고정 천장등처럼 보이는 광원이 켜져 있어 스마트폰 단독 조명 조건을 위반한다.",
         "여러 개의 큰 돌조각이 공중에 떠 있으나 이를 붙들거나 정지시킬 지지물이 없고, 요구된 부유 대상인 미세 흙먼지를 넘어선 물체가 무중력처럼 제시된다."
        ],
        "physics": "미세 먼지는 붕괴로 교란된 공기에 떠 있는 상태로 읽힐 수 있다. 그러나 중앙의 주먹 크기 이상 돌조각 여러 개는 접촉 지지 없이 공중에 있으며, 열린 붕괴부가 발생 원인처럼 보이더라도 완전히 정지한 상태를 유지할 물리적 지지는 없다. 바닥의 큰 돌들은 잔해 위에 안정적으로 놓여 있다."
       },
       {
        "label": "B",
        "direction": "왼쪽 손이 든 스마트폰의 후면 플래시가 화면 오른쪽을 향해 중앙의 먼지장과 오른쪽 검은 광맥 표면을 정확히 비춘다. 먼지에는 뚜렷한 이동 방향이 없어 정지된 장처럼 보인다. 휴대전화 화면은 사용자 쪽을 향하고 카메라에는 후면과 플래시가 보여 사용 방향도 맞다.",
        "built_space": "무릎 높이에 가까운 터널 바닥 시점의 클로즈업이다. 아래에는 젖은 흙바닥과 붕괴 잔해가 있고, 뒤쪽에 수직 철제 지지대 하나가 보이며, 오른쪽 전경에는 젖고 검은 광맥 벽이 크게 자리한다. 가까운 프레이밍 때문에 나머지 지지 구조가 제외된 것은 자연스럽다. 고정등은 보이지 않고 스마트폰 플래시 하나만 광원으로 기능한다.",
        "entities": "주 피사체인 미세 먼지 입자층, 그 아래의 터널 바닥, 붕괴 잔해, 오른쪽의 검은 광맥과 물결 모양 표면 질감, 바닥 가까이의 옅은 기체가 모두 보인다. 스마트폰과 이를 잡은 손은 조명 작동에 필요한 요소로 허용되며 문자는 없다. 다만 왼쪽 가장자리에 손·팔뿐 아니라 머리와 코·입의 옆얼굴 윤곽까지 나타나 ‘사람과 얼굴 없음’ 조건을 위반한다. 어둡고 잘려 있어 인물의 나이·성별·민족성은 판별할 수 없다.",
        "hard_violations": [
         "허용된 손이나 팔을 넘어 왼쪽 가장자리에 사람의 머리와 옆얼굴 윤곽이 보여 ‘사람 없음, 얼굴 없음’ 조건을 위반한다."
        ],
        "physics": "스마트폰은 왼손이 실제로 감싸 쥐고 있어 지지되며, 후면 플래시가 먼지 쪽을 향한다. 미세 먼지는 붕괴 뒤 공기 중에 부유한 입자로 읽히고 단일 프레임 안에서는 정지 상태가 성립한다. 바닥의 돌과 흙은 바닥 및 다른 잔해에 의해 지지되고, 옅은 기체는 바닥을 따라 퍼져 있다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "이전 숏과 유사한 넓은 붕괴 장면을 반복해 먼지 입자의 클로즈업을 놓쳤고, 켜진 천장 광원과 공중의 큰 파편도 ‘스마트폰만의 조명’ 및 정지한 흙먼지라는 핵심 순간에 어긋난다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "무릎 높이의 클로즈업에서 스마트폰 빛이 정지한 먼지층과 오른쪽 검은 광맥을 비추고 바닥을 함께 보여 핵심 구도는 가장 정확하지만, 허용 범위를 넘은 사람의 얼굴 윤곽이 나타난다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "시선이나 무기는 없다. 먼지뿐 아니라 여러 돌조각이 중앙에서 아래쪽 바닥을 향해 떨어지는 붕괴 장면처럼 보인다. 빛은 스마트폰이 아니라 화면 상단의 밝은 천장 광원에서 아래로 향하며 먼지장을 비춘다.",
        "built_space": "붕괴한 광산 터널을 정면으로 넓게 보여 준다. 좌우에 각각 수직 철제 지지대와 사선 보강재가 보이고, 중앙에는 무너진 개구부, 아래에는 큰 암석과 잔해가 쌓인 바닥, 오른쪽에는 젖고 검은 암벽이 있다. 상단 중앙에는 켜진 고정 조명처럼 보이는 밝은 광원이 하나 있고 뒤쪽에도 작은 밝은 점이 보여, 모든 고정등이 꺼졌다는 공간 상태와 맞지 않는다.",
        "entities": "사람과 문자는 없다. 붕괴 잔해, 터널 바닥, 먼지, 오른쪽의 검은 광맥 계열 암벽은 보이지만 광맥 표면의 파문은 분명하지 않다. 먼지보다 큰 암석 파편이 다수 공중에 있어 요구된 ‘정지한 흙먼지 입자’의 순수한 주제가 흐려진다. 스마트폰은 화면에 없으며 그 빛으로 식별되는 조명도 아니다.",
        "hard_violations": [
         "모든 고정등이 꺼져 있어야 하는데 화면 상단의 고정 천장등처럼 보이는 광원이 켜져 있어 스마트폰 단독 조명 조건을 위반한다.",
         "여러 개의 큰 돌조각이 공중에 떠 있으나 이를 붙들거나 정지시킬 지지물이 없고, 요구된 부유 대상인 미세 흙먼지를 넘어선 물체가 무중력처럼 제시된다."
        ],
        "physics": "미세 먼지는 붕괴로 교란된 공기에 떠 있는 상태로 읽힐 수 있다. 그러나 중앙의 주먹 크기 이상 돌조각 여러 개는 접촉 지지 없이 공중에 있으며, 열린 붕괴부가 발생 원인처럼 보이더라도 완전히 정지한 상태를 유지할 물리적 지지는 없다. 바닥의 큰 돌들은 잔해 위에 안정적으로 놓여 있다."
       },
       {
        "label": "A",
        "direction": "왼쪽 손이 든 스마트폰의 후면 플래시가 화면 오른쪽을 향해 중앙의 먼지장과 오른쪽 검은 광맥 표면을 정확히 비춘다. 먼지에는 뚜렷한 이동 방향이 없어 정지된 장처럼 보인다. 휴대전화 화면은 사용자 쪽을 향하고 카메라에는 후면과 플래시가 보여 사용 방향도 맞다.",
        "built_space": "무릎 높이에 가까운 터널 바닥 시점의 클로즈업이다. 아래에는 젖은 흙바닥과 붕괴 잔해가 있고, 뒤쪽에 수직 철제 지지대 하나가 보이며, 오른쪽 전경에는 젖고 검은 광맥 벽이 크게 자리한다. 가까운 프레이밍 때문에 나머지 지지 구조가 제외된 것은 자연스럽다. 고정등은 보이지 않고 스마트폰 플래시 하나만 광원으로 기능한다.",
        "entities": "주 피사체인 미세 먼지 입자층, 그 아래의 터널 바닥, 붕괴 잔해, 오른쪽의 검은 광맥과 물결 모양 표면 질감, 바닥 가까이의 옅은 기체가 모두 보인다. 스마트폰과 이를 잡은 손은 조명 작동에 필요한 요소로 허용되며 문자는 없다. 다만 왼쪽 가장자리에 손·팔뿐 아니라 머리와 코·입의 옆얼굴 윤곽까지 나타나 ‘사람과 얼굴 없음’ 조건을 위반한다. 어둡고 잘려 있어 인물의 나이·성별·민족성은 판별할 수 없다.",
        "hard_violations": [
         "허용된 손이나 팔을 넘어 왼쪽 가장자리에 사람의 머리와 옆얼굴 윤곽이 보여 ‘사람 없음, 얼굴 없음’ 조건을 위반한다."
        ],
        "physics": "스마트폰은 왼손이 실제로 감싸 쥐고 있어 지지되며, 후면 플래시가 먼지 쪽을 향한다. 미세 먼지는 붕괴 뒤 공기 중에 부유한 입자로 읽히고 단일 프레임 안에서는 정지 상태가 성립한다. 바닥의 돌과 흙은 바닥 및 다른 잔해에 의해 지지되고, 옅은 기체는 바닥을 따라 퍼져 있다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 0.533
   },
   "adjusted": {
    "A": 1.75,
    "B": 0.283
   },
   "violations": {
    "B": [
     "[gemini-pro] 클로즈업(close-up) 프레이밍 지시를 어기고 레퍼런스와 동일한 와이드 샷으로 연출됨",
     "[gemini-pro] 유일한 광원으로 지시된 스마트폰이 등장하지 않음",
     "[gpt] 모든 고정등이 꺼져 있어야 하는데 화면 상단의 고정 천장등처럼 보이는 광원이 켜져 있어 스마트폰 단독 조명 조건을 위반한다.",
     "[gpt] 여러 개의 큰 돌조각이 공중에 떠 있으나 이를 붙들거나 정지시킬 지지물이 없고, 요구된 부유 대상인 미세 흙먼지를 넘어선 물체가 무중력처럼 제시된다."
    ],
    "A": [
     "[gpt] 허용된 손이나 팔을 넘어 왼쪽 가장자리에 사람의 머리와 옆얼굴 윤곽이 보여 ‘사람 없음, 얼굴 없음’ 조건을 위반한다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 1750,
   "B": 283
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1750,
    "verdict_ko": "스마트폰 조명, 허공에 정지한 흙먼지 입자, 클로즈업 프레이밍 및 시점 등 프롬프트의 구체적인 지침들을 완벽하게 구현했습니다.  ★위반: [gpt] 허용된 손이나 팔을 넘어 왼쪽 가장자리에 사람의 머리와 옆얼굴 윤곽이 보여 ‘사람 없음, 얼굴 없음’ 조건을 위반한다."
   },
   {
    "label": "B",
    "score": 283,
    "verdict_ko": "요구된 클로즈업 카메라 구도와 스마트폰 조명 조건을 완전히 무시하고 이전 샷(레퍼런스)의 와이드 구도를 그대로 복제하여 심각한 위반에 해당합니다.  ★위반: [gemini-pro] 클로즈업(close-up) 프레이밍 지시를 어기고 레퍼런스와 동일한 와이드 샷으로 연출됨 / [gemini-pro] 유일한 광원으로 지시된 스마트폰이 등장하지 않음 / [gpt] 모든 고정등이 꺼져 있어야 하는데 화면 상단의 고정 천장등처럼 보이는 광원이 켜져 있어 스마트폰 단독 조명 조건을 위반한다. / [gpt] 여러 개의 큰 돌조각이 공중에 떠 있으나 이를 붙들거나 정지시킬 지지물이 없고, 요구된 부유 대상인 미세 흙먼지를 넘어선 물체가 무중력처럼 제시된다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features, lighting mood and each person's clothing are LOCKED to this photo; never copy its camera framing. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S2sh12_sel.png",
    "asset_id": "4d7c34fe-10e9-46d9-b32f-41831ad6c501",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b27e-77ce-7ffe-8d75-4efb8891cdd5",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S2sh12"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S2sh21::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:20:53.819572+00:00",
  "fingerprint": "0d5ea355816c2dd2da064b34f3ec5cd20b0963f761bd8a60f41c58e28dac5e05",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S2sh21_sel.png",
  "source_sha256": "0a41186b6b64e3c20d39ae2ae628bfeaaa39e331712cfb7a356c28bf08a6b007",
  "file": "S2sh21_cine.png",
  "staged_sha256": "ae80db7da585533477ac6ac68880a872857981ab13cf6bb5e17994c1b6cf79c3",
  "latency_ms": 15544
 },
 "S2sh22::signage": {
  "fp": "c545effad1823389",
  "inscriptions": [
   {
    "text_native": "21:14:07",
    "source": "scene_text_quoted",
    "reason_ko": "스마트폰 시계 화면에 멈춰 표시된 시간이 명시되어 있습니다.",
    "source_quote": "21:14:07"
   }
  ],
  "cues": [],
  "dropped": []
 },
 "S2sh22": {
  "input_fingerprint": "d0874f512d8e83c1",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): '21:14:07'에서 숫자가 멈춘 스마트폰 시계 화면의 클로즈업.\n\nLOCATION (lock): Inside the dark, collapsed mine passage where the trapped survivor lies under rubble, lit only by the smartphone screen. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: smartphone screen showing 21:14:07 in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: smartphone (clock stopped at 21:14:07) — The active screen face is visible obliquely to camera and displays the frozen clock digits 21:14:07; used as Primary time-stop insert with bezel retained to signal device mediation; collapsed debris (lying around the smartphone); used as Narrow peripheral context that fixes the phone within the collapsed tunnel.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The smartphone display provides the only stated localized light, isolated within otherwise low-key darkness.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone clock is frozen at 21:14:07, with its light remaining the tunnel’s only illumination. Collapse rubble, the sealed blast door, fallen MIDGE, floor-hugging gas, and motionless airborne dust remain in place.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"21:14:07\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): '21:14:07'에서 숫자가 멈춘 스마트폰 시계 화면의 클로즈업.\n\nLOCATION (lock): Inside the dark, collapsed mine passage where the trapped survivor lies under rubble, lit only by the smartphone screen. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: smartphone screen showing 21:14:07 in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: smartphone (clock stopped at 21:14:07) — The active screen face is visible obliquely to camera and displays the frozen clock digits 21:14:07; used as Primary time-stop insert with bezel retained to signal device mediation; collapsed debris (lying around the smartphone); used as Narrow peripheral context that fixes the phone within the collapsed tunnel.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The smartphone display provides the only stated localized light, isolated within otherwise low-key darkness.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone clock is frozen at 21:14:07, with its light remaining the tunnel’s only illumination. Collapse rubble, the sealed blast door, fallen MIDGE, floor-hugging gas, and motionless airborne dust remain in place.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"21:14:07\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): '21:14:07'에서 숫자가 멈춘 스마트폰 시계 화면의 클로즈업.\n\nLOCATION (lock): Inside the dark, collapsed mine passage where the trapped survivor lies under rubble, lit only by the smartphone screen. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: smartphone screen showing 21:14:07 in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: smartphone (clock stopped at 21:14:07) — The active screen face is visible obliquely to camera and displays the frozen clock digits 21:14:07; used as Primary time-stop insert with bezel retained to signal device mediation; collapsed debris (lying around the smartphone); used as Narrow peripheral context that fixes the phone within the collapsed tunnel.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The smartphone display provides the only stated localized light, isolated within otherwise low-key darkness.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone clock is frozen at 21:14:07, with its light remaining the tunnel’s only illumination. Collapse rubble, the sealed blast door, fallen MIDGE, floor-hugging gas, and motionless airborne dust remain in place.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"21:14:07\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라가 바닥의 바위 위에 놓인 스마트폰을 비스듬히 내려다보고 있습니다.",
    "built_space": "어두운 탄광 바닥으로 보이는 거칠고 검은 바위와 파편들이 주변에 흩어져 있습니다.",
    "entities": "화면에 '21:14:07'이 표시된 스마트폰이 있으며, 프롬프트에 명시되지 않은 사람의 손이 우측에 있습니다.",
    "hard_violations": [
     "[gemini-pro] 인물 등장 금지(NO PEOPLE) 지시사항을 위반하고 화면 우측에 사람의 손이 렌더링됨",
     "[gpt] 스마트폰을 들거나 조작하는 행동이 명시되지 않았는데 사람의 손과 팔을 새로 등장시켜 ‘NO PEOPLE IN THIS SHOT’의 무인 staging을 위반했다."
    ],
    "physics": "스마트폰은 바닥의 바위 위에 놓여 지탱되고 있으며, 사람의 손은 바위와 스마트폰 가장자리에 얹혀 있습니다."
   },
   {
    "label": "B",
    "direction": "카메라가 바닥의 바위 위에 놓인 스마트폰을 비스듬히 내려다보고 있습니다.",
    "built_space": "어두운 갱도 바닥을 이루는 거친 바위와 흙먼지, 파편들이 주변에 흩어져 있습니다.",
    "entities": "화면에 '21:14:07' 텍스트가 명확하게 렌더링된 스마트폰과 주변의 바위 잔해들입니다.",
    "hard_violations": [],
    "physics": "스마트폰은 바닥의 울퉁불퉁한 바위와 파편들 위에 지탱되어 놓여 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "프롬프트에서 인물이 등장하지 않아야 한다고 엄격히 제한했으나 화면에 사람의 손이 나타나 결정적인 하드 위반에 해당합니다."
       },
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "요구된 클로즈업 앵글과 '21:14:07' 텍스트를 정확히 렌더링했으며, 인물 없이 잔해 속 스마트폰이라는 설정을 완벽히 구현했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "카메라가 바닥의 바위 위에 놓인 스마트폰을 비스듬히 내려다보고 있습니다.",
        "built_space": "어두운 탄광 바닥으로 보이는 거칠고 검은 바위와 파편들이 주변에 흩어져 있습니다.",
        "entities": "화면에 '21:14:07'이 표시된 스마트폰이 있으며, 프롬프트에 명시되지 않은 사람의 손이 우측에 있습니다.",
        "hard_violations": [
         "인물 등장 금지(NO PEOPLE) 지시사항을 위반하고 화면 우측에 사람의 손이 렌더링됨"
        ],
        "physics": "스마트폰은 바닥의 바위 위에 놓여 지탱되고 있으며, 사람의 손은 바위와 스마트폰 가장자리에 얹혀 있습니다."
       },
       {
        "label": "B",
        "direction": "카메라가 바닥의 바위 위에 놓인 스마트폰을 비스듬히 내려다보고 있습니다.",
        "built_space": "어두운 갱도 바닥을 이루는 거친 바위와 흙먼지, 파편들이 주변에 흩어져 있습니다.",
        "entities": "화면에 '21:14:07' 텍스트가 명확하게 렌더링된 스마트폰과 주변의 바위 잔해들입니다.",
        "hard_violations": [],
        "physics": "스마트폰은 바닥의 울퉁불퉁한 바위와 파편들 위에 지탱되어 놓여 있습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "프롬프트에서 인물이 등장하지 않아야 한다고 엄격히 제한했으나 화면에 사람의 손이 나타나 결정적인 하드 위반에 해당합니다."
       },
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "요구된 클로즈업 앵글과 '21:14:07' 텍스트를 정확히 렌더링했으며, 인물 없이 잔해 속 스마트폰이라는 설정을 완벽히 구현했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "카메라가 바닥의 바위 위에 놓인 스마트폰을 비스듬히 내려다보고 있습니다.",
        "built_space": "어두운 탄광 바닥으로 보이는 거칠고 검은 바위와 파편들이 주변에 흩어져 있습니다.",
        "entities": "화면에 '21:14:07'이 표시된 스마트폰이 있으며, 프롬프트에 명시되지 않은 사람의 손이 우측에 있습니다.",
        "hard_violations": [
         "인물 등장 금지(NO PEOPLE) 지시사항을 위반하고 화면 우측에 사람의 손이 렌더링됨"
        ],
        "physics": "스마트폰은 바닥의 바위 위에 놓여 지탱되고 있으며, 사람의 손은 바위와 스마트폰 가장자리에 얹혀 있습니다."
       },
       {
        "label": "B",
        "direction": "카메라가 바닥의 바위 위에 놓인 스마트폰을 비스듬히 내려다보고 있습니다.",
        "built_space": "어두운 갱도 바닥을 이루는 거친 바위와 흙먼지, 파편들이 주변에 흩어져 있습니다.",
        "entities": "화면에 '21:14:07' 텍스트가 명확하게 렌더링된 스마트폰과 주변의 바위 잔해들입니다.",
        "hard_violations": [],
        "physics": "스마트폰은 바닥의 울퉁불퉁한 바위와 파편들 위에 지탱되어 놓여 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "스마트폰이 잔해 위에 자연스럽게 놓인 채 화면과 베젤이 사선으로 보이고, 정확한 ‘21:14:07’이 화면 중앙에 배치되어 인서트 클로즈업과 무인 staging을 충실히 따른다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "시간 표시는 정확하고 클로즈업도 가깝지만, 촬영 순간에 요구되지 않은 손을 화면에 크게 추가하여 ‘사람 없음’과 고정된 장소 상태를 위반한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "시선·무기·이동체는 없다. 스마트폰의 활성 화면은 카메라를 향해 비스듬히 놓여 있으며 ‘21:14:07’이 정방향으로 명확히 보인다.",
        "built_space": "붕괴한 갱도 바닥의 석탄 조각, 흙, 암석 잔해만 좁은 주변 맥락으로 보인다. 스마트폰은 프레임 중간 중앙의 전경을 차지한다. 봉인된 방폭문이나 쓰러진 인물은 인서트 프레이밍 밖이라 보이지 않으며, 중복된 고정 설비나 불가능한 반사는 없다.",
        "entities": "베젤이 유지된 실제 크기의 스마트폰 한 대와 붕괴 잔해가 보인다. 화면에는 요구된 유일한 가독 문구 ‘21:14:07’이 정확히 표시되며 다른 읽을 수 있는 글자나 인물은 없다.",
        "hard_violations": [],
        "physics": "스마트폰은 여러 암석과 흙더미 위에 접촉해 지지되고 있어 떠 있지 않는다. 잔해도 모두 바닥과 다른 돌 위에 놓여 있으며, 공중에서 움직이는 물체는 없다."
       },
       {
        "label": "B",
        "direction": "시선·무기·이동체는 없다. 스마트폰 화면은 카메라를 향해 비스듬히 놓였고 ‘21:14:07’이 정방향으로 보인다. 오른쪽 위의 손은 전화기 가장자리 쪽으로 뻗어 접촉하고 있다.",
        "built_space": "붕괴한 갱도 바닥의 석탄, 흙, 암석 잔해가 스마트폰 주변을 채운다. 전화기는 중앙에서 하단까지 크게 놓였고, 큰 손이 우상단을 점유한다. 고정 설비의 중복이나 불가능한 반사는 없지만, 손 때문에 요구된 순수한 시간 정지 인서트의 공간 구성이 바뀌었다.",
        "entities": "스마트폰 한 대, 잔해, 사람의 손과 팔 일부가 보인다. 화면의 유일한 글자는 정확한 ‘21:14:07’이며 다른 가독 문구는 없다. 그러나 프롬프트가 요구하지 않은 사람의 신체 일부가 추가되었다.",
        "hard_violations": [
         "스마트폰을 들거나 조작하는 행동이 명시되지 않았는데 사람의 손과 팔을 새로 등장시켜 ‘NO PEOPLE IN THIS SHOT’의 무인 staging을 위반했다."
        ],
        "physics": "스마트폰의 하단과 측면은 암석 잔해 위에 놓여 지지되고, 손도 전화기 가장자리에 닿아 있다. 손과 전화기 모두 물리적으로 떠 있지는 않으며 해부학적으로도 가능하지만, 이 접촉 행동 자체가 명시된 정지 상태에 불필요하게 추가되었다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "스마트폰이 잔해 위에 자연스럽게 놓인 채 화면과 베젤이 사선으로 보이고, 정확한 ‘21:14:07’이 화면 중앙에 배치되어 인서트 클로즈업과 무인 staging을 충실히 따른다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "시간 표시는 정확하고 클로즈업도 가깝지만, 촬영 순간에 요구되지 않은 손을 화면에 크게 추가하여 ‘사람 없음’과 고정된 장소 상태를 위반한다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "시선·무기·이동체는 없다. 스마트폰의 활성 화면은 카메라를 향해 비스듬히 놓여 있으며 ‘21:14:07’이 정방향으로 명확히 보인다.",
        "built_space": "붕괴한 갱도 바닥의 석탄 조각, 흙, 암석 잔해만 좁은 주변 맥락으로 보인다. 스마트폰은 프레임 중간 중앙의 전경을 차지한다. 봉인된 방폭문이나 쓰러진 인물은 인서트 프레이밍 밖이라 보이지 않으며, 중복된 고정 설비나 불가능한 반사는 없다.",
        "entities": "베젤이 유지된 실제 크기의 스마트폰 한 대와 붕괴 잔해가 보인다. 화면에는 요구된 유일한 가독 문구 ‘21:14:07’이 정확히 표시되며 다른 읽을 수 있는 글자나 인물은 없다.",
        "hard_violations": [],
        "physics": "스마트폰은 여러 암석과 흙더미 위에 접촉해 지지되고 있어 떠 있지 않는다. 잔해도 모두 바닥과 다른 돌 위에 놓여 있으며, 공중에서 움직이는 물체는 없다."
       },
       {
        "label": "A",
        "direction": "시선·무기·이동체는 없다. 스마트폰 화면은 카메라를 향해 비스듬히 놓였고 ‘21:14:07’이 정방향으로 보인다. 오른쪽 위의 손은 전화기 가장자리 쪽으로 뻗어 접촉하고 있다.",
        "built_space": "붕괴한 갱도 바닥의 석탄, 흙, 암석 잔해가 스마트폰 주변을 채운다. 전화기는 중앙에서 하단까지 크게 놓였고, 큰 손이 우상단을 점유한다. 고정 설비의 중복이나 불가능한 반사는 없지만, 손 때문에 요구된 순수한 시간 정지 인서트의 공간 구성이 바뀌었다.",
        "entities": "스마트폰 한 대, 잔해, 사람의 손과 팔 일부가 보인다. 화면의 유일한 글자는 정확한 ‘21:14:07’이며 다른 가독 문구는 없다. 그러나 프롬프트가 요구하지 않은 사람의 신체 일부가 추가되었다.",
        "hard_violations": [
         "스마트폰을 들거나 조작하는 행동이 명시되지 않았는데 사람의 손과 팔을 새로 등장시켜 ‘NO PEOPLE IN THIS SHOT’의 무인 staging을 위반했다."
        ],
        "physics": "스마트폰의 하단과 측면은 암석 잔해 위에 놓여 지지되고, 손도 전화기 가장자리에 닿아 있다. 손과 전화기 모두 물리적으로 떠 있지는 않으며 해부학적으로도 가능하지만, 이 접촉 행동 자체가 명시된 정지 상태에 불필요하게 추가되었다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 0.667,
    "B": 2.0
   },
   "adjusted": {
    "A": 0.417,
    "B": 2.0
   },
   "violations": {
    "A": [
     "[gemini-pro] 인물 등장 금지(NO PEOPLE) 지시사항을 위반하고 화면 우측에 사람의 손이 렌더링됨",
     "[gpt] 스마트폰을 들거나 조작하는 행동이 명시되지 않았는데 사람의 손과 팔을 새로 등장시켜 ‘NO PEOPLE IN THIS SHOT’의 무인 staging을 위반했다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "A": 417,
   "B": 2000
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 417,
    "verdict_ko": "프롬프트에서 인물이 등장하지 않아야 한다고 엄격히 제한했으나 화면에 사람의 손이 나타나 결정적인 하드 위반에 해당합니다.  ★위반: [gemini-pro] 인물 등장 금지(NO PEOPLE) 지시사항을 위반하고 화면 우측에 사람의 손이 렌더링됨 / [gpt] 스마트폰을 들거나 조작하는 행동이 명시되지 않았는데 사람의 손과 팔을 새로 등장시켜 ‘NO PEOPLE IN THIS SHOT’의 무인 staging을 위반했다."
   },
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "요구된 클로즈업 앵글과 '21:14:07' 텍스트를 정확히 렌더링했으며, 인물 없이 잔해 속 스마트폰이라는 설정을 완벽히 구현했습니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features, lighting mood and each person's clothing are LOCKED to this photo; never copy its camera framing. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S2sh21_sel.png",
    "asset_id": "36545fad-5e64-4b09-b989-dce1d0b820d9",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b283-57bb-745f-a4a6-d21e27c37477",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S2sh21"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S2sh22::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:21:49.521803+00:00",
  "fingerprint": "6cadfc5e06fb2c0432759a96096b2bf74d53ea9f2ed444870dd0bcdbd4faf1c6",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S2sh22_sel.png",
  "source_sha256": "784bcaeefc7791f086b2f9c4ec5e4d876b81f423d8efcbf806b91c502c070794",
  "file": "S2sh22_cine.png",
  "staged_sha256": "42704b511e110aedbd00a2087391b14e4abb3ed871ba8eb4e52eaf84e2f8be75",
  "latency_ms": 16563
 },
 "S3sh26::signage": {
  "fp": "e5b5f650c8de241f",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S3sh26": {
  "input_fingerprint": "a46fb71d6086192e",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 굳어 있는 '토니(앤서니 로저스)'의 얼굴 옆 허공에 얼어붙은 듯 맺혀 있는 물 한 방울.\n\nLOCATION (lock): Inside the time-frozen, rubble-filled section of the abandoned mine tunnel beside the trapped survivor, with dust suspended in the air. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) rigid profile in the middle-left of the frame, midground; suspended water droplet in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: suspended water droplet (motionless beside Tony's face); used as Primary suspended detail and match-cut anchor; motionless dust (remaining suspended in the tunnel air); used as Sparse depth markers around the droplet and rigid profile.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained low-key illumination appropriate to the enclosed tunnel separates the droplet and Tony's rigid profile without specifying an additional source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the collapsed concrete tunnel, rubble-strewn surroundings, motionless dust, and dim metallic color palette. Exclude falling debris, flowing dust, and any unrelated electronic display from the previous scene state.\n\nIMMOBILE CHARACTER POSE — CANONICAL (identical wherever this character appears in ANY panel; on any conflict THIS POSE WINS): Tony is immobilized beneath the rubble with his head and fixed eyes directed toward the smartphone, while one bloody hand extends from the debris and its index finger remains suspended directly above the send button. The rest of his body remains trapped under the rubble.\n\nIMMOBILE BODIES OBEY GRAVITY: a person who is dead or unconscious\nexerts NO muscular effort. Every part of their body — head, torso,\narms, hands, fingers, legs — rests fully on whatever supports it\n(floor, wall, furniture, their own lap) and hangs or slumps with\ngravity. NEVER show any part of an immobile person's body lifted,\nraised, held up in the air, or posed as if presenting something:\nan object in their grip stays clenched in a hand that itself lies\nfallen on a support — the hand does not hold the object up. If the\ncanonical pose leaves a body part unspecified, resolve it as the\nmost gravity-compliant, fully-supported position.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone remains frozen at 21:14:07 with the unsent draft addressed to MARA, while dust hangs motionless in the tunnel. A fine crack has formed among the suspended dust, and a water droplet emerging from it is stopped near Tony. 토니(앤서니 로저스): He is immobilized beneath the rubble with his head and fixed eyes directed toward the smartphone, while one bloody hand extends from the debris and its index finger remains suspended directly above the send button. The rest of his body remains trapped under the rubble. His rescue gear remains attached.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 굳어 있는 '토니(앤서니 로저스)'의 얼굴 옆 허공에 얼어붙은 듯 맺혀 있는 물 한 방울.\n\nLOCATION (lock): Inside the time-frozen, rubble-filled section of the abandoned mine tunnel beside the trapped survivor, with dust suspended in the air. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) rigid profile in the middle-left of the frame, midground; suspended water droplet in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: suspended water droplet (motionless beside Tony's face); used as Primary suspended detail and match-cut anchor; motionless dust (remaining suspended in the tunnel air); used as Sparse depth markers around the droplet and rigid profile.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained low-key illumination appropriate to the enclosed tunnel separates the droplet and Tony's rigid profile without specifying an additional source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the collapsed concrete tunnel, rubble-strewn surroundings, motionless dust, and dim metallic color palette. Exclude falling debris, flowing dust, and any unrelated electronic display from the previous scene state.\n\nIMMOBILE CHARACTER POSE — CANONICAL (identical wherever this character appears in ANY panel; on any conflict THIS POSE WINS): Tony is immobilized beneath the rubble with his head and fixed eyes directed toward the smartphone, while one bloody hand extends from the debris and its index finger remains suspended directly above the send button. The rest of his body remains trapped under the rubble.\n\nIMMOBILE BODIES OBEY GRAVITY: a person who is dead or unconscious\nexerts NO muscular effort. Every part of their body — head, torso,\narms, hands, fingers, legs — rests fully on whatever supports it\n(floor, wall, furniture, their own lap) and hangs or slumps with\ngravity. NEVER show any part of an immobile person's body lifted,\nraised, held up in the air, or posed as if presenting something:\nan object in their grip stays clenched in a hand that itself lies\nfallen on a support — the hand does not hold the object up. If the\ncanonical pose leaves a body part unspecified, resolve it as the\nmost gravity-compliant, fully-supported position.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone remains frozen at 21:14:07 with the unsent draft addressed to MARA, while dust hangs motionless in the tunnel. A fine crack has formed among the suspended dust, and a water droplet emerging from it is stopped near Tony. 토니(앤서니 로저스): He is immobilized beneath the rubble with his head and fixed eyes directed toward the smartphone, while one bloody hand extends from the debris and its index finger remains suspended directly above the send button. The rest of his body remains trapped under the rubble. His rescue gear remains attached.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 굳어 있는 '토니(앤서니 로저스)'의 얼굴 옆 허공에 얼어붙은 듯 맺혀 있는 물 한 방울.\n\nLOCATION (lock): Inside the time-frozen, rubble-filled section of the abandoned mine tunnel beside the trapped survivor, with dust suspended in the air. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) rigid profile in the middle-left of the frame, midground; suspended water droplet in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: suspended water droplet (motionless beside Tony's face); used as Primary suspended detail and match-cut anchor; motionless dust (remaining suspended in the tunnel air); used as Sparse depth markers around the droplet and rigid profile.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained low-key illumination appropriate to the enclosed tunnel separates the droplet and Tony's rigid profile without specifying an additional source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the collapsed concrete tunnel, rubble-strewn surroundings, motionless dust, and dim metallic color palette. Exclude falling debris, flowing dust, and any unrelated electronic display from the previous scene state.\n\nIMMOBILE CHARACTER POSE — CANONICAL (identical wherever this character appears in ANY panel; on any conflict THIS POSE WINS): Tony is immobilized beneath the rubble with his head and fixed eyes directed toward the smartphone, while one bloody hand extends from the debris and its index finger remains suspended directly above the send button. The rest of his body remains trapped under the rubble.\n\nIMMOBILE BODIES OBEY GRAVITY: a person who is dead or unconscious\nexerts NO muscular effort. Every part of their body — head, torso,\narms, hands, fingers, legs — rests fully on whatever supports it\n(floor, wall, furniture, their own lap) and hangs or slumps with\ngravity. NEVER show any part of an immobile person's body lifted,\nraised, held up in the air, or posed as if presenting something:\nan object in their grip stays clenched in a hand that itself lies\nfallen on a support — the hand does not hold the object up. If the\ncanonical pose leaves a body part unspecified, resolve it as the\nmost gravity-compliant, fully-supported position.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone remains frozen at 21:14:07 with the unsent draft addressed to MARA, while dust hangs motionless in the tunnel. A fine crack has formed among the suspended dust, and a water droplet emerging from it is stopped near Tony. 토니(앤서니 로저스): He is immobilized beneath the rubble with his head and fixed eyes directed toward the smartphone, while one bloody hand extends from the debris and its index finger remains suspended directly above the send button. The rest of his body remains trapped under the rubble. His rescue gear remains attached.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "토니의 시선은 정면에서 약간 아래쪽을 향하고 있다.",
    "built_space": "화면 우측에 젖은 암벽이 있고 바닥에 잔해가 있는 폐광 터널 공간으로 레퍼런스와 일치한다.",
    "entities": "토니는 레퍼런스 이미지와 동일한 헬멧과 의상을 착용하고 있으며, 화면 중앙 허공에 물방울이 맺혀 있다.",
    "hard_violations": [
     "[gpt] 헬멧 측면에 읽을 수 있는 문자 표기가 노출되어 ‘읽을 수 있는 글자 금지’ 조건을 위반한다."
    ],
    "physics": "물방울과 먼지 입자들이 허공에 떠 있으며, 토니의 몸은 바닥의 잔해에 의해 지탱되고 있다."
   },
   {
    "label": "B",
    "direction": "토니의 시선은 우측 하단에 있는 스마트폰을 향하고 있다.",
    "built_space": "우측의 젖은 암벽과 바닥의 잔해로 이루어진 터널 공간이다.",
    "entities": "토니가 등장하나 레퍼런스에 있는 헬멧과 헤드랜턴, 목 가리개가 없다. 중앙에 구형의 물방울이 떠 있고 우측 하단에 스마트폰이 있다.",
    "hard_violations": [
     "[gpt] 공중의 큰 구형 물방울에 보이는 발생 지점·낙하 궤적·접촉 지지물이 없어 물리적으로 아무것도 지지하지 않는다.",
     "[gpt] 토니가 잔해 아래에 고정된 정해진 위치가 아니라 잔해 옆에서 몸을 세운 듯 배치되어 연속 장면의 고정 스테이징을 충족하지 못한다."
    ],
    "physics": "물방울과 먼지가 공중에 떠 있고, 토니는 잔해에 기대어 있다. 스마트폰은 바닥 잔해에 놓여 지탱되고 있다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "레퍼런스의 헬멧과 복장 설정을 정확히 유지했으며, 요구된 클로즈업 구도와 허공의 물방울을 훌륭하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "레퍼런스에 고정된 헬멧과 목 가리개를 완전히 누락했으며, 화면에 스마트폰을 포함했으나 프롬프트에 명시된 손이 보이지 않습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선은 정면에서 약간 아래쪽을 향하고 있다.",
        "built_space": "화면 우측에 젖은 암벽이 있고 바닥에 잔해가 있는 폐광 터널 공간으로 레퍼런스와 일치한다.",
        "entities": "토니는 레퍼런스 이미지와 동일한 헬멧과 의상을 착용하고 있으며, 화면 중앙 허공에 물방울이 맺혀 있다.",
        "hard_violations": [],
        "physics": "물방울과 먼지 입자들이 허공에 떠 있으며, 토니의 몸은 바닥의 잔해에 의해 지탱되고 있다."
       },
       {
        "label": "B",
        "direction": "토니의 시선은 우측 하단에 있는 스마트폰을 향하고 있다.",
        "built_space": "우측의 젖은 암벽과 바닥의 잔해로 이루어진 터널 공간이다.",
        "entities": "토니가 등장하나 레퍼런스에 있는 헬멧과 헤드랜턴, 목 가리개가 없다. 중앙에 구형의 물방울이 떠 있고 우측 하단에 스마트폰이 있다.",
        "hard_violations": [],
        "physics": "물방울과 먼지가 공중에 떠 있고, 토니는 잔해에 기대어 있다. 스마트폰은 바닥 잔해에 놓여 지탱되고 있다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "레퍼런스의 헬멧과 복장 설정을 정확히 유지했으며, 요구된 클로즈업 구도와 허공의 물방울을 훌륭하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "레퍼런스에 고정된 헬멧과 목 가리개를 완전히 누락했으며, 화면에 스마트폰을 포함했으나 프롬프트에 명시된 손이 보이지 않습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선은 정면에서 약간 아래쪽을 향하고 있다.",
        "built_space": "화면 우측에 젖은 암벽이 있고 바닥에 잔해가 있는 폐광 터널 공간으로 레퍼런스와 일치한다.",
        "entities": "토니는 레퍼런스 이미지와 동일한 헬멧과 의상을 착용하고 있으며, 화면 중앙 허공에 물방울이 맺혀 있다.",
        "hard_violations": [],
        "physics": "물방울과 먼지 입자들이 허공에 떠 있으며, 토니의 몸은 바닥의 잔해에 의해 지탱되고 있다."
       },
       {
        "label": "B",
        "direction": "토니의 시선은 우측 하단에 있는 스마트폰을 향하고 있다.",
        "built_space": "우측의 젖은 암벽과 바닥의 잔해로 이루어진 터널 공간이다.",
        "entities": "토니가 등장하나 레퍼런스에 있는 헬멧과 헤드랜턴, 목 가리개가 없다. 중앙에 구형의 물방울이 떠 있고 우측 하단에 스마트폰이 있다.",
        "hard_violations": [],
        "physics": "물방울과 먼지가 공중에 떠 있고, 토니는 잔해에 기대어 있다. 스마트폰은 바닥 잔해에 놓여 지탱되고 있다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "스마트폰을 향한 시선은 맞지만 토니가 잔해 아래에 고정된 상태로 보이지 않고, 지나치게 큰 구형 물방울에는 낙하·발생 경로가 전혀 드러나지 않는다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "중간 왼쪽의 경직된 옆얼굴, 중앙의 정지 물방울, 잔해 아래의 위치와 구조 장비를 가장 충실히 구현했지만 헬멧의 읽을 수 있는 문자가 명시적 금지 조건을 위반한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 눈은 오른쪽 아래를 향하며 프레임 우하단의 스마트폰 화면에 대체로 닿는다. 물방울은 진행 방향 없이 토니의 코앞 중앙에 구형으로 떠 있다.",
        "built_space": "젖고 검은 터널 벽이 오른쪽에 있고 바닥에는 콘크리트 파편과 진흙이 보이며 공중에는 많은 먼지가 있다. 스마트폰 한 대가 우하단에 놓였지만 토니의 머리나 어깨를 붙잡는 잔해는 보이지 않아 ‘잔해 아래에 갇힌’ 배치가 명확하지 않다.",
        "entities": "토니는 미국인 30대 남성으로 보이고 얼굴과 머리색은 참고 인물과 대체로 유사하지만 헬멧 등 고정된 구조 장비가 보이지 않는다. 스마트폰 한 대와 먼지는 존재한다. 중앙 물체는 액체처럼 반사하지만 실제 물 한 방울보다 현저히 큰 구형 방울이며, 미세 균열은 보이지 않는다.",
        "hard_violations": [
         "공중의 큰 구형 물방울에 보이는 발생 지점·낙하 궤적·접촉 지지물이 없어 물리적으로 아무것도 지지하지 않는다.",
         "토니가 잔해 아래에 고정된 정해진 위치가 아니라 잔해 옆에서 몸을 세운 듯 배치되어 연속 장면의 고정 스테이징을 충족하지 못한다."
        ],
        "physics": "토니의 몸은 프레임 밖으로 이어져 바닥 지지는 추정할 수 있으나, 보이는 머리와 어깨에는 그를 잔해 아래에 고정하는 접촉점이 없다. 스마트폰은 프레임 가장자리 아래 표면에 놓인 것으로 보인다. 물방울은 완전한 구형으로 공중에 떠 있고 균열이나 낙하 흔적이 없어 아무것도 지지하지 않는다."
       },
       {
        "label": "B",
        "direction": "토니의 고정된 눈은 화면 오른쪽을 향해 프레임 밖 스마트폰이 있을 법한 방향을 본다. 물방울은 토니 얼굴 옆 중앙에 있으며 아래로 늘어진 형태와 위쪽의 작은 액체 조각이 수직 낙하 경로를 암시한다.",
        "built_space": "오른쪽에는 참고 사진과 같은 젖은 검은 터널 벽 한 면이 있고, 아래에는 콘크리트 파편과 진흙, 위와 왼쪽에는 토니를 둘러싼 붕괴 잔해가 있다. 토니는 그 잔해 아래 좁은 틈에 자리하며 먼지는 비교적 성기게 공중에 정지해 있다. 스마트폰은 올바른 클로즈업 크롭 밖에 있다.",
        "entities": "토니는 참고 인물과 유사한 30대 미국인 남성이며 검은 헬멧, 헤드램프, 오염된 구조복과 가슴 장비를 착용했다. 중앙에는 광택과 굴절을 가진 물방울이 있고 주변에 먼지가 있다. 다만 방울은 실제 한 방울보다 다소 크고 미세 균열은 식별되지 않으며, 헬멧 측면에 읽을 수 있는 문자가 있다.",
        "hard_violations": [
         "헬멧 측면에 읽을 수 있는 문자 표기가 노출되어 ‘읽을 수 있는 글자 금지’ 조건을 위반한다."
        ],
        "physics": "토니의 몸과 헬멧은 위·왼쪽의 콘크리트 잔해 아래 끼어 있어 갇힌 상태와 접촉 지지가 보인다. 프레임 밖 손과 스마트폰의 접촉은 이 크롭에서 판정할 수 없다. 물방울은 공중에 정지했지만 아래로 늘어진 형상과 바로 위의 작은 액체 조각이 위에서 시작된 낙하를 시각적으로 암시한다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "스마트폰을 향한 시선은 맞지만 토니가 잔해 아래에 고정된 상태로 보이지 않고, 지나치게 큰 구형 물방울에는 낙하·발생 경로가 전혀 드러나지 않는다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "중간 왼쪽의 경직된 옆얼굴, 중앙의 정지 물방울, 잔해 아래의 위치와 구조 장비를 가장 충실히 구현했지만 헬멧의 읽을 수 있는 문자가 명시적 금지 조건을 위반한다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "토니의 눈은 오른쪽 아래를 향하며 프레임 우하단의 스마트폰 화면에 대체로 닿는다. 물방울은 진행 방향 없이 토니의 코앞 중앙에 구형으로 떠 있다.",
        "built_space": "젖고 검은 터널 벽이 오른쪽에 있고 바닥에는 콘크리트 파편과 진흙이 보이며 공중에는 많은 먼지가 있다. 스마트폰 한 대가 우하단에 놓였지만 토니의 머리나 어깨를 붙잡는 잔해는 보이지 않아 ‘잔해 아래에 갇힌’ 배치가 명확하지 않다.",
        "entities": "토니는 미국인 30대 남성으로 보이고 얼굴과 머리색은 참고 인물과 대체로 유사하지만 헬멧 등 고정된 구조 장비가 보이지 않는다. 스마트폰 한 대와 먼지는 존재한다. 중앙 물체는 액체처럼 반사하지만 실제 물 한 방울보다 현저히 큰 구형 방울이며, 미세 균열은 보이지 않는다.",
        "hard_violations": [
         "공중의 큰 구형 물방울에 보이는 발생 지점·낙하 궤적·접촉 지지물이 없어 물리적으로 아무것도 지지하지 않는다.",
         "토니가 잔해 아래에 고정된 정해진 위치가 아니라 잔해 옆에서 몸을 세운 듯 배치되어 연속 장면의 고정 스테이징을 충족하지 못한다."
        ],
        "physics": "토니의 몸은 프레임 밖으로 이어져 바닥 지지는 추정할 수 있으나, 보이는 머리와 어깨에는 그를 잔해 아래에 고정하는 접촉점이 없다. 스마트폰은 프레임 가장자리 아래 표면에 놓인 것으로 보인다. 물방울은 완전한 구형으로 공중에 떠 있고 균열이나 낙하 흔적이 없어 아무것도 지지하지 않는다."
       },
       {
        "label": "A",
        "direction": "토니의 고정된 눈은 화면 오른쪽을 향해 프레임 밖 스마트폰이 있을 법한 방향을 본다. 물방울은 토니 얼굴 옆 중앙에 있으며 아래로 늘어진 형태와 위쪽의 작은 액체 조각이 수직 낙하 경로를 암시한다.",
        "built_space": "오른쪽에는 참고 사진과 같은 젖은 검은 터널 벽 한 면이 있고, 아래에는 콘크리트 파편과 진흙, 위와 왼쪽에는 토니를 둘러싼 붕괴 잔해가 있다. 토니는 그 잔해 아래 좁은 틈에 자리하며 먼지는 비교적 성기게 공중에 정지해 있다. 스마트폰은 올바른 클로즈업 크롭 밖에 있다.",
        "entities": "토니는 참고 인물과 유사한 30대 미국인 남성이며 검은 헬멧, 헤드램프, 오염된 구조복과 가슴 장비를 착용했다. 중앙에는 광택과 굴절을 가진 물방울이 있고 주변에 먼지가 있다. 다만 방울은 실제 한 방울보다 다소 크고 미세 균열은 식별되지 않으며, 헬멧 측면에 읽을 수 있는 문자가 있다.",
        "hard_violations": [
         "헬멧 측면에 읽을 수 있는 문자 표기가 노출되어 ‘읽을 수 있는 글자 금지’ 조건을 위반한다."
        ],
        "physics": "토니의 몸과 헬멧은 위·왼쪽의 콘크리트 잔해 아래 끼어 있어 갇힌 상태와 접촉 지지가 보인다. 프레임 밖 손과 스마트폰의 접촉은 이 크롭에서 판정할 수 없다. 물방울은 공중에 정지했지만 아래로 늘어진 형상과 바로 위의 작은 액체 조각이 위에서 시작된 낙하를 시각적으로 암시한다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.167
   },
   "adjusted": {
    "A": 1.75,
    "B": 0.917
   },
   "violations": {
    "B": [
     "[gpt] 공중의 큰 구형 물방울에 보이는 발생 지점·낙하 궤적·접촉 지지물이 없어 물리적으로 아무것도 지지하지 않는다.",
     "[gpt] 토니가 잔해 아래에 고정된 정해진 위치가 아니라 잔해 옆에서 몸을 세운 듯 배치되어 연속 장면의 고정 스테이징을 충족하지 못한다."
    ],
    "A": [
     "[gpt] 헬멧 측면에 읽을 수 있는 문자 표기가 노출되어 ‘읽을 수 있는 글자 금지’ 조건을 위반한다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 1750,
   "B": 917
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1750,
    "verdict_ko": "레퍼런스의 헬멧과 복장 설정을 정확히 유지했으며, 요구된 클로즈업 구도와 허공의 물방울을 훌륭하게 구현했습니다.  ★위반: [gpt] 헬멧 측면에 읽을 수 있는 문자 표기가 노출되어 ‘읽을 수 있는 글자 금지’ 조건을 위반한다."
   },
   {
    "label": "B",
    "score": 917,
    "verdict_ko": "레퍼런스에 고정된 헬멧과 목 가리개를 완전히 누락했으며, 화면에 스마트폰을 포함했으나 프롬프트에 명시된 손이 보이지 않습니다.  ★위반: [gpt] 공중의 큰 구형 물방울에 보이는 발생 지점·낙하 궤적·접촉 지지물이 없어 물리적으로 아무것도 지지하지 않는다. / [gpt] 토니가 잔해 아래에 고정된 정해진 위치가 아니라 잔해 옆에서 몸을 세운 듯 배치되어 연속 장면의 고정 스테이징을 충족하지 못한다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features, lighting mood and each person's clothing are LOCKED to this photo; never copy its camera framing. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S2sh21_sel.png",
    "asset_id": "36545fad-5e64-4b09-b989-dce1d0b820d9",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b286-e316-7ca4-bd26-dec2c6d2f458",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S2sh21"
  }
 },
 "S3sh26::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:23:01.808495+00:00",
  "fingerprint": "1c16fb38882b1fb5bdd70b2f5d6dca2819936b3b28521820dd70ca1dea59fa30",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S3sh26_sel.png",
  "source_sha256": "b9cc2c549c309c45fd6057469cc16e2acf14a87bde17e5d67c501fc9ff1ad7bf",
  "file": "S3sh26_cine.png",
  "staged_sha256": "4057ab7d21a7dd7965f4c60e14c62e8a8f1179854f4e927ae3efa4dffd891706",
  "latency_ms": 15976
 },
 "S3sh30::signage": {
  "fp": "5a414a0bbb7aafad",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S3sh30": {
  "input_fingerprint": "67b17739fcf7dcb5",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 굵어진 뿌리의 팽창 압력을 이기지 못하고 쩍 갈라진 콘크리트 벽면.\n\nLOCATION (lock): Inside the sealed mine passage, at a concrete wall beside the corroding blast-door structure where expanding roots are splitting the masonry. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- KEY BACKGROUND ELEMENTS: concrete wall (split open under root pressure); used as Primary ruptured surface carrying the diagonal fracture; thick root (expanded to rope-like thickness and pressing through the wall) — Its side contour crosses the pressure point diagonally before continuing beyond the close frame; used as Visible cause of the concrete rupture; broken concrete fragments (dislodged around the opened split); used as Secondary scale reference at the fracture edges.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained low-key illumination emphasizes the wall break and root contour without introducing a new visible light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The blast door is heavily rusted and invaded by roots now as thick as rope. Root pressure has split the concrete, while the suspended dust, stopped water droplet, and smartphone clock at 21:14:07 remain unchanged.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 굵어진 뿌리의 팽창 압력을 이기지 못하고 쩍 갈라진 콘크리트 벽면.\n\nLOCATION (lock): Inside the sealed mine passage, at a concrete wall beside the corroding blast-door structure where expanding roots are splitting the masonry. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- KEY BACKGROUND ELEMENTS: concrete wall (split open under root pressure); used as Primary ruptured surface carrying the diagonal fracture; thick root (expanded to rope-like thickness and pressing through the wall) — Its side contour crosses the pressure point diagonally before continuing beyond the close frame; used as Visible cause of the concrete rupture; broken concrete fragments (dislodged around the opened split); used as Secondary scale reference at the fracture edges.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained low-key illumination emphasizes the wall break and root contour without introducing a new visible light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The blast door is heavily rusted and invaded by roots now as thick as rope. Root pressure has split the concrete, while the suspended dust, stopped water droplet, and smartphone clock at 21:14:07 remain unchanged.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 굵어진 뿌리의 팽창 압력을 이기지 못하고 쩍 갈라진 콘크리트 벽면.\n\nLOCATION (lock): Inside the sealed mine passage, at a concrete wall beside the corroding blast-door structure where expanding roots are splitting the masonry. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- KEY BACKGROUND ELEMENTS: concrete wall (split open under root pressure); used as Primary ruptured surface carrying the diagonal fracture; thick root (expanded to rope-like thickness and pressing through the wall) — Its side contour crosses the pressure point diagonally before continuing beyond the close frame; used as Visible cause of the concrete rupture; broken concrete fragments (dislodged around the opened split); used as Secondary scale reference at the fracture edges.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained low-key illumination emphasizes the wall break and root contour without introducing a new visible light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The blast door is heavily rusted and invaded by roots now as thick as rope. Root pressure has split the concrete, while the suspended dust, stopped water droplet, and smartphone clock at 21:14:07 remain unchanged.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "굵은 뿌리들이 콘크리트 벽을 뚫고 프레임의 우측 하단 대각선 방향으로 뻗어나가며 화면 밖으로 이어집니다.",
    "built_space": "화면 좌측과 중앙에 균열이 간 콘크리트 벽이 있고, 우측 배경에 녹슨 방폭문 구조물이 배치되어 있습니다. 인물은 없습니다.",
    "entities": "밧줄처럼 굵어진 뿌리, 파손된 콘크리트 파편, 배경의 녹슨 철문, 그리고 공중에 떠 있는 다수의 물방울과 먼지 입자들이 정확히 묘사되었습니다.",
    "hard_violations": [],
    "physics": "뿌리는 갈라진 콘크리트 틈에 단단히 박혀 지탱되고 있으며, 주변의 물방울과 먼지들은 중력에 영향을 받지 않고 공중에 완전히 정지된 상태를 보여줍니다."
   },
   {
    "label": "B",
    "direction": "하나의 굵은 뿌리가 벽체에서 빠져나와 프레임 내부의 바닥을 향해 대각선 아래로 뻗어 안착합니다.",
    "built_space": "화면 좌측에 거대한 녹슨 방폭문의 일부가 있고, 중앙에 파손된 콘크리트 벽면, 하단에 바닥면이 노출되어 있습니다.",
    "entities": "단일한 굵은 뿌리, 부서진 콘크리트 조각들, 녹슨 방폭문이 묘사되어 있으나, 공중에 정지된 뚜렷한 물방울 형태는 식별하기 어렵고 잔물결이나 먼지처럼 보입니다.",
    "hard_violations": [],
    "physics": "뿌리는 벽체에서 나와 바닥면에 닿아 지탱되고 있으며, 뿌리 표면의 물기가 아래로 흐르는 듯한 흔적이 있어 시간이 정지된 물리적 상태를 완벽히 구현하지 못했습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "정지된 물방울과 먼지의 상태를 성공적으로 유지했으며, 프레임을 가로질러 뻗어나가는 뿌리의 구도와 콘크리트 파열의 묘사가 촬영 지시문과 정확히 일치합니다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "콘크리트가 파열된 질감은 우수하나, 뿌리가 프레임 밖으로 이어지지 않고 바닥에 닿아 있어 클로즈업 프레임 지시를 어겼으며, 물방울이 정지되지 않고 흐르는 듯한 느낌을 줍니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "굵은 뿌리들이 콘크리트 벽을 뚫고 프레임의 우측 하단 대각선 방향으로 뻗어나가며 화면 밖으로 이어집니다.",
        "built_space": "화면 좌측과 중앙에 균열이 간 콘크리트 벽이 있고, 우측 배경에 녹슨 방폭문 구조물이 배치되어 있습니다. 인물은 없습니다.",
        "entities": "밧줄처럼 굵어진 뿌리, 파손된 콘크리트 파편, 배경의 녹슨 철문, 그리고 공중에 떠 있는 다수의 물방울과 먼지 입자들이 정확히 묘사되었습니다.",
        "hard_violations": [],
        "physics": "뿌리는 갈라진 콘크리트 틈에 단단히 박혀 지탱되고 있으며, 주변의 물방울과 먼지들은 중력에 영향을 받지 않고 공중에 완전히 정지된 상태를 보여줍니다."
       },
       {
        "label": "B",
        "direction": "하나의 굵은 뿌리가 벽체에서 빠져나와 프레임 내부의 바닥을 향해 대각선 아래로 뻗어 안착합니다.",
        "built_space": "화면 좌측에 거대한 녹슨 방폭문의 일부가 있고, 중앙에 파손된 콘크리트 벽면, 하단에 바닥면이 노출되어 있습니다.",
        "entities": "단일한 굵은 뿌리, 부서진 콘크리트 조각들, 녹슨 방폭문이 묘사되어 있으나, 공중에 정지된 뚜렷한 물방울 형태는 식별하기 어렵고 잔물결이나 먼지처럼 보입니다.",
        "hard_violations": [],
        "physics": "뿌리는 벽체에서 나와 바닥면에 닿아 지탱되고 있으며, 뿌리 표면의 물기가 아래로 흐르는 듯한 흔적이 있어 시간이 정지된 물리적 상태를 완벽히 구현하지 못했습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "정지된 물방울과 먼지의 상태를 성공적으로 유지했으며, 프레임을 가로질러 뻗어나가는 뿌리의 구도와 콘크리트 파열의 묘사가 촬영 지시문과 정확히 일치합니다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "콘크리트가 파열된 질감은 우수하나, 뿌리가 프레임 밖으로 이어지지 않고 바닥에 닿아 있어 클로즈업 프레임 지시를 어겼으며, 물방울이 정지되지 않고 흐르는 듯한 느낌을 줍니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "굵은 뿌리들이 콘크리트 벽을 뚫고 프레임의 우측 하단 대각선 방향으로 뻗어나가며 화면 밖으로 이어집니다.",
        "built_space": "화면 좌측과 중앙에 균열이 간 콘크리트 벽이 있고, 우측 배경에 녹슨 방폭문 구조물이 배치되어 있습니다. 인물은 없습니다.",
        "entities": "밧줄처럼 굵어진 뿌리, 파손된 콘크리트 파편, 배경의 녹슨 철문, 그리고 공중에 떠 있는 다수의 물방울과 먼지 입자들이 정확히 묘사되었습니다.",
        "hard_violations": [],
        "physics": "뿌리는 갈라진 콘크리트 틈에 단단히 박혀 지탱되고 있으며, 주변의 물방울과 먼지들은 중력에 영향을 받지 않고 공중에 완전히 정지된 상태를 보여줍니다."
       },
       {
        "label": "B",
        "direction": "하나의 굵은 뿌리가 벽체에서 빠져나와 프레임 내부의 바닥을 향해 대각선 아래로 뻗어 안착합니다.",
        "built_space": "화면 좌측에 거대한 녹슨 방폭문의 일부가 있고, 중앙에 파손된 콘크리트 벽면, 하단에 바닥면이 노출되어 있습니다.",
        "entities": "단일한 굵은 뿌리, 부서진 콘크리트 조각들, 녹슨 방폭문이 묘사되어 있으나, 공중에 정지된 뚜렷한 물방울 형태는 식별하기 어렵고 잔물결이나 먼지처럼 보입니다.",
        "hard_violations": [],
        "physics": "뿌리는 벽체에서 나와 바닥면에 닿아 지탱되고 있으며, 뿌리 표면의 물기가 아래로 흐르는 듯한 흔적이 있어 시간이 정지된 물리적 상태를 완벽히 구현하지 못했습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "갈라진 콘크리트와 대각선 뿌리 압박은 명확하지만 프레임이 다소 넓고 차가운 조명이 이전 숏의 절제된 광색과 덜 일치한다."
       },
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "콘크리트 파열점과 이를 대각선으로 관통해 화면 밖까지 이어지는 굵은 뿌리를 정확한 클로즈업으로 포착하고 녹슨 방폭문과 저조도 분위기도 충실히 유지한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "시선이나 무기는 없다. 굵은 뿌리가 화면 오른쪽 아래에서 왼쪽 위의 콘크리트 파열점을 향해 비스듬히 밀고 들어가며, 진행 방향이 실제 갈라진 지점에 닿는다.",
        "built_space": "광산 통로의 콘크리트 벽 한 면과 그 왼쪽 가장자리에 녹슨 방폭문 구조물 한 개가 보인다. 벽 중앙에는 하나의 큰 파열부와 주변 균열·탈락 조각이 있으며 사람은 없다. 반사면이나 거울은 없다.",
        "entities": "콘크리트 벽, 압력으로 열린 대각선 균열, 밧줄보다 굵어진 실제 나무뿌리, 파열 가장자리와 바닥의 콘크리트 파편, 부분적으로 보이는 녹슨 방폭문이 모두 식별된다. 사람·얼굴·읽을 수 있는 글자는 없다. 다만 뿌리가 밧줄 굵기보다는 작은 나무줄기처럼 매우 굵고, 조명이 기준 이미지보다 푸른 편이다.",
        "hard_violations": [],
        "physics": "뿌리는 오른쪽 화면 밖과 벽 내부로 이어져 고정되어 있고 파열점에 직접 접촉하므로 지지와 압력 원인이 보인다. 콘크리트 파편은 균열 가장자리나 바닥에 놓여 있으며 일부 작은 물방울과 먼지는 장면에 명시된 정지 상태로 공중에 보인다. 지지 없이 떠 있는 인물이나 소품은 없다."
       },
       {
        "label": "B",
        "direction": "시선이나 무기는 없다. 주된 굵은 뿌리는 화면 오른쪽 아래에서 왼쪽 위의 벽 파열점으로 향해 밀고 들어가며, 그 측면 윤곽이 압력점을 대각선으로 가로지른 뒤 프레임 밖으로 계속된다. 보조 뿌리들도 같은 파열부에 모인다.",
        "built_space": "가까운 콘크리트 벽 한 면과 오른쪽 배경의 녹슨 방폭문 구조물 한 개가 보인다. 벽에는 하나의 주된 열린 파열부와 방사형 균열, 가장자리의 탈락 조각이 있다. 사람은 없으며 반사면이나 광학적으로 불가능한 반사도 없다.",
        "entities": "콘크리트 벽, 크게 열린 균열, 굵고 물리적인 뿌리, 여러 콘크리트 파편, 녹슨 방폭문, 부유 먼지와 물방울이 모두 보인다. 사람·얼굴·의상·읽을 수 있는 글자는 없다. 장소의 습기 찬 콘크리트와 부식 금속 재질도 기준 이미지와 잘 맞는다.",
        "hard_violations": [],
        "physics": "굵은 뿌리는 벽 내부와 오른쪽 프레임 밖으로 연속되어 고정되고 벽 파열면에 밀착해 압력을 전달한다. 균열에 걸린 파편은 주변 콘크리트에 받쳐져 있고, 중앙의 작은 파편은 막 탈락해 낙하하는 순간으로 읽혀 파열이 발사 원인을 제공한다. 공중의 먼지와 물방울은 프롬프트가 잠긴 정지 상태로 명시한 요소이며, 지지 없이 떠 있는 인물이나 사용 소품은 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "갈라진 콘크리트와 대각선 뿌리 압박은 명확하지만 프레임이 다소 넓고 차가운 조명이 이전 숏의 절제된 광색과 덜 일치한다."
       },
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "콘크리트 파열점과 이를 대각선으로 관통해 화면 밖까지 이어지는 굵은 뿌리를 정확한 클로즈업으로 포착하고 녹슨 방폭문과 저조도 분위기도 충실히 유지한다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "시선이나 무기는 없다. 굵은 뿌리가 화면 오른쪽 아래에서 왼쪽 위의 콘크리트 파열점을 향해 비스듬히 밀고 들어가며, 진행 방향이 실제 갈라진 지점에 닿는다.",
        "built_space": "광산 통로의 콘크리트 벽 한 면과 그 왼쪽 가장자리에 녹슨 방폭문 구조물 한 개가 보인다. 벽 중앙에는 하나의 큰 파열부와 주변 균열·탈락 조각이 있으며 사람은 없다. 반사면이나 거울은 없다.",
        "entities": "콘크리트 벽, 압력으로 열린 대각선 균열, 밧줄보다 굵어진 실제 나무뿌리, 파열 가장자리와 바닥의 콘크리트 파편, 부분적으로 보이는 녹슨 방폭문이 모두 식별된다. 사람·얼굴·읽을 수 있는 글자는 없다. 다만 뿌리가 밧줄 굵기보다는 작은 나무줄기처럼 매우 굵고, 조명이 기준 이미지보다 푸른 편이다.",
        "hard_violations": [],
        "physics": "뿌리는 오른쪽 화면 밖과 벽 내부로 이어져 고정되어 있고 파열점에 직접 접촉하므로 지지와 압력 원인이 보인다. 콘크리트 파편은 균열 가장자리나 바닥에 놓여 있으며 일부 작은 물방울과 먼지는 장면에 명시된 정지 상태로 공중에 보인다. 지지 없이 떠 있는 인물이나 소품은 없다."
       },
       {
        "label": "A",
        "direction": "시선이나 무기는 없다. 주된 굵은 뿌리는 화면 오른쪽 아래에서 왼쪽 위의 벽 파열점으로 향해 밀고 들어가며, 그 측면 윤곽이 압력점을 대각선으로 가로지른 뒤 프레임 밖으로 계속된다. 보조 뿌리들도 같은 파열부에 모인다.",
        "built_space": "가까운 콘크리트 벽 한 면과 오른쪽 배경의 녹슨 방폭문 구조물 한 개가 보인다. 벽에는 하나의 주된 열린 파열부와 방사형 균열, 가장자리의 탈락 조각이 있다. 사람은 없으며 반사면이나 광학적으로 불가능한 반사도 없다.",
        "entities": "콘크리트 벽, 크게 열린 균열, 굵고 물리적인 뿌리, 여러 콘크리트 파편, 녹슨 방폭문, 부유 먼지와 물방울이 모두 보인다. 사람·얼굴·의상·읽을 수 있는 글자는 없다. 장소의 습기 찬 콘크리트와 부식 금속 재질도 기준 이미지와 잘 맞는다.",
        "hard_violations": [],
        "physics": "굵은 뿌리는 벽 내부와 오른쪽 프레임 밖으로 연속되어 고정되고 벽 파열면에 밀착해 압력을 전달한다. 균열에 걸린 파편은 주변 콘크리트에 받쳐져 있고, 중앙의 작은 파편은 막 탈락해 낙하하는 순간으로 읽혀 파열이 발사 원인을 제공한다. 공중의 먼지와 물방울은 프롬프트가 잠긴 정지 상태로 명시한 요소이며, 지지 없이 떠 있는 인물이나 사용 소품은 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.444
   },
   "adjusted": {
    "A": 2.0,
    "B": 1.444
   },
   "violations": {},
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 1444
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "정지된 물방울과 먼지의 상태를 성공적으로 유지했으며, 프레임을 가로질러 뻗어나가는 뿌리의 구도와 콘크리트 파열의 묘사가 촬영 지시문과 정확히 일치합니다."
   },
   {
    "label": "B",
    "score": 1444,
    "verdict_ko": "콘크리트가 파열된 질감은 우수하나, 뿌리가 프레임 밖으로 이어지지 않고 바닥에 닿아 있어 클로즈업 프레임 지시를 어겼으며, 물방울이 정지되지 않고 흐르는 듯한 느낌을 줍니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S3sh26_sel.png",
    "asset_id": "16a170d8-273c-453c-9642-aced81dd440c",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b28b-5e83-75de-b96a-e3adfc464ca3",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S3sh26"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S3sh30::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:23:59.798484+00:00",
  "fingerprint": "03257c1894b905d75c45255b4784ec6bcc7de199c7db8438aea267c66038e922",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S3sh30_sel.png",
  "source_sha256": "0a448d3459a9bf731dec9707eb3010a2eceb91e7e18ce9ee2c634048e692921d",
  "file": "S3sh30_cine.png",
  "staged_sha256": "038e6643d6c5498165932467457d4546cef0b0180d596a70049ab18f63e98767",
  "latency_ms": 13502
 },
 "S3sh31::signage": {
  "fp": "de0629ba768e5549",
  "inscriptions": [
   {
    "text_native": "21:14:07",
    "source": "scene_text_quoted",
    "reason_ko": "스마트폰 화면에 표시된 시간이 장면에 명시적으로 지정되어 있습니다.",
    "source_quote": "21:14:07"
   }
  ],
  "cues": [],
  "dropped": []
 },
 "S3sh31": {
  "input_fingerprint": "2e7242f0d57cc6ed",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 주변의 풍화된 잔해 위로 여전히 '21:14:07'을 띄운 채 놓인 스마트폰 화면.\n\nLOCATION (lock): Inside the collapsed mine passage, among heavily weathered rubble surrounding the motionless survivor and phone. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: smartphone screen showing 21:14:07 in the lower-center of the frame, foreground; weathered debris surrounding phone in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: smartphone (still displaying 21:14:07) — The screen face is visible from a steep high angle and clearly shows the unchanged time 21:14:07; used as Primary time-reveal endpoint after the downward tilt; weathered debris (surrounding the unchanged smartphone); used as Environmental contrast demonstrating change around the static display.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Neutral low-key illumination appropriate to the tunnel keeps the screen legible against the restrained, weathered surroundings without specifying a new source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone still displays 21:14:07 amid weathered rubble. Thick roots have forced through the rusted blast door and cracked concrete, while the airborne dust and water droplet remain stopped.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"21:14:07\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 주변의 풍화된 잔해 위로 여전히 '21:14:07'을 띄운 채 놓인 스마트폰 화면.\n\nLOCATION (lock): Inside the collapsed mine passage, among heavily weathered rubble surrounding the motionless survivor and phone. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: smartphone screen showing 21:14:07 in the lower-center of the frame, foreground; weathered debris surrounding phone in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: smartphone (still displaying 21:14:07) — The screen face is visible from a steep high angle and clearly shows the unchanged time 21:14:07; used as Primary time-reveal endpoint after the downward tilt; weathered debris (surrounding the unchanged smartphone); used as Environmental contrast demonstrating change around the static display.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Neutral low-key illumination appropriate to the tunnel keeps the screen legible against the restrained, weathered surroundings without specifying a new source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone still displays 21:14:07 amid weathered rubble. Thick roots have forced through the rusted blast door and cracked concrete, while the airborne dust and water droplet remain stopped.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"21:14:07\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 주변의 풍화된 잔해 위로 여전히 '21:14:07'을 띄운 채 놓인 스마트폰 화면.\n\nLOCATION (lock): Inside the collapsed mine passage, among heavily weathered rubble surrounding the motionless survivor and phone. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: smartphone screen showing 21:14:07 in the lower-center of the frame, foreground; weathered debris surrounding phone in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: smartphone (still displaying 21:14:07) — The screen face is visible from a steep high angle and clearly shows the unchanged time 21:14:07; used as Primary time-reveal endpoint after the downward tilt; weathered debris (surrounding the unchanged smartphone); used as Environmental contrast demonstrating change around the static display.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Neutral low-key illumination appropriate to the tunnel keeps the screen legible against the restrained, weathered surroundings without specifying a new source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone still displays 21:14:07 amid weathered rubble. Thick roots have forced through the rusted blast door and cracked concrete, while the airborne dust and water droplet remain stopped.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"21:14:07\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라는 가파른 하이 앵글로 아래를 향해 있으며, 위를 향해 놓인 스마트폰 화면을 정확히 비추고 있습니다.",
    "built_space": "레퍼런스 이미지의 구조와 일치합니다. 왼쪽에 갈라진 콘크리트 벽과 굵은 뿌리가 있고, 오른쪽 배경에 수직으로 닫힌 녹슨 방폭문이 그대로 위치해 있습니다. 전경은 잔해들로 이루어져 있습니다.",
    "entities": "스마트폰 화면에 '21:14:07'이 정확히 출력되어 있습니다. 풍화된 잔해, 굵은 뿌리, 정지된 물방울과 먼지가 존재하며, 인물은 없습니다.",
    "hard_violations": [
     "[gpt] 유일하게 허용된 핵심 화면 문구에 콜론이 하나 더 나타나 ‘21:14:07’을 정확히 표시하지 못한다."
    ],
    "physics": "스마트폰은 불규칙한 잔해 위에 안정적으로 놓여 있습니다. 물방울과 먼지는 프롬프트의 지시대로 공중에 뜬 채 정지된 상태를 잘 보여줍니다."
   },
   {
    "label": "B",
    "direction": "카메라는 스마트폰을 향해 비스듬히 아래를 내려다보고 있으나, 요구된 '가파른(steep)' 각도보다는 완만합니다.",
    "built_space": "왼쪽의 콘크리트 벽과 뿌리는 존재하지만, 레퍼런스에서 오른쪽 배경 벽에 고정되어 있던 녹슨 방폭문이 떨어져 나와 우측 바닥에 평평하게 놓여 있어 공간적 일관성이 깨졌습니다.",
    "entities": "스마트폰 화면에 '21:14:07'이 정확히 표시되어 있습니다. 뿌리, 바위, 잔해, 정지된 물방울이 묘사되어 있으며, 인물은 없습니다.",
    "hard_violations": [
     "[gemini-pro] 레퍼런스에 확립된 고정된 배경 구조물(녹슨 문)의 위치를 임의로 바닥으로 변경함 (Location Lock 위반)"
    ],
    "physics": "스마트폰이 바닥에 안정적으로 놓여 있으며, 물방울은 공중에 정지된 상태로 떠 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "레퍼런스의 공간 구조(왼쪽 벽의 뿌리와 오른쪽 배경의 녹슨 문)를 정확히 유지하면서, 프롬프트가 요구한 가파른 하이 앵글과 정지된 물방울, 스마트폰의 시간('21:14:07')을 완벽하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "스마트폰의 화면 텍스트와 정지된 물방울은 잘 표현되었으나, 레퍼런스에서 수직으로 서 있던 녹슨 문이 바닥에 눕혀져 있어 공간의 연속성(Location Lock)을 위반했으며, 요구된 앵글보다 완만한 각도로 촬영되었습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 가파른 하이 앵글로 아래를 향해 있으며, 위를 향해 놓인 스마트폰 화면을 정확히 비추고 있습니다.",
        "built_space": "레퍼런스 이미지의 구조와 일치합니다. 왼쪽에 갈라진 콘크리트 벽과 굵은 뿌리가 있고, 오른쪽 배경에 수직으로 닫힌 녹슨 방폭문이 그대로 위치해 있습니다. 전경은 잔해들로 이루어져 있습니다.",
        "entities": "스마트폰 화면에 '21:14:07'이 정확히 출력되어 있습니다. 풍화된 잔해, 굵은 뿌리, 정지된 물방울과 먼지가 존재하며, 인물은 없습니다.",
        "hard_violations": [],
        "physics": "스마트폰은 불규칙한 잔해 위에 안정적으로 놓여 있습니다. 물방울과 먼지는 프롬프트의 지시대로 공중에 뜬 채 정지된 상태를 잘 보여줍니다."
       },
       {
        "label": "B",
        "direction": "카메라는 스마트폰을 향해 비스듬히 아래를 내려다보고 있으나, 요구된 '가파른(steep)' 각도보다는 완만합니다.",
        "built_space": "왼쪽의 콘크리트 벽과 뿌리는 존재하지만, 레퍼런스에서 오른쪽 배경 벽에 고정되어 있던 녹슨 방폭문이 떨어져 나와 우측 바닥에 평평하게 놓여 있어 공간적 일관성이 깨졌습니다.",
        "entities": "스마트폰 화면에 '21:14:07'이 정확히 표시되어 있습니다. 뿌리, 바위, 잔해, 정지된 물방울이 묘사되어 있으며, 인물은 없습니다.",
        "hard_violations": [
         "레퍼런스에 확립된 고정된 배경 구조물(녹슨 문)의 위치를 임의로 바닥으로 변경함 (Location Lock 위반)"
        ],
        "physics": "스마트폰이 바닥에 안정적으로 놓여 있으며, 물방울은 공중에 정지된 상태로 떠 있습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "레퍼런스의 공간 구조(왼쪽 벽의 뿌리와 오른쪽 배경의 녹슨 문)를 정확히 유지하면서, 프롬프트가 요구한 가파른 하이 앵글과 정지된 물방울, 스마트폰의 시간('21:14:07')을 완벽하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "스마트폰의 화면 텍스트와 정지된 물방울은 잘 표현되었으나, 레퍼런스에서 수직으로 서 있던 녹슨 문이 바닥에 눕혀져 있어 공간의 연속성(Location Lock)을 위반했으며, 요구된 앵글보다 완만한 각도로 촬영되었습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 가파른 하이 앵글로 아래를 향해 있으며, 위를 향해 놓인 스마트폰 화면을 정확히 비추고 있습니다.",
        "built_space": "레퍼런스 이미지의 구조와 일치합니다. 왼쪽에 갈라진 콘크리트 벽과 굵은 뿌리가 있고, 오른쪽 배경에 수직으로 닫힌 녹슨 방폭문이 그대로 위치해 있습니다. 전경은 잔해들로 이루어져 있습니다.",
        "entities": "스마트폰 화면에 '21:14:07'이 정확히 출력되어 있습니다. 풍화된 잔해, 굵은 뿌리, 정지된 물방울과 먼지가 존재하며, 인물은 없습니다.",
        "hard_violations": [],
        "physics": "스마트폰은 불규칙한 잔해 위에 안정적으로 놓여 있습니다. 물방울과 먼지는 프롬프트의 지시대로 공중에 뜬 채 정지된 상태를 잘 보여줍니다."
       },
       {
        "label": "B",
        "direction": "카메라는 스마트폰을 향해 비스듬히 아래를 내려다보고 있으나, 요구된 '가파른(steep)' 각도보다는 완만합니다.",
        "built_space": "왼쪽의 콘크리트 벽과 뿌리는 존재하지만, 레퍼런스에서 오른쪽 배경 벽에 고정되어 있던 녹슨 방폭문이 떨어져 나와 우측 바닥에 평평하게 놓여 있어 공간적 일관성이 깨졌습니다.",
        "entities": "스마트폰 화면에 '21:14:07'이 정확히 표시되어 있습니다. 뿌리, 바위, 잔해, 정지된 물방울이 묘사되어 있으며, 인물은 없습니다.",
        "hard_violations": [
         "레퍼런스에 확립된 고정된 배경 구조물(녹슨 문)의 위치를 임의로 바닥으로 변경함 (Location Lock 위반)"
        ],
        "physics": "스마트폰이 바닥에 안정적으로 놓여 있으며, 물방울은 공중에 정지된 상태로 떠 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "스마트폰을 전경 하단 중앙의 클로즈업으로 배치하고 화면에 정확히 ‘21:14:07’을 표시해 핵심 시간 공개와 프레이밍을 가장 충실히 구현했다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "장소의 녹슨 방폭문과 뿌리는 더 잘 이어지지만, 핵심 화면이 ‘21::14:07’처럼 보여 지정 문구를 정확히 재현하지 못한 점이 결정적이다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "시선·무기·지시물은 없다. 스마트폰 화면은 가파른 상부 시점의 카메라를 향해 열려 있고, 표시 시간이 관객에게 명확히 보인다.",
        "built_space": "무너진 광산 통로의 갈라진 콘크리트, 다량의 잔해, 벽을 뚫고 나온 굵은 뿌리와 오른쪽의 녹슨 금속문 일부가 보인다. 스마트폰은 하단 중앙 전경, 풍화 잔해는 중앙 중경에 놓여 요구된 공간 배치와 맞는다. 인물이나 불가능한 반사는 없다.",
        "entities": "실물 크기의 스마트폰, 정확한 ‘21:14:07’ 표시, 풍화된 콘크리트 잔해, 굵은 뿌리, 녹슨 방폭문 일부, 공중의 먼지와 물방울이 확인된다. 사람·얼굴·신체는 없다. 화면 가장자리의 미세한 UI 흔적은 판독되지 않는다.",
        "hard_violations": [],
        "physics": "스마트폰은 평평한 콘크리트 바닥과 잔해 위에 완전히 지지되어 있다. 뿌리는 갈라진 벽과 잔해 사이에 박혀 있다. 물방울과 먼지는 낙하·비산 중 한 순간이 촬영된 모습으로, 정지된 시간 상태에 부합하며 다른 물체가 부자연스럽게 떠 있지 않다."
       },
       {
        "label": "B",
        "direction": "시선·무기·지시물은 없다. 스마트폰의 기능면은 높은 카메라 쪽을 향해 있어 화면 자체는 잘 보이지만, 표시 문자열은 ‘21::14:07’처럼 읽힌다.",
        "built_space": "갈라진 콘크리트 벽, 많은 잔해, 통로를 가로지르는 굵은 뿌리, 뒤쪽의 녹슨 방폭문 한 개가 보이며 참고 이미지의 공간을 잘 계승한다. 스마트폰은 하단 중앙에 있고 잔해가 중앙을 둘러싼다. 인물과 불가능한 반사는 없다.",
        "entities": "스마트폰, 풍화된 잔해, 굵은 뿌리, 녹슨 방폭문, 먼지와 물방울은 존재하고 사람은 없다. 다만 가장 중요한 화면 문구가 요구된 정확한 ‘21:14:07’이 아니라 콜론이 하나 더 들어간 ‘21::14:07’처럼 보인다.",
        "hard_violations": [
         "유일하게 허용된 핵심 화면 문구에 콜론이 하나 더 나타나 ‘21:14:07’을 정확히 표시하지 못한다."
        ],
        "physics": "스마트폰은 잔해와 젖은 바닥에 받쳐져 있고 뿌리는 벽과 잔해에 고정되어 있다. 공중의 물방울과 먼지는 비산 또는 낙하가 멈춘 순간으로 읽혀 지정된 정지 상태와 맞으며, 그 밖에 지지 없이 떠 있는 물체는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "스마트폰을 전경 하단 중앙의 클로즈업으로 배치하고 화면에 정확히 ‘21:14:07’을 표시해 핵심 시간 공개와 프레이밍을 가장 충실히 구현했다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "장소의 녹슨 방폭문과 뿌리는 더 잘 이어지지만, 핵심 화면이 ‘21::14:07’처럼 보여 지정 문구를 정확히 재현하지 못한 점이 결정적이다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "시선·무기·지시물은 없다. 스마트폰 화면은 가파른 상부 시점의 카메라를 향해 열려 있고, 표시 시간이 관객에게 명확히 보인다.",
        "built_space": "무너진 광산 통로의 갈라진 콘크리트, 다량의 잔해, 벽을 뚫고 나온 굵은 뿌리와 오른쪽의 녹슨 금속문 일부가 보인다. 스마트폰은 하단 중앙 전경, 풍화 잔해는 중앙 중경에 놓여 요구된 공간 배치와 맞는다. 인물이나 불가능한 반사는 없다.",
        "entities": "실물 크기의 스마트폰, 정확한 ‘21:14:07’ 표시, 풍화된 콘크리트 잔해, 굵은 뿌리, 녹슨 방폭문 일부, 공중의 먼지와 물방울이 확인된다. 사람·얼굴·신체는 없다. 화면 가장자리의 미세한 UI 흔적은 판독되지 않는다.",
        "hard_violations": [],
        "physics": "스마트폰은 평평한 콘크리트 바닥과 잔해 위에 완전히 지지되어 있다. 뿌리는 갈라진 벽과 잔해 사이에 박혀 있다. 물방울과 먼지는 낙하·비산 중 한 순간이 촬영된 모습으로, 정지된 시간 상태에 부합하며 다른 물체가 부자연스럽게 떠 있지 않다."
       },
       {
        "label": "A",
        "direction": "시선·무기·지시물은 없다. 스마트폰의 기능면은 높은 카메라 쪽을 향해 있어 화면 자체는 잘 보이지만, 표시 문자열은 ‘21::14:07’처럼 읽힌다.",
        "built_space": "갈라진 콘크리트 벽, 많은 잔해, 통로를 가로지르는 굵은 뿌리, 뒤쪽의 녹슨 방폭문 한 개가 보이며 참고 이미지의 공간을 잘 계승한다. 스마트폰은 하단 중앙에 있고 잔해가 중앙을 둘러싼다. 인물과 불가능한 반사는 없다.",
        "entities": "스마트폰, 풍화된 잔해, 굵은 뿌리, 녹슨 방폭문, 먼지와 물방울은 존재하고 사람은 없다. 다만 가장 중요한 화면 문구가 요구된 정확한 ‘21:14:07’이 아니라 콜론이 하나 더 들어간 ‘21::14:07’처럼 보인다.",
        "hard_violations": [
         "유일하게 허용된 핵심 화면 문구에 콜론이 하나 더 나타나 ‘21:14:07’을 정확히 표시하지 못한다."
        ],
        "physics": "스마트폰은 잔해와 젖은 바닥에 받쳐져 있고 뿌리는 벽과 잔해에 고정되어 있다. 공중의 물방울과 먼지는 비산 또는 낙하가 멈춘 순간으로 읽혀 지정된 정지 상태와 맞으며, 그 밖에 지지 없이 떠 있는 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.667,
    "B": 1.6
   },
   "adjusted": {
    "A": 1.417,
    "B": 1.35
   },
   "violations": {
    "B": [
     "[gemini-pro] 레퍼런스에 확립된 고정된 배경 구조물(녹슨 문)의 위치를 임의로 바닥으로 변경함 (Location Lock 위반)"
    ],
    "A": [
     "[gpt] 유일하게 허용된 핵심 화면 문구에 콜론이 하나 더 나타나 ‘21:14:07’을 정확히 표시하지 못한다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "agreed": false
  },
  "totals": {
   "A": 1417,
   "B": 1350
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1417,
    "verdict_ko": "레퍼런스의 공간 구조(왼쪽 벽의 뿌리와 오른쪽 배경의 녹슨 문)를 정확히 유지하면서, 프롬프트가 요구한 가파른 하이 앵글과 정지된 물방울, 스마트폰의 시간('21:14:07')을 완벽하게 구현했습니다.  ★위반: [gpt] 유일하게 허용된 핵심 화면 문구에 콜론이 하나 더 나타나 ‘21:14:07’을 정확히 표시하지 못한다."
   },
   {
    "label": "B",
    "score": 1350,
    "verdict_ko": "스마트폰의 화면 텍스트와 정지된 물방울은 잘 표현되었으나, 레퍼런스에서 수직으로 서 있던 녹슨 문이 바닥에 눕혀져 있어 공간의 연속성(Location Lock)을 위반했으며, 요구된 앵글보다 완만한 각도로 촬영되었습니다.  ★위반: [gemini-pro] 레퍼런스에 확립된 고정된 배경 구조물(녹슨 문)의 위치를 임의로 바닥으로 변경함 (Location Lock 위반)"
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S3sh30_sel.png",
    "asset_id": "4c739f5a-2da1-46d1-80ff-e5c80cc45923",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b28e-d6dc-72c1-a0b7-f42fe889718c",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S3sh30"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S3sh31::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:24:56.488883+00:00",
  "fingerprint": "25ffcf596c791912a676d3df550a59f99cceb720c3a9e2771a5c1fc791d4de2e",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S3sh31_sel.png",
  "source_sha256": "68f3ed87a08d79b82a79c04b2290a45d2a5549e4c860fe79196749e324fbcad5",
  "file": "S3sh31_cine.png",
  "staged_sha256": "e3d67a92cfd37d2e53a0bf7b2e4aefd5c4f4a6706d3a0f1b5a3cede1961d821e",
  "latency_ms": 17866
 },
 "S4sh32::signage": {
  "fp": "22f465860347ee8e",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S4sh32": {
  "input_fingerprint": "acddcd4801ac3fb5",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 막혔던 숨을 토해내며 입을 크게 벌린 '토니(앤서니 로저스)'의 흙먼지 묻은 얼굴.\n\nLOCATION (lock): Inside the ruined mine passage reclaimed by massive roots, amid broken concrete and the remains of the sealed doorway. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) breathing face in the middle-center of the frame, foreground.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained low-key illumination keeps the dust on Tony's face tactile while preserving the dark, uncertain tunnel mood without naming an unstated source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone clock still reads 21:14:07 at the instant motion resumes. A waist-high tree root has torn through the concrete and iron blast door, leaving the tunnel heavily weathered and split. 토니(앤서니 로저스): His dust-covered face abruptly moves again as he opens his mouth and draws breath. His rescue gear and chest-mounted equipment remain attached.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 막혔던 숨을 토해내며 입을 크게 벌린 '토니(앤서니 로저스)'의 흙먼지 묻은 얼굴.\n\nLOCATION (lock): Inside the ruined mine passage reclaimed by massive roots, amid broken concrete and the remains of the sealed doorway. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) breathing face in the middle-center of the frame, foreground.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained low-key illumination keeps the dust on Tony's face tactile while preserving the dark, uncertain tunnel mood without naming an unstated source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone clock still reads 21:14:07 at the instant motion resumes. A waist-high tree root has torn through the concrete and iron blast door, leaving the tunnel heavily weathered and split. 토니(앤서니 로저스): His dust-covered face abruptly moves again as he opens his mouth and draws breath. His rescue gear and chest-mounted equipment remain attached.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 막혔던 숨을 토해내며 입을 크게 벌린 '토니(앤서니 로저스)'의 흙먼지 묻은 얼굴.\n\nLOCATION (lock): Inside the ruined mine passage reclaimed by massive roots, amid broken concrete and the remains of the sealed doorway. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) breathing face in the middle-center of the frame, foreground.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained low-key illumination keeps the dust on Tony's face tactile while preserving the dark, uncertain tunnel mood without naming an unstated source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone clock still reads 21:14:07 at the instant motion resumes. A waist-high tree root has torn through the concrete and iron blast door, leaving the tunnel heavily weathered and split. 토니(앤서니 로저스): His dust-covered face abruptly moves again as he opens his mouth and draws breath. His rescue gear and chest-mounted equipment remain attached.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "토니의 시선은 정면 위쪽을 향함.",
    "built_space": "폐광 통로. 이전 샷의 카메라 구도와 배경을 완벽하게 동일한 비율로 복사하는 바람에, 아래를 향하던 카메라 앵글의 바닥 잔해들이 인물 목 뒤에 수직으로 세워진 것처럼 배치되는 공간적 모순이 발생함.",
    "entities": "토니(앤서니 로저스)의 흙먼지 묻은 얼굴과 벌린 입, 그리고 가슴에 부착된 장비가 구현됨.",
    "hard_violations": [
     "[gemini-pro] physically impossible staging (아래를 향해 찍힌 바닥 잔해 이미지를 서 있는 인물의 수직 배경으로 사용하여 공간 왜곡 발생)",
     "[gemini-pro] 지지하는 것 없이 허공에 멈춰 있는 물방울 (동작이 재개되었다는 프롬프트에도 불구하고 이전 샷의 정지된 물방울 위치를 2D 텍스처처럼 인물 주위에 그대로 복사함)"
    ],
    "physics": "동작이 재개되는 순간임에도 이전 샷 레퍼런스에 있던 허공의 물방울들이 물리적 지지나 낙하 모션 없이 인물 턱 주변 허공에 똑같은 좌표로 떠 있음."
   },
   {
    "label": "B",
    "direction": "토니의 시선은 정면 위쪽의 허공을 향해 있음.",
    "built_space": "폐광 통로 내부. 인물이 전경을 크게 차지하여 뒤쪽의 부서진 콘크리트 벽과 철문이 수직 배경으로 자연스럽게 위치하며, 공간의 깊이감이 유지됨.",
    "entities": "토니(앤서니 로저스)는 30대 미국인 남성으로 흙먼지가 묻은 얼굴과 입을 크게 벌린 표정이 잘 표현됨. 가슴에는 녹색 안테나가 있는 장비와 하네스를 착용함.",
    "hard_violations": [],
    "physics": "인물은 꼿꼿이 몸을 세우고 있으며 화면 밖에 있는 하체에 의해 자연스럽게 지지되고 있음. 부자연스럽게 허공에 뜬 물체가 없음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "이전 샷의 배경 구도를 일부 답습한 아쉬움이 있으나, 클로즈업 샷의 스케일을 키워 공간의 물리적 오류를 방지했으며 얼굴의 흙먼지와 조명, 거칠게 숨을 들이마시는 표정을 사실적으로 잘 구현했습니다."
       },
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "이전 샷의 배경 2D 이미지를 그대로 복사하여 수평인 바닥 잔해가 수직 벽처럼 인물 뒤에 묘사되는 물리적 공간 오류가 발생했으며, 허공의 물방울까지 복사된 심각한 하드 위반입니다."
       }
      ],
      "readings": [
       {
        "label": "B",
        "direction": "토니의 시선은 정면 위쪽의 허공을 향해 있음.",
        "built_space": "폐광 통로 내부. 인물이 전경을 크게 차지하여 뒤쪽의 부서진 콘크리트 벽과 철문이 수직 배경으로 자연스럽게 위치하며, 공간의 깊이감이 유지됨.",
        "entities": "토니(앤서니 로저스)는 30대 미국인 남성으로 흙먼지가 묻은 얼굴과 입을 크게 벌린 표정이 잘 표현됨. 가슴에는 녹색 안테나가 있는 장비와 하네스를 착용함.",
        "hard_violations": [],
        "physics": "인물은 꼿꼿이 몸을 세우고 있으며 화면 밖에 있는 하체에 의해 자연스럽게 지지되고 있음. 부자연스럽게 허공에 뜬 물체가 없음."
       },
       {
        "label": "A",
        "direction": "토니의 시선은 정면 위쪽을 향함.",
        "built_space": "폐광 통로. 이전 샷의 카메라 구도와 배경을 완벽하게 동일한 비율로 복사하는 바람에, 아래를 향하던 카메라 앵글의 바닥 잔해들이 인물 목 뒤에 수직으로 세워진 것처럼 배치되는 공간적 모순이 발생함.",
        "entities": "토니(앤서니 로저스)의 흙먼지 묻은 얼굴과 벌린 입, 그리고 가슴에 부착된 장비가 구현됨.",
        "hard_violations": [
         "physically impossible staging (아래를 향해 찍힌 바닥 잔해 이미지를 서 있는 인물의 수직 배경으로 사용하여 공간 왜곡 발생)",
         "지지하는 것 없이 허공에 멈춰 있는 물방울 (동작이 재개되었다는 프롬프트에도 불구하고 이전 샷의 정지된 물방울 위치를 2D 텍스처처럼 인물 주위에 그대로 복사함)"
        ],
        "physics": "동작이 재개되는 순간임에도 이전 샷 레퍼런스에 있던 허공의 물방울들이 물리적 지지나 낙하 모션 없이 인물 턱 주변 허공에 똑같은 좌표로 떠 있음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "이전 샷의 배경 구도를 일부 답습한 아쉬움이 있으나, 클로즈업 샷의 스케일을 키워 공간의 물리적 오류를 방지했으며 얼굴의 흙먼지와 조명, 거칠게 숨을 들이마시는 표정을 사실적으로 잘 구현했습니다."
       },
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "이전 샷의 배경 2D 이미지를 그대로 복사하여 수평인 바닥 잔해가 수직 벽처럼 인물 뒤에 묘사되는 물리적 공간 오류가 발생했으며, 허공의 물방울까지 복사된 심각한 하드 위반입니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "토니의 시선은 정면 위쪽의 허공을 향해 있음.",
        "built_space": "폐광 통로 내부. 인물이 전경을 크게 차지하여 뒤쪽의 부서진 콘크리트 벽과 철문이 수직 배경으로 자연스럽게 위치하며, 공간의 깊이감이 유지됨.",
        "entities": "토니(앤서니 로저스)는 30대 미국인 남성으로 흙먼지가 묻은 얼굴과 입을 크게 벌린 표정이 잘 표현됨. 가슴에는 녹색 안테나가 있는 장비와 하네스를 착용함.",
        "hard_violations": [],
        "physics": "인물은 꼿꼿이 몸을 세우고 있으며 화면 밖에 있는 하체에 의해 자연스럽게 지지되고 있음. 부자연스럽게 허공에 뜬 물체가 없음."
       },
       {
        "label": "A",
        "direction": "토니의 시선은 정면 위쪽을 향함.",
        "built_space": "폐광 통로. 이전 샷의 카메라 구도와 배경을 완벽하게 동일한 비율로 복사하는 바람에, 아래를 향하던 카메라 앵글의 바닥 잔해들이 인물 목 뒤에 수직으로 세워진 것처럼 배치되는 공간적 모순이 발생함.",
        "entities": "토니(앤서니 로저스)의 흙먼지 묻은 얼굴과 벌린 입, 그리고 가슴에 부착된 장비가 구현됨.",
        "hard_violations": [
         "physically impossible staging (아래를 향해 찍힌 바닥 잔해 이미지를 서 있는 인물의 수직 배경으로 사용하여 공간 왜곡 발생)",
         "지지하는 것 없이 허공에 멈춰 있는 물방울 (동작이 재개되었다는 프롬프트에도 불구하고 이전 샷의 정지된 물방울 위치를 2D 텍스처처럼 인물 주위에 그대로 복사함)"
        ],
        "physics": "동작이 재개되는 순간임에도 이전 샷 레퍼런스에 있던 허공의 물방울들이 물리적 지지나 낙하 모션 없이 인물 턱 주변 허공에 똑같은 좌표로 떠 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "얼굴을 화면 중앙 전경에 더 타이트한 클로즈업으로 배치하고 입을 크게 벌려 막혔던 숨을 토해내는 순간을 가장 명확하게 구현했다."
       },
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "장소와 인물 정체성은 잘 맞지만 A보다 프레이밍이 조금 넓고 입 벌림과 호흡 동작의 강도가 덜하며, 가슴 장비도 참고 이미지와 다소 다르다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 얼굴은 정면에 가깝고 눈은 화면 왼쪽 바깥을 향한다. 지시된 응시 대상은 없으며, 무기나 겨누는 물체도 없다. 크게 열린 입은 카메라 앞쪽으로 숨을 내뱉는 동작으로 읽힌다.",
        "built_space": "토니는 화면 중앙 전경을 차지한다. 뒤에는 갈라진 콘크리트 벽, 여러 갈래의 거대한 뿌리, 콘크리트 잔해와 녹슨 봉쇄문 일부가 보이며 이전 숏의 폐광 통로 구조와 재질이 이어진다. 프레임 안에 중복된 고정 설비나 불가능한 반사는 없다.",
        "entities": "보이는 사람은 30대 중반의 미국인 남성 토니 한 명뿐이다. 얼굴과 짧은 자연색 머리, 체격 및 어두운 구조복은 캐릭터 참고와 대체로 일치하고 얼굴에는 흙먼지가 선명하다. 다만 수염이 참고보다 조금 짙다. 구조용 하네스와 가슴 장비가 프레임 하단에 부착되어 있으며 스마트폰은 올바른 클로즈업 밖에 있어 보이지 않는다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "토니의 상반신은 수직으로 안정되어 있고 프레임 밖의 하체나 주변 지면이 체중을 지지하는 자연스러운 자세로 보인다. 공중에 뜬 신체나 물체는 없다. 가슴 장비는 스트랩과 하네스에 물리적으로 고정되어 있다."
       },
       {
        "label": "B",
        "direction": "토니는 화면 왼쪽 바깥을 바라보고 있으며 지정된 별도 대상은 없다. 열린 입은 전방으로 숨을 내쉬거나 들이마시는 순간으로 읽히고, 입 왼쪽의 작은 물방울들은 왼쪽 전방으로 이동하는 듯하다. 무기나 겨누는 물체는 없다.",
        "built_space": "토니는 중앙 전경에서 잔해에 기대거나 낮게 자리한 듯 보인다. 배경에는 파손된 콘크리트 벽, 굵은 뿌리 여러 갈래, 녹슨 금속과 봉쇄문 잔해가 있어 이전 숏의 폐광 통로와 잘 연결된다. 중복된 고정 설비나 불가능한 반사는 보이지 않는다.",
        "entities": "유일한 인물은 토니에 부합하는 30대 중반의 미국인 남성으로, 짧은 갈색 머리와 얼굴형, 가벼운 수염은 참고 인물과 잘 맞는다. 얼굴은 흙먼지로 덮였고 구조복과 하네스가 보인다. 가슴 장비는 부착되어 있으나 참고 이미지의 장비보다 무전기 형태가 강해 외형 일치도가 떨어진다. 스마트폰은 프레임 밖이며 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "상반신은 잔해 쪽에 기대어 지지되는 자연스러운 자세이며 공중에 뜬 신체는 없다. 가슴 장비는 하네스에 고정되어 있다. 얼굴 왼쪽의 방울은 호흡이나 침 튀김으로 발사된 짧은 비행 상태로 해석 가능해 물리적으로 불가능하지 않다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "얼굴을 화면 중앙 전경에 더 타이트한 클로즈업으로 배치하고 입을 크게 벌려 막혔던 숨을 토해내는 순간을 가장 명확하게 구현했다."
       },
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "장소와 인물 정체성은 잘 맞지만 A보다 프레이밍이 조금 넓고 입 벌림과 호흡 동작의 강도가 덜하며, 가슴 장비도 참고 이미지와 다소 다르다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "토니의 얼굴은 정면에 가깝고 눈은 화면 왼쪽 바깥을 향한다. 지시된 응시 대상은 없으며, 무기나 겨누는 물체도 없다. 크게 열린 입은 카메라 앞쪽으로 숨을 내뱉는 동작으로 읽힌다.",
        "built_space": "토니는 화면 중앙 전경을 차지한다. 뒤에는 갈라진 콘크리트 벽, 여러 갈래의 거대한 뿌리, 콘크리트 잔해와 녹슨 봉쇄문 일부가 보이며 이전 숏의 폐광 통로 구조와 재질이 이어진다. 프레임 안에 중복된 고정 설비나 불가능한 반사는 없다.",
        "entities": "보이는 사람은 30대 중반의 미국인 남성 토니 한 명뿐이다. 얼굴과 짧은 자연색 머리, 체격 및 어두운 구조복은 캐릭터 참고와 대체로 일치하고 얼굴에는 흙먼지가 선명하다. 다만 수염이 참고보다 조금 짙다. 구조용 하네스와 가슴 장비가 프레임 하단에 부착되어 있으며 스마트폰은 올바른 클로즈업 밖에 있어 보이지 않는다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "토니의 상반신은 수직으로 안정되어 있고 프레임 밖의 하체나 주변 지면이 체중을 지지하는 자연스러운 자세로 보인다. 공중에 뜬 신체나 물체는 없다. 가슴 장비는 스트랩과 하네스에 물리적으로 고정되어 있다."
       },
       {
        "label": "A",
        "direction": "토니는 화면 왼쪽 바깥을 바라보고 있으며 지정된 별도 대상은 없다. 열린 입은 전방으로 숨을 내쉬거나 들이마시는 순간으로 읽히고, 입 왼쪽의 작은 물방울들은 왼쪽 전방으로 이동하는 듯하다. 무기나 겨누는 물체는 없다.",
        "built_space": "토니는 중앙 전경에서 잔해에 기대거나 낮게 자리한 듯 보인다. 배경에는 파손된 콘크리트 벽, 굵은 뿌리 여러 갈래, 녹슨 금속과 봉쇄문 잔해가 있어 이전 숏의 폐광 통로와 잘 연결된다. 중복된 고정 설비나 불가능한 반사는 보이지 않는다.",
        "entities": "유일한 인물은 토니에 부합하는 30대 중반의 미국인 남성으로, 짧은 갈색 머리와 얼굴형, 가벼운 수염은 참고 인물과 잘 맞는다. 얼굴은 흙먼지로 덮였고 구조복과 하네스가 보인다. 가슴 장비는 부착되어 있으나 참고 이미지의 장비보다 무전기 형태가 강해 외형 일치도가 떨어진다. 스마트폰은 프레임 밖이며 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "상반신은 잔해 쪽에 기대어 지지되는 자연스러운 자세이며 공중에 뜬 신체는 없다. 가슴 장비는 하네스에 고정되어 있다. 얼굴 왼쪽의 방울은 호흡이나 침 튀김으로 발사된 짧은 비행 상태로 해석 가능해 물리적으로 불가능하지 않다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.175,
    "B": 2.0
   },
   "adjusted": {
    "A": 0.925,
    "B": 2.0
   },
   "violations": {
    "A": [
     "[gemini-pro] physically impossible staging (아래를 향해 찍힌 바닥 잔해 이미지를 서 있는 인물의 수직 배경으로 사용하여 공간 왜곡 발생)",
     "[gemini-pro] 지지하는 것 없이 허공에 멈춰 있는 물방울 (동작이 재개되었다는 프롬프트에도 불구하고 이전 샷의 정지된 물방울 위치를 2D 텍스처처럼 인물 주위에 그대로 복사함)"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "B": 2000,
   "A": 925
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "이전 샷의 배경 구도를 일부 답습한 아쉬움이 있으나, 클로즈업 샷의 스케일을 키워 공간의 물리적 오류를 방지했으며 얼굴의 흙먼지와 조명, 거칠게 숨을 들이마시는 표정을 사실적으로 잘 구현했습니다."
   },
   {
    "label": "A",
    "score": 925,
    "verdict_ko": "이전 샷의 배경 2D 이미지를 그대로 복사하여 수평인 바닥 잔해가 수직 벽처럼 인물 뒤에 묘사되는 물리적 공간 오류가 발생했으며, 허공의 물방울까지 복사된 심각한 하드 위반입니다.  ★위반: [gemini-pro] physically impossible staging (아래를 향해 찍힌 바닥 잔해 이미지를 서 있는 인물의 수직 배경으로 사용하여 공간 왜곡 발생) / [gemini-pro] 지지하는 것 없이 허공에 멈춰 있는 물방울 (동작이 재개되었다는 프롬프트에도 불구하고 이전 샷의 정지된 물방울 위치를 2D 텍스처처럼 인물 주위에 그대로 복사함)"
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S3sh31_sel.png",
    "asset_id": "3d2698bf-200b-40d2-937a-334e75c7e3db",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b292-a7d4-7706-9460-7266dd2a4252",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S3sh31"
  }
 },
 "S4sh32::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:26:46.224086+00:00",
  "fingerprint": "4a7e4920f1e74d7b54f2c6418147bd9fbf6bf5fa2b08239f0357fb7438471c9c",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S4sh32_sel.png",
  "source_sha256": "1fbd8e869ed066abcd57b5addf7aba06c0de0d28daf5a30ff93740b34d8382cb",
  "file": "S4sh32_cine.png",
  "staged_sha256": "d095d512c422de7762adf4e15341358b6af2467a003c3afdb1b14bd062968223",
  "latency_ms": 15417
 },
 "S4sh33::signage": {
  "fp": "5615ca252b492396",
  "inscriptions": [
   {
    "text_native": "21:14:08",
    "source": "scene_text_quoted",
    "reason_ko": "스마트폰 화면에 표시되는 정확한 시간 숫자가 지문에 명시되어 있다.",
    "source_quote": "21:14:08"
   }
  ],
  "cues": [],
  "dropped": []
 },
 "S4sh33": {
  "input_fingerprint": "f710a86e66ec51a9",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): '21:14:08'로 숫자가 넘어간 순간의 스마트폰 화면 클로즈업.\n\nLOCATION (lock): Inside the root-invaded ruins of the collapsed mine tunnel, beside the newly revived survivor. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: smartphone display showing 21:14:08 and 17 percent in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: smartphone (display changed to 21:14:08 with battery at 17 percent) — The screen face is squarely visible to camera and shows 21:14:08 together with the 17 percent battery reading; used as Primary insert marking the resumption of time.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The active display supplies localized screen illumination within an otherwise restrained low-key frame.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone clock advances to 21:14:08 and its battery remains at 17 percent. The massive root, cracked concrete, and torn iron door remain unchanged around it.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"21:14:08\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): '21:14:08'로 숫자가 넘어간 순간의 스마트폰 화면 클로즈업.\n\nLOCATION (lock): Inside the root-invaded ruins of the collapsed mine tunnel, beside the newly revived survivor. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: smartphone display showing 21:14:08 and 17 percent in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: smartphone (display changed to 21:14:08 with battery at 17 percent) — The screen face is squarely visible to camera and shows 21:14:08 together with the 17 percent battery reading; used as Primary insert marking the resumption of time.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The active display supplies localized screen illumination within an otherwise restrained low-key frame.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone clock advances to 21:14:08 and its battery remains at 17 percent. The massive root, cracked concrete, and torn iron door remain unchanged around it.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"21:14:08\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): '21:14:08'로 숫자가 넘어간 순간의 스마트폰 화면 클로즈업.\n\nLOCATION (lock): Inside the root-invaded ruins of the collapsed mine tunnel, beside the newly revived survivor. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: smartphone display showing 21:14:08 and 17 percent in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: smartphone (display changed to 21:14:08 with battery at 17 percent) — The screen face is squarely visible to camera and shows 21:14:08 together with the 17 percent battery reading; used as Primary insert marking the resumption of time.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The active display supplies localized screen illumination within an otherwise restrained low-key frame.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone clock advances to 21:14:08 and its battery remains at 17 percent. The massive root, cracked concrete, and torn iron door remain unchanged around it.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"21:14:08\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라는 흙이 묻은 손에 들려 있는 스마트폰의 정면을 똑바로 향하고 있습니다.",
    "built_space": "레퍼런스 이미지의 배경과 일치합니다. 붕괴된 터널 내부로, 왼쪽에는 거대한 나무뿌리가 있고 오른쪽 뒤편으로는 녹슨 철문이 위치해 있으며 주변에 잔해가 널려 있습니다.",
    "entities": "스마트폰 화면 중앙에는 '21:14:08'과 '17 percent'가 정확히 표시되어 있습니다. 화면을 잡고 있는 흙 묻은 맨손이 보이며, 그 외에 다른 인물이나 얼굴은 프레임에 존재하지 않습니다.",
    "hard_violations": [],
    "physics": "더러운 손의 엄지와 검지가 스마트폰의 하단부를 단단히 쥐고 지탱하고 있습니다."
   },
   {
    "label": "B",
    "direction": "카메라는 흙 묻은 손에 쥐어진 스마트폰 화면의 정면을 비추고 있습니다.",
    "built_space": "레퍼런스와 동일하게 붕괴된 터널을 보여주며, 왼쪽에 굵은 나무뿌리, 오른쪽에 녹슨 철문, 바닥에 파편들이 배치되어 있어 공간적 연속성이 유지됩니다.",
    "entities": "스마트폰 화면에 '21:14:08'과 '17 percent', 배터리 아이콘이 표시되어 있으나, 그 아래위에 읽을 수 없는 깨진 텍스트들이 추가로 존재합니다. 폰을 쥔 손이 보이며, 화면 하단부에 프롬프트에 없는 인물의 얼굴(눈과 코 부분)이 반사되어 뚜렷하게 보입니다.",
    "hard_violations": [
     "[gemini-pro] invented people (스마트폰 화면에 반사된 인물의 얼굴)",
     "[gemini-pro] leaked markers/diagrams/text (화면 상의 지시되지 않은 깨진 텍스트들)",
     "[gpt] 화면 상단의 추가로 읽히는 UI 문구가 나타나, 지정 문구 외에는 읽을 수 있는 글자를 두지 말라는 조건을 위반한다."
    ],
    "physics": "흙 묻은 손이 스마트폰의 오른쪽 측면과 하단을 감싸 쥐고 안정적으로 들고 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "프롬프트가 요구한 '21:14:08' 화면을 정확하게 렌더링했으며, 이전 샷의 배경을 완벽히 유지한 채 규정대로 인물의 얼굴을 노출하지 않고 화면을 든 손만 자연스럽게 표현하여 지시사항을 완벽히 이행했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "배경의 연속성은 훌륭하나, 프롬프트에서 엄격히 금지한 인물의 얼굴이 스마트폰 화면에 반사되어 나타났으며 요구하지 않은 정체불명의 텍스트까지 화면에 추가되어 핵심 규칙을 위반했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 흙이 묻은 손에 들려 있는 스마트폰의 정면을 똑바로 향하고 있습니다.",
        "built_space": "레퍼런스 이미지의 배경과 일치합니다. 붕괴된 터널 내부로, 왼쪽에는 거대한 나무뿌리가 있고 오른쪽 뒤편으로는 녹슨 철문이 위치해 있으며 주변에 잔해가 널려 있습니다.",
        "entities": "스마트폰 화면 중앙에는 '21:14:08'과 '17 percent'가 정확히 표시되어 있습니다. 화면을 잡고 있는 흙 묻은 맨손이 보이며, 그 외에 다른 인물이나 얼굴은 프레임에 존재하지 않습니다.",
        "hard_violations": [],
        "physics": "더러운 손의 엄지와 검지가 스마트폰의 하단부를 단단히 쥐고 지탱하고 있습니다."
       },
       {
        "label": "B",
        "direction": "카메라는 흙 묻은 손에 쥐어진 스마트폰 화면의 정면을 비추고 있습니다.",
        "built_space": "레퍼런스와 동일하게 붕괴된 터널을 보여주며, 왼쪽에 굵은 나무뿌리, 오른쪽에 녹슨 철문, 바닥에 파편들이 배치되어 있어 공간적 연속성이 유지됩니다.",
        "entities": "스마트폰 화면에 '21:14:08'과 '17 percent', 배터리 아이콘이 표시되어 있으나, 그 아래위에 읽을 수 없는 깨진 텍스트들이 추가로 존재합니다. 폰을 쥔 손이 보이며, 화면 하단부에 프롬프트에 없는 인물의 얼굴(눈과 코 부분)이 반사되어 뚜렷하게 보입니다.",
        "hard_violations": [
         "invented people (스마트폰 화면에 반사된 인물의 얼굴)",
         "leaked markers/diagrams/text (화면 상의 지시되지 않은 깨진 텍스트들)"
        ],
        "physics": "흙 묻은 손이 스마트폰의 오른쪽 측면과 하단을 감싸 쥐고 안정적으로 들고 있습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 10,
        "verdict_ko": "프롬프트가 요구한 '21:14:08' 화면을 정확하게 렌더링했으며, 이전 샷의 배경을 완벽히 유지한 채 규정대로 인물의 얼굴을 노출하지 않고 화면을 든 손만 자연스럽게 표현하여 지시사항을 완벽히 이행했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "배경의 연속성은 훌륭하나, 프롬프트에서 엄격히 금지한 인물의 얼굴이 스마트폰 화면에 반사되어 나타났으며 요구하지 않은 정체불명의 텍스트까지 화면에 추가되어 핵심 규칙을 위반했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 흙이 묻은 손에 들려 있는 스마트폰의 정면을 똑바로 향하고 있습니다.",
        "built_space": "레퍼런스 이미지의 배경과 일치합니다. 붕괴된 터널 내부로, 왼쪽에는 거대한 나무뿌리가 있고 오른쪽 뒤편으로는 녹슨 철문이 위치해 있으며 주변에 잔해가 널려 있습니다.",
        "entities": "스마트폰 화면 중앙에는 '21:14:08'과 '17 percent'가 정확히 표시되어 있습니다. 화면을 잡고 있는 흙 묻은 맨손이 보이며, 그 외에 다른 인물이나 얼굴은 프레임에 존재하지 않습니다.",
        "hard_violations": [],
        "physics": "더러운 손의 엄지와 검지가 스마트폰의 하단부를 단단히 쥐고 지탱하고 있습니다."
       },
       {
        "label": "B",
        "direction": "카메라는 흙 묻은 손에 쥐어진 스마트폰 화면의 정면을 비추고 있습니다.",
        "built_space": "레퍼런스와 동일하게 붕괴된 터널을 보여주며, 왼쪽에 굵은 나무뿌리, 오른쪽에 녹슨 철문, 바닥에 파편들이 배치되어 있어 공간적 연속성이 유지됩니다.",
        "entities": "스마트폰 화면에 '21:14:08'과 '17 percent', 배터리 아이콘이 표시되어 있으나, 그 아래위에 읽을 수 없는 깨진 텍스트들이 추가로 존재합니다. 폰을 쥔 손이 보이며, 화면 하단부에 프롬프트에 없는 인물의 얼굴(눈과 코 부분)이 반사되어 뚜렷하게 보입니다.",
        "hard_violations": [
         "invented people (스마트폰 화면에 반사된 인물의 얼굴)",
         "leaked markers/diagrams/text (화면 상의 지시되지 않은 깨진 텍스트들)"
        ],
        "physics": "흙 묻은 손이 스마트폰의 오른쪽 측면과 하단을 감싸 쥐고 안정적으로 들고 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "스마트폰 화면을 더 크게 잡은 정확한 인서트 클로즈업이며, 중앙에 21:14:08과 배터리 17 percent가 선명하고 손의 지지도 자연스럽다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "시간과 배터리 수치는 맞지만 화면 비중이 작아 인서트 클로즈업이 약하고, 상단의 추가 문구와 불필요한 UI 글자가 ‘다른 읽을 수 있는 글자 금지’를 위반한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "스마트폰의 기능면인 화면이 카메라를 향하며, 손으로 보는 방향과 촬영 방향 모두 화면에 맞춰져 있다. 무기·시선·이동체는 없다.",
        "built_space": "배경에 갈라진 콘크리트, 여러 갈래로 뻗은 거대한 뿌리, 뒤쪽의 녹슨 철문 한 개가 보이며 참조 장소와 대체로 일치한다. 사람은 없고 손과 스마트폰만 전경에 있다. 반사는 화면 표면에서 약하게 나타나며 광학적으로 가능하다.",
        "entities": "실물 크기의 케이스 낀 스마트폰과 이를 잡은 더러운 손이 보인다. 화면에는 21:14:08과 17 percent가 표시되지만, 상단에 추가로 읽히는 작은 문구와 기타 UI가 있다. 얼굴이나 몸, 별도 인물은 없다.",
        "hard_violations": [
         "화면 상단의 추가로 읽히는 UI 문구가 나타나, 지정 문구 외에는 읽을 수 있는 글자를 두지 말라는 조건을 위반한다."
        ],
        "physics": "스마트폰은 아래와 오른쪽에서 손가락과 손바닥이 실제로 감싸 쥐고 있어 지지된다. 공중에 떠 있는 물체나 지지 없는 신체는 없다."
       },
       {
        "label": "B",
        "direction": "스마트폰 화면이 카메라를 거의 정면으로 향하고 있으며, 샷 텍스트가 관객에게 보여 주도록 지정한 시간과 배터리 정보가 직접 카메라를 향한다. 무기·시선·이동체는 없다.",
        "built_space": "배경에 갈라진 콘크리트, 거대한 뿌리와 잔해, 뒤쪽의 녹슨 철문 한 개가 보이며 참조 장소의 재질과 배치를 유지한다. 인물은 없고 손과 스마트폰만 전경에 있다. 화면의 미약한 반사와 먼지, 균열은 카메라 각도상 가능하다.",
        "entities": "긁히고 먼지 묻은 실제 스마트폰, 이를 잡은 더러운 손, 화면 중앙의 정확한 21:14:08 및 17 percent 표시가 보인다. 얼굴·몸·추가 인물·별도 소품은 없다.",
        "hard_violations": [],
        "physics": "스마트폰은 하단과 오른쪽을 감싼 손 및 화면 위에 닿은 엄지로 확실히 지지된다. 손목 아래는 프레임 밖이지만 정상적인 손-held 자세이며 떠 있는 물체는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "스마트폰 화면을 더 크게 잡은 정확한 인서트 클로즈업이며, 중앙에 21:14:08과 배터리 17 percent가 선명하고 손의 지지도 자연스럽다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "시간과 배터리 수치는 맞지만 화면 비중이 작아 인서트 클로즈업이 약하고, 상단의 추가 문구와 불필요한 UI 글자가 ‘다른 읽을 수 있는 글자 금지’를 위반한다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "스마트폰의 기능면인 화면이 카메라를 향하며, 손으로 보는 방향과 촬영 방향 모두 화면에 맞춰져 있다. 무기·시선·이동체는 없다.",
        "built_space": "배경에 갈라진 콘크리트, 여러 갈래로 뻗은 거대한 뿌리, 뒤쪽의 녹슨 철문 한 개가 보이며 참조 장소와 대체로 일치한다. 사람은 없고 손과 스마트폰만 전경에 있다. 반사는 화면 표면에서 약하게 나타나며 광학적으로 가능하다.",
        "entities": "실물 크기의 케이스 낀 스마트폰과 이를 잡은 더러운 손이 보인다. 화면에는 21:14:08과 17 percent가 표시되지만, 상단에 추가로 읽히는 작은 문구와 기타 UI가 있다. 얼굴이나 몸, 별도 인물은 없다.",
        "hard_violations": [
         "화면 상단의 추가로 읽히는 UI 문구가 나타나, 지정 문구 외에는 읽을 수 있는 글자를 두지 말라는 조건을 위반한다."
        ],
        "physics": "스마트폰은 아래와 오른쪽에서 손가락과 손바닥이 실제로 감싸 쥐고 있어 지지된다. 공중에 떠 있는 물체나 지지 없는 신체는 없다."
       },
       {
        "label": "A",
        "direction": "스마트폰 화면이 카메라를 거의 정면으로 향하고 있으며, 샷 텍스트가 관객에게 보여 주도록 지정한 시간과 배터리 정보가 직접 카메라를 향한다. 무기·시선·이동체는 없다.",
        "built_space": "배경에 갈라진 콘크리트, 거대한 뿌리와 잔해, 뒤쪽의 녹슨 철문 한 개가 보이며 참조 장소의 재질과 배치를 유지한다. 인물은 없고 손과 스마트폰만 전경에 있다. 화면의 미약한 반사와 먼지, 균열은 카메라 각도상 가능하다.",
        "entities": "긁히고 먼지 묻은 실제 스마트폰, 이를 잡은 더러운 손, 화면 중앙의 정확한 21:14:08 및 17 percent 표시가 보인다. 얼굴·몸·추가 인물·별도 소품은 없다.",
        "hard_violations": [],
        "physics": "스마트폰은 하단과 오른쪽을 감싼 손 및 화면 위에 닿은 엄지로 확실히 지지된다. 손목 아래는 프레임 밖이지만 정상적인 손-held 자세이며 떠 있는 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.067
   },
   "adjusted": {
    "A": 2.0,
    "B": 0.817
   },
   "violations": {
    "B": [
     "[gemini-pro] invented people (스마트폰 화면에 반사된 인물의 얼굴)",
     "[gemini-pro] leaked markers/diagrams/text (화면 상의 지시되지 않은 깨진 텍스트들)",
     "[gpt] 화면 상단의 추가로 읽히는 UI 문구가 나타나, 지정 문구 외에는 읽을 수 있는 글자를 두지 말라는 조건을 위반한다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 817
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "프롬프트가 요구한 '21:14:08' 화면을 정확하게 렌더링했으며, 이전 샷의 배경을 완벽히 유지한 채 규정대로 인물의 얼굴을 노출하지 않고 화면을 든 손만 자연스럽게 표현하여 지시사항을 완벽히 이행했습니다."
   },
   {
    "label": "B",
    "score": 817,
    "verdict_ko": "배경의 연속성은 훌륭하나, 프롬프트에서 엄격히 금지한 인물의 얼굴이 스마트폰 화면에 반사되어 나타났으며 요구하지 않은 정체불명의 텍스트까지 화면에 추가되어 핵심 규칙을 위반했습니다.  ★위반: [gemini-pro] invented people (스마트폰 화면에 반사된 인물의 얼굴) / [gemini-pro] leaked markers/diagrams/text (화면 상의 지시되지 않은 깨진 텍스트들) / [gpt] 화면 상단의 추가로 읽히는 UI 문구가 나타나, 지정 문구 외에는 읽을 수 있는 글자를 두지 말라는 조건을 위반한다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S4sh32_sel.png",
    "asset_id": "a2faf465-d725-471d-a231-af1e4970fa0d",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b299-5c6d-7dac-94fb-fc506be37544",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S4sh32"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S4sh33::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:27:38.018103+00:00",
  "fingerprint": "f51d28568004592d05398f8cc7aecabea318aefcfa84e152952c7a2fd1fcec3b",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S4sh33_sel.png",
  "source_sha256": "f58d3eeaa671102a0bf22ad42a7ed64ae3dbe8a35286ece6d1985e46074f0d56",
  "file": "S4sh33_cine.png",
  "staged_sha256": "0fdc8e0d3add95af77dd0c654e45dbf77615883894b3214a700e04fe34d5c780",
  "latency_ms": 18411
 },
 "S4sh36::signage": {
  "fp": "2dad11a92f2535da",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S4sh36": {
  "input_fingerprint": "63e87b1bd00ae352",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 갈라진 콘크리트와 찢겨진 철문 사이를 뱀처럼 휘감은 거대한 나무뿌리.\n\nLOCATION (lock): Inside the ruined mine tunnel at the breached blast-door area, where enormous roots coil through cracked concrete and torn steel. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: enormous tree root (waist-thick and winding through the damaged section) — Its long curved axis crosses the frame diagonally between the concrete and door; used as Primary focal line traced by the lateral camera movement; split concrete (cracked apart by the root) — The broken opening faces the camera obliquely, revealing the root passing through it; used as Frames one side of the root and establishes the scale of the damage; torn iron door (ripped open by the root) — Its damaged face is seen at an oblique upward angle beside the root; used as Forms the opposing edge of the environmental reveal.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Low-to-mid-key ambient light appropriate to the tunnel preserves restrained mineral blacks and weathered neutral tones.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A waist-high tree root coils through the cracked concrete and the iron door it has ripped apart. The smartphone remains active at 21:14:08 with 17 percent battery.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 갈라진 콘크리트와 찢겨진 철문 사이를 뱀처럼 휘감은 거대한 나무뿌리.\n\nLOCATION (lock): Inside the ruined mine tunnel at the breached blast-door area, where enormous roots coil through cracked concrete and torn steel. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: enormous tree root (waist-thick and winding through the damaged section) — Its long curved axis crosses the frame diagonally between the concrete and door; used as Primary focal line traced by the lateral camera movement; split concrete (cracked apart by the root) — The broken opening faces the camera obliquely, revealing the root passing through it; used as Frames one side of the root and establishes the scale of the damage; torn iron door (ripped open by the root) — Its damaged face is seen at an oblique upward angle beside the root; used as Forms the opposing edge of the environmental reveal.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Low-to-mid-key ambient light appropriate to the tunnel preserves restrained mineral blacks and weathered neutral tones.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A waist-high tree root coils through the cracked concrete and the iron door it has ripped apart. The smartphone remains active at 21:14:08 with 17 percent battery.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): night.\n\nSHOT TEXT (authoritative, Korean): 갈라진 콘크리트와 찢겨진 철문 사이를 뱀처럼 휘감은 거대한 나무뿌리.\n\nLOCATION (lock): Inside the ruined mine tunnel at the breached blast-door area, where enormous roots coil through cracked concrete and torn steel. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: enormous tree root (waist-thick and winding through the damaged section) — Its long curved axis crosses the frame diagonally between the concrete and door; used as Primary focal line traced by the lateral camera movement; split concrete (cracked apart by the root) — The broken opening faces the camera obliquely, revealing the root passing through it; used as Frames one side of the root and establishes the scale of the damage; torn iron door (ripped open by the root) — Its damaged face is seen at an oblique upward angle beside the root; used as Forms the opposing edge of the environmental reveal.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Low-to-mid-key ambient light appropriate to the tunnel preserves restrained mineral blacks and weathered neutral tones.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A waist-high tree root coils through the cracked concrete and the iron door it has ripped apart. The smartphone remains active at 21:14:08 with 17 percent battery.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라는 터널 내부의 무너진 벽과 철문을 향하고 있으며, 거대한 뿌리가 화면을 대각선으로 가로지름.",
    "built_space": "왼쪽에 갈라진 콘크리트 벽이 있고 오른쪽에 녹슨 철문이 위치하여 레퍼런스의 배경 구조를 따르고 있음.",
    "entities": "거대한 나무뿌리, 갈라진 콘크리트, 철문이 보임. 스마트폰은 화면에 나타나지 않음.",
    "hard_violations": [
     "[gemini-pro] 오른쪽 철문에 겹쳐져 있는 단절된 나무뿌리 조각 (물리적으로 불가능한 배치 및 클리핑 현상)"
    ],
    "physics": "왼쪽의 굵은 뿌리는 바닥과 벽에 지지되어 있으나, 오른쪽 철문 표면에 붙어 있는 잔뿌리는 주 뿌리와 완전히 단절되어 있으며 물리적으로 불가능하게 금속에 겹쳐져 있거나 허공에 떠 있음."
   },
   {
    "label": "B",
    "direction": "카메라는 터널 내부의 손상된 벽과 철문을 바라보며, 나무뿌리가 좌측 상단에서 우측 하단으로 화면을 대각선으로 가로지름.",
    "built_space": "왼쪽에 훼손된 콘크리트 벽면과 오른쪽에 크게 찢겨진 구멍이 있는 녹슨 철문이 배치되어 있으며, 레퍼런스의 공간적 특징과 일치함.",
    "entities": "거대한 나무뿌리, 갈라진 콘크리트, 찢겨진 철문이 정확히 묘사됨. Carried state에 언급된 스마트폰은 카메라 앵글 상 화면에 나타나지 않음.",
    "hard_violations": [],
    "physics": "거대한 나무뿌리는 콘크리트 벽의 파손된 틈에서 나와 바닥의 잔해들 위로 중력에 맞게 자연스럽게 늘어져 지지받고 있음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "와이드 샷 구도 내에서 갈라진 콘크리트, 찢겨진 철문, 그리고 그 사이를 가로지르는 거대한 나무뿌리의 물리적 형태와 질감을 레퍼런스에 맞춰 매우 사실적으로 구현했습니다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "지정된 배경 요소를 포함했으나 우측 철문에 주 뿌리와 단절된 채 허공에 떠 있거나 금속에 비정상적으로 겹쳐진 잔뿌리가 묘사되어 물리적 치명적 오류가 발생했습니다."
       }
      ],
      "readings": [
       {
        "label": "B",
        "direction": "카메라는 터널 내부의 손상된 벽과 철문을 바라보며, 나무뿌리가 좌측 상단에서 우측 하단으로 화면을 대각선으로 가로지름.",
        "built_space": "왼쪽에 훼손된 콘크리트 벽면과 오른쪽에 크게 찢겨진 구멍이 있는 녹슨 철문이 배치되어 있으며, 레퍼런스의 공간적 특징과 일치함.",
        "entities": "거대한 나무뿌리, 갈라진 콘크리트, 찢겨진 철문이 정확히 묘사됨. Carried state에 언급된 스마트폰은 카메라 앵글 상 화면에 나타나지 않음.",
        "hard_violations": [],
        "physics": "거대한 나무뿌리는 콘크리트 벽의 파손된 틈에서 나와 바닥의 잔해들 위로 중력에 맞게 자연스럽게 늘어져 지지받고 있음."
       },
       {
        "label": "A",
        "direction": "카메라는 터널 내부의 무너진 벽과 철문을 향하고 있으며, 거대한 뿌리가 화면을 대각선으로 가로지름.",
        "built_space": "왼쪽에 갈라진 콘크리트 벽이 있고 오른쪽에 녹슨 철문이 위치하여 레퍼런스의 배경 구조를 따르고 있음.",
        "entities": "거대한 나무뿌리, 갈라진 콘크리트, 철문이 보임. 스마트폰은 화면에 나타나지 않음.",
        "hard_violations": [
         "오른쪽 철문에 겹쳐져 있는 단절된 나무뿌리 조각 (물리적으로 불가능한 배치 및 클리핑 현상)"
        ],
        "physics": "왼쪽의 굵은 뿌리는 바닥과 벽에 지지되어 있으나, 오른쪽 철문 표면에 붙어 있는 잔뿌리는 주 뿌리와 완전히 단절되어 있으며 물리적으로 불가능하게 금속에 겹쳐져 있거나 허공에 떠 있음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "와이드 샷 구도 내에서 갈라진 콘크리트, 찢겨진 철문, 그리고 그 사이를 가로지르는 거대한 나무뿌리의 물리적 형태와 질감을 레퍼런스에 맞춰 매우 사실적으로 구현했습니다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "지정된 배경 요소를 포함했으나 우측 철문에 주 뿌리와 단절된 채 허공에 떠 있거나 금속에 비정상적으로 겹쳐진 잔뿌리가 묘사되어 물리적 치명적 오류가 발생했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "카메라는 터널 내부의 손상된 벽과 철문을 바라보며, 나무뿌리가 좌측 상단에서 우측 하단으로 화면을 대각선으로 가로지름.",
        "built_space": "왼쪽에 훼손된 콘크리트 벽면과 오른쪽에 크게 찢겨진 구멍이 있는 녹슨 철문이 배치되어 있으며, 레퍼런스의 공간적 특징과 일치함.",
        "entities": "거대한 나무뿌리, 갈라진 콘크리트, 찢겨진 철문이 정확히 묘사됨. Carried state에 언급된 스마트폰은 카메라 앵글 상 화면에 나타나지 않음.",
        "hard_violations": [],
        "physics": "거대한 나무뿌리는 콘크리트 벽의 파손된 틈에서 나와 바닥의 잔해들 위로 중력에 맞게 자연스럽게 늘어져 지지받고 있음."
       },
       {
        "label": "A",
        "direction": "카메라는 터널 내부의 무너진 벽과 철문을 향하고 있으며, 거대한 뿌리가 화면을 대각선으로 가로지름.",
        "built_space": "왼쪽에 갈라진 콘크리트 벽이 있고 오른쪽에 녹슨 철문이 위치하여 레퍼런스의 배경 구조를 따르고 있음.",
        "entities": "거대한 나무뿌리, 갈라진 콘크리트, 철문이 보임. 스마트폰은 화면에 나타나지 않음.",
        "hard_violations": [
         "오른쪽 철문에 겹쳐져 있는 단절된 나무뿌리 조각 (물리적으로 불가능한 배치 및 클리핑 현상)"
        ],
        "physics": "왼쪽의 굵은 뿌리는 바닥과 벽에 지지되어 있으나, 오른쪽 철문 표면에 붙어 있는 잔뿌리는 주 뿌리와 완전히 단절되어 있으며 물리적으로 불가능하게 금속에 겹쳐져 있거나 허공에 떠 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "넓은 구도와 대각선의 굵은 뿌리, 갈라진 콘크리트, 녹슨 철문은 맞지만 주 뿌리가 철문의 찢어진 부분을 직접 관통하거나 휘감는 관계가 약하다."
       },
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "허리 굵기의 뿌리가 갈라진 콘크리트에서 철문의 찢어진 구멍까지 대각선으로 이어져, 요구된 장소와 환경적 동작을 가장 충실하게 구현한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "사람의 시선이나 무기, 이동체는 없다. 가장 굵은 뿌리의 축은 좌측 상단 콘크리트 틈에서 우측 하단 잔해 쪽으로 향하지만, 철문의 찢어진 구멍을 직접 통과하지 않고 문 앞 바닥으로 내려간다.",
        "built_space": "좌측에 갈라져 벌어진 콘크리트 벽과 하나의 파손 개구부, 우측에 녹슨 철문 한 짝과 찢어진 구멍 하나가 보인다. 굵은 뿌리는 콘크리트 개구부를 통과하나 철문은 그 뒤에 분리되어 보여, 뿌리가 양쪽 구조물을 함께 찢고 관통했다는 배치가 충분히 성립하지 않는다. 반사면은 없다.",
        "entities": "사람·얼굴·신체는 없고 읽을 수 있는 글자도 없다. 거대한 실제 나무뿌리, 갈라진 콘크리트, 찢기고 녹슨 철문, 바닥 잔해가 모두 재질감 있게 표현되었다. 스마트폰은 보이지 않으며 이 환경 중심 와이드숏에서는 필수 가시 요소가 아니다.",
        "hard_violations": [],
        "physics": "굵은 뿌리는 좌측 콘크리트 틈에 박혀 있고 우측 하단의 잔해와 바닥에 닿아 지지된다. 가는 가지들도 벽의 틈과 바닥 방향으로 이어진다. 공중에 근거 없이 떠 있는 물체는 없다."
       },
       {
        "label": "B",
        "direction": "사람의 시선이나 무기, 이동체는 없다. 주 뿌리의 긴 축은 좌측 콘크리트 개구부에서 우측 하단으로 대각선을 이루고, 우측 상단에서 내려오는 굽은 뿌리 구간은 철문의 찢어진 구멍을 정확히 향해 관통한다.",
        "built_space": "좌측에 갈라진 콘크리트 벽과 하나의 큰 파손 개구부, 우측에 녹슨 철문 한 짝과 하나의 크게 찢어진 구멍이 보인다. 뿌리가 콘크리트와 철문 양쪽의 파손부를 실제로 통과하여 서로 마주 보는 손상 가장자리를 형성하며, 카메라는 이를 비스듬히 바라본다. 불가능한 반사는 없다.",
        "entities": "사람·얼굴·신체는 없고 읽을 수 있는 글자도 없다. 허리 굵기의 실제 나무뿌리, 갈라진 콘크리트, 찢어진 녹슨 철문과 광물성 잔해가 모두 확인되며, 이전 장면의 어둡고 중성적인 터널 재질과 조명도 유지된다. 스마트폰은 보이지 않으며 환경만 보여주는 프레이밍과 충돌하지 않는다.",
        "hard_violations": [],
        "physics": "좌측의 굵은 뿌리는 콘크리트 개구부에 끼어 있고 하단 잔해와 프레임 밖 지면으로 이어져 지지된다. 철문을 관통하는 우측 뿌리도 상단 프레임 밖의 고정된 연장부와 문의 파손 가장자리에 의해 지지되어 보인다. 떠 있는 물체나 지지 없는 구조는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "넓은 구도와 대각선의 굵은 뿌리, 갈라진 콘크리트, 녹슨 철문은 맞지만 주 뿌리가 철문의 찢어진 부분을 직접 관통하거나 휘감는 관계가 약하다."
       },
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "허리 굵기의 뿌리가 갈라진 콘크리트에서 철문의 찢어진 구멍까지 대각선으로 이어져, 요구된 장소와 환경적 동작을 가장 충실하게 구현한다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "사람의 시선이나 무기, 이동체는 없다. 가장 굵은 뿌리의 축은 좌측 상단 콘크리트 틈에서 우측 하단 잔해 쪽으로 향하지만, 철문의 찢어진 구멍을 직접 통과하지 않고 문 앞 바닥으로 내려간다.",
        "built_space": "좌측에 갈라져 벌어진 콘크리트 벽과 하나의 파손 개구부, 우측에 녹슨 철문 한 짝과 찢어진 구멍 하나가 보인다. 굵은 뿌리는 콘크리트 개구부를 통과하나 철문은 그 뒤에 분리되어 보여, 뿌리가 양쪽 구조물을 함께 찢고 관통했다는 배치가 충분히 성립하지 않는다. 반사면은 없다.",
        "entities": "사람·얼굴·신체는 없고 읽을 수 있는 글자도 없다. 거대한 실제 나무뿌리, 갈라진 콘크리트, 찢기고 녹슨 철문, 바닥 잔해가 모두 재질감 있게 표현되었다. 스마트폰은 보이지 않으며 이 환경 중심 와이드숏에서는 필수 가시 요소가 아니다.",
        "hard_violations": [],
        "physics": "굵은 뿌리는 좌측 콘크리트 틈에 박혀 있고 우측 하단의 잔해와 바닥에 닿아 지지된다. 가는 가지들도 벽의 틈과 바닥 방향으로 이어진다. 공중에 근거 없이 떠 있는 물체는 없다."
       },
       {
        "label": "A",
        "direction": "사람의 시선이나 무기, 이동체는 없다. 주 뿌리의 긴 축은 좌측 콘크리트 개구부에서 우측 하단으로 대각선을 이루고, 우측 상단에서 내려오는 굽은 뿌리 구간은 철문의 찢어진 구멍을 정확히 향해 관통한다.",
        "built_space": "좌측에 갈라진 콘크리트 벽과 하나의 큰 파손 개구부, 우측에 녹슨 철문 한 짝과 하나의 크게 찢어진 구멍이 보인다. 뿌리가 콘크리트와 철문 양쪽의 파손부를 실제로 통과하여 서로 마주 보는 손상 가장자리를 형성하며, 카메라는 이를 비스듬히 바라본다. 불가능한 반사는 없다.",
        "entities": "사람·얼굴·신체는 없고 읽을 수 있는 글자도 없다. 허리 굵기의 실제 나무뿌리, 갈라진 콘크리트, 찢어진 녹슨 철문과 광물성 잔해가 모두 확인되며, 이전 장면의 어둡고 중성적인 터널 재질과 조명도 유지된다. 스마트폰은 보이지 않으며 환경만 보여주는 프레이밍과 충돌하지 않는다.",
        "hard_violations": [],
        "physics": "좌측의 굵은 뿌리는 콘크리트 개구부에 끼어 있고 하단 잔해와 프레임 밖 지면으로 이어져 지지된다. 철문을 관통하는 우측 뿌리도 상단 프레임 밖의 고정된 연장부와 문의 파손 가장자리에 의해 지지되어 보인다. 떠 있는 물체나 지지 없는 구조는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.375,
    "B": 1.778
   },
   "adjusted": {
    "A": 1.125,
    "B": 1.778
   },
   "violations": {
    "A": [
     "[gemini-pro] 오른쪽 철문에 겹쳐져 있는 단절된 나무뿌리 조각 (물리적으로 불가능한 배치 및 클리핑 현상)"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "agreed": false
  },
  "totals": {
   "B": 1778,
   "A": 1125
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 1778,
    "verdict_ko": "와이드 샷 구도 내에서 갈라진 콘크리트, 찢겨진 철문, 그리고 그 사이를 가로지르는 거대한 나무뿌리의 물리적 형태와 질감을 레퍼런스에 맞춰 매우 사실적으로 구현했습니다."
   },
   {
    "label": "A",
    "score": 1125,
    "verdict_ko": "지정된 배경 요소를 포함했으나 우측 철문에 주 뿌리와 단절된 채 허공에 떠 있거나 금속에 비정상적으로 겹쳐진 잔뿌리가 묘사되어 물리적 치명적 오류가 발생했습니다.  ★위반: [gemini-pro] 오른쪽 철문에 겹쳐져 있는 단절된 나무뿌리 조각 (물리적으로 불가능한 배치 및 클리핑 현상)"
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S4sh33_sel.png",
    "asset_id": "8f790b9a-13df-44cb-b47b-11823ab47a1c",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b29c-c905-7921-ad8c-9d7240921e40",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S4sh33"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S4sh36::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:28:43.716747+00:00",
  "fingerprint": "e82243b4e4483344fd9ada168544cb6b442c436029f45b26d4b46c47ffc676e4",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S4sh36_sel.png",
  "source_sha256": "7b259ede56e5983a324eb7ead4b2af6b9f6dfef65f7ff541ce0472cd2fd3a27d",
  "file": "S4sh36_cine.png",
  "staged_sha256": "17256be44592ff26252b19d9c86e22e513c342c35b7bfcc4dd4d6290f237a91c",
  "latency_ms": 14928
 },
 "S5sh41::signage": {
  "fp": "8e3883ca441072fa",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S5sh41": {
  "input_fingerprint": "d5fefd44bc2b0bb2",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day, bright sunlight.\n\nSHOT TEXT (authoritative, Korean): 울창한 수관 위로 위태롭게 기울어진 채 솟아 있는 콘크리트 고층건물의 뼈대.\n\nLOCATION (lock): Outside at a forest overlook near the mine exit, looking across the canopy toward the tilted skeleton of a distant concrete high-rise. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nSTRUCTURE LOOK AUTHORITY: the attached STRUCTURE LOOK photograph is the identity of the fixed structure at this location — wherever that structure appears in the frame, its shape, proportions, openings, materials and colors are LOCKED to it. The LOCATION PHOTOGRAPH remains the authority for this shot's sub-space, surroundings, time of day and lighting. If the two conflict on the structure itself, the STRUCTURE LOOK photo wins; for everything else, the LOCATION PHOTOGRAPH wins.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: leaning high-rise skeleton in the upper-center of the frame, background; forest canopy in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: concrete high-rise skeleton (skeletal and leaning above the trees) — Its open structural face is visible obliquely, with the entire frame tilting diagonally against the canopy; used as Primary distant focal point held at the completion of the pan; dense forest canopy (extending across the valley); used as Provides the lower visual field and scale reference beneath the building.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Strong daylight renders the forest and distant concrete structure with grounded contrast and a restrained natural palette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Strong daylight falls on an endless forest and a river crossing the valley. Far above the canopy, the skeletal frame of a concrete high-rise rises at an angle, while large roots choke the mine exit.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day, bright sunlight.\n\nSHOT TEXT (authoritative, Korean): 울창한 수관 위로 위태롭게 기울어진 채 솟아 있는 콘크리트 고층건물의 뼈대.\n\nLOCATION (lock): Outside at a forest overlook near the mine exit, looking across the canopy toward the tilted skeleton of a distant concrete high-rise. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nSTRUCTURE LOOK AUTHORITY: the attached STRUCTURE LOOK photograph is the identity of the fixed structure at this location — wherever that structure appears in the frame, its shape, proportions, openings, materials and colors are LOCKED to it. The LOCATION PHOTOGRAPH remains the authority for this shot's sub-space, surroundings, time of day and lighting. If the two conflict on the structure itself, the STRUCTURE LOOK photo wins; for everything else, the LOCATION PHOTOGRAPH wins.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: leaning high-rise skeleton in the upper-center of the frame, background; forest canopy in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: concrete high-rise skeleton (skeletal and leaning above the trees) — Its open structural face is visible obliquely, with the entire frame tilting diagonally against the canopy; used as Primary distant focal point held at the completion of the pan; dense forest canopy (extending across the valley); used as Provides the lower visual field and scale reference beneath the building.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Strong daylight renders the forest and distant concrete structure with grounded contrast and a restrained natural palette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Strong daylight falls on an endless forest and a river crossing the valley. Far above the canopy, the skeletal frame of a concrete high-rise rises at an angle, while large roots choke the mine exit.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day, bright sunlight.\n\nSHOT TEXT (authoritative, Korean): 울창한 수관 위로 위태롭게 기울어진 채 솟아 있는 콘크리트 고층건물의 뼈대.\n\nLOCATION (lock): Outside at a forest overlook near the mine exit, looking across the canopy toward the tilted skeleton of a distant concrete high-rise. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nSTRUCTURE LOOK AUTHORITY: the attached STRUCTURE LOOK photograph is the identity of the fixed structure at this location — wherever that structure appears in the frame, its shape, proportions, openings, materials and colors are LOCKED to it. The LOCATION PHOTOGRAPH remains the authority for this shot's sub-space, surroundings, time of day and lighting. If the two conflict on the structure itself, the STRUCTURE LOOK photo wins; for everything else, the LOCATION PHOTOGRAPH wins.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: leaning high-rise skeleton in the upper-center of the frame, background; forest canopy in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: concrete high-rise skeleton (skeletal and leaning above the trees) — Its open structural face is visible obliquely, with the entire frame tilting diagonally against the canopy; used as Primary distant focal point held at the completion of the pan; dense forest canopy (extending across the valley); used as Provides the lower visual field and scale reference beneath the building.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Strong daylight renders the forest and distant concrete structure with grounded contrast and a restrained natural palette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Strong daylight falls on an endless forest and a river crossing the valley. Far above the canopy, the skeletal frame of a concrete high-rise rises at an angle, while large roots choke the mine exit.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "멀리 있는 콘크리트 고층건물 뼈대가 왼쪽으로 비스듬히 기울어져 있음.",
    "built_space": "왼쪽 근경에 나무뿌리로 덮인 광산 출구가 있고, 중앙에는 이끼 낀 기차 선로와 콘크리트 바닥이 있으며, 오른쪽에는 녹슨 금속 기둥과 나무가 배치되어 있음.",
    "entities": "지시된 사각형의 콘크리트 고층건물 뼈대와 숲 수관, 뿌리가 얽힌 광산 출구 등 프롬프트의 모든 요소가 명확히 구현됨.",
    "hard_violations": [],
    "physics": "기울어진 건물은 먼 산비탈에 단단히 박혀 지탱되고 있으며, 근경의 구조물과 뿌리, 선로가 물리적으로 자연스럽게 지면에 놓여 있음."
   },
   {
    "label": "B",
    "direction": "멀리 있는 금속 철탑 구조물이 오른쪽으로 비스듬히 기울어져 있음.",
    "built_space": "왼쪽 근경에 나무뿌리로 덮인 광산 출구가 있고, 중앙에 이끼 낀 기차 선로가 깔려 있으며 우측은 숲으로 덮여 있음.",
    "entities": "광산 출구와 숲 수관은 묘사되었으나, 요구된 '콘크리트 고층건물'이 아닌 금속 골조 철탑이 생성됨.",
    "hard_violations": [],
    "physics": "기울어진 철탑은 산비탈에 지탱되고 있고, 선로와 흙이 물리적 무리 없이 배치되어 있음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "STRUCTURE LOOK 레퍼런스와 프롬프트의 '콘크리트 고층건물' 지시를 충실히 반영하였으며, 주변 환경도 LOCATION 레퍼런스의 요소와 텍스트의 요구사항을 잘 조합하여 구성했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "STRUCTURE LOOK 레퍼런스와 프롬프트의 '콘크리트 고층건물' 지시를 무시하고 LOCATION 레퍼런스에 있는 금속 골조 철탑을 그대로 생성하여 지시사항을 크게 위반했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "멀리 있는 콘크리트 고층건물 뼈대가 왼쪽으로 비스듬히 기울어져 있음.",
        "built_space": "왼쪽 근경에 나무뿌리로 덮인 광산 출구가 있고, 중앙에는 이끼 낀 기차 선로와 콘크리트 바닥이 있으며, 오른쪽에는 녹슨 금속 기둥과 나무가 배치되어 있음.",
        "entities": "지시된 사각형의 콘크리트 고층건물 뼈대와 숲 수관, 뿌리가 얽힌 광산 출구 등 프롬프트의 모든 요소가 명확히 구현됨.",
        "hard_violations": [],
        "physics": "기울어진 건물은 먼 산비탈에 단단히 박혀 지탱되고 있으며, 근경의 구조물과 뿌리, 선로가 물리적으로 자연스럽게 지면에 놓여 있음."
       },
       {
        "label": "B",
        "direction": "멀리 있는 금속 철탑 구조물이 오른쪽으로 비스듬히 기울어져 있음.",
        "built_space": "왼쪽 근경에 나무뿌리로 덮인 광산 출구가 있고, 중앙에 이끼 낀 기차 선로가 깔려 있으며 우측은 숲으로 덮여 있음.",
        "entities": "광산 출구와 숲 수관은 묘사되었으나, 요구된 '콘크리트 고층건물'이 아닌 금속 골조 철탑이 생성됨.",
        "hard_violations": [],
        "physics": "기울어진 철탑은 산비탈에 지탱되고 있고, 선로와 흙이 물리적 무리 없이 배치되어 있음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "STRUCTURE LOOK 레퍼런스와 프롬프트의 '콘크리트 고층건물' 지시를 충실히 반영하였으며, 주변 환경도 LOCATION 레퍼런스의 요소와 텍스트의 요구사항을 잘 조합하여 구성했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "STRUCTURE LOOK 레퍼런스와 프롬프트의 '콘크리트 고층건물' 지시를 무시하고 LOCATION 레퍼런스에 있는 금속 골조 철탑을 그대로 생성하여 지시사항을 크게 위반했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "멀리 있는 콘크리트 고층건물 뼈대가 왼쪽으로 비스듬히 기울어져 있음.",
        "built_space": "왼쪽 근경에 나무뿌리로 덮인 광산 출구가 있고, 중앙에는 이끼 낀 기차 선로와 콘크리트 바닥이 있으며, 오른쪽에는 녹슨 금속 기둥과 나무가 배치되어 있음.",
        "entities": "지시된 사각형의 콘크리트 고층건물 뼈대와 숲 수관, 뿌리가 얽힌 광산 출구 등 프롬프트의 모든 요소가 명확히 구현됨.",
        "hard_violations": [],
        "physics": "기울어진 건물은 먼 산비탈에 단단히 박혀 지탱되고 있으며, 근경의 구조물과 뿌리, 선로가 물리적으로 자연스럽게 지면에 놓여 있음."
       },
       {
        "label": "B",
        "direction": "멀리 있는 금속 철탑 구조물이 오른쪽으로 비스듬히 기울어져 있음.",
        "built_space": "왼쪽 근경에 나무뿌리로 덮인 광산 출구가 있고, 중앙에 이끼 낀 기차 선로가 깔려 있으며 우측은 숲으로 덮여 있음.",
        "entities": "광산 출구와 숲 수관은 묘사되었으나, 요구된 '콘크리트 고층건물'이 아닌 금속 골조 철탑이 생성됨.",
        "hard_violations": [],
        "physics": "기울어진 철탑은 산비탈에 지탱되고 있고, 선로와 흙이 물리적 무리 없이 배치되어 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "상단 중앙의 기울어진 콘크리트 격자형 고층 골조, 숲 수관과 강, 뿌리에 잠식된 광산 출구를 정확한 와이드 숏으로 구현해 핵심 구조 정체성과 배치가 가장 충실하다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "와이드 구도와 숲·강·광산 출구 배치는 맞지만, 핵심 건물이 잠긴 콘크리트 격자 골조가 아니라 가느다란 금속 철탑처럼 보여 구조물 정체성이 크게 어긋난다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "사람, 시선, 무기, 지시 물체는 없다. 고층 구조물은 바닥이 오른쪽 아래에 있고 꼭대기가 왼쪽 위로 기울어 수관 위로 대각선으로 솟으며, 강은 계곡 중앙에서 화면 아래쪽으로 굽어 흐른다.",
        "built_space": "왼쪽 아래에 뿌리가 덮인 광산 출구 1개, 전경에 나란한 폐광 레일 2개, 상단 중앙에 먼 고층 골조 1개가 보인다. 사람은 없고 반사면도 없어 광학적으로 불가능한 반사는 없다. 다만 고층 골조가 콘크리트 건물보다는 철제 탑 구조로 표현됐다.",
        "entities": "울창한 수관, 계곡의 강, 뿌리에 잠식된 광산 출구, 기울어진 원거리 골조가 모두 있으며 사람과 읽을 수 있는 글자는 없다. 그러나 잠긴 구조물의 재료·개구부·비례는 참조의 콘크리트 격자형 고층 골조와 일치하지 않고, 케이블과 가는 철재가 많은 산업용 철탑에 가깝다.",
        "hard_violations": [],
        "physics": "기울어진 골조의 하단은 숲이 덮인 암반 능선에 이어져 지지점이 보이며 공중에 떠 있지 않다. 광산 위 뿌리는 토사와 암반에 붙어 있고 레일은 지면에 놓여 있다."
       },
       {
        "label": "B",
        "direction": "사람, 시선, 무기, 지시 물체는 없다. 고층 골조는 바닥이 오른쪽 아래, 꼭대기가 왼쪽 위를 향하도록 위태롭게 기울어져 있으며 숲 수관 위 상단 중앙에 솟아 있다. 강은 골조 아래 계곡을 따라 화면 아래쪽으로 굽어 흐른다.",
        "built_space": "왼쪽 아래에 뿌리가 얽힌 광산 출구 1개, 전경에 폐광 레일 한 쌍, 상단 중앙에 기울어진 고층 골조 1개가 있다. 오른쪽의 큰 나무뿌리가 레일과 출구 주변을 잠식한다. 사람이나 불가능한 반사는 없으며 각 요소의 원근 배치도 작동한다.",
        "entities": "콘크리트 격자와 반복된 직사각형 개구부를 지닌 기울어진 고층 골조, 울창한 수관, 계곡의 강, 광산 출구와 큰 뿌리가 모두 확인된다. 사람·얼굴·추가 인물 및 읽을 수 있는 글자가 없고, 구조물의 재료와 형태가 STRUCTURE LOOK 참조에 가깝다.",
        "hard_violations": [],
        "physics": "고층 골조 하단이 숲이 덮인 암반 능선에 박혀 있어 기울어진 몸체를 받치는 기반이 보이고, 요구된 불안정한 기울기를 현실적인 정지 상태로 표현한다. 뿌리는 나무와 지면에 연결되고 레일은 콘크리트·토사 위에 놓여 있어 떠 있는 물체가 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "상단 중앙의 기울어진 콘크리트 격자형 고층 골조, 숲 수관과 강, 뿌리에 잠식된 광산 출구를 정확한 와이드 숏으로 구현해 핵심 구조 정체성과 배치가 가장 충실하다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "와이드 구도와 숲·강·광산 출구 배치는 맞지만, 핵심 건물이 잠긴 콘크리트 격자 골조가 아니라 가느다란 금속 철탑처럼 보여 구조물 정체성이 크게 어긋난다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "사람, 시선, 무기, 지시 물체는 없다. 고층 구조물은 바닥이 오른쪽 아래에 있고 꼭대기가 왼쪽 위로 기울어 수관 위로 대각선으로 솟으며, 강은 계곡 중앙에서 화면 아래쪽으로 굽어 흐른다.",
        "built_space": "왼쪽 아래에 뿌리가 덮인 광산 출구 1개, 전경에 나란한 폐광 레일 2개, 상단 중앙에 먼 고층 골조 1개가 보인다. 사람은 없고 반사면도 없어 광학적으로 불가능한 반사는 없다. 다만 고층 골조가 콘크리트 건물보다는 철제 탑 구조로 표현됐다.",
        "entities": "울창한 수관, 계곡의 강, 뿌리에 잠식된 광산 출구, 기울어진 원거리 골조가 모두 있으며 사람과 읽을 수 있는 글자는 없다. 그러나 잠긴 구조물의 재료·개구부·비례는 참조의 콘크리트 격자형 고층 골조와 일치하지 않고, 케이블과 가는 철재가 많은 산업용 철탑에 가깝다.",
        "hard_violations": [],
        "physics": "기울어진 골조의 하단은 숲이 덮인 암반 능선에 이어져 지지점이 보이며 공중에 떠 있지 않다. 광산 위 뿌리는 토사와 암반에 붙어 있고 레일은 지면에 놓여 있다."
       },
       {
        "label": "A",
        "direction": "사람, 시선, 무기, 지시 물체는 없다. 고층 골조는 바닥이 오른쪽 아래, 꼭대기가 왼쪽 위를 향하도록 위태롭게 기울어져 있으며 숲 수관 위 상단 중앙에 솟아 있다. 강은 골조 아래 계곡을 따라 화면 아래쪽으로 굽어 흐른다.",
        "built_space": "왼쪽 아래에 뿌리가 얽힌 광산 출구 1개, 전경에 폐광 레일 한 쌍, 상단 중앙에 기울어진 고층 골조 1개가 있다. 오른쪽의 큰 나무뿌리가 레일과 출구 주변을 잠식한다. 사람이나 불가능한 반사는 없으며 각 요소의 원근 배치도 작동한다.",
        "entities": "콘크리트 격자와 반복된 직사각형 개구부를 지닌 기울어진 고층 골조, 울창한 수관, 계곡의 강, 광산 출구와 큰 뿌리가 모두 확인된다. 사람·얼굴·추가 인물 및 읽을 수 있는 글자가 없고, 구조물의 재료와 형태가 STRUCTURE LOOK 참조에 가깝다.",
        "hard_violations": [],
        "physics": "고층 골조 하단이 숲이 덮인 암반 능선에 박혀 있어 기울어진 몸체를 받치는 기반이 보이고, 요구된 불안정한 기울기를 현실적인 정지 상태로 표현한다. 뿌리는 나무와 지면에 연결되고 레일은 콘크리트·토사 위에 놓여 있어 떠 있는 물체가 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.111
   },
   "adjusted": {
    "A": 2.0,
    "B": 1.111
   },
   "violations": {},
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 1111
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "STRUCTURE LOOK 레퍼런스와 프롬프트의 '콘크리트 고층건물' 지시를 충실히 반영하였으며, 주변 환경도 LOCATION 레퍼런스의 요소와 텍스트의 요구사항을 잘 조합하여 구성했습니다."
   },
   {
    "label": "B",
    "score": 1111,
    "verdict_ko": "STRUCTURE LOOK 레퍼런스와 프롬프트의 '콘크리트 고층건물' 지시를 무시하고 LOCATION 레퍼런스에 있는 금속 골조 철탑을 그대로 생성하여 지시사항을 크게 위반했습니다."
   }
  ],
  "refs": [
   {
    "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its spatial layout, surroundings, fixed features, time of day and lighting mood are spatial truth; stage the moment inside this place. If a STRUCTURE LOOK photograph is also attached, that photo wins for the fixed structure itself — this photograph wins for everything around it. Never copy its camera framing.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/episodes/c804efc3-0697-4c22-98b4-6992a70c2b20/images/background_chain/L03B01.png",
    "asset_id": "5e4f2e93-5dee-42da-aed6-df71cc3d14c1",
    "role": "location_plate"
   },
   {
    "label": "STRUCTURE LOOK — the confirmed photograph of the fixed structure at this location: wherever the structure appears in the frame, its shape, proportions, materials, colors and openings are LOCKED to this photo. Never copy its camera framing, time of day or lighting — the shot text and the LOCATION PHOTOGRAPH are the authorities for those.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/background_chain/seed_bg_mine_complex_sel.png",
    "asset_id": "3f107fbd-85c0-4e79-84fd-608eeb3a7598",
    "role": "structure_seed_look"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2a0-ac6e-75ff-b90d-9794738a3389",
  "ref_mode": "플레이트+seed만 (배경 전용)",
  "share_plan": {
   "ref_plan": "background"
  },
  "lane_policy": "ab_select_bypass:bg_only"
 },
 "S5sh41::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:29:59.127439+00:00",
  "fingerprint": "575858adefb0ed9385002eaed9c6c510a393fa7e164d6c11d0e40865650077fb",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S5sh41_sel.png",
  "source_sha256": "0813abdee429e3e12cb83f36174d2e747798cca6235be285cd1f95b84878fd16",
  "file": "S5sh41_cine.png",
  "staged_sha256": "491b5e0d0e56beefa258fa6e33d363421ac9f8c2f02ecd46dda5e1ce7ab381cc",
  "latency_ms": 12495
 },
 "S5sh43::signage": {
  "fp": "d8c2e0791a2d710d",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S5sh43": {
  "input_fingerprint": "4ec6105392ba6989",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day, bright sunlight.\n\nSHOT TEXT (authoritative, Korean): 믿을 수 없다는 듯 폐허가 된 숲을 멍하니 응시하는 '토니(앤서니 로저스)'의 뒷모습.\n\nLOCATION (lock): Outside in the open area immediately beyond the mine exit, overlooking an immense forest and river valley in strong daylight. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward expansive forest and valley; expansive forest and valley in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: expansive forest (stretching across the former developed landscape); used as Large background field that reduces Tony to an isolated human scale; river across the valley (visible crossing the valley) — Its course runs laterally through the distant landscape; used as Adds depth and defines the valley beyond Tony; leaning high-rise skeleton (visible above the distant trees) — Its tilted structural side is visible beyond the canopy; used as A distant anomaly held within Tony's sightline.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Strong daylight keeps Tony clearly separated against the expansive forest while retaining restrained greens and weathered neutrals.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the dense daylight forest canopy, broad valley wilderness, and the single tilted concrete high-rise skeleton rising above the trees. Exclude highways, active industrial buildings, additional intact towers, and any mine-interior elements.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The former highway and industrial park are absent, replaced by an immense forest and a river-filled valley. The tilted high-rise skeleton remains visible beyond the canopy in strong daylight. 토니(앤서니 로저스): He stands facing the vast forest with his smartphone raised, staring at the landscape in disbelief. His rescue harness and chest equipment remain on him.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day, bright sunlight.\n\nSHOT TEXT (authoritative, Korean): 믿을 수 없다는 듯 폐허가 된 숲을 멍하니 응시하는 '토니(앤서니 로저스)'의 뒷모습.\n\nLOCATION (lock): Outside in the open area immediately beyond the mine exit, overlooking an immense forest and river valley in strong daylight. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward expansive forest and valley; expansive forest and valley in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: expansive forest (stretching across the former developed landscape); used as Large background field that reduces Tony to an isolated human scale; river across the valley (visible crossing the valley) — Its course runs laterally through the distant landscape; used as Adds depth and defines the valley beyond Tony; leaning high-rise skeleton (visible above the distant trees) — Its tilted structural side is visible beyond the canopy; used as A distant anomaly held within Tony's sightline.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Strong daylight keeps Tony clearly separated against the expansive forest while retaining restrained greens and weathered neutrals.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the dense daylight forest canopy, broad valley wilderness, and the single tilted concrete high-rise skeleton rising above the trees. Exclude highways, active industrial buildings, additional intact towers, and any mine-interior elements.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The former highway and industrial park are absent, replaced by an immense forest and a river-filled valley. The tilted high-rise skeleton remains visible beyond the canopy in strong daylight. 토니(앤서니 로저스): He stands facing the vast forest with his smartphone raised, staring at the landscape in disbelief. His rescue harness and chest equipment remain on him.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day, bright sunlight.\n\nSHOT TEXT (authoritative, Korean): 믿을 수 없다는 듯 폐허가 된 숲을 멍하니 응시하는 '토니(앤서니 로저스)'의 뒷모습.\n\nLOCATION (lock): Outside in the open area immediately beyond the mine exit, overlooking an immense forest and river valley in strong daylight. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward expansive forest and valley; expansive forest and valley in the middle-center of the frame, background.\n- KEY BACKGROUND ELEMENTS: expansive forest (stretching across the former developed landscape); used as Large background field that reduces Tony to an isolated human scale; river across the valley (visible crossing the valley) — Its course runs laterally through the distant landscape; used as Adds depth and defines the valley beyond Tony; leaning high-rise skeleton (visible above the distant trees) — Its tilted structural side is visible beyond the canopy; used as A distant anomaly held within Tony's sightline.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Strong daylight keeps Tony clearly separated against the expansive forest while retaining restrained greens and weathered neutrals.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the dense daylight forest canopy, broad valley wilderness, and the single tilted concrete high-rise skeleton rising above the trees. Exclude highways, active industrial buildings, additional intact towers, and any mine-interior elements.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The former highway and industrial park are absent, replaced by an immense forest and a river-filled valley. The tilted high-rise skeleton remains visible beyond the canopy in strong daylight. 토니(앤서니 로저스): He stands facing the vast forest with his smartphone raised, staring at the landscape in disbelief. His rescue harness and chest equipment remain on him.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "토니는 프레임 중앙의 넓은 숲과 계곡을 향해 똑바로 서서 시선을 던지고 있으며, 오른손에 든 스마트폰 역시 앞쪽 풍경을 향하고 있습니다.",
    "built_space": "동굴 입구, 끊어진 철로, 식물 뿌리, 멀리 기울어진 고층 건물 등 제공된 배경 기준 이미지의 모든 요소가 동일한 위치와 크기로 완벽하게 일치하게 배치되어 있으며, 토니는 동굴 밖 콘크리트 구조물 위에 서 있습니다.",
    "entities": "기준 이미지의 더러워진 전술복, 하네스, 밧줄을 착용한 토니의 모습이 보이며, 헬멧 없이 자연스러운 머리카락을 드러내고 있습니다. 손에는 스마트폰이 들려 있습니다.",
    "hard_violations": [],
    "physics": "토니의 두 발은 콘크리트 바닥을 안정적으로 딛고 서 있으며, 오른손은 스마트폰을 공중에 들고 있는 상태로 자연스러운 지지 상태를 보여줍니다."
   },
   {
    "label": "B",
    "direction": "토니의 몸이 약간 오른쪽으로 틀어져 있으며, 시선과 들고 있는 스마트폰이 정면의 풍경보다는 다소 측면이나 손 안의 기기를 향해 있습니다.",
    "built_space": "제공된 배경 기준 이미지의 폐광 입구, 철로, 멀리 있는 기울어진 건물과 숲의 형태가 모두 올바른 위치에 완벽하게 일치하여 묘사되었습니다.",
    "entities": "기준 이미지에 명시된 토니의 복장, 하네스, 가슴 장비가 잘 표현되었고, 헬멧을 벗은 머리 스타일이 드러나며 스마트폰을 들고 있습니다.",
    "hard_violations": [],
    "physics": "두 발로 바닥을 단단히 디디고 서 있는 자세이며, 손으로 기기를 안정적으로 쥐고 있어 물리적으로 어색함이 없습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "프롬프트가 요구한 완벽한 뒷모습과 넓은 숲을 응시하며 스마트폰을 들어 올린 자세를 정확하게 구현했으며, 배경 기준 이미지의 디테일도 완벽하게 유지했습니다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "배경과 인물의 외형은 훌륭하게 구현되었으나, 완전한 뒷모습이라기보다 측면에 가까우며 시선이 풍경보다는 스마트폰을 향하고 있어 지시사항과 다소 차이가 있습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니는 프레임 중앙의 넓은 숲과 계곡을 향해 똑바로 서서 시선을 던지고 있으며, 오른손에 든 스마트폰 역시 앞쪽 풍경을 향하고 있습니다.",
        "built_space": "동굴 입구, 끊어진 철로, 식물 뿌리, 멀리 기울어진 고층 건물 등 제공된 배경 기준 이미지의 모든 요소가 동일한 위치와 크기로 완벽하게 일치하게 배치되어 있으며, 토니는 동굴 밖 콘크리트 구조물 위에 서 있습니다.",
        "entities": "기준 이미지의 더러워진 전술복, 하네스, 밧줄을 착용한 토니의 모습이 보이며, 헬멧 없이 자연스러운 머리카락을 드러내고 있습니다. 손에는 스마트폰이 들려 있습니다.",
        "hard_violations": [],
        "physics": "토니의 두 발은 콘크리트 바닥을 안정적으로 딛고 서 있으며, 오른손은 스마트폰을 공중에 들고 있는 상태로 자연스러운 지지 상태를 보여줍니다."
       },
       {
        "label": "B",
        "direction": "토니의 몸이 약간 오른쪽으로 틀어져 있으며, 시선과 들고 있는 스마트폰이 정면의 풍경보다는 다소 측면이나 손 안의 기기를 향해 있습니다.",
        "built_space": "제공된 배경 기준 이미지의 폐광 입구, 철로, 멀리 있는 기울어진 건물과 숲의 형태가 모두 올바른 위치에 완벽하게 일치하여 묘사되었습니다.",
        "entities": "기준 이미지에 명시된 토니의 복장, 하네스, 가슴 장비가 잘 표현되었고, 헬멧을 벗은 머리 스타일이 드러나며 스마트폰을 들고 있습니다.",
        "hard_violations": [],
        "physics": "두 발로 바닥을 단단히 디디고 서 있는 자세이며, 손으로 기기를 안정적으로 쥐고 있어 물리적으로 어색함이 없습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "프롬프트가 요구한 완벽한 뒷모습과 넓은 숲을 응시하며 스마트폰을 들어 올린 자세를 정확하게 구현했으며, 배경 기준 이미지의 디테일도 완벽하게 유지했습니다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "배경과 인물의 외형은 훌륭하게 구현되었으나, 완전한 뒷모습이라기보다 측면에 가까우며 시선이 풍경보다는 스마트폰을 향하고 있어 지시사항과 다소 차이가 있습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "토니는 프레임 중앙의 넓은 숲과 계곡을 향해 똑바로 서서 시선을 던지고 있으며, 오른손에 든 스마트폰 역시 앞쪽 풍경을 향하고 있습니다.",
        "built_space": "동굴 입구, 끊어진 철로, 식물 뿌리, 멀리 기울어진 고층 건물 등 제공된 배경 기준 이미지의 모든 요소가 동일한 위치와 크기로 완벽하게 일치하게 배치되어 있으며, 토니는 동굴 밖 콘크리트 구조물 위에 서 있습니다.",
        "entities": "기준 이미지의 더러워진 전술복, 하네스, 밧줄을 착용한 토니의 모습이 보이며, 헬멧 없이 자연스러운 머리카락을 드러내고 있습니다. 손에는 스마트폰이 들려 있습니다.",
        "hard_violations": [],
        "physics": "토니의 두 발은 콘크리트 바닥을 안정적으로 딛고 서 있으며, 오른손은 스마트폰을 공중에 들고 있는 상태로 자연스러운 지지 상태를 보여줍니다."
       },
       {
        "label": "B",
        "direction": "토니의 몸이 약간 오른쪽으로 틀어져 있으며, 시선과 들고 있는 스마트폰이 정면의 풍경보다는 다소 측면이나 손 안의 기기를 향해 있습니다.",
        "built_space": "제공된 배경 기준 이미지의 폐광 입구, 철로, 멀리 있는 기울어진 건물과 숲의 형태가 모두 올바른 위치에 완벽하게 일치하여 묘사되었습니다.",
        "entities": "기준 이미지에 명시된 토니의 복장, 하네스, 가슴 장비가 잘 표현되었고, 헬멧을 벗은 머리 스타일이 드러나며 스마트폰을 들고 있습니다.",
        "hard_violations": [],
        "physics": "두 발로 바닥을 단단히 디디고 서 있는 자세이며, 손으로 기기를 안정적으로 쥐고 있어 물리적으로 어색함이 없습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "토니를 중간 왼쪽의 와이드 숏으로 배치하고 스마트폰을 든 채 숲·강·기울어진 고층 골조를 바라보게 해 핵심 행동과 장소를 가장 정확히 구현했다."
       },
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "장소와 와이드 구도는 잘 맞지만 토니의 머리와 시선이 풍경보다 낮게 든 스마트폰으로 향해 ‘폐허가 된 숲을 멍하니 응시’하는 결정적 순간이 약하다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 몸은 계곡과 기울어진 고층 골조 쪽을 향하지만, 고개와 시선은 아래쪽에 든 스마트폰을 향한 것으로 보인다. 따라서 시선이 숲과 계곡에 명확히 닿지 않는다. 스마트폰은 세로로 들려 있고 기능 면은 토니 쪽을 향한 것으로 보인다.",
        "built_space": "광산 출구 바로 바깥의 열린 콘크리트·흙 지대이며, 왼쪽에 광산 입구 1개, 전경에 낡은 레일 2줄, 오른쪽에 큰 나무와 노출된 뿌리가 보인다. 토니는 출구 앞 콘크리트 가장자리에 서 있다. 배경에는 숲으로 덮인 계곡, 횡방향으로 굽이치는 강 1개, 기울어진 고층 골조 1개가 있으며 추가 타워나 산업시설은 없다.",
        "entities": "등을 보이는 인물은 자연색의 짧은 갈색 머리를 지닌 30대 중반 미국인 남성으로 읽히며 토니의 체격과 작업복에 부합한다. 구조용 하네스, 로프와 장비 주머니가 유지되고 스마트폰도 있다. 다른 사람이나 불필요한 물체는 추가되지 않았다.",
        "hard_violations": [],
        "physics": "토니의 두 발이 콘크리트 지면에 닿아 체중을 지지하며, 오른손이 스마트폰을 실제로 쥐고 있다. 장비와 로프는 하네스와 허리에 부착되어 있어 떠 있는 물체가 없고 자세도 물리적으로 가능하다."
       },
       {
        "label": "B",
        "direction": "토니의 몸과 고개가 정면의 광대한 숲·강 계곡 및 기울어진 고층 골조를 향한다. 오른손으로 스마트폰을 눈높이 가까이 올렸지만 머리는 풍경 쪽을 향해 있어, 스마트폰을 든 채 믿기지 않는 풍경을 응시하는 관계가 성립한다. 휴대전화 화면은 토니의 눈 쪽이고 카메라는 뒷면을 보는 배치다.",
        "built_space": "광산 출구 바로 밖의 열린 지대이며 왼쪽에 광산 입구 1개, 전경에 낡은 레일 2줄과 콘크리트 바닥, 오른쪽에 큰 나무와 노출 뿌리가 있다. 토니는 출구 너머의 평평한 콘크리트 가장자리에 중간 왼쪽으로 서 있다. 배경에는 광대한 숲, 계곡을 가로질러 굽이치는 강 1개, 수관 위로 솟은 기울어진 고층 골조 1개만 보여 장소 고정 사항과 맞는다.",
        "entities": "유일한 인물은 뒤에서 보이는 자연색 짧은 갈색 머리의 30대 중반 미국인 남성 토니이며, 참조와 맞는 체격·오염된 작업복·구조 하네스·로프·허리 장비를 착용했다. 오른손에는 스마트폰이 있다. 다른 인물, 온전한 추가 타워, 산업시설이나 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "토니는 두 발을 콘크리트 바닥에 안정적으로 딛고 있으며 약간 벌어진 다리가 몸을 지지한다. 오른손이 들어 올린 스마트폰을 확실히 잡고 있고, 로프와 장비는 하네스 및 벨트에 매달려 있다. 지지 없이 떠 있는 신체나 물체는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "토니를 중간 왼쪽의 와이드 숏으로 배치하고 스마트폰을 든 채 숲·강·기울어진 고층 골조를 바라보게 해 핵심 행동과 장소를 가장 정확히 구현했다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "장소와 와이드 구도는 잘 맞지만 토니의 머리와 시선이 풍경보다 낮게 든 스마트폰으로 향해 ‘폐허가 된 숲을 멍하니 응시’하는 결정적 순간이 약하다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "토니의 몸은 계곡과 기울어진 고층 골조 쪽을 향하지만, 고개와 시선은 아래쪽에 든 스마트폰을 향한 것으로 보인다. 따라서 시선이 숲과 계곡에 명확히 닿지 않는다. 스마트폰은 세로로 들려 있고 기능 면은 토니 쪽을 향한 것으로 보인다.",
        "built_space": "광산 출구 바로 바깥의 열린 콘크리트·흙 지대이며, 왼쪽에 광산 입구 1개, 전경에 낡은 레일 2줄, 오른쪽에 큰 나무와 노출된 뿌리가 보인다. 토니는 출구 앞 콘크리트 가장자리에 서 있다. 배경에는 숲으로 덮인 계곡, 횡방향으로 굽이치는 강 1개, 기울어진 고층 골조 1개가 있으며 추가 타워나 산업시설은 없다.",
        "entities": "등을 보이는 인물은 자연색의 짧은 갈색 머리를 지닌 30대 중반 미국인 남성으로 읽히며 토니의 체격과 작업복에 부합한다. 구조용 하네스, 로프와 장비 주머니가 유지되고 스마트폰도 있다. 다른 사람이나 불필요한 물체는 추가되지 않았다.",
        "hard_violations": [],
        "physics": "토니의 두 발이 콘크리트 지면에 닿아 체중을 지지하며, 오른손이 스마트폰을 실제로 쥐고 있다. 장비와 로프는 하네스와 허리에 부착되어 있어 떠 있는 물체가 없고 자세도 물리적으로 가능하다."
       },
       {
        "label": "A",
        "direction": "토니의 몸과 고개가 정면의 광대한 숲·강 계곡 및 기울어진 고층 골조를 향한다. 오른손으로 스마트폰을 눈높이 가까이 올렸지만 머리는 풍경 쪽을 향해 있어, 스마트폰을 든 채 믿기지 않는 풍경을 응시하는 관계가 성립한다. 휴대전화 화면은 토니의 눈 쪽이고 카메라는 뒷면을 보는 배치다.",
        "built_space": "광산 출구 바로 밖의 열린 지대이며 왼쪽에 광산 입구 1개, 전경에 낡은 레일 2줄과 콘크리트 바닥, 오른쪽에 큰 나무와 노출 뿌리가 있다. 토니는 출구 너머의 평평한 콘크리트 가장자리에 중간 왼쪽으로 서 있다. 배경에는 광대한 숲, 계곡을 가로질러 굽이치는 강 1개, 수관 위로 솟은 기울어진 고층 골조 1개만 보여 장소 고정 사항과 맞는다.",
        "entities": "유일한 인물은 뒤에서 보이는 자연색 짧은 갈색 머리의 30대 중반 미국인 남성 토니이며, 참조와 맞는 체격·오염된 작업복·구조 하네스·로프·허리 장비를 착용했다. 오른손에는 스마트폰이 있다. 다른 인물, 온전한 추가 타워, 산업시설이나 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "토니는 두 발을 콘크리트 바닥에 안정적으로 딛고 있으며 약간 벌어진 다리가 몸을 지지한다. 오른손이 들어 올린 스마트폰을 확실히 잡고 있고, 로프와 장비는 하네스 및 벨트에 매달려 있다. 지지 없이 떠 있는 신체나 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.556
   },
   "adjusted": {
    "A": 2.0,
    "B": 1.556
   },
   "violations": {},
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 1556
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "프롬프트가 요구한 완벽한 뒷모습과 넓은 숲을 응시하며 스마트폰을 들어 올린 자세를 정확하게 구현했으며, 배경 기준 이미지의 디테일도 완벽하게 유지했습니다."
   },
   {
    "label": "B",
    "score": 1556,
    "verdict_ko": "배경과 인물의 외형은 훌륭하게 구현되었으나, 완전한 뒷모습이라기보다 측면에 가까우며 시선이 풍경보다는 스마트폰을 향하고 있어 지시사항과 다소 차이가 있습니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S5sh41_sel.png",
    "asset_id": "381059db-896e-48f9-a132-83d93277b0c4",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2a5-3bf0-75c8-bc69-76372cd95042",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S5sh41"
  },
  "lane_policy": "ab_select_bypass:prev"
 },
 "S5sh43::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:30:54.356895+00:00",
  "fingerprint": "b6744245ed592b8cdaead3bdf4b4289e02690a3f9ff32efc2c6b61886429788d",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S5sh43_sel.png",
  "source_sha256": "e4a85320b9188c867e91e7d3662b819c5a51391cef8cd76ab21b40bab87ec7f0",
  "file": "S5sh43_cine.png",
  "staged_sha256": "8a4a1e8d4cc0c687f3bfe8bb9b289d6b4e9d40f3389a0c2fa8f5eac03e4af180",
  "latency_ms": 21712
 },
 "S6sh49::signage": {
  "fp": "e296e2ce72e32151",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "era_assess::3c5e470bdfe94bad": {
  "subjects": [],
  "subject_text": "갱도 출구 주변 펜실베이니아 원시 숲\n강한 낮빛 아래 끝없이 펼쳐진 울창한 숲. 계곡의 강과 뿌리에 둘러싸인 갱도 입구, 멀리 기울어진 고층건물 골조가 보인다.",
  "identity": "canonical",
  "scope_id": "L03",
  "scope_role": "location_exterior",
  "scope_sha": "0309819f98fc3a9a"
 },
 "S6sh49::bgfirst_bg": {
  "input_fingerprint": "49b8088c2b7f14c6",
  "prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 거대한 원통 모양으로 매끄럽게 파여 나간 기이한 건물 잔해를 향해 고개를 들고 시선을 고정한 채 서 있는 '토니(앤서니 로저스)'.\n\nLOCATION (lock): Outside in a forest clearing around a ruined building whose mass has been smoothly removed in a gigantic cylindrical void.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward cylindrical void; cylindrical void in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: building remains with cylindrical void (an immense smooth cylindrical section has been carved away) — The damaged face is seen obliquely so the circular depth and missing volume read behind Tony; used as Primary environmental revelation aligned with Tony's raised sightline; surrounding forest (present around the building remains); used as Provides environmental and scale context around the anomalous damage.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight presents the discovery in restrained forest greens, concrete neutrals, and low-to-mid-key cinematic contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.\n\nThe THIRD attached image (STRUCTURE LOOK) is the identity source of the fixed structure at this location: its faces, openings, levels, materials and signage are truth. Where it conflicts with the LOCATION PHOTOGRAPH about the structure itself, the STRUCTURE LOOK wins; the photograph still governs the surroundings, time of day and lighting.",
  "effective_prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 거대한 원통 모양으로 매끄럽게 파여 나간 기이한 건물 잔해를 향해 고개를 들고 시선을 고정한 채 서 있는 '토니(앤서니 로저스)'.\n\nLOCATION (lock): Outside in a forest clearing around a ruined building whose mass has been smoothly removed in a gigantic cylindrical void.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward cylindrical void; cylindrical void in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: building remains with cylindrical void (an immense smooth cylindrical section has been carved away) — The damaged face is seen obliquely so the circular depth and missing volume read behind Tony; used as Primary environmental revelation aligned with Tony's raised sightline; surrounding forest (present around the building remains); used as Provides environmental and scale context around the anomalous damage.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight presents the discovery in restrained forest greens, concrete neutrals, and low-to-mid-key cinematic contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.\n\nThe THIRD attached image (STRUCTURE LOOK) is the identity source of the fixed structure at this location: its faces, openings, levels, materials and signage are truth. Where it conflicts with the LOCATION PHOTOGRAPH about the structure itself, the STRUCTURE LOOK wins; the photograph still governs the surroundings, time of day and lighting.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S6sh49__bgfirst_bg.png",
  "asset_id": "5c56b1c3-47a5-4284-9d79-85995ac89673",
  "input_asset_ids": [
   "51cfa0df-a0ed-472a-a51f-e9ffc31c94ff",
   "de7512bc-fe56-4901-b01a-ebc1afe72e59",
   "3f107fbd-85c0-4e79-84fd-608eeb3a7598"
  ]
 },
 "S6sh49": {
  "input_fingerprint": "89e34ec30f938f08",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 거대한 원통 모양으로 매끄럽게 파여 나간 기이한 건물 잔해를 향해 고개를 들고 시선을 고정한 채 서 있는 '토니(앤서니 로저스)'.\n\nLOCATION (lock): Outside in a forest clearing around a ruined building whose mass has been smoothly removed in a gigantic cylindrical void. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nSTRUCTURE LOOK AUTHORITY: the attached STRUCTURE LOOK photograph is the identity of the fixed structure at this location — wherever that structure appears in the frame, its shape, proportions, openings, materials and colors are LOCKED to it. The LOCATION PHOTOGRAPH remains the authority for this shot's sub-space, surroundings, time of day and lighting. If the two conflict on the structure itself, the STRUCTURE LOOK photo wins; for everything else, the LOCATION PHOTOGRAPH wins.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward cylindrical void; cylindrical void in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: building remains with cylindrical void (an immense smooth cylindrical section has been carved away) — The damaged face is seen obliquely so the circular depth and missing volume read behind Tony; used as Primary environmental revelation aligned with Tony's raised sightline; surrounding forest (present around the building remains); used as Provides environmental and scale context around the anomalous damage.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight presents the discovery in restrained forest greens, concrete neutrals, and low-to-mid-key cinematic contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A building ruin has been removed in a huge, smooth cylindrical section. The surrounding region remains overgrown forest beneath a daylight sky. 토니(앤서니 로저스): On the eighth day, he stands looking up at the strangely cylindrical building ruin. His rescue harness and surviving equipment remain with him.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Stage the shot. The FIRST attached image (SHOT BACKGROUND) is the finished empty background of this shot — keep it EXACTLY: its camera, perspective, architecture, lighting and every fixed feature stay untouched. The SECOND attached image (LAYOUT SKETCH) tells you ONLY where the people go: each sketched person's position, screen size, pose and the gaze/motion arrows. Ignore the sketch's background lines. The CHARACTER REFERENCE photographs show the real people.\nPlace the real people into the background at exactly the sketched positions, sizes and poses, following the arrow directions. No sketch lines or arrows may remain.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 거대한 원통 모양으로 매끄럽게 파여 나간 기이한 건물 잔해를 향해 고개를 들고 시선을 고정한 채 서 있는 '토니(앤서니 로저스)'.\n\nLOCATION (lock): Outside in a forest clearing around a ruined building whose mass has been smoothly removed in a gigantic cylindrical void. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward cylindrical void; cylindrical void in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: building remains with cylindrical void (an immense smooth cylindrical section has been carved away) — The damaged face is seen obliquely so the circular depth and missing volume read behind Tony; used as Primary environmental revelation aligned with Tony's raised sightline; surrounding forest (present around the building remains); used as Provides environmental and scale context around the anomalous damage.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight presents the discovery in restrained forest greens, concrete neutrals, and low-to-mid-key cinematic contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A building ruin has been removed in a huge, smooth cylindrical section. The surrounding region remains overgrown forest beneath a daylight sky. 토니(앤서니 로저스): On the eighth day, he stands looking up at the strangely cylindrical building ruin. His rescue harness and surviving equipment remain with him.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 거대한 원통 모양으로 매끄럽게 파여 나간 기이한 건물 잔해를 향해 고개를 들고 시선을 고정한 채 서 있는 '토니(앤서니 로저스)'.\n\nLOCATION (lock): Outside in a forest clearing around a ruined building whose mass has been smoothly removed in a gigantic cylindrical void. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nSTRUCTURE LOOK AUTHORITY: the attached STRUCTURE LOOK photograph is the identity of the fixed structure at this location — wherever that structure appears in the frame, its shape, proportions, openings, materials and colors are LOCKED to it. The LOCATION PHOTOGRAPH remains the authority for this shot's sub-space, surroundings, time of day and lighting. If the two conflict on the structure itself, the STRUCTURE LOOK photo wins; for everything else, the LOCATION PHOTOGRAPH wins.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, midground, looks toward cylindrical void; cylindrical void in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: building remains with cylindrical void (an immense smooth cylindrical section has been carved away) — The damaged face is seen obliquely so the circular depth and missing volume read behind Tony; used as Primary environmental revelation aligned with Tony's raised sightline; surrounding forest (present around the building remains); used as Provides environmental and scale context around the anomalous damage.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight presents the discovery in restrained forest greens, concrete neutrals, and low-to-mid-key cinematic contrast.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A building ruin has been removed in a huge, smooth cylindrical section. The surrounding region remains overgrown forest beneath a daylight sky. 토니(앤서니 로저스): On the eighth day, he stands looking up at the strangely cylindrical building ruin. His rescue harness and surviving equipment remain with him.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S6sh49__bgfirst_bg.png",
     "asset_id": "5c56b1c3-47a5-4284-9d79-85995ac89673",
     "role": "bgfirst_bg"
    },
    {
     "label": "LAYOUT SKETCH (people placement only)",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/conti/conti_S6sh49.png",
     "asset_id": "51cfa0df-a0ed-472a-a51f-e9ffc31c94ff",
     "role": "conti_light"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    }
   ],
   "B": [
    {
     "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its spatial layout, surroundings, fixed features, time of day and lighting mood are spatial truth; stage the moment inside this place. If a STRUCTURE LOOK photograph is also attached, that photo wins for the fixed structure itself — this photograph wins for everything around it. Never copy its camera framing.",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/episodes/c804efc3-0697-4c22-98b4-6992a70c2b20/images/background_chain/L03B02.png",
     "asset_id": "de7512bc-fe56-4901-b01a-ebc1afe72e59",
     "role": "location_plate"
    },
    {
     "label": "STRUCTURE LOOK — the confirmed photograph of the fixed structure at this location: wherever the structure appears in the frame, its shape, proportions, materials, colors and openings are LOCKED to this photo. Never copy its camera framing, time of day or lighting — the shot text and the LOCATION PHOTOGRAPH are the authorities for those.",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/background_chain/seed_bg_mine_complex_sel.png",
     "asset_id": "3f107fbd-85c0-4e79-84fd-608eeb3a7598",
     "role": "structure_seed_look"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "judge_flip_skipped": "judge_fn_owns_order",
  "readings": [
   {
    "label": "A",
    "direction": "화면 중좌측의 토니가 오른쪽 배경에 있는 건물의 거대한 원통형 구멍을 향해 고개를 들고 시선을 고정하고 있음.",
    "built_space": "숲속 공터에 위치한 폐허가 된 콘크리트 건물. 원통형 구멍의 깊이와 건물의 전반적인 형태가 레퍼런스 사진과 상당히 유사하게 묘사됨.",
    "entities": "토니의 복장이 레퍼런스와 전혀 다름. 지정된 헬멧, 얼굴 가리개, 가슴 부착 통신 장비 등이 모두 누락되었고 다른 형태의 일반적인 재킷을 입고 있음.",
    "hard_violations": [],
    "physics": "두 발로 숲의 지면을 안정적으로 딛고 서 있으며, 신체 지탱이나 자세에 물리적인 오류가 없음."
   },
   {
    "label": "B",
    "direction": "화면 중좌측의 토니가 우측 배경에 위치한 건물의 거대한 원통형 공간을 향해 고개를 들어 응시하고 있음.",
    "built_space": "폐허가 된 콘크리트 건물과 네모난 창문들의 구조는 일치하나, 뚫려 있는 원통형 구멍이 얇은 벽을 관통하는 것이 아니라 비정상적으로 깊은 터널처럼 왜곡되어 비례가 달라짐.",
    "entities": "토니가 캐릭터 레퍼런스에 제시된 헬멧, 발라클라바, 가슴 부착 카메라/무전기, 하네스, 흙이 묻은 전술복 등을 매우 정확하게 착용하고 있음.",
    "hard_violations": [],
    "physics": "자연스러운 자세로 두 발을 땅에 딛고 서 있으며, 신체나 장비의 중력 및 지탱 묘사에 어긋나는 부분이 없음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "건물의 원통형 구멍이 터널처럼 과장되게 깊어진 점은 아쉬우나, 캐릭터의 복잡한 헬멧과 전술 장비 등 복장을 레퍼런스와 완벽하게 일치시켰습니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "건물 폐허의 형태와 구멍의 깊이는 레퍼런스에 더 가깝게 구현되었지만, 캐릭터의 필수적인 헬멧과 장비가 완전히 누락되어 프롬프트 지시를 심각하게 어겼습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "화면 중좌측의 토니가 오른쪽 배경에 있는 건물의 거대한 원통형 구멍을 향해 고개를 들고 시선을 고정하고 있음.",
        "built_space": "숲속 공터에 위치한 폐허가 된 콘크리트 건물. 원통형 구멍의 깊이와 건물의 전반적인 형태가 레퍼런스 사진과 상당히 유사하게 묘사됨.",
        "entities": "토니의 복장이 레퍼런스와 전혀 다름. 지정된 헬멧, 얼굴 가리개, 가슴 부착 통신 장비 등이 모두 누락되었고 다른 형태의 일반적인 재킷을 입고 있음.",
        "hard_violations": [],
        "physics": "두 발로 숲의 지면을 안정적으로 딛고 서 있으며, 신체 지탱이나 자세에 물리적인 오류가 없음."
       },
       {
        "label": "B",
        "direction": "화면 중좌측의 토니가 우측 배경에 위치한 건물의 거대한 원통형 공간을 향해 고개를 들어 응시하고 있음.",
        "built_space": "폐허가 된 콘크리트 건물과 네모난 창문들의 구조는 일치하나, 뚫려 있는 원통형 구멍이 얇은 벽을 관통하는 것이 아니라 비정상적으로 깊은 터널처럼 왜곡되어 비례가 달라짐.",
        "entities": "토니가 캐릭터 레퍼런스에 제시된 헬멧, 발라클라바, 가슴 부착 카메라/무전기, 하네스, 흙이 묻은 전술복 등을 매우 정확하게 착용하고 있음.",
        "hard_violations": [],
        "physics": "자연스러운 자세로 두 발을 땅에 딛고 서 있으며, 신체나 장비의 중력 및 지탱 묘사에 어긋나는 부분이 없음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "건물의 원통형 구멍이 터널처럼 과장되게 깊어진 점은 아쉬우나, 캐릭터의 복잡한 헬멧과 전술 장비 등 복장을 레퍼런스와 완벽하게 일치시켰습니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "건물 폐허의 형태와 구멍의 깊이는 레퍼런스에 더 가깝게 구현되었지만, 캐릭터의 필수적인 헬멧과 장비가 완전히 누락되어 프롬프트 지시를 심각하게 어겼습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "화면 중좌측의 토니가 오른쪽 배경에 있는 건물의 거대한 원통형 구멍을 향해 고개를 들고 시선을 고정하고 있음.",
        "built_space": "숲속 공터에 위치한 폐허가 된 콘크리트 건물. 원통형 구멍의 깊이와 건물의 전반적인 형태가 레퍼런스 사진과 상당히 유사하게 묘사됨.",
        "entities": "토니의 복장이 레퍼런스와 전혀 다름. 지정된 헬멧, 얼굴 가리개, 가슴 부착 통신 장비 등이 모두 누락되었고 다른 형태의 일반적인 재킷을 입고 있음.",
        "hard_violations": [],
        "physics": "두 발로 숲의 지면을 안정적으로 딛고 서 있으며, 신체 지탱이나 자세에 물리적인 오류가 없음."
       },
       {
        "label": "B",
        "direction": "화면 중좌측의 토니가 우측 배경에 위치한 건물의 거대한 원통형 공간을 향해 고개를 들어 응시하고 있음.",
        "built_space": "폐허가 된 콘크리트 건물과 네모난 창문들의 구조는 일치하나, 뚫려 있는 원통형 구멍이 얇은 벽을 관통하는 것이 아니라 비정상적으로 깊은 터널처럼 왜곡되어 비례가 달라짐.",
        "entities": "토니가 캐릭터 레퍼런스에 제시된 헬멧, 발라클라바, 가슴 부착 카메라/무전기, 하네스, 흙이 묻은 전술복 등을 매우 정확하게 착용하고 있음.",
        "hard_violations": [],
        "physics": "자연스러운 자세로 두 발을 땅에 딛고 서 있으며, 신체나 장비의 중력 및 지탱 묘사에 어긋나는 부분이 없음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "토니가 중좌측에서 장비를 유지한 채 우측의 깊은 원통형 절삭부를 올려다보고, 사선으로 드러난 원통 깊이도 가장 충실하지만 고정 구조물은 STRUCTURE LOOK 사진의 원형 폐허와 일치하지 않는다."
       },
       {
        "label": "B",
        "score": 5,
        "verdict_ko": "인물의 중좌측 배치와 올려다보는 행동은 맞지만 원통이 거의 정면·중앙에 놓여 사선 깊이가 약하고, 헬멧과 흉부 구조 장비가 없으며 STRUCTURE LOOK의 구조 정체성도 맞지 않는다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니는 고개와 눈을 위쪽 오른편으로 들어 원통형 공동의 상부 왼쪽 테두리와 그 위의 손상된 콘크리트 면을 바라본다. 시선이 공동의 중심보다는 상단 가장자리에 닿지만, 명시된 원통형 잔해를 향한 올려다보기로 읽힌다. 무기나 이동 물체는 없다.",
        "built_space": "숲 공터에 하나의 폐허 건물과 하나의 거대한 원통형 관통 공동이 보인다. 공동은 화면 중우측에서 사선으로 보여 가까운 타원형 입구, 긴 매끈한 내벽, 먼 출구가 모두 드러난다. 건물에는 위층의 여러 직사각형 개구부, 아래층의 세로 개구부들, 공동 오른쪽의 녹슨 수직 철골 하나가 보인다. 토니는 공동 앞 중좌측의 지면에 서 있어 구조물과의 위치 관계가 자연스럽다. 다만 이 고정 구조물은 LOCATION 사진과는 매우 가깝지만 STRUCTURE LOOK 사진의 계곡 속 원형·방사형 폐허 형태와는 일치하지 않는다. 반사면은 없다.",
        "entities": "사람은 토니 한 명뿐이며 30대 중반의 성인 남성으로 보인다. 얼굴, 짧은 자연색 머리, 체격은 캐릭터 참조와 대체로 맞고, 헬멧·어두운 작업복·장갑·흉부 장비·구조 하네스·로프와 파우치가 보인다. 거대한 매끈한 원통형 공동, 콘크리트 폐허, 울창한 숲이 모두 존재하며 읽을 수 있는 글자나 로고는 없다.",
        "hard_violations": [],
        "physics": "토니의 양발이 이끼 낀 지면에 확실히 닿아 체중을 지지하고 있으며 약간 벌린 자세로 자연스럽게 서 있다. 로프와 장비는 허리 하네스 및 몸의 스트랩에 연결되어 매달려 있고 공중에 뜬 물체가 없다."
       },
       {
        "label": "B",
        "direction": "토니는 등을 보인 채 머리를 위쪽 오른편으로 젖혀 원통형 공동의 상부 왼쪽 가장자리와 위쪽 콘크리트 면을 바라본다. 목표는 공동의 상부 테두리로 읽혀 행동은 대체로 맞지만 시선이 공동 내부 중심에 직접 고정된 모습은 아니다. 무기나 이동 물체는 없다.",
        "built_space": "숲 공터에 하나의 폐허 건물과 하나의 거대한 원통형 관통 공동이 있다. 공동은 화면 중앙에 가깝고 거의 정면 원형으로 보이며, 매끈한 내벽과 반대편 숲이 보인다. 위층과 아래층에 여러 직사각형 개구부가 있고 공동 오른쪽에 녹슨 수직 철골 하나가 있다. 토니는 공동 앞 중좌측 지면에 서 있다. 배치는 물리적으로 가능하지만, 요구된 중우측 배치와 손상면의 사선 관찰이 약해 원통의 제거된 부피보다 둥근 터널처럼 읽힌다. 또한 고정 구조물은 STRUCTURE LOOK 사진의 원형·방사형 폐허와 일치하지 않는다. 반사면은 없다.",
        "entities": "사람은 토니 한 명뿐이며 뒤에서 본 30대 성인 남성으로 보인다. 얼굴이 보이지 않아 참조 얼굴의 정확한 동일성을 확인할 수 없고, 짧은 갈색 머리와 체격은 대체로 부합한다. 어두운 작업복, 허리 로프와 파우치는 있으나 참조의 헬멧과 뚜렷한 흉부 구조 하네스·장비가 보이지 않는다. 콘크리트 폐허, 원통형 공동, 숲은 존재하고 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "토니의 두 발이 지면에 닿아 몸을 지지하며 서 있는 자세가 가능하다. 로프는 허리 장비에 연결되어 아래로 늘어지고 파우치도 벨트에 부착되어 있다. 공중에 뜨거나 지지되지 않은 몸이나 물체는 없다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "토니가 중좌측에서 장비를 유지한 채 우측의 깊은 원통형 절삭부를 올려다보고, 사선으로 드러난 원통 깊이도 가장 충실하지만 고정 구조물은 STRUCTURE LOOK 사진의 원형 폐허와 일치하지 않는다."
       },
       {
        "label": "A",
        "score": 5,
        "verdict_ko": "인물의 중좌측 배치와 올려다보는 행동은 맞지만 원통이 거의 정면·중앙에 놓여 사선 깊이가 약하고, 헬멧과 흉부 구조 장비가 없으며 STRUCTURE LOOK의 구조 정체성도 맞지 않는다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "토니는 고개와 눈을 위쪽 오른편으로 들어 원통형 공동의 상부 왼쪽 테두리와 그 위의 손상된 콘크리트 면을 바라본다. 시선이 공동의 중심보다는 상단 가장자리에 닿지만, 명시된 원통형 잔해를 향한 올려다보기로 읽힌다. 무기나 이동 물체는 없다.",
        "built_space": "숲 공터에 하나의 폐허 건물과 하나의 거대한 원통형 관통 공동이 보인다. 공동은 화면 중우측에서 사선으로 보여 가까운 타원형 입구, 긴 매끈한 내벽, 먼 출구가 모두 드러난다. 건물에는 위층의 여러 직사각형 개구부, 아래층의 세로 개구부들, 공동 오른쪽의 녹슨 수직 철골 하나가 보인다. 토니는 공동 앞 중좌측의 지면에 서 있어 구조물과의 위치 관계가 자연스럽다. 다만 이 고정 구조물은 LOCATION 사진과는 매우 가깝지만 STRUCTURE LOOK 사진의 계곡 속 원형·방사형 폐허 형태와는 일치하지 않는다. 반사면은 없다.",
        "entities": "사람은 토니 한 명뿐이며 30대 중반의 성인 남성으로 보인다. 얼굴, 짧은 자연색 머리, 체격은 캐릭터 참조와 대체로 맞고, 헬멧·어두운 작업복·장갑·흉부 장비·구조 하네스·로프와 파우치가 보인다. 거대한 매끈한 원통형 공동, 콘크리트 폐허, 울창한 숲이 모두 존재하며 읽을 수 있는 글자나 로고는 없다.",
        "hard_violations": [],
        "physics": "토니의 양발이 이끼 낀 지면에 확실히 닿아 체중을 지지하고 있으며 약간 벌린 자세로 자연스럽게 서 있다. 로프와 장비는 허리 하네스 및 몸의 스트랩에 연결되어 매달려 있고 공중에 뜬 물체가 없다."
       },
       {
        "label": "A",
        "direction": "토니는 등을 보인 채 머리를 위쪽 오른편으로 젖혀 원통형 공동의 상부 왼쪽 가장자리와 위쪽 콘크리트 면을 바라본다. 목표는 공동의 상부 테두리로 읽혀 행동은 대체로 맞지만 시선이 공동 내부 중심에 직접 고정된 모습은 아니다. 무기나 이동 물체는 없다.",
        "built_space": "숲 공터에 하나의 폐허 건물과 하나의 거대한 원통형 관통 공동이 있다. 공동은 화면 중앙에 가깝고 거의 정면 원형으로 보이며, 매끈한 내벽과 반대편 숲이 보인다. 위층과 아래층에 여러 직사각형 개구부가 있고 공동 오른쪽에 녹슨 수직 철골 하나가 있다. 토니는 공동 앞 중좌측 지면에 서 있다. 배치는 물리적으로 가능하지만, 요구된 중우측 배치와 손상면의 사선 관찰이 약해 원통의 제거된 부피보다 둥근 터널처럼 읽힌다. 또한 고정 구조물은 STRUCTURE LOOK 사진의 원형·방사형 폐허와 일치하지 않는다. 반사면은 없다.",
        "entities": "사람은 토니 한 명뿐이며 뒤에서 본 30대 성인 남성으로 보인다. 얼굴이 보이지 않아 참조 얼굴의 정확한 동일성을 확인할 수 없고, 짧은 갈색 머리와 체격은 대체로 부합한다. 어두운 작업복, 허리 로프와 파우치는 있으나 참조의 헬멧과 뚜렷한 흉부 구조 하네스·장비가 보이지 않는다. 콘크리트 폐허, 원통형 공동, 숲은 존재하고 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "토니의 두 발이 지면에 닿아 몸을 지지하며 서 있는 자세가 가능하다. 로프는 허리 장비에 연결되어 아래로 늘어지고 파우치도 벨트에 부착되어 있다. 공중에 뜨거나 지지되지 않은 몸이나 물체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.214,
    "B": 2.0
   },
   "adjusted": {
    "A": 1.214,
    "B": 2.0
   },
   "violations": {},
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "B": 2000,
   "A": 1214
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "건물의 원통형 구멍이 터널처럼 과장되게 깊어진 점은 아쉬우나, 캐릭터의 복잡한 헬멧과 전술 장비 등 복장을 레퍼런스와 완벽하게 일치시켰습니다."
   },
   {
    "label": "A",
    "score": 1214,
    "verdict_ko": "건물 폐허의 형태와 구멍의 깊이는 레퍼런스에 더 가깝게 구현되었지만, 캐릭터의 필수적인 헬멧과 장비가 완전히 누락되어 프롬프트 지시를 심각하게 어겼습니다."
   }
  ],
  "refs": [
   {
    "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its spatial layout, surroundings, fixed features, time of day and lighting mood are spatial truth; stage the moment inside this place. If a STRUCTURE LOOK photograph is also attached, that photo wins for the fixed structure itself — this photograph wins for everything around it. Never copy its camera framing.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/episodes/c804efc3-0697-4c22-98b4-6992a70c2b20/images/background_chain/L03B02.png",
    "asset_id": "de7512bc-fe56-4901-b01a-ebc1afe72e59",
    "role": "location_plate"
   },
   {
    "label": "STRUCTURE LOOK — the confirmed photograph of the fixed structure at this location: wherever the structure appears in the frame, its shape, proportions, materials, colors and openings are LOCKED to this photo. Never copy its camera framing, time of day or lighting — the shot text and the LOCATION PHOTOGRAPH are the authorities for those.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/background_chain/seed_bg_mine_complex_sel.png",
    "asset_id": "3f107fbd-85c0-4e79-84fd-608eeb3a7598",
    "role": "structure_seed_look"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2a9-4337-733a-a24e-fbd075eb0dda",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S6sh49__bgfirst_bg.png",
   "bg_asset_id": "5c56b1c3-47a5-4284-9d79-85995ac89673",
   "bg_record_key": "S6sh49::bgfirst_bg",
   "chain_winner": false,
   "authority": "plate",
   "seed_attached": true
  },
  "ref_mode": "플레이트+엔티티 (2택1: 무콘티 승)",
  "share_plan": {
   "ref_plan": "background"
  },
  "lane_policy": "ab_select_ready"
 },
 "S6sh49::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:33:55.420386+00:00",
  "fingerprint": "3ae8fd474e20a47fc98a6af717663cac359f8e870336eb10f3d076763604de21",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S6sh49_sel.png",
  "source_sha256": "5bd33a02411cc42be1d0d050d75118e55c72225fb9b1ca9a192a59dc6e307ff9",
  "file": "S6sh49_cine.png",
  "staged_sha256": "ed31a1828b21e4624d2f290167c9718c2eac740d2ccb16cd2b25590768815c81",
  "latency_ms": 20952
 },
 "S6sh53::signage": {
  "fp": "9362184dee9458c0",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S6sh53::bgfirst_bg": {
  "input_fingerprint": "17927a4c9b6b716a",
  "prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 소리가 난 수풀 쪽을 노려보며 바닥에 바짝 엎드린 '토니(앤서니 로저스)'의 긴장한 전신.\n\nLOCATION (lock): Outside on the forest floor beside a berry-bearing thicket, where dense brush conceals the source of a snapping branch.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the lower-left of the frame, midground, looks toward brush where the sound originated; brush where the sound originated in the upper-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: brush at the sound source (the direction from which a branch-breaking sound originated); used as Threat anchor placed beyond Tony's raised head and sightline; forest ground (supporting Tony's prone body); used as Creates the low diagonal plane connecting Tony to the brush.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight is held to restrained forest greens and weathered neutrals, with grounded contrast supporting the sudden tension.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "effective_prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 소리가 난 수풀 쪽을 노려보며 바닥에 바짝 엎드린 '토니(앤서니 로저스)'의 긴장한 전신.\n\nLOCATION (lock): Outside on the forest floor beside a berry-bearing thicket, where dense brush conceals the source of a snapping branch.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the lower-left of the frame, midground, looks toward brush where the sound originated; brush where the sound originated in the upper-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: brush at the sound source (the direction from which a branch-breaking sound originated); used as Threat anchor placed beyond Tony's raised head and sightline; forest ground (supporting Tony's prone body); used as Creates the low diagonal plane connecting Tony to the brush.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight is held to restrained forest greens and weathered neutrals, with grounded contrast supporting the sudden tension.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S6sh53__bgfirst_bg.png",
  "asset_id": "7d273cbd-b1cd-461a-9fe6-3121ac794bdf",
  "input_asset_ids": [
   "24426d18-9230-48aa-99ac-b4af3c2dbaed",
   "5094d3b6-9a8e-40d5-9d02-a151027a874e"
  ]
 },
 "S6sh53": {
  "input_fingerprint": "60d3e37fccb84469",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 소리가 난 수풀 쪽을 노려보며 바닥에 바짝 엎드린 '토니(앤서니 로저스)'의 긴장한 전신.\n\nLOCATION (lock): Outside on the forest floor beside a berry-bearing thicket, where dense brush conceals the source of a snapping branch. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the lower-left of the frame, midground, looks toward brush where the sound originated; brush where the sound originated in the upper-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: brush at the sound source (the direction from which a branch-breaking sound originated); used as Threat anchor placed beyond Tony's raised head and sightline; forest ground (supporting Tony's prone body); used as Creates the low diagonal plane connecting Tony to the brush.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight is held to restrained forest greens and weathered neutrals, with grounded contrast supporting the sudden tension.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Dense brush and forest floor surround the position, with the smartphone battery at 4 percent. Red berries remain nearby after being examined with the camera. 토니(앤서니 로저스): He is pressed low against the ground, tensely watching the brush where the branch snapped. He retains his smartphone and rescue harness after twelve days of survival.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Stage the shot. The FIRST attached image (SHOT BACKGROUND) is the finished empty background of this shot — keep it EXACTLY: its camera, perspective, architecture, lighting and every fixed feature stay untouched. The SECOND attached image (LAYOUT SKETCH) tells you ONLY where the people go: each sketched person's position, screen size, pose and the gaze/motion arrows. Ignore the sketch's background lines. The CHARACTER REFERENCE photographs show the real people.\nPlace the real people into the background at exactly the sketched positions, sizes and poses, following the arrow directions. No sketch lines or arrows may remain.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 소리가 난 수풀 쪽을 노려보며 바닥에 바짝 엎드린 '토니(앤서니 로저스)'의 긴장한 전신.\n\nLOCATION (lock): Outside on the forest floor beside a berry-bearing thicket, where dense brush conceals the source of a snapping branch. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the lower-left of the frame, midground, looks toward brush where the sound originated; brush where the sound originated in the upper-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: brush at the sound source (the direction from which a branch-breaking sound originated); used as Threat anchor placed beyond Tony's raised head and sightline; forest ground (supporting Tony's prone body); used as Creates the low diagonal plane connecting Tony to the brush.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight is held to restrained forest greens and weathered neutrals, with grounded contrast supporting the sudden tension.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Dense brush and forest floor surround the position, with the smartphone battery at 4 percent. Red berries remain nearby after being examined with the camera. 토니(앤서니 로저스): He is pressed low against the ground, tensely watching the brush where the branch snapped. He retains his smartphone and rescue harness after twelve days of survival.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 소리가 난 수풀 쪽을 노려보며 바닥에 바짝 엎드린 '토니(앤서니 로저스)'의 긴장한 전신.\n\nLOCATION (lock): Outside on the forest floor beside a berry-bearing thicket, where dense brush conceals the source of a snapping branch. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 토니(앤서니 로저스) in the lower-left of the frame, midground, looks toward brush where the sound originated; brush where the sound originated in the upper-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: brush at the sound source (the direction from which a branch-breaking sound originated); used as Threat anchor placed beyond Tony's raised head and sightline; forest ground (supporting Tony's prone body); used as Creates the low diagonal plane connecting Tony to the brush.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight is held to restrained forest greens and weathered neutrals, with grounded contrast supporting the sudden tension.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Dense brush and forest floor surround the position, with the smartphone battery at 4 percent. Red berries remain nearby after being examined with the camera. 토니(앤서니 로저스): He is pressed low against the ground, tensely watching the brush where the branch snapped. He retains his smartphone and rescue harness after twelve days of survival.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S6sh53__bgfirst_bg.png",
     "asset_id": "7d273cbd-b1cd-461a-9fe6-3121ac794bdf",
     "role": "bgfirst_bg"
    },
    {
     "label": "LAYOUT SKETCH (people placement only)",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/conti/conti_S6sh53.png",
     "asset_id": "24426d18-9230-48aa-99ac-b4af3c2dbaed",
     "role": "conti_light"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    }
   ],
   "B": [
    {
     "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/episodes/c804efc3-0697-4c22-98b4-6992a70c2b20/images/background_chain/L03B03.png",
     "asset_id": "5094d3b6-9a8e-40d5-9d02-a151027a874e",
     "role": "location_plate"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "judge_flip_skipped": "judge_fn_owns_order",
  "readings": [
   {
    "label": "A",
    "direction": "토니의 시선은 화면 우측에 부러진 나뭇가지가 있는 수풀을 향하고 있습니다.",
    "built_space": "배경에 위치 참조 사진과 일치하는 터널 입구와 직사각형 문들이 늘어선 폐허 건물이 배치되어 있습니다.",
    "entities": "토니는 레퍼런스와 일치하는 복장, 하네스, 헬멧을 착용하고 있으며 스마트폰을 들고 있습니다. 우측 수풀에는 열매와 부러진 가지가 보입니다.",
    "hard_violations": [],
    "physics": "토니는 지면에 밀착해 엎드려 바닥의 지지를 받고 있으며, 손으로 스마트폰을 안정적으로 쥐고 있습니다."
   },
   {
    "label": "B",
    "direction": "토니의 시선은 화면 우측의 붉은 열매가 열린 수풀을 향하고 있습니다.",
    "built_space": "폐허 건물이 배경에 흐리게 나타나지만, 참조 사진의 구조적 디테일(터널, 선로 등)은 명확하지 않습니다.",
    "entities": "토니는 복장과 하네스는 갖추었으나 레퍼런스의 헬멧을 착용하지 않았습니다. 스마트폰을 들고 있으며 우측에는 열매 맺힌 수풀이 있습니다.",
    "hard_violations": [],
    "physics": "토니는 지면에 엎드려 몸을 지탱하고 있으며, 손으로 스마트폰을 쥐고 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "요청된 와이드 샷과 '전신' 프레이밍을 훌륭하게 구현했으며, 레퍼런스의 장소와 캐릭터 복장(헬멧 포함)을 충실히 반영했습니다."
       },
       {
        "label": "B",
        "score": 5,
        "verdict_ko": "프레임이 너무 좁아 텍스트가 요구한 '전신'을 보여주지 못했으며, 캐릭터가 헬멧을 착용하지 않았습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선은 화면 우측에 부러진 나뭇가지가 있는 수풀을 향하고 있습니다.",
        "built_space": "배경에 위치 참조 사진과 일치하는 터널 입구와 직사각형 문들이 늘어선 폐허 건물이 배치되어 있습니다.",
        "entities": "토니는 레퍼런스와 일치하는 복장, 하네스, 헬멧을 착용하고 있으며 스마트폰을 들고 있습니다. 우측 수풀에는 열매와 부러진 가지가 보입니다.",
        "hard_violations": [],
        "physics": "토니는 지면에 밀착해 엎드려 바닥의 지지를 받고 있으며, 손으로 스마트폰을 안정적으로 쥐고 있습니다."
       },
       {
        "label": "B",
        "direction": "토니의 시선은 화면 우측의 붉은 열매가 열린 수풀을 향하고 있습니다.",
        "built_space": "폐허 건물이 배경에 흐리게 나타나지만, 참조 사진의 구조적 디테일(터널, 선로 등)은 명확하지 않습니다.",
        "entities": "토니는 복장과 하네스는 갖추었으나 레퍼런스의 헬멧을 착용하지 않았습니다. 스마트폰을 들고 있으며 우측에는 열매 맺힌 수풀이 있습니다.",
        "hard_violations": [],
        "physics": "토니는 지면에 엎드려 몸을 지탱하고 있으며, 손으로 스마트폰을 쥐고 있습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "요청된 와이드 샷과 '전신' 프레이밍을 훌륭하게 구현했으며, 레퍼런스의 장소와 캐릭터 복장(헬멧 포함)을 충실히 반영했습니다."
       },
       {
        "label": "B",
        "score": 5,
        "verdict_ko": "프레임이 너무 좁아 텍스트가 요구한 '전신'을 보여주지 못했으며, 캐릭터가 헬멧을 착용하지 않았습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "토니의 시선은 화면 우측에 부러진 나뭇가지가 있는 수풀을 향하고 있습니다.",
        "built_space": "배경에 위치 참조 사진과 일치하는 터널 입구와 직사각형 문들이 늘어선 폐허 건물이 배치되어 있습니다.",
        "entities": "토니는 레퍼런스와 일치하는 복장, 하네스, 헬멧을 착용하고 있으며 스마트폰을 들고 있습니다. 우측 수풀에는 열매와 부러진 가지가 보입니다.",
        "hard_violations": [],
        "physics": "토니는 지면에 밀착해 엎드려 바닥의 지지를 받고 있으며, 손으로 스마트폰을 안정적으로 쥐고 있습니다."
       },
       {
        "label": "B",
        "direction": "토니의 시선은 화면 우측의 붉은 열매가 열린 수풀을 향하고 있습니다.",
        "built_space": "폐허 건물이 배경에 흐리게 나타나지만, 참조 사진의 구조적 디테일(터널, 선로 등)은 명확하지 않습니다.",
        "entities": "토니는 복장과 하네스는 갖추었으나 레퍼런스의 헬멧을 착용하지 않았습니다. 스마트폰을 들고 있으며 우측에는 열매 맺힌 수풀이 있습니다.",
        "hard_violations": [],
        "physics": "토니는 지면에 엎드려 몸을 지탱하고 있으며, 손으로 스마트폰을 쥐고 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "정확한 폐허 위치와 넓은 구도, 우상단 수풀을 향한 시선이 가장 충실하지만 토니의 발끝이 잘려 ‘긴장한 전신’ 조건은 완수하지 못했다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "토니의 긴장된 포복과 시선은 명확하나 수풀이 배경이 아닌 큰 전경 요소이고 프레임이 더 좁으며 헬멧도 없어 기준에서 더 멀다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니는 고개와 눈을 화면 오른쪽 위의 붉은 열매 수풀에 정확히 고정하고 있다. 스마트폰은 손에 쥐었으며 뒷면이 카메라 쪽을 향한다. 별도의 무기나 다른 지향 물체는 없다.",
        "built_space": "야외 숲바닥이며 토니는 화면 왼쪽 아래에서 엎드려 있다. 뒤에는 참고 장소를 닮은 낡은 콘크리트 건물 일부가 보이고, 식별 가능한 위층 개구부 약 2개와 아래층의 어두운 출입구·개구부 약 3개가 보인다. 다만 위협 수풀은 우상단 배경이 아니라 토니와 가까운 큰 전경 요소로 확대되어 있다. 반사면은 없다.",
        "entities": "토니 한 명만 있으며 30대 중반의 백인계 미국 남성으로 보이고 얼굴·머리·체격은 인물 참고와 대체로 가깝다. 참고 복장과 유사한 오염된 전술복, 구조용 하네스와 로프, 장갑, 스마트폰이 보이지만 참고의 헬멧은 없다. 붉은 열매 수풀, 울창한 숲바닥, 폐허 건물이 모두 보이며 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "토니의 가슴·골반·다리와 양쪽 팔꿈치가 숲바닥에 닿아 몸을 지지하고, 머리와 상체는 팔꿈치 지지로 조금 들어 올려져 있다. 스마트폰은 장갑 낀 손으로 확실히 쥐고 있다. 부유하거나 지지 없는 물체는 없지만 양쪽 다리 끝이 왼쪽 프레임 밖으로 잘려 전신을 확인할 수 없다."
       },
       {
        "label": "B",
        "direction": "토니는 고개와 눈을 오른쪽으로 돌려 화면 우상단의 빽빽한 수풀과 그 안의 부러진 가지들을 바라본다. 시선은 소리 근원으로 읽히는 수풀에 도달한다. 스마트폰은 오른손에 들려 화면 면이 카메라 쪽으로 향하지만 토니는 이를 보고 있지 않으며, 현재는 사용보다 보유 상태로 읽힌다.",
        "built_space": "참고 장소와 매우 가까운 야외 폐허 배치로, 왼쪽의 큰 어두운 입구 1개와 이어지는 아래층 세로 개구부 약 5개, 위층 작은 사각 개구부 약 5개가 보인다. 토니는 화면 왼쪽 아래 숲길의 전경부터 중경에 걸쳐 엎드렸고, 위협 수풀은 오른쪽 중·배경에 놓였다. 건물의 풍화 콘크리트, 왼쪽 암벽, 덩굴과 숲길의 관계도 참고 장소에 부합하며 반사면은 없다.",
        "entities": "토니 한 명만 보이며 30대 중반의 백인계 남성 얼굴, 체격, 어두운 머리와 수염이 참고 인물에 가깝다. 참고와 같은 헬멧, 오염된 전술복, 구조용 하네스·카라비너, 장갑과 스마트폰을 유지한다. 붉은 열매 수풀과 부러진 가지, 숲바닥, 폐허가 보이고 추가 인물이나 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "토니의 흉부·복부·골반·다리가 바닥에 닿고 굽힌 팔꿈치가 들어 올린 머리와 상체를 지지하므로 낮은 포복 자세가 물리적으로 성립한다. 스마트폰은 오른손이 확실히 쥐고 있다. 바닥의 긴 가지와 수풀의 부러진 가지도 표면 또는 줄기에 지지되어 있다. 다만 다리와 발 일부가 아래·왼쪽 프레임 경계에서 잘려 완전한 전신은 아니다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "정확한 폐허 위치와 넓은 구도, 우상단 수풀을 향한 시선이 가장 충실하지만 토니의 발끝이 잘려 ‘긴장한 전신’ 조건은 완수하지 못했다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "토니의 긴장된 포복과 시선은 명확하나 수풀이 배경이 아닌 큰 전경 요소이고 프레임이 더 좁으며 헬멧도 없어 기준에서 더 멀다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "토니는 고개와 눈을 화면 오른쪽 위의 붉은 열매 수풀에 정확히 고정하고 있다. 스마트폰은 손에 쥐었으며 뒷면이 카메라 쪽을 향한다. 별도의 무기나 다른 지향 물체는 없다.",
        "built_space": "야외 숲바닥이며 토니는 화면 왼쪽 아래에서 엎드려 있다. 뒤에는 참고 장소를 닮은 낡은 콘크리트 건물 일부가 보이고, 식별 가능한 위층 개구부 약 2개와 아래층의 어두운 출입구·개구부 약 3개가 보인다. 다만 위협 수풀은 우상단 배경이 아니라 토니와 가까운 큰 전경 요소로 확대되어 있다. 반사면은 없다.",
        "entities": "토니 한 명만 있으며 30대 중반의 백인계 미국 남성으로 보이고 얼굴·머리·체격은 인물 참고와 대체로 가깝다. 참고 복장과 유사한 오염된 전술복, 구조용 하네스와 로프, 장갑, 스마트폰이 보이지만 참고의 헬멧은 없다. 붉은 열매 수풀, 울창한 숲바닥, 폐허 건물이 모두 보이며 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "토니의 가슴·골반·다리와 양쪽 팔꿈치가 숲바닥에 닿아 몸을 지지하고, 머리와 상체는 팔꿈치 지지로 조금 들어 올려져 있다. 스마트폰은 장갑 낀 손으로 확실히 쥐고 있다. 부유하거나 지지 없는 물체는 없지만 양쪽 다리 끝이 왼쪽 프레임 밖으로 잘려 전신을 확인할 수 없다."
       },
       {
        "label": "A",
        "direction": "토니는 고개와 눈을 오른쪽으로 돌려 화면 우상단의 빽빽한 수풀과 그 안의 부러진 가지들을 바라본다. 시선은 소리 근원으로 읽히는 수풀에 도달한다. 스마트폰은 오른손에 들려 화면 면이 카메라 쪽으로 향하지만 토니는 이를 보고 있지 않으며, 현재는 사용보다 보유 상태로 읽힌다.",
        "built_space": "참고 장소와 매우 가까운 야외 폐허 배치로, 왼쪽의 큰 어두운 입구 1개와 이어지는 아래층 세로 개구부 약 5개, 위층 작은 사각 개구부 약 5개가 보인다. 토니는 화면 왼쪽 아래 숲길의 전경부터 중경에 걸쳐 엎드렸고, 위협 수풀은 오른쪽 중·배경에 놓였다. 건물의 풍화 콘크리트, 왼쪽 암벽, 덩굴과 숲길의 관계도 참고 장소에 부합하며 반사면은 없다.",
        "entities": "토니 한 명만 보이며 30대 중반의 백인계 남성 얼굴, 체격, 어두운 머리와 수염이 참고 인물에 가깝다. 참고와 같은 헬멧, 오염된 전술복, 구조용 하네스·카라비너, 장갑과 스마트폰을 유지한다. 붉은 열매 수풀과 부러진 가지, 숲바닥, 폐허가 보이고 추가 인물이나 읽을 수 있는 문자는 없다.",
        "hard_violations": [],
        "physics": "토니의 흉부·복부·골반·다리가 바닥에 닿고 굽힌 팔꿈치가 들어 올린 머리와 상체를 지지하므로 낮은 포복 자세가 물리적으로 성립한다. 스마트폰은 오른손이 확실히 쥐고 있다. 바닥의 긴 가지와 수풀의 부러진 가지도 표면 또는 줄기에 지지되어 있다. 다만 다리와 발 일부가 아래·왼쪽 프레임 경계에서 잘려 완전한 전신은 아니다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 1.306
   },
   "adjusted": {
    "A": 2.0,
    "B": 1.306
   },
   "violations": {},
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 1306
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "요청된 와이드 샷과 '전신' 프레이밍을 훌륭하게 구현했으며, 레퍼런스의 장소와 캐릭터 복장(헬멧 포함)을 충실히 반영했습니다."
   },
   {
    "label": "B",
    "score": 1306,
    "verdict_ko": "프레임이 너무 좁아 텍스트가 요구한 '전신'을 보여주지 못했으며, 캐릭터가 헬멧을 착용하지 않았습니다."
   }
  ],
  "refs": [
   {
    "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/episodes/c804efc3-0697-4c22-98b4-6992a70c2b20/images/background_chain/L03B03.png",
    "asset_id": "5094d3b6-9a8e-40d5-9d02-a151027a874e",
    "role": "location_plate"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2b4-881f-75be-8521-a0584e216a94",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S6sh53__bgfirst_bg.png",
   "bg_asset_id": "7d273cbd-b1cd-461a-9fe6-3121ac794bdf",
   "bg_record_key": "S6sh53::bgfirst_bg",
   "chain_winner": true,
   "authority": "plate"
  },
  "ref_mode": "재투영 배경+콘티+엔티티 (2택1: 체인 승)",
  "share_plan": {
   "ref_plan": "background"
  }
 },
 "S6sh53::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:37:11.012576+00:00",
  "fingerprint": "798e13d215cbf2f7e7b2588556799de8921bc3165414d28bc6b1d7df2c66fd01",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S6sh53_sel.png",
  "source_sha256": "57bd2feec5e13139ac9a22dbc6004746a6c992ce3115b6bd81490b4a0250ed17",
  "file": "S6sh53_cine.png",
  "staged_sha256": "d66f2e9688ee5336eebaf0e7203ea4f456a715308700509fec136a4847628127",
  "latency_ms": 15405
 },
 "S7sh1::signage": {
  "fp": "5a643b4c23e9bdf6",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "era_assess::972befbb64d5703c": {
  "subjects": [],
  "subject_text": "거대 수목이 우거진 숲속 능선\n거대한 나무들이 빽빽하게 선 낮의 숲속 능선. 넓게 벌어진 수관과 굵은 가지, 나무 사이의 좁은 골짜기와 덤불이 이어진다.",
  "identity": "canonical",
  "scope_id": "L04",
  "scope_role": "location_exterior",
  "scope_sha": "24ac46c326f3f438"
 },
 "groupbg::숲속_능선매복지": {
  "input_fingerprint": "dd2be03d738b8019",
  "meta": {
   "model": "gpt-image-2",
   "size": "1536x864",
   "pack": "11.202607220237",
   "contract": "bgfirst_full_v3",
   "group_sig": {
    "key": "숲속_능선매복지",
    "tags": [
     "S7sh1",
     "S7sh14",
     "S7sh19",
     "S8sh1",
     "S8sh6",
     "S8sh7"
    ]
   },
   "context_sig": "738d9b3bc780fbc2"
  },
  "prompt": "Create ONE empty live-action location background photograph — NO PEOPLE, no figures, no body parts, no silhouettes, no shadows or reflections of people anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods and produce in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch of a shot that happens at this location: use it ONLY as spatial evidence — what this place contains, how its ground, structures and landmarks are arranged and proportioned. Ignore the sketched people and arrows entirely, and do NOT copy its line style: render a fully photographic, physically plausible real place that fits THE LOCATION text below.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nTHE LOCATION — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated: Outside in an open gap among giant trees along a forested ridge, high enough for a long horizontal leap between trunks.\n\nLOCATION DETAIL (descriptive evidence about this place from the production's location records — evidence, not a staging order): use it to understand what kind of place this is and what it permanently contains. Carry over ONLY its enduring physical features — layout, structures, ground and materials, fixed equipment and installations. Do NOT treat any lighting, weather or time-of-day wording, momentary object states, depicted events or subjective impressions in it as fixed facts of the place; the TIME OF DAY lock below is the sole authority for time and lighting.\n거대 수목이 우거진 숲속 능선: 나무 사이 간격이 넓고 굵은 나뭇가지들이 교차하는 산악 고지대. 지면 덤불 사이로 은폐된 실전용 함정 장치들과 특이한 형태로 파손된 옛 건물 잔해가 함께 존재한다. (특징: 수십 미터 간격으로 떨어진 거대한 수목과 굵은 가지들; 좁은 골짜기 바닥에 깔린 가느다란 금속선과 위장 그물; 표면이 거대한 원통 모양으로 매끄럽고 깔끔하게 도려내진 구조물 잔해; 녹색 계열의 잎과 거친 산악 지면)\n\nSCENE EVIDENCE (verbatim quotes from the screenplay about this place — treat them as evidence of what the location physically contains and looks like; stage the PLACE those moments happen in, but do NOT depict the momentary actions, people or staged props themselves):\n- S7. EXT. 숲속 능선 — 낮\n- 윌마가 정확히 토니의 은신처로 총구를 돌린다.\n- S8. EXT. 숲속 능선 — 낮\n- 토니가 구조용 칼을 바닥에 내려놓고 양손을 든 채 나온다.\n\nTIME OF DAY (lock): day.\n\nRender ONE photorealistic empty location photograph, 16:9, neutral enough that every shot of this place can be staged from it later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "effective_prompt": "Create ONE empty live-action location background photograph — NO PEOPLE, no figures, no body parts, no silhouettes, no shadows or reflections of people anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods and produce in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch of a shot that happens at this location: use it ONLY as spatial evidence — what this place contains, how its ground, structures and landmarks are arranged and proportioned. Ignore the sketched people and arrows entirely, and do NOT copy its line style: render a fully photographic, physically plausible real place that fits THE LOCATION text below.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nTHE LOCATION — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated: Outside in an open gap among giant trees along a forested ridge, high enough for a long horizontal leap between trunks.\n\nLOCATION DETAIL (descriptive evidence about this place from the production's location records — evidence, not a staging order): use it to understand what kind of place this is and what it permanently contains. Carry over ONLY its enduring physical features — layout, structures, ground and materials, fixed equipment and installations. Do NOT treat any lighting, weather or time-of-day wording, momentary object states, depicted events or subjective impressions in it as fixed facts of the place; the TIME OF DAY lock below is the sole authority for time and lighting.\n거대 수목이 우거진 숲속 능선: 나무 사이 간격이 넓고 굵은 나뭇가지들이 교차하는 산악 고지대. 지면 덤불 사이로 은폐된 실전용 함정 장치들과 특이한 형태로 파손된 옛 건물 잔해가 함께 존재한다. (특징: 수십 미터 간격으로 떨어진 거대한 수목과 굵은 가지들; 좁은 골짜기 바닥에 깔린 가느다란 금속선과 위장 그물; 표면이 거대한 원통 모양으로 매끄럽고 깔끔하게 도려내진 구조물 잔해; 녹색 계열의 잎과 거친 산악 지면)\n\nSCENE EVIDENCE (verbatim quotes from the screenplay about this place — treat them as evidence of what the location physically contains and looks like; stage the PLACE those moments happen in, but do NOT depict the momentary actions, people or staged props themselves):\n- S7. EXT. 숲속 능선 — 낮\n- 윌마가 정확히 토니의 은신처로 총구를 돌린다.\n- S8. EXT. 숲속 능선 — 낮\n- 토니가 구조용 칼을 바닥에 내려놓고 양손을 든 채 나온다.\n\nTIME OF DAY (lock): day.\n\nRender ONE photorealistic empty location photograph, 16:9, neutral enough that every shot of this place can be staged from it later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/groupbg_숲속_능선매복지_881053.png",
  "asset_id": "07a4fd67-5a38-4ab0-91f1-5caad760a5e3",
  "input_asset_ids": [
   "17df0856-6b44-4862-8efe-daeac6037a4c"
  ],
  "origin_tag": "S7sh1",
  "place_text": "Outside in an open gap among giant trees along a forested ridge, high enough for a long horizontal leap between trunks.",
  "origin_inputs": {
   "place_text": "Outside in an open gap among giant trees along a forested ridge, high enough for a long horizontal leap between trunks.",
   "time_of_day_en": "day",
   "conti_asset_id": "17df0856-6b44-4862-8efe-daeac6037a4c"
  }
 },
 "S7sh1::bgfirst_bg": {
  "input_fingerprint": "2babf43190f26b46",
  "prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 두 발을 뻗고 숲속 거대한 나무들 사이를 수평으로 도약해 허공에 뜬 mid-action 자세의 '윌마 디어링' 전신, 비행 반대 방향으로 옷자락이 강하게 휘날리는\n\nLOCATION (lock): Outside in an open gap among giant trees along a forested ridge, high enough for a long horizontal leap between trunks.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, midground, moves toward arrival tree at frame right; departure tree in the middle-left of the frame, background; arrival tree in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: departure tree (the tree Wilma has just left) — Its trunk is seen along the left side of the lateral flight path; used as Marks the beginning of the leap and provides scale; arrival tree (the next tree on Wilma's route) — Its trunk faces the camera obliquely at the right end of her trajectory; used as Defines Wilma's destination and movement direction; jumper belt (faintly vibrating beneath Wilma's arms) — Its fitted sides are visible beneath her airborne torso; used as Supplies the only visible mechanism associated with the otherwise unsupported leap.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight gives the airborne figure crisp separation within a restrained forest-green and weathered-neutral palette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "effective_prompt": "Create the EMPTY BACKGROUND PLATE for one film shot — NO PEOPLE, no figures, no body parts, no sketch lines, no arrows anywhere.\n\"Empty\" means no people only: KEEP the location's inherent occupants and stock that define the place — animals in an animal shelter, pen or farm, goods in a market, moored boats in a harbour — unless the shot text explicitly removes them.\nThe FIRST attached image is a thin-line storyboard sketch: use ONLY its camera angle, horizon, perspective and the placement/size of buildings and set masses — ignore the sketched people and arrows entirely. The SECOND attached image (LOCATION PHOTOGRAPH) is the real place: take its architecture, materials, signage and fixed features, and RE-PROJECT them into the sketch's camera. If the photograph's camera differs from the sketch's, the sketch's camera wins.\nHUMAN-SCALE CALIBRATION: derive every structure's true size from human-scale elements — a door ≈ 2m, a window ≈ 1–1.5m wide, one storey ≈ 2.5–3m; never inflate a small structure or shrink a large one.\n\nSHOT TEXT this background must serve (Korean): 두 발을 뻗고 숲속 거대한 나무들 사이를 수평으로 도약해 허공에 뜬 mid-action 자세의 '윌마 디어링' 전신, 비행 반대 방향으로 옷자락이 강하게 휘날리는\n\nLOCATION (lock): Outside in an open gap among giant trees along a forested ridge, high enough for a long horizontal leap between trunks.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, midground, moves toward arrival tree at frame right; departure tree in the middle-left of the frame, background; arrival tree in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: departure tree (the tree Wilma has just left) — Its trunk is seen along the left side of the lateral flight path; used as Marks the beginning of the leap and provides scale; arrival tree (the next tree on Wilma's route) — Its trunk faces the camera obliquely at the right end of her trajectory; used as Defines Wilma's destination and movement direction; jumper belt (faintly vibrating beneath Wilma's arms) — Its fitted sides are visible beneath her airborne torso; used as Supplies the only visible mechanism associated with the otherwise unsupported leap.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight gives the airborne figure crisp separation within a restrained forest-green and weathered-neutral palette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nRender ONE photorealistic empty location photograph, 16:9, that this shot can be staged inside later. No readable writing anywhere: surfaces that would carry writing may be present, but stage any wording out of legibility — an oblique angle, distance, shallow focus. No captions, watermarks or overlay text.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S7sh1__bgfirst_bg.png",
  "asset_id": "c7670afb-e940-409e-9361-faf7e1059820",
  "input_asset_ids": [
   "17df0856-6b44-4862-8efe-daeac6037a4c",
   "07a4fd67-5a38-4ab0-91f1-5caad760a5e3"
  ]
 },
 "S7sh1": {
  "input_fingerprint": "ff9d875bdb290072",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 두 발을 뻗고 숲속 거대한 나무들 사이를 수평으로 도약해 허공에 뜬 mid-action 자세의 '윌마 디어링' 전신, 비행 반대 방향으로 옷자락이 강하게 휘날리는\n\nLOCATION (lock): Outside in an open gap among giant trees along a forested ridge, high enough for a long horizontal leap between trunks. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, midground, moves toward arrival tree at frame right; departure tree in the middle-left of the frame, background; arrival tree in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: departure tree (the tree Wilma has just left) — Its trunk is seen along the left side of the lateral flight path; used as Marks the beginning of the leap and provides scale; arrival tree (the next tree on Wilma's route) — Its trunk faces the camera obliquely at the right end of her trajectory; used as Defines Wilma's destination and movement direction; jumper belt (faintly vibrating beneath Wilma's arms) — Its fitted sides are visible beneath her airborne torso; used as Supplies the only visible mechanism associated with the otherwise unsupported leap.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight gives the airborne figure crisp separation within a restrained forest-green and weathered-neutral palette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The forest ridge is filled with enormous trees, and the smartphone battery remains at 4 percent. Jumper belts emit a faint vibration as the chase crosses the gaps between trees. 윌마 디어링: She is airborne in a long horizontal leap between trees, wearing a faintly vibrating jumper belt. Her clothing streams opposite her direction of travel.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Stage the shot. The FIRST attached image (SHOT BACKGROUND) is the finished empty background of this shot — keep it EXACTLY: its camera, perspective, architecture, lighting and every fixed feature stay untouched. The SECOND attached image (LAYOUT SKETCH) tells you ONLY where the people go: each sketched person's position, screen size, pose and the gaze/motion arrows. Ignore the sketch's background lines. The CHARACTER REFERENCE photographs show the real people.\nPlace the real people into the background at exactly the sketched positions, sizes and poses, following the arrow directions. No sketch lines or arrows may remain.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 두 발을 뻗고 숲속 거대한 나무들 사이를 수평으로 도약해 허공에 뜬 mid-action 자세의 '윌마 디어링' 전신, 비행 반대 방향으로 옷자락이 강하게 휘날리는\n\nLOCATION (lock): Outside in an open gap among giant trees along a forested ridge, high enough for a long horizontal leap between trunks. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, midground, moves toward arrival tree at frame right; departure tree in the middle-left of the frame, background; arrival tree in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: departure tree (the tree Wilma has just left) — Its trunk is seen along the left side of the lateral flight path; used as Marks the beginning of the leap and provides scale; arrival tree (the next tree on Wilma's route) — Its trunk faces the camera obliquely at the right end of her trajectory; used as Defines Wilma's destination and movement direction; jumper belt (faintly vibrating beneath Wilma's arms) — Its fitted sides are visible beneath her airborne torso; used as Supplies the only visible mechanism associated with the otherwise unsupported leap.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight gives the airborne figure crisp separation within a restrained forest-green and weathered-neutral palette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The forest ridge is filled with enormous trees, and the smartphone battery remains at 4 percent. Jumper belts emit a faint vibration as the chase crosses the gaps between trees. 윌마 디어링: She is airborne in a long horizontal leap between trees, wearing a faintly vibrating jumper belt. Her clothing streams opposite her direction of travel.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 두 발을 뻗고 숲속 거대한 나무들 사이를 수평으로 도약해 허공에 뜬 mid-action 자세의 '윌마 디어링' 전신, 비행 반대 방향으로 옷자락이 강하게 휘날리는\n\nLOCATION (lock): Outside in an open gap among giant trees along a forested ridge, high enough for a long horizontal leap between trunks. The shot takes place here — the attached LOCATION PHOTOGRAPH shows the exact spot.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, midground, moves toward arrival tree at frame right; departure tree in the middle-left of the frame, background; arrival tree in the middle-right of the frame, background.\n- KEY BACKGROUND ELEMENTS: departure tree (the tree Wilma has just left) — Its trunk is seen along the left side of the lateral flight path; used as Marks the beginning of the leap and provides scale; arrival tree (the next tree on Wilma's route) — Its trunk faces the camera obliquely at the right end of her trajectory; used as Defines Wilma's destination and movement direction; jumper belt (faintly vibrating beneath Wilma's arms) — Its fitted sides are visible beneath her airborne torso; used as Supplies the only visible mechanism associated with the otherwise unsupported leap.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight gives the airborne figure crisp separation within a restrained forest-green and weathered-neutral palette.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The forest ridge is filled with enormous trees, and the smartphone battery remains at 4 percent. Jumper belts emit a faint vibration as the chase crosses the gaps between trees. 윌마 디어링: She is airborne in a long horizontal leap between trees, wearing a faintly vibrating jumper belt. Her clothing streams opposite her direction of travel.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S7sh1__bgfirst_bg.png",
     "asset_id": "c7670afb-e940-409e-9361-faf7e1059820",
     "role": "bgfirst_bg"
    },
    {
     "label": "LAYOUT SKETCH (people placement only)",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/conti/conti_S7sh1.png",
     "asset_id": "17df0856-6b44-4862-8efe-daeac6037a4c",
     "role": "conti_light"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ],
   "B": [
    {
     "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/groupbg_숲속_능선매복지_881053.png",
     "asset_id": "07a4fd67-5a38-4ab0-91f1-5caad760a5e3",
     "role": "bgfirst_group_bg"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "judge_flip_skipped": "judge_fn_owns_order",
  "readings": [
   {
    "label": "A",
    "direction": "인물은 오른쪽을 향해 양 다리를 뻗으며 도약하고 있습니다.",
    "built_space": "레퍼런스와 동일한 숲속 지형과 나무, 구조물들이 올바르게 배치되어 있습니다.",
    "entities": "윌마 디어링의 얼굴과 점프수트 기본 형태는 일치하나, 레퍼런스에 존재하지 않는 거대한 녹색 망토(옷자락)가 추가되었습니다.",
    "hard_violations": [
     "[gemini-pro] invented objects (레퍼런스에 없는 망토 추가)",
     "[gemini-pro] physically impossible staging (오른발이 우측 나무 기둥 안으로 파묻혀 있음)"
    ],
    "physics": "공중에서 다리를 찢고 비행 중이나, 앞으로 뻗은 발이 목표 나무의 표면을 뚫고 들어가 물리적으로 불가능한 상태입니다."
   },
   {
    "label": "B",
    "direction": "인물은 오른쪽을 향해 시선을 두고 화면 우측의 나무를 향해 도약하고 있습니다.",
    "built_space": "레퍼런스와 동일한 숲속 틈새입니다. 좌측에 출발 나무, 우측에 도착 나무, 우측 하단에 콘크리트 구조물이 정확히 배치되어 있습니다.",
    "entities": "윌마 디어링의 얼굴, 체형, 녹색 점프수트 의상이 레퍼런스와 일치하며, 허리에 착용한 점퍼 벨트의 끈이 뒤로 휘날리고 있습니다.",
    "hard_violations": [],
    "physics": "공중에 떠 있는 도약 자세이며, 프롬프트에 명시된 점퍼 벨트가 비행을 지지하는 것으로 보입니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "지정된 숲속 위치에서 도약하는 인물의 모습을 자연스럽게 구현했으며, 캐릭터의 의상과 외형이 레퍼런스와 일치합니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "레퍼런스에 없는 거대한 망토를 의상에 추가했으며, 앞으로 뻗은 오른발이 나무 기둥에 파묻히는 물리적 오류가 발생했습니다."
       }
      ],
      "readings": [
       {
        "label": "B",
        "direction": "인물은 오른쪽을 향해 시선을 두고 화면 우측의 나무를 향해 도약하고 있습니다.",
        "built_space": "레퍼런스와 동일한 숲속 틈새입니다. 좌측에 출발 나무, 우측에 도착 나무, 우측 하단에 콘크리트 구조물이 정확히 배치되어 있습니다.",
        "entities": "윌마 디어링의 얼굴, 체형, 녹색 점프수트 의상이 레퍼런스와 일치하며, 허리에 착용한 점퍼 벨트의 끈이 뒤로 휘날리고 있습니다.",
        "hard_violations": [],
        "physics": "공중에 떠 있는 도약 자세이며, 프롬프트에 명시된 점퍼 벨트가 비행을 지지하는 것으로 보입니다."
       },
       {
        "label": "A",
        "direction": "인물은 오른쪽을 향해 양 다리를 뻗으며 도약하고 있습니다.",
        "built_space": "레퍼런스와 동일한 숲속 지형과 나무, 구조물들이 올바르게 배치되어 있습니다.",
        "entities": "윌마 디어링의 얼굴과 점프수트 기본 형태는 일치하나, 레퍼런스에 존재하지 않는 거대한 녹색 망토(옷자락)가 추가되었습니다.",
        "hard_violations": [
         "invented objects (레퍼런스에 없는 망토 추가)",
         "physically impossible staging (오른발이 우측 나무 기둥 안으로 파묻혀 있음)"
        ],
        "physics": "공중에서 다리를 찢고 비행 중이나, 앞으로 뻗은 발이 목표 나무의 표면을 뚫고 들어가 물리적으로 불가능한 상태입니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "지정된 숲속 위치에서 도약하는 인물의 모습을 자연스럽게 구현했으며, 캐릭터의 의상과 외형이 레퍼런스와 일치합니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "레퍼런스에 없는 거대한 망토를 의상에 추가했으며, 앞으로 뻗은 오른발이 나무 기둥에 파묻히는 물리적 오류가 발생했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "인물은 오른쪽을 향해 시선을 두고 화면 우측의 나무를 향해 도약하고 있습니다.",
        "built_space": "레퍼런스와 동일한 숲속 틈새입니다. 좌측에 출발 나무, 우측에 도착 나무, 우측 하단에 콘크리트 구조물이 정확히 배치되어 있습니다.",
        "entities": "윌마 디어링의 얼굴, 체형, 녹색 점프수트 의상이 레퍼런스와 일치하며, 허리에 착용한 점퍼 벨트의 끈이 뒤로 휘날리고 있습니다.",
        "hard_violations": [],
        "physics": "공중에 떠 있는 도약 자세이며, 프롬프트에 명시된 점퍼 벨트가 비행을 지지하는 것으로 보입니다."
       },
       {
        "label": "A",
        "direction": "인물은 오른쪽을 향해 양 다리를 뻗으며 도약하고 있습니다.",
        "built_space": "레퍼런스와 동일한 숲속 지형과 나무, 구조물들이 올바르게 배치되어 있습니다.",
        "entities": "윌마 디어링의 얼굴과 점프수트 기본 형태는 일치하나, 레퍼런스에 존재하지 않는 거대한 녹색 망토(옷자락)가 추가되었습니다.",
        "hard_violations": [
         "invented objects (레퍼런스에 없는 망토 추가)",
         "physically impossible staging (오른발이 우측 나무 기둥 안으로 파묻혀 있음)"
        ],
        "physics": "공중에서 다리를 찢고 비행 중이나, 앞으로 뻗은 발이 목표 나무의 표면을 뚫고 들어가 물리적으로 불가능한 상태입니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "오른쪽 도착 나무를 향한 수평 도약, 중앙 배치, 뒤쪽으로 강하게 날리는 머리카락과 옷자락, 점퍼 벨트가 핵심 동작을 가장 충실히 구현했다."
       },
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "오른쪽 나무로 향하는 도약과 장소·복장은 잘 맞지만 상체가 세워진 달리기형 자세이고 옷자락의 강한 역방향 휘날림이 부족하다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마의 얼굴과 시선, 진행 방향, 벌어진 두 다리와 앞쪽 부츠가 모두 프레임 오른쪽의 굵은 도착 나무를 향한다. 두 팔과 벨트 끈은 진행 반대쪽인 왼쪽으로 밀려 있다. 오른발은 도착 나무 앞까지 왔지만 아직 확실히 디디지는 않았다.",
        "built_space": "야외 능선의 열린 숲이다. 출발 지점 역할의 나무줄기가 중간 왼쪽에 있고, 훨씬 굵은 도착 나무줄기가 중간 오른쪽에 있다. 배경의 산등성이, 철망, 케이블, 부서진 콘크리트 원통 등 장소 사진의 고정 요소도 같은 배치로 보인다. 윌마는 두 줄기 사이의 중경 중앙에 있다.",
        "entities": "사람은 윌마 한 명뿐이며 20대 후반으로 보이는 백인 미국인 여성이다. 묶은 금발, 체격, 짙은 녹색 작업복과 부츠가 인물 참고와 잘 맞는다. 허리에는 장치형 점퍼 벨트가 보이지만 진동 자체는 정지 화면에서 명확하지 않다. 스마트폰은 보이지 않으며 이 프레이밍에서는 요구되지 않는다.",
        "hard_violations": [],
        "physics": "두 발 모두 지면에서 떨어져 있고 출발 나무에서 오른쪽 도착 나무로 건너가는 보폭이 형성되어 있다. 허리의 점퍼 벨트가 공중 체공을 가능하게 하는 유일한 가시적 장치이며, 뒤로 뻗은 팔과 왼쪽으로 끌리는 끈이 오른쪽 운동을 뒷받침한다. 다만 상체가 비교적 수직이라 긴 수평 비행보다는 크게 벌린 달리기 점프에 가깝고, 밀착형 작업복 자체는 거의 휘날리지 않는다."
       },
       {
        "label": "B",
        "direction": "윌마의 얼굴과 시선, 앞으로 뻗은 두 손, 몸통, 두 발의 긴 축이 모두 프레임 오른쪽의 굵은 도착 나무를 정확히 향한다. 머리카락과 긴 옷자락은 비행 반대 방향인 왼쪽으로 강하게 흐른다. 앞 부츠는 도착 나무 바로 앞을 향하지만 아직 닿지 않았다.",
        "built_space": "출발 나무는 비행 경로 왼쪽의 중간 왼쪽 배경에, 카메라에 비스듬히 면한 굵은 도착 나무는 중간 오른쪽 배경에 배치되어 있다. 윌마는 그 사이의 중경 중앙에 있다. 산등성이, 철망, 바닥 케이블과 우측 콘크리트 원통을 포함해 장소 사진의 숲과 폐허 요소가 일관되게 유지된다.",
        "entities": "등장 인물은 윌마 한 명뿐이며 젊은 성인 백인 여성으로 보인다. 금발과 얼굴, 체격, 녹색 복장과 부츠는 참고 인물과 대체로 맞고 허리 양옆에 점퍼 벨트가 보인다. 다만 참고의 단정히 묶은 머리와 밀착형 작업복보다 머리 길이와 뒤로 펄럭이는 긴 외피가 더 과장되어 있다. 읽을 수 있는 글자나 추가 인물은 없다.",
        "hard_violations": [],
        "physics": "몸 전체가 거의 수평이고 두 다리가 길게 뻗어 출발점에서 도착점까지 이어지는 도약 궤적을 만든다. 허리를 감싼 점퍼 벨트가 체공을 설명하는 가시적 장치이며, 왼쪽으로 날리는 머리카락과 옷자락이 오른쪽 속도를 물리적으로 읽히게 한다. 뻗은 손과 앞발은 오른쪽 나무를 착지·접촉 지점으로 삼고 있어 출발과 도착이 모두 분명하다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "오른쪽 도착 나무를 향한 수평 도약, 중앙 배치, 뒤쪽으로 강하게 날리는 머리카락과 옷자락, 점퍼 벨트가 핵심 동작을 가장 충실히 구현했다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "오른쪽 나무로 향하는 도약과 장소·복장은 잘 맞지만 상체가 세워진 달리기형 자세이고 옷자락의 강한 역방향 휘날림이 부족하다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "윌마의 얼굴과 시선, 진행 방향, 벌어진 두 다리와 앞쪽 부츠가 모두 프레임 오른쪽의 굵은 도착 나무를 향한다. 두 팔과 벨트 끈은 진행 반대쪽인 왼쪽으로 밀려 있다. 오른발은 도착 나무 앞까지 왔지만 아직 확실히 디디지는 않았다.",
        "built_space": "야외 능선의 열린 숲이다. 출발 지점 역할의 나무줄기가 중간 왼쪽에 있고, 훨씬 굵은 도착 나무줄기가 중간 오른쪽에 있다. 배경의 산등성이, 철망, 케이블, 부서진 콘크리트 원통 등 장소 사진의 고정 요소도 같은 배치로 보인다. 윌마는 두 줄기 사이의 중경 중앙에 있다.",
        "entities": "사람은 윌마 한 명뿐이며 20대 후반으로 보이는 백인 미국인 여성이다. 묶은 금발, 체격, 짙은 녹색 작업복과 부츠가 인물 참고와 잘 맞는다. 허리에는 장치형 점퍼 벨트가 보이지만 진동 자체는 정지 화면에서 명확하지 않다. 스마트폰은 보이지 않으며 이 프레이밍에서는 요구되지 않는다.",
        "hard_violations": [],
        "physics": "두 발 모두 지면에서 떨어져 있고 출발 나무에서 오른쪽 도착 나무로 건너가는 보폭이 형성되어 있다. 허리의 점퍼 벨트가 공중 체공을 가능하게 하는 유일한 가시적 장치이며, 뒤로 뻗은 팔과 왼쪽으로 끌리는 끈이 오른쪽 운동을 뒷받침한다. 다만 상체가 비교적 수직이라 긴 수평 비행보다는 크게 벌린 달리기 점프에 가깝고, 밀착형 작업복 자체는 거의 휘날리지 않는다."
       },
       {
        "label": "A",
        "direction": "윌마의 얼굴과 시선, 앞으로 뻗은 두 손, 몸통, 두 발의 긴 축이 모두 프레임 오른쪽의 굵은 도착 나무를 정확히 향한다. 머리카락과 긴 옷자락은 비행 반대 방향인 왼쪽으로 강하게 흐른다. 앞 부츠는 도착 나무 바로 앞을 향하지만 아직 닿지 않았다.",
        "built_space": "출발 나무는 비행 경로 왼쪽의 중간 왼쪽 배경에, 카메라에 비스듬히 면한 굵은 도착 나무는 중간 오른쪽 배경에 배치되어 있다. 윌마는 그 사이의 중경 중앙에 있다. 산등성이, 철망, 바닥 케이블과 우측 콘크리트 원통을 포함해 장소 사진의 숲과 폐허 요소가 일관되게 유지된다.",
        "entities": "등장 인물은 윌마 한 명뿐이며 젊은 성인 백인 여성으로 보인다. 금발과 얼굴, 체격, 녹색 복장과 부츠는 참고 인물과 대체로 맞고 허리 양옆에 점퍼 벨트가 보인다. 다만 참고의 단정히 묶은 머리와 밀착형 작업복보다 머리 길이와 뒤로 펄럭이는 긴 외피가 더 과장되어 있다. 읽을 수 있는 글자나 추가 인물은 없다.",
        "hard_violations": [],
        "physics": "몸 전체가 거의 수평이고 두 다리가 길게 뻗어 출발점에서 도착점까지 이어지는 도약 궤적을 만든다. 허리를 감싼 점퍼 벨트가 체공을 설명하는 가시적 장치이며, 왼쪽으로 날리는 머리카락과 옷자락이 오른쪽 속도를 물리적으로 읽히게 한다. 뻗은 손과 앞발은 오른쪽 나무를 착지·접촉 지점으로 삼고 있어 출발과 도착이 모두 분명하다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.444,
    "B": 1.778
   },
   "adjusted": {
    "A": 1.194,
    "B": 1.778
   },
   "violations": {
    "A": [
     "[gemini-pro] invented objects (레퍼런스에 없는 망토 추가)",
     "[gemini-pro] physically impossible staging (오른발이 우측 나무 기둥 안으로 파묻혀 있음)"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "agreed": false
  },
  "totals": {
   "B": 1778,
   "A": 1194
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 1778,
    "verdict_ko": "지정된 숲속 위치에서 도약하는 인물의 모습을 자연스럽게 구현했으며, 캐릭터의 의상과 외형이 레퍼런스와 일치합니다."
   },
   {
    "label": "A",
    "score": 1194,
    "verdict_ko": "레퍼런스에 없는 거대한 망토를 의상에 추가했으며, 앞으로 뻗은 오른발이 나무 기둥에 파묻히는 물리적 오류가 발생했습니다.  ★위반: [gemini-pro] invented objects (레퍼런스에 없는 망토 추가) / [gemini-pro] physically impossible staging (오른발이 우측 나무 기둥 안으로 파묻혀 있음)"
   }
  ],
  "refs": [
   {
    "label": "LOCATION PHOTOGRAPH — the exact place of this shot: its architecture, materials, fixed features and lighting mood are spatial truth; stage the moment inside this place. Never copy its camera framing.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/groupbg_숲속_능선매복지_881053.png",
    "asset_id": "07a4fd67-5a38-4ab0-91f1-5caad760a5e3",
    "role": "bgfirst_group_bg"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2c0-68e1-7f08-8c7b-3541e38aee5f",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S7sh1__bgfirst_bg.png",
   "bg_asset_id": "c7670afb-e940-409e-9361-faf7e1059820",
   "bg_record_key": "S7sh1::bgfirst_bg",
   "chain_winner": false,
   "authority": "groupbg",
   "group_key": "숲속_능선매복지",
   "groupbg_asset_id": "07a4fd67-5a38-4ab0-91f1-5caad760a5e3"
  },
  "ref_mode": "그룹 배경+엔티티 (2택1: 무콘티 승)",
  "share_plan": {
   "ref_plan": "background"
  }
 },
 "S7sh1::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:41:25.027656+00:00",
  "fingerprint": "d53479eff44c470ecde28332ef2c55d61fa1b9167d7f860b50b6f966631ee9e8",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S7sh1_sel.png",
  "source_sha256": "1d48908527be3e2e316cdc799e86c1ef2783bd9e9f62ee6cacca8eca299f0c4a",
  "file": "S7sh1_cine.png",
  "staged_sha256": "284b9156d6d06cb9a038e9664484ad8d89d8ebf97ae412c879f6d3940e06b38b",
  "latency_ms": 17196
 },
 "S7sh14::signage": {
  "fp": "6f65e61a08c4316b",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S7sh14": {
  "input_fingerprint": "04adc17a33946b73",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 강렬한 불꽃과 함께 산산조각 나 허공에 흩뿌려진 파편들의 폭발 중간 순간의 '미지(MIDGE)'.\n\nLOCATION (lock): Outside among the trees and undergrowth of the forested ridge, directly in front of an armed pursuer where the small drone explodes midair. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 미지(MIDGE) in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: forest trees behind the impact (visible behind the airborne fragmentation) — Their vertical trunks remain oblique to the firing line; used as Provides depth and scale behind Midge's suspended fragments; oblique firing axis (defined by the shot that strikes Midge) — It runs diagonally through the fragmentation rather than directly into the lens; used as Preserves the physical cause and direction of the impact.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The impact's intense flame briefly drives the contrast against the daylight forest, while the surrounding palette remains restrained and rugged.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): MIDGE is exploding into fragments in midair after being hit by the rocket pistol. A thin metal tripwire remains across the ravine floor, and the triggered camouflage net hangs overhead.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 미지(MIDGE) (손바닥 크기, 소형 비행 드론, 전방 촬영 카메라, 내장 점검등) — wearing: Non-human or non-outfit entity. Uses entity reference image as-is. — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 강렬한 불꽃과 함께 산산조각 나 허공에 흩뿌려진 파편들의 폭발 중간 순간의 '미지(MIDGE)'.\n\nLOCATION (lock): Outside among the trees and undergrowth of the forested ridge, directly in front of an armed pursuer where the small drone explodes midair. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 미지(MIDGE) in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: forest trees behind the impact (visible behind the airborne fragmentation) — Their vertical trunks remain oblique to the firing line; used as Provides depth and scale behind Midge's suspended fragments; oblique firing axis (defined by the shot that strikes Midge) — It runs diagonally through the fragmentation rather than directly into the lens; used as Preserves the physical cause and direction of the impact.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The impact's intense flame briefly drives the contrast against the daylight forest, while the surrounding palette remains restrained and rugged.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): MIDGE is exploding into fragments in midair after being hit by the rocket pistol. A thin metal tripwire remains across the ravine floor, and the triggered camouflage net hangs overhead.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 미지(MIDGE) (손바닥 크기, 소형 비행 드론, 전방 촬영 카메라, 내장 점검등) — wearing: Non-human or non-outfit entity. Uses entity reference image as-is. — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 강렬한 불꽃과 함께 산산조각 나 허공에 흩뿌려진 파편들의 폭발 중간 순간의 '미지(MIDGE)'.\n\nLOCATION (lock): Outside among the trees and undergrowth of the forested ridge, directly in front of an armed pursuer where the small drone explodes midair. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 미지(MIDGE) in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: forest trees behind the impact (visible behind the airborne fragmentation) — Their vertical trunks remain oblique to the firing line; used as Provides depth and scale behind Midge's suspended fragments; oblique firing axis (defined by the shot that strikes Midge) — It runs diagonally through the fragmentation rather than directly into the lens; used as Preserves the physical cause and direction of the impact.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: The impact's intense flame briefly drives the contrast against the daylight forest, while the surrounding palette remains restrained and rugged.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): MIDGE is exploding into fragments in midair after being hit by the rocket pistol. A thin metal tripwire remains across the ravine floor, and the triggered camouflage net hangs overhead.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 미지(MIDGE) (손바닥 크기, 소형 비행 드론, 전방 촬영 카메라, 내장 점검등) — wearing: Non-human or non-outfit entity. Uses entity reference image as-is. — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "왼쪽 아래에서 화면 중앙의 드론 폭발 지점을 향해 날아온 발사체의 연기 궤적이 사선으로 분명하게 향하고 있음.",
    "built_space": "숲 속 배경의 나무 기둥들이 보이며, 왼쪽 뒤편에 위장망이 걸려 있음. 클로즈업 프레이밍으로 인해 레퍼런스의 전체적인 지형(콘크리트 링, 먼 산)은 시야에서 배제됨.",
    "entities": "중앙에 공중에서 화염과 함께 폭발하며 산산조각 나는 소형 드론 '미지(MIDGE)'가 있으며, 명시된 전면 촬영 카메라 렌즈와 내부 부품들이 뚜렷하게 확인됨.",
    "hard_violations": [],
    "physics": "드론이 공중에서 피격당해 폭발하는 찰나의 순간이 물리적으로 설득력 있게 표현되어 있으며, 파편들이 허공에 자연스럽게 흩뿌려지고 있음."
   },
   {
    "label": "B",
    "direction": "화면 중앙의 폭발 지점에서 오른쪽 나무 기둥을 향해 불꽃과 연기의 궤적이 가로지르고 있음.",
    "built_space": "레퍼런스 이미지와 동일한 와이드 앵글로 숲의 전경, 우측의 큰 나무와 콘크리트 링, 바닥의 인계철선, 상단에 걸린 위장망, 멀리 보이는 산맥까지 모두 그대로 담아냄.",
    "entities": "중앙에서 파편들이 폭발하고 있으나, 드론 본연의 형태나 전면 카메라 같은 디테일은 거의 식별할 수 없고 단순한 잔해물로 보임.",
    "hard_violations": [
     "[gemini-pro] 샷 스케일 위반: 프롬프트에서 명시적으로 요구한 '클로즈업' 프레이밍을 무시하고 레퍼런스 이미지의 와이드 샷 프레이밍을 그대로 복사함."
    ],
    "physics": "공중에서 잔해들이 폭발하고 파편이 튀고 있으나, 타격의 원인이 되는 사선 궤적의 방향성이 모호함."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "요구된 '클로즈업' 샷 스케일을 정확히 준수하여 공중에서 폭발하는 드론(전면 카메라 렌즈 포함)과 사선 방향의 피격 궤적을 생생하게 잘 구현했습니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "레퍼런스 이미지의 카메라 프레이밍을 절대 복사하지 말라는 지시와 '클로즈업' 샷 스케일 요구를 완전히 무시한 채 와이드 샷으로 렌더링하는 치명적인 오류를 범했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "왼쪽 아래에서 화면 중앙의 드론 폭발 지점을 향해 날아온 발사체의 연기 궤적이 사선으로 분명하게 향하고 있음.",
        "built_space": "숲 속 배경의 나무 기둥들이 보이며, 왼쪽 뒤편에 위장망이 걸려 있음. 클로즈업 프레이밍으로 인해 레퍼런스의 전체적인 지형(콘크리트 링, 먼 산)은 시야에서 배제됨.",
        "entities": "중앙에 공중에서 화염과 함께 폭발하며 산산조각 나는 소형 드론 '미지(MIDGE)'가 있으며, 명시된 전면 촬영 카메라 렌즈와 내부 부품들이 뚜렷하게 확인됨.",
        "hard_violations": [],
        "physics": "드론이 공중에서 피격당해 폭발하는 찰나의 순간이 물리적으로 설득력 있게 표현되어 있으며, 파편들이 허공에 자연스럽게 흩뿌려지고 있음."
       },
       {
        "label": "B",
        "direction": "화면 중앙의 폭발 지점에서 오른쪽 나무 기둥을 향해 불꽃과 연기의 궤적이 가로지르고 있음.",
        "built_space": "레퍼런스 이미지와 동일한 와이드 앵글로 숲의 전경, 우측의 큰 나무와 콘크리트 링, 바닥의 인계철선, 상단에 걸린 위장망, 멀리 보이는 산맥까지 모두 그대로 담아냄.",
        "entities": "중앙에서 파편들이 폭발하고 있으나, 드론 본연의 형태나 전면 카메라 같은 디테일은 거의 식별할 수 없고 단순한 잔해물로 보임.",
        "hard_violations": [
         "샷 스케일 위반: 프롬프트에서 명시적으로 요구한 '클로즈업' 프레이밍을 무시하고 레퍼런스 이미지의 와이드 샷 프레이밍을 그대로 복사함."
        ],
        "physics": "공중에서 잔해들이 폭발하고 파편이 튀고 있으나, 타격의 원인이 되는 사선 궤적의 방향성이 모호함."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "요구된 '클로즈업' 샷 스케일을 정확히 준수하여 공중에서 폭발하는 드론(전면 카메라 렌즈 포함)과 사선 방향의 피격 궤적을 생생하게 잘 구현했습니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "레퍼런스 이미지의 카메라 프레이밍을 절대 복사하지 말라는 지시와 '클로즈업' 샷 스케일 요구를 완전히 무시한 채 와이드 샷으로 렌더링하는 치명적인 오류를 범했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "왼쪽 아래에서 화면 중앙의 드론 폭발 지점을 향해 날아온 발사체의 연기 궤적이 사선으로 분명하게 향하고 있음.",
        "built_space": "숲 속 배경의 나무 기둥들이 보이며, 왼쪽 뒤편에 위장망이 걸려 있음. 클로즈업 프레이밍으로 인해 레퍼런스의 전체적인 지형(콘크리트 링, 먼 산)은 시야에서 배제됨.",
        "entities": "중앙에 공중에서 화염과 함께 폭발하며 산산조각 나는 소형 드론 '미지(MIDGE)'가 있으며, 명시된 전면 촬영 카메라 렌즈와 내부 부품들이 뚜렷하게 확인됨.",
        "hard_violations": [],
        "physics": "드론이 공중에서 피격당해 폭발하는 찰나의 순간이 물리적으로 설득력 있게 표현되어 있으며, 파편들이 허공에 자연스럽게 흩뿌려지고 있음."
       },
       {
        "label": "B",
        "direction": "화면 중앙의 폭발 지점에서 오른쪽 나무 기둥을 향해 불꽃과 연기의 궤적이 가로지르고 있음.",
        "built_space": "레퍼런스 이미지와 동일한 와이드 앵글로 숲의 전경, 우측의 큰 나무와 콘크리트 링, 바닥의 인계철선, 상단에 걸린 위장망, 멀리 보이는 산맥까지 모두 그대로 담아냄.",
        "entities": "중앙에서 파편들이 폭발하고 있으나, 드론 본연의 형태나 전면 카메라 같은 디테일은 거의 식별할 수 없고 단순한 잔해물로 보임.",
        "hard_violations": [
         "샷 스케일 위반: 프롬프트에서 명시적으로 요구한 '클로즈업' 프레이밍을 무시하고 레퍼런스 이미지의 와이드 샷 프레이밍을 그대로 복사함."
        ],
        "physics": "공중에서 잔해들이 폭발하고 파편이 튀고 있으나, 타격의 원인이 되는 사선 궤적의 방향성이 모호함."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "요구된 클로즈업과 중앙 전경 배치, 대각선으로 미지에 꽂히는 타격축, 공중 폭발과 파편의 순간을 가장 정확히 구현했다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "타격 방향과 폭발 순간 및 장소 연속성은 맞지만, 클로즈업이 아니라 숲 바닥과 구조물을 넓게 담은 와이드 프레이밍이라 핵심 구도 지시를 크게 어겼다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "오른쪽 화면 밖에서 시작한 밝은 발사 궤적이 오른쪽 위에서 중앙의 폭발 지점으로 비스듬히 들어가며 미지의 파편화 중심에 닿는다. 파편은 폭발 중심에서 사방으로 퍼지고 있다.",
        "built_space": "야외 숲 능선의 넓은 바닥, 여러 수직 나무줄기, 화면 위에 처진 위장망 1개, 바닥을 가로지르는 가는 금속선, 오른쪽 가장자리의 콘크리트 원통 구조물이 보인다. 이전 장면의 장소와 고정 요소는 대체로 이어지지만, 바닥과 구조물을 과도하게 많이 보여 클로즈업에 맞지 않는다.",
        "entities": "사람은 없다. 중앙에는 불꽃과 함께 금속·전자 부품으로 보이는 파편들이 있으나 미지의 전방 카메라나 소형 드론 형상이 뚜렷하게 식별되지는 않는다. 위장망과 바닥의 가는 트립와이어는 보이며 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "파편들은 중앙 폭발의 충격으로 방사상 비산하는 것으로 읽혀 명확한 운동 원인이 있다. 발사 궤적도 폭발 중심에 연결된다. 위장망은 나무와 케이블에 매달려 있고 트립와이어는 지면 가까이 고정되어 있어 떠 있는 물체는 없다."
       },
       {
        "label": "B",
        "direction": "왼쪽 아래 화면 밖에서 들어온 연기 섞인 발사 궤적이 대각선으로 중앙의 미지 폭발부에 정확히 닿는다. 카메라 모듈 주변 파편과 불꽃은 충돌점을 중심으로 사방으로 뻗는다.",
        "built_space": "가까이 당긴 숲속 화면으로, 뒤쪽에 여러 나무줄기가 깊이감 있게 서 있고 왼쪽에는 케이블에 매달린 위장망 1개가 보인다. 숲 바닥과 트립와이어는 올바른 클로즈업 때문에 프레임 밖이다. 나무줄기는 발사선과 비스듬한 관계를 이루며 장소의 수목과 주광 상태도 이어진다.",
        "entities": "사람은 없다. 중앙 전경에는 미지의 전방 촬영 카메라 모듈이 선명하게 남아 있고 그 주위의 소형 드론 외피, 배선, 금속 부품이 폭발하며 분해되고 있다. 위장망이 보이고 읽을 수 있는 글자나 로고는 없다.",
        "hard_violations": [],
        "physics": "중앙의 화염과 압력파가 드론 외피와 배선, 금속 조각을 바깥으로 밀어내는 명확한 비산 원인이다. 남은 카메라 모듈도 폭발 중인 기체 구조에 붙어 있으며, 파편들이 이유 없이 정지하거나 떠 있는 것으로 보이지 않는다. 위장망은 케이블로 지지된다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "요구된 클로즈업과 중앙 전경 배치, 대각선으로 미지에 꽂히는 타격축, 공중 폭발과 파편의 순간을 가장 정확히 구현했다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "타격 방향과 폭발 순간 및 장소 연속성은 맞지만, 클로즈업이 아니라 숲 바닥과 구조물을 넓게 담은 와이드 프레이밍이라 핵심 구도 지시를 크게 어겼다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "오른쪽 화면 밖에서 시작한 밝은 발사 궤적이 오른쪽 위에서 중앙의 폭발 지점으로 비스듬히 들어가며 미지의 파편화 중심에 닿는다. 파편은 폭발 중심에서 사방으로 퍼지고 있다.",
        "built_space": "야외 숲 능선의 넓은 바닥, 여러 수직 나무줄기, 화면 위에 처진 위장망 1개, 바닥을 가로지르는 가는 금속선, 오른쪽 가장자리의 콘크리트 원통 구조물이 보인다. 이전 장면의 장소와 고정 요소는 대체로 이어지지만, 바닥과 구조물을 과도하게 많이 보여 클로즈업에 맞지 않는다.",
        "entities": "사람은 없다. 중앙에는 불꽃과 함께 금속·전자 부품으로 보이는 파편들이 있으나 미지의 전방 카메라나 소형 드론 형상이 뚜렷하게 식별되지는 않는다. 위장망과 바닥의 가는 트립와이어는 보이며 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "파편들은 중앙 폭발의 충격으로 방사상 비산하는 것으로 읽혀 명확한 운동 원인이 있다. 발사 궤적도 폭발 중심에 연결된다. 위장망은 나무와 케이블에 매달려 있고 트립와이어는 지면 가까이 고정되어 있어 떠 있는 물체는 없다."
       },
       {
        "label": "A",
        "direction": "왼쪽 아래 화면 밖에서 들어온 연기 섞인 발사 궤적이 대각선으로 중앙의 미지 폭발부에 정확히 닿는다. 카메라 모듈 주변 파편과 불꽃은 충돌점을 중심으로 사방으로 뻗는다.",
        "built_space": "가까이 당긴 숲속 화면으로, 뒤쪽에 여러 나무줄기가 깊이감 있게 서 있고 왼쪽에는 케이블에 매달린 위장망 1개가 보인다. 숲 바닥과 트립와이어는 올바른 클로즈업 때문에 프레임 밖이다. 나무줄기는 발사선과 비스듬한 관계를 이루며 장소의 수목과 주광 상태도 이어진다.",
        "entities": "사람은 없다. 중앙 전경에는 미지의 전방 촬영 카메라 모듈이 선명하게 남아 있고 그 주위의 소형 드론 외피, 배선, 금속 부품이 폭발하며 분해되고 있다. 위장망이 보이고 읽을 수 있는 글자나 로고는 없다.",
        "hard_violations": [],
        "physics": "중앙의 화염과 압력파가 드론 외피와 배선, 금속 조각을 바깥으로 밀어내는 명확한 비산 원인이다. 남은 카메라 모듈도 폭발 중인 기체 구조에 붙어 있으며, 파편들이 이유 없이 정지하거나 떠 있는 것으로 보이지 않는다. 위장망은 케이블로 지지된다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 0.889
   },
   "adjusted": {
    "A": 2.0,
    "B": 0.639
   },
   "violations": {
    "B": [
     "[gemini-pro] 샷 스케일 위반: 프롬프트에서 명시적으로 요구한 '클로즈업' 프레이밍을 무시하고 레퍼런스 이미지의 와이드 샷 프레이밍을 그대로 복사함."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 639
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "요구된 '클로즈업' 샷 스케일을 정확히 준수하여 공중에서 폭발하는 드론(전면 카메라 렌즈 포함)과 사선 방향의 피격 궤적을 생생하게 잘 구현했습니다."
   },
   {
    "label": "B",
    "score": 639,
    "verdict_ko": "레퍼런스 이미지의 카메라 프레이밍을 절대 복사하지 말라는 지시와 '클로즈업' 샷 스케일 요구를 완전히 무시한 채 와이드 샷으로 렌더링하는 치명적인 오류를 범했습니다.  ★위반: [gemini-pro] 샷 스케일 위반: 프롬프트에서 명시적으로 요구한 '클로즈업' 프레이밍을 무시하고 레퍼런스 이미지의 와이드 샷 프레이밍을 그대로 복사함."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S7sh1_sel.png",
    "asset_id": "79a71e16-413f-4f77-8ddf-e3e84e5529ed",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2d0-65d2-7f4c-a750-83b067cdd007",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S7sh1"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S7sh14::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:42:23.275170+00:00",
  "fingerprint": "63fb6e3d4056ab608cdf41b7235a725bb9819a79a0975e87c19869a348744b96",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S7sh14_sel.png",
  "source_sha256": "8e03dfcd4532ec8273c01aafdc0181beafabfdee05caf11b4951bd2e0f461849",
  "file": "S7sh14_cine.png",
  "staged_sha256": "7cd0a07636f0bf16e9c7cefca0b93605b9a1382147327c948db8d6d80afdee9f",
  "latency_ms": 16381
 },
 "S7sh19::signage": {
  "fp": "f1393c61a0bc8521",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S7sh19": {
  "input_fingerprint": "a2982b461e0a9893",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 은신처 안의 '토니(앤서니 로저스)' 얼굴을 정면으로 겨냥한 검은 총구와 그 뒤편 '윌마 디어링'의 차가운 눈빛.\n\nLOCATION (lock): Outside at a concealed patch of undergrowth on the forested ridge, facing the surrounding trees where the gunwoman has located the hiding place. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, foreground, looks toward 윌마 디어링; 윌마 디어링 in the middle-center of the frame, midground, looks toward 토니(앤서니 로저스); hiding-place opening in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: gun muzzle and barrel (leveled directly at Tony's face) — The muzzle faces toward Tony and passes obliquely beside the camera, with Wilma aligned behind the barrel; used as Near-foreground threat line connecting the two characters; hiding-place opening (open toward Wilma's position) — Its near edge frames Tony while the outward opening reveals Wilma; used as Creates a through-object frame and separates foreground concealment from the confrontation outside; forest beyond the hiding place (visible behind Wilma); used as Supplies limited spatial context without competing with the opponent-to-opponent axis.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight entering the hiding-place opening creates controlled low-to-mid-key separation between the near muzzle, Tony's face edge, and Wilma beyond.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): MIDGE’s fragments remain scattered around the ravine, and the tripwire and sprung camouflage net remain in place. The second pursuer’s dislodged gun lies at the fight site. 윌마 디어링: She stands armed with her gun aimed directly into the concealed position. Her jumper belt remains fastened. 토니(앤서니 로저스): He remains inside the concealment with his rescue harness and surviving equipment, facing the gun barrel aimed at his position.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마 디어링 right now, so 윌마 디어링's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마 디어링: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 은신처 안의 '토니(앤서니 로저스)' 얼굴을 정면으로 겨냥한 검은 총구와 그 뒤편 '윌마 디어링'의 차가운 눈빛.\n\nLOCATION (lock): Outside at a concealed patch of undergrowth on the forested ridge, facing the surrounding trees where the gunwoman has located the hiding place. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, foreground, looks toward 윌마 디어링; 윌마 디어링 in the middle-center of the frame, midground, looks toward 토니(앤서니 로저스); hiding-place opening in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: gun muzzle and barrel (leveled directly at Tony's face) — The muzzle faces toward Tony and passes obliquely beside the camera, with Wilma aligned behind the barrel; used as Near-foreground threat line connecting the two characters; hiding-place opening (open toward Wilma's position) — Its near edge frames Tony while the outward opening reveals Wilma; used as Creates a through-object frame and separates foreground concealment from the confrontation outside; forest beyond the hiding place (visible behind Wilma); used as Supplies limited spatial context without competing with the opponent-to-opponent axis.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight entering the hiding-place opening creates controlled low-to-mid-key separation between the near muzzle, Tony's face edge, and Wilma beyond.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): MIDGE’s fragments remain scattered around the ravine, and the tripwire and sprung camouflage net remain in place. The second pursuer’s dislodged gun lies at the fight site. 윌마 디어링: She stands armed with her gun aimed directly into the concealed position. Her jumper belt remains fastened. 토니(앤서니 로저스): He remains inside the concealment with his rescue harness and surviving equipment, facing the gun barrel aimed at his position.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마 디어링 right now, so 윌마 디어링's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마 디어링: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 은신처 안의 '토니(앤서니 로저스)' 얼굴을 정면으로 겨냥한 검은 총구와 그 뒤편 '윌마 디어링'의 차가운 눈빛.\n\nLOCATION (lock): Outside at a concealed patch of undergrowth on the forested ridge, facing the surrounding trees where the gunwoman has located the hiding place. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 토니(앤서니 로저스) in the middle-left of the frame, foreground, looks toward 윌마 디어링; 윌마 디어링 in the middle-center of the frame, midground, looks toward 토니(앤서니 로저스); hiding-place opening in the middle-center of the frame, midground.\n- KEY BACKGROUND ELEMENTS: gun muzzle and barrel (leveled directly at Tony's face) — The muzzle faces toward Tony and passes obliquely beside the camera, with Wilma aligned behind the barrel; used as Near-foreground threat line connecting the two characters; hiding-place opening (open toward Wilma's position) — Its near edge frames Tony while the outward opening reveals Wilma; used as Creates a through-object frame and separates foreground concealment from the confrontation outside; forest beyond the hiding place (visible behind Wilma); used as Supplies limited spatial context without competing with the opponent-to-opponent axis.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight entering the hiding-place opening creates controlled low-to-mid-key separation between the near muzzle, Tony's face edge, and Wilma beyond.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): MIDGE’s fragments remain scattered around the ravine, and the tripwire and sprung camouflage net remain in place. The second pursuer’s dislodged gun lies at the fight site. 윌마 디어링: She stands armed with her gun aimed directly into the concealed position. Her jumper belt remains fastened. 토니(앤서니 로저스): He remains inside the concealment with his rescue harness and surviving equipment, facing the gun barrel aimed at his position.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마 디어링 right now, so 윌마 디어링's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마 디어링: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "윌마의 시선과 총구가 토니의 얼굴 쪽을 겨냥하고 있으나, 총기가 위치한 공간적 깊이가 심하게 왜곡되어 있습니다.",
    "built_space": "나무와 덤불로 이루어진 숲 환경이지만, 이전 샷에서 요구된 은신처의 구조물(예: 위장막)이 나타나지 않았습니다.",
    "entities": "토니와 윌마의 얼굴 및 복장은 레퍼런스와 잘 일치합니다. 윌마가 쥐고 있는 총기도 레퍼런스의 디자인과 유사합니다.",
    "hard_violations": [
     "[gemini-pro] 물리적으로 불가능한 해부학 및 원근법 (중경에 있는 인물의 어깨에서 이어질 수 없는 거대한 크기의 손과 총이 전경에 떠 있음)"
    ],
    "physics": "전경에 배치된 거대한 손과 총은 중경에 있는 윌마의 신체와 물리적으로 연결될 수 없는 위치와 크기를 가지고 있어 해부학적 타당성이 전혀 없습니다."
   },
   {
    "label": "B",
    "direction": "윌마의 시선과 그녀가 실제로 들고 있는 소형 총기는 토니를 향하고 있습니다. 그러나 화면 중간을 가로지르는 검은 원통형 물체는 특정 대상을 겨냥하지 않은 채 허공에 놓여 있습니다.",
    "built_space": "흙더미와 위장막(camo net)이 배치되어 있어 이전 샷에서 제시된 은신처의 설정과 공간적 연속성을 잘 유지하고 있습니다.",
    "entities": "토니와 윌마의 인물 특성 및 복장이 레퍼런스와 일치하며, 윌마가 쥐고 있는 총기 역시 레퍼런스와 일치합니다.",
    "hard_violations": [
     "[gemini-pro] 지지대 없이 공중에 떠 있는 정체불명의 거대한 검은색 원통형 물체 (물리 법칙 위반 및 프롬프트에 없는 물체 생성)",
     "[gpt] 윌마가 손에 쥔 총기 외에 화면 오른쪽에서 별도의 긴 총열이 추가로 등장해 총기/총열이 중복되었다.",
     "[gpt] 화면을 가로지르는 긴 검은 총열은 손, 총몸, 고정대 등 어떤 지지도 없이 떠 있어 물리적으로 성립하지 않는다."
    ],
    "physics": "윌마는 양손으로 자신의 총을 지탱하고 있으나, 전경을 가로지르는 거대한 검은색 총열(또는 파이프)은 누구의 손에도 쥐어져 있지 않고 지지대 없이 허공에 수평으로 떠 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 0,
        "verdict_ko": "이전 샷의 위장막과 은신처 배경을 잘 구현했으나, 프롬프트의 전경 총열 지시를 오인하여 아무 지지대 없이 화면을 가로지르는 거대한 검은색 원통을 허공에 생성한 치명적인 결함이 있습니다."
       },
       {
        "label": "A",
        "score": 0,
        "verdict_ko": "중경에 위치한 윌마의 몸집과 전혀 맞지 않게 거대한 손과 총이 전경에 떠 있어, 물리적으로 불가능한 해부학과 원근법 왜곡을 보여줍니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마의 시선과 총구가 토니의 얼굴 쪽을 겨냥하고 있으나, 총기가 위치한 공간적 깊이가 심하게 왜곡되어 있습니다.",
        "built_space": "나무와 덤불로 이루어진 숲 환경이지만, 이전 샷에서 요구된 은신처의 구조물(예: 위장막)이 나타나지 않았습니다.",
        "entities": "토니와 윌마의 얼굴 및 복장은 레퍼런스와 잘 일치합니다. 윌마가 쥐고 있는 총기도 레퍼런스의 디자인과 유사합니다.",
        "hard_violations": [
         "물리적으로 불가능한 해부학 및 원근법 (중경에 있는 인물의 어깨에서 이어질 수 없는 거대한 크기의 손과 총이 전경에 떠 있음)"
        ],
        "physics": "전경에 배치된 거대한 손과 총은 중경에 있는 윌마의 신체와 물리적으로 연결될 수 없는 위치와 크기를 가지고 있어 해부학적 타당성이 전혀 없습니다."
       },
       {
        "label": "B",
        "direction": "윌마의 시선과 그녀가 실제로 들고 있는 소형 총기는 토니를 향하고 있습니다. 그러나 화면 중간을 가로지르는 검은 원통형 물체는 특정 대상을 겨냥하지 않은 채 허공에 놓여 있습니다.",
        "built_space": "흙더미와 위장막(camo net)이 배치되어 있어 이전 샷에서 제시된 은신처의 설정과 공간적 연속성을 잘 유지하고 있습니다.",
        "entities": "토니와 윌마의 인물 특성 및 복장이 레퍼런스와 일치하며, 윌마가 쥐고 있는 총기 역시 레퍼런스와 일치합니다.",
        "hard_violations": [
         "지지대 없이 공중에 떠 있는 정체불명의 거대한 검은색 원통형 물체 (물리 법칙 위반 및 프롬프트에 없는 물체 생성)"
        ],
        "physics": "윌마는 양손으로 자신의 총을 지탱하고 있으나, 전경을 가로지르는 거대한 검은색 총열(또는 파이프)은 누구의 손에도 쥐어져 있지 않고 지지대 없이 허공에 수평으로 떠 있습니다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 0,
        "verdict_ko": "이전 샷의 위장막과 은신처 배경을 잘 구현했으나, 프롬프트의 전경 총열 지시를 오인하여 아무 지지대 없이 화면을 가로지르는 거대한 검은색 원통을 허공에 생성한 치명적인 결함이 있습니다."
       },
       {
        "label": "A",
        "score": 0,
        "verdict_ko": "중경에 위치한 윌마의 몸집과 전혀 맞지 않게 거대한 손과 총이 전경에 떠 있어, 물리적으로 불가능한 해부학과 원근법 왜곡을 보여줍니다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "A",
        "direction": "윌마의 시선과 총구가 토니의 얼굴 쪽을 겨냥하고 있으나, 총기가 위치한 공간적 깊이가 심하게 왜곡되어 있습니다.",
        "built_space": "나무와 덤불로 이루어진 숲 환경이지만, 이전 샷에서 요구된 은신처의 구조물(예: 위장막)이 나타나지 않았습니다.",
        "entities": "토니와 윌마의 얼굴 및 복장은 레퍼런스와 잘 일치합니다. 윌마가 쥐고 있는 총기도 레퍼런스의 디자인과 유사합니다.",
        "hard_violations": [
         "물리적으로 불가능한 해부학 및 원근법 (중경에 있는 인물의 어깨에서 이어질 수 없는 거대한 크기의 손과 총이 전경에 떠 있음)"
        ],
        "physics": "전경에 배치된 거대한 손과 총은 중경에 있는 윌마의 신체와 물리적으로 연결될 수 없는 위치와 크기를 가지고 있어 해부학적 타당성이 전혀 없습니다."
       },
       {
        "label": "B",
        "direction": "윌마의 시선과 그녀가 실제로 들고 있는 소형 총기는 토니를 향하고 있습니다. 그러나 화면 중간을 가로지르는 검은 원통형 물체는 특정 대상을 겨냥하지 않은 채 허공에 놓여 있습니다.",
        "built_space": "흙더미와 위장막(camo net)이 배치되어 있어 이전 샷에서 제시된 은신처의 설정과 공간적 연속성을 잘 유지하고 있습니다.",
        "entities": "토니와 윌마의 인물 특성 및 복장이 레퍼런스와 일치하며, 윌마가 쥐고 있는 총기 역시 레퍼런스와 일치합니다.",
        "hard_violations": [
         "지지대 없이 공중에 떠 있는 정체불명의 거대한 검은색 원통형 물체 (물리 법칙 위반 및 프롬프트에 없는 물체 생성)"
        ],
        "physics": "윌마는 양손으로 자신의 총을 지탱하고 있으나, 전경을 가로지르는 거대한 검은색 총열(또는 파이프)은 누구의 손에도 쥐어져 있지 않고 지지대 없이 허공에 수평으로 떠 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "토니의 얼굴을 향한 단일 총구, 그 뒤에 정렬된 윌마의 차가운 시선, 은신처 개구부를 통한 근접 구도를 가장 충실히 구현했다."
       },
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "인물과 숲은 대체로 맞지만 윌마가 쥔 총과 별개인 거대한 총열이 지지 없이 화면을 가로질러, 핵심 위협선과 물리적 staging이 붕괴한다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니는 화면 오른쪽의 윌마 쪽을 곁눈질하고 윌마도 토니를 응시한다. 윌마가 손에 든 소형 총은 토니 쪽을 겨누지만, 동시에 화면 오른쪽 밖에서 나온 별도의 긴 검은 총열이 토니의 코와 얼굴 높이를 향해 수평으로 뻗어 있다. 따라서 어느 총이 권위 있는 위협선인지 모호하다.",
        "built_space": "나뭇가지·뿌리·흙과 위장망으로 둘러싸인 개구부가 하나 보인다. 토니는 개구부 안쪽 왼쪽 전경에 앉거나 웅크리고 있고, 윌마는 개구부 밖 중앙 중경에 서 있으며 뒤에는 낮의 숲이 보인다. 공간 배치는 대체로 성립하지만, 개구부를 가로지르는 긴 총열의 발사자나 고정점은 보이지 않는다.",
        "entities": "토니와 윌마 두 사람만 있으며 얼굴·성별·연령대·머리색과 복장은 각 참고 이미지에 대체로 부합한다. 토니의 헬멧과 장비, 윌마의 녹색 점프수트와 잠긴 벨트가 보인다. 윌마가 쥔 총은 미래형 권총 계열이지만 참고 총기와 세부 형상이 다소 다르고, 그와 별개인 거대한 검은 총열이 추가되어 있다.",
        "hard_violations": [
         "윌마가 손에 쥔 총기 외에 화면 오른쪽에서 별도의 긴 총열이 추가로 등장해 총기/총열이 중복되었다.",
         "화면을 가로지르는 긴 검은 총열은 손, 총몸, 고정대 등 어떤 지지도 없이 떠 있어 물리적으로 성립하지 않는다."
        ],
        "physics": "토니의 몸은 은신처 바닥과 가장자리에 기대고 있으며 장갑 낀 팔도 흙 가장자리에 놓여 있어 지지가 보인다. 윌마는 두 손으로 자신의 소형 총을 잡아 조준한다. 그러나 얼굴 앞을 가로지르는 별도의 긴 총열에는 이를 붙드는 손이나 연결된 총몸이 전혀 없어 아무것도 지지하지 않는다."
       },
       {
        "label": "B",
        "direction": "윌마의 눈은 화면 왼쪽 전경의 토니를 차갑게 응시한다. 윌마가 든 총의 총구도 토니의 왼쪽 얼굴·관자 부근을 직접 겨누며, 총구 뒤에 윌마의 얼굴이 정렬된다. 토니의 시선은 오른쪽의 윌마 방향이지만 약간 아래로 비껴 있어 완전한 눈맞춤은 아니다.",
        "built_space": "굵은 나뭇가지와 잎, 위장망이 만든 은신처 개구부 하나가 화면 중앙을 둘러싼다. 토니는 개구부 안쪽 왼쪽 전경, 윌마는 바깥 중앙 중경에 서 있고, 개구부 뒤로 낮의 숲과 나무들이 제한적으로 보인다. 별도의 건축 고정물이나 불가능한 반사는 없으며 인물 위치와 개구부 방향이 맞는다.",
        "entities": "토니와 윌마 두 사람만 등장한다. 토니는 참고와 일치하는 30대 백인 남성 얼굴, 헬멧, 오염된 전술복과 구조 장비를 갖췄고, 윌마는 참고와 일치하는 젊은 백인 여성 얼굴과 묶은 금발, 녹색 점프수트 및 잠긴 벨트를 갖췄다. 총은 참고의 회색·검정 미래형 휴대 총기와 유사한 광학 조준기와 총열 구조를 지녔으나 세부 비례는 완전히 동일하지 않다.",
        "hard_violations": [],
        "physics": "윌마의 장갑 낀 오른손이 권총 손잡이를 확실히 잡고 팔을 뻗어 조준하므로 총의 지지와 사용 자세가 성립한다. 토니는 은신처 안 바닥에 몸을 둔 상태로 보이며 떠 있는 신체나 물체가 없다. 두 인물 모두 중력과 주변 지형에 맞는 자세다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "토니의 얼굴을 향한 단일 총구, 그 뒤에 정렬된 윌마의 차가운 시선, 은신처 개구부를 통한 근접 구도를 가장 충실히 구현했다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "인물과 숲은 대체로 맞지만 윌마가 쥔 총과 별개인 거대한 총열이 지지 없이 화면을 가로질러, 핵심 위협선과 물리적 staging이 붕괴한다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "토니는 화면 오른쪽의 윌마 쪽을 곁눈질하고 윌마도 토니를 응시한다. 윌마가 손에 든 소형 총은 토니 쪽을 겨누지만, 동시에 화면 오른쪽 밖에서 나온 별도의 긴 검은 총열이 토니의 코와 얼굴 높이를 향해 수평으로 뻗어 있다. 따라서 어느 총이 권위 있는 위협선인지 모호하다.",
        "built_space": "나뭇가지·뿌리·흙과 위장망으로 둘러싸인 개구부가 하나 보인다. 토니는 개구부 안쪽 왼쪽 전경에 앉거나 웅크리고 있고, 윌마는 개구부 밖 중앙 중경에 서 있으며 뒤에는 낮의 숲이 보인다. 공간 배치는 대체로 성립하지만, 개구부를 가로지르는 긴 총열의 발사자나 고정점은 보이지 않는다.",
        "entities": "토니와 윌마 두 사람만 있으며 얼굴·성별·연령대·머리색과 복장은 각 참고 이미지에 대체로 부합한다. 토니의 헬멧과 장비, 윌마의 녹색 점프수트와 잠긴 벨트가 보인다. 윌마가 쥔 총은 미래형 권총 계열이지만 참고 총기와 세부 형상이 다소 다르고, 그와 별개인 거대한 검은 총열이 추가되어 있다.",
        "hard_violations": [
         "윌마가 손에 쥔 총기 외에 화면 오른쪽에서 별도의 긴 총열이 추가로 등장해 총기/총열이 중복되었다.",
         "화면을 가로지르는 긴 검은 총열은 손, 총몸, 고정대 등 어떤 지지도 없이 떠 있어 물리적으로 성립하지 않는다."
        ],
        "physics": "토니의 몸은 은신처 바닥과 가장자리에 기대고 있으며 장갑 낀 팔도 흙 가장자리에 놓여 있어 지지가 보인다. 윌마는 두 손으로 자신의 소형 총을 잡아 조준한다. 그러나 얼굴 앞을 가로지르는 별도의 긴 총열에는 이를 붙드는 손이나 연결된 총몸이 전혀 없어 아무것도 지지하지 않는다."
       },
       {
        "label": "A",
        "direction": "윌마의 눈은 화면 왼쪽 전경의 토니를 차갑게 응시한다. 윌마가 든 총의 총구도 토니의 왼쪽 얼굴·관자 부근을 직접 겨누며, 총구 뒤에 윌마의 얼굴이 정렬된다. 토니의 시선은 오른쪽의 윌마 방향이지만 약간 아래로 비껴 있어 완전한 눈맞춤은 아니다.",
        "built_space": "굵은 나뭇가지와 잎, 위장망이 만든 은신처 개구부 하나가 화면 중앙을 둘러싼다. 토니는 개구부 안쪽 왼쪽 전경, 윌마는 바깥 중앙 중경에 서 있고, 개구부 뒤로 낮의 숲과 나무들이 제한적으로 보인다. 별도의 건축 고정물이나 불가능한 반사는 없으며 인물 위치와 개구부 방향이 맞는다.",
        "entities": "토니와 윌마 두 사람만 등장한다. 토니는 참고와 일치하는 30대 백인 남성 얼굴, 헬멧, 오염된 전술복과 구조 장비를 갖췄고, 윌마는 참고와 일치하는 젊은 백인 여성 얼굴과 묶은 금발, 녹색 점프수트 및 잠긴 벨트를 갖췄다. 총은 참고의 회색·검정 미래형 휴대 총기와 유사한 광학 조준기와 총열 구조를 지녔으나 세부 비례는 완전히 동일하지 않다.",
        "hard_violations": [],
        "physics": "윌마의 장갑 낀 오른손이 권총 손잡이를 확실히 잡고 팔을 뻗어 조준하므로 총의 지지와 사용 자세가 성립한다. 토니는 은신처 안 바닥에 몸을 둔 상태로 보이며 떠 있는 신체나 물체가 없다. 두 인물 모두 중력과 주변 지형에 맞는 자세다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.0,
    "B": 0.222
   },
   "adjusted": {
    "A": 0.75,
    "B": -0.028
   },
   "violations": {
    "A": [
     "[gemini-pro] 물리적으로 불가능한 해부학 및 원근법 (중경에 있는 인물의 어깨에서 이어질 수 없는 거대한 크기의 손과 총이 전경에 떠 있음)"
    ],
    "B": [
     "[gemini-pro] 지지대 없이 공중에 떠 있는 정체불명의 거대한 검은색 원통형 물체 (물리 법칙 위반 및 프롬프트에 없는 물체 생성)",
     "[gpt] 윌마가 손에 쥔 총기 외에 화면 오른쪽에서 별도의 긴 총열이 추가로 등장해 총기/총열이 중복되었다.",
     "[gpt] 화면을 가로지르는 긴 검은 총열은 손, 총몸, 고정대 등 어떤 지지도 없이 떠 있어 물리적으로 성립하지 않는다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "agreed": false
  },
  "totals": {
   "B": -28,
   "A": 750
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": -28,
    "verdict_ko": "이전 샷의 위장막과 은신처 배경을 잘 구현했으나, 프롬프트의 전경 총열 지시를 오인하여 아무 지지대 없이 화면을 가로지르는 거대한 검은색 원통을 허공에 생성한 치명적인 결함이 있습니다.  ★위반: [gemini-pro] 지지대 없이 공중에 떠 있는 정체불명의 거대한 검은색 원통형 물체 (물리 법칙 위반 및 프롬프트에 없는 물체 생성) / [gpt] 윌마가 손에 쥔 총기 외에 화면 오른쪽에서 별도의 긴 총열이 추가로 등장해 총기/총열이 중복되었다. / [gpt] 화면을 가로지르는 긴 검은 총열은 손, 총몸, 고정대 등 어떤 지지도 없이 떠 있어 물리적으로 성립하지 않는다."
   },
   {
    "label": "A",
    "score": 750,
    "verdict_ko": "중경에 위치한 윌마의 몸집과 전혀 맞지 않게 거대한 손과 총이 전경에 떠 있어, 물리적으로 불가능한 해부학과 원근법 왜곡을 보여줍니다.  ★위반: [gemini-pro] 물리적으로 불가능한 해부학 및 원근법 (중경에 있는 인물의 어깨에서 이어질 수 없는 거대한 크기의 손과 총이 전경에 떠 있음)"
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S7sh14_sel.png",
    "asset_id": "9f4880ef-5fa8-4054-8fde-d1fcd38cddc5",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   },
   {
    "label": "PROP REFERENCE — 원미래 휴대 총기: the exact object appearing in this shot; match its look, material and wear exactly.",
    "path": "<bytes:735057>",
    "asset_id": "c1ead6e4-0009-4c88-b335-55e1096a2fa1",
    "role": "prop_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2d3-fcd8-77b5-af6c-49ae211104ae",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S7sh14"
  }
 },
 "S7sh19::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:43:44.700527+00:00",
  "fingerprint": "4fc84b87d38e9c62bdad42fc782532eec2f584f4d46cf4f89e2d87550aceda2e",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S7sh19_sel.png",
  "source_sha256": "9b6aca7073e9a3d95e075b442fd453cf042c1392fdbdd03727d2c50b3fffedca",
  "file": "S7sh19_cine.png",
  "staged_sha256": "a47adc69f9176758dbc003cd41054631f172eb18184f63750fd390ec3668c0fe",
  "latency_ms": 17451
 },
 "S8sh1::signage": {
  "fp": "ead5a0710ac971f2",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S8sh1": {
  "input_fingerprint": "70b4a5a8493bdf77",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 흙바닥에 버려진 금속 칼 곁에서 양손을 어깨 높이로 들어 올린 채 서 있는 '토니(앤서니 로저스)'.\n\nLOCATION (lock): Outside on a small patch of bare earth within the forested ridge, near the abandoned rescue knife and the recent ambush site. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: rescue knife (placed on the ground beside the camera position) — Its side and blade axis are visible on the ground, angled away from Tony; used as Large near-foreground surrender evidence and starting point for the rising move; forest ground (supporting the discarded knife and Tony's position); used as Creates the low perspective line from the knife toward Tony; forest trees (surrounding the confrontation area) — Their trunks rise behind Tony and reinforce the upward camera angle; used as Background scale and vertical structure around Tony's raised arms.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight supports a restrained forest-green and weathered-neutral palette with grounded contrast around the surrender gesture.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 토니(앤서니 로저스) — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Tony’s metal rescue knife lies on the dirt where he set it down. MIDGE’s fragments, the sprung camouflage net, tripwire, and the dislodged pursuer’s gun remain around the fight site. 토니(앤서니 로저스): He stands in the open with both hands raised to shoulder height and his rescue harness still on. He is no longer carrying the knife.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 흙바닥에 버려진 금속 칼 곁에서 양손을 어깨 높이로 들어 올린 채 서 있는 '토니(앤서니 로저스)'.\n\nLOCATION (lock): Outside on a small patch of bare earth within the forested ridge, near the abandoned rescue knife and the recent ambush site. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: rescue knife (placed on the ground beside the camera position) — Its side and blade axis are visible on the ground, angled away from Tony; used as Large near-foreground surrender evidence and starting point for the rising move; forest ground (supporting the discarded knife and Tony's position); used as Creates the low perspective line from the knife toward Tony; forest trees (surrounding the confrontation area) — Their trunks rise behind Tony and reinforce the upward camera angle; used as Background scale and vertical structure around Tony's raised arms.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight supports a restrained forest-green and weathered-neutral palette with grounded contrast around the surrender gesture.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 토니(앤서니 로저스) — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Tony’s metal rescue knife lies on the dirt where he set it down. MIDGE’s fragments, the sprung camouflage net, tripwire, and the dislodged pursuer’s gun remain around the fight site. 토니(앤서니 로저스): He stands in the open with both hands raised to shoulder height and his rescue harness still on. He is no longer carrying the knife.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 흙바닥에 버려진 금속 칼 곁에서 양손을 어깨 높이로 들어 올린 채 서 있는 '토니(앤서니 로저스)'.\n\nLOCATION (lock): Outside on a small patch of bare earth within the forested ridge, near the abandoned rescue knife and the recent ambush site. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: rescue knife (placed on the ground beside the camera position) — Its side and blade axis are visible on the ground, angled away from Tony; used as Large near-foreground surrender evidence and starting point for the rising move; forest ground (supporting the discarded knife and Tony's position); used as Creates the low perspective line from the knife toward Tony; forest trees (surrounding the confrontation area) — Their trunks rise behind Tony and reinforce the upward camera angle; used as Background scale and vertical structure around Tony's raised arms.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight supports a restrained forest-green and weathered-neutral palette with grounded contrast around the surrender gesture.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The clothing, hair and overall look of 토니(앤서니 로저스) — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): Tony’s metal rescue knife lies on the dirt where he set it down. MIDGE’s fragments, the sprung camouflage net, tripwire, and the dislodged pursuer’s gun remain around the fight site. 토니(앤서니 로저스): He stands in the open with both hands raised to shoulder height and his rescue harness still on. He is no longer carrying the knife.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라는 바닥의 칼에서 토니를 향해 위로 향하는 낮은 구도이며, 토니는 양손을 들고 정면을 응시함.",
    "built_space": "숲속 흙바닥에 금속 칼이 전경에 배치되어 있고, 뒤로 굵은 활엽수 기둥들이 서 있어 레퍼런스 환경과 유사하게 구성됨.",
    "entities": "토니의 얼굴, 복장, 헬멧이 레퍼런스와 일치하며 전경에 구조용 칼이 존재함. 그러나 프롬프트가 요구한 '서 있는' 상태가 아님.",
    "hard_violations": [
     "[gemini-pro] 물리적으로 불가능한 무대 연출: 토니의 무릎 아래 하반신이 지면을 뚫고 흙바닥 아래로 완전히 파묻혀 잘려나간 것처럼 렌더링됨."
    ],
    "physics": "토니의 다리가 바닥 표면을 그대로 통과해 파묻혀 있어 신체를 지탱하는 방식이 물리적으로 불가능함. 전경의 칼은 바닥에 놓여 있음."
   },
   {
    "label": "B",
    "direction": "카메라는 바닥의 칼에서 토니를 향하는 낮은 앵글을 잡고 있으며, 토니는 양손을 어깨 높이로 들고 정면을 향함.",
    "built_space": "숲속 흙바닥 전경에 금속 칼이 놓여 있으나, 배경의 나무들이 레퍼런스의 식생과 달리 얇은 침엽수 위주로 구성됨.",
    "entities": "전경에 금속 칼이 묘사되었고 토니가 두 발로 서서 항복 자세를 취하고 있음. 단, 토니가 레퍼런스 이미지에 명시된 헬멧을 착용하지 않았음.",
    "hard_violations": [],
    "physics": "토니는 두 발로 바닥을 확실히 딛고 안정적으로 서 있으며, 구조용 칼과 주변 물체들도 중력에 맞게 지면에 정상적으로 놓여 있음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "레퍼런스의 복장과 헬멧은 잘 반영했으나, '서 있는(standing)' 지시를 위반하고 하반신이 흙바닥에 완전히 파묻히는 물리적으로 불가능한 렌더링 오류가 발생하여 실격 사유가 됨."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "전경의 칼에서부터 서 있는 토니를 바라보는 지정된 낮은 앵글 구도를 정확히 구현했으나, 레퍼런스의 헬멧이 누락되었고 숲의 식생(침엽수림)이 이전 샷과 일치하지 않음."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 바닥의 칼에서 토니를 향해 위로 향하는 낮은 구도이며, 토니는 양손을 들고 정면을 응시함.",
        "built_space": "숲속 흙바닥에 금속 칼이 전경에 배치되어 있고, 뒤로 굵은 활엽수 기둥들이 서 있어 레퍼런스 환경과 유사하게 구성됨.",
        "entities": "토니의 얼굴, 복장, 헬멧이 레퍼런스와 일치하며 전경에 구조용 칼이 존재함. 그러나 프롬프트가 요구한 '서 있는' 상태가 아님.",
        "hard_violations": [
         "물리적으로 불가능한 무대 연출: 토니의 무릎 아래 하반신이 지면을 뚫고 흙바닥 아래로 완전히 파묻혀 잘려나간 것처럼 렌더링됨."
        ],
        "physics": "토니의 다리가 바닥 표면을 그대로 통과해 파묻혀 있어 신체를 지탱하는 방식이 물리적으로 불가능함. 전경의 칼은 바닥에 놓여 있음."
       },
       {
        "label": "B",
        "direction": "카메라는 바닥의 칼에서 토니를 향하는 낮은 앵글을 잡고 있으며, 토니는 양손을 어깨 높이로 들고 정면을 향함.",
        "built_space": "숲속 흙바닥 전경에 금속 칼이 놓여 있으나, 배경의 나무들이 레퍼런스의 식생과 달리 얇은 침엽수 위주로 구성됨.",
        "entities": "전경에 금속 칼이 묘사되었고 토니가 두 발로 서서 항복 자세를 취하고 있음. 단, 토니가 레퍼런스 이미지에 명시된 헬멧을 착용하지 않았음.",
        "hard_violations": [],
        "physics": "토니는 두 발로 바닥을 확실히 딛고 안정적으로 서 있으며, 구조용 칼과 주변 물체들도 중력에 맞게 지면에 정상적으로 놓여 있음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 2,
        "verdict_ko": "레퍼런스의 복장과 헬멧은 잘 반영했으나, '서 있는(standing)' 지시를 위반하고 하반신이 흙바닥에 완전히 파묻히는 물리적으로 불가능한 렌더링 오류가 발생하여 실격 사유가 됨."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "전경의 칼에서부터 서 있는 토니를 바라보는 지정된 낮은 앵글 구도를 정확히 구현했으나, 레퍼런스의 헬멧이 누락되었고 숲의 식생(침엽수림)이 이전 샷과 일치하지 않음."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 바닥의 칼에서 토니를 향해 위로 향하는 낮은 구도이며, 토니는 양손을 들고 정면을 응시함.",
        "built_space": "숲속 흙바닥에 금속 칼이 전경에 배치되어 있고, 뒤로 굵은 활엽수 기둥들이 서 있어 레퍼런스 환경과 유사하게 구성됨.",
        "entities": "토니의 얼굴, 복장, 헬멧이 레퍼런스와 일치하며 전경에 구조용 칼이 존재함. 그러나 프롬프트가 요구한 '서 있는' 상태가 아님.",
        "hard_violations": [
         "물리적으로 불가능한 무대 연출: 토니의 무릎 아래 하반신이 지면을 뚫고 흙바닥 아래로 완전히 파묻혀 잘려나간 것처럼 렌더링됨."
        ],
        "physics": "토니의 다리가 바닥 표면을 그대로 통과해 파묻혀 있어 신체를 지탱하는 방식이 물리적으로 불가능함. 전경의 칼은 바닥에 놓여 있음."
       },
       {
        "label": "B",
        "direction": "카메라는 바닥의 칼에서 토니를 향하는 낮은 앵글을 잡고 있으며, 토니는 양손을 어깨 높이로 들고 정면을 향함.",
        "built_space": "숲속 흙바닥 전경에 금속 칼이 놓여 있으나, 배경의 나무들이 레퍼런스의 식생과 달리 얇은 침엽수 위주로 구성됨.",
        "entities": "전경에 금속 칼이 묘사되었고 토니가 두 발로 서서 항복 자세를 취하고 있음. 단, 토니가 레퍼런스 이미지에 명시된 헬멧을 착용하지 않았음.",
        "hard_violations": [],
        "physics": "토니는 두 발로 바닥을 확실히 딛고 안정적으로 서 있으며, 구조용 칼과 주변 물체들도 중력에 맞게 지면에 정상적으로 놓여 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "전신 와이드 숏과 칼에서 토니로 이어지는 낮은 원근선, 선 채 양손을 든 순간을 가장 정확히 구현했지만 헬멧 및 전투 현장 잔해의 연속성이 부족하다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "헬멧·구조 하네스·위장망과 잔해는 더 충실하지만 토니의 하체를 흙둔덕 뒤로 가린 더 타이트한 프레이밍이라 권위적인 와이드 숏 지시에서 A보다 뒤진다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "토니는 정면의 카메라 쪽을 바라보며 두 손바닥도 카메라 쪽으로 보인다. 겨누는 무기는 없다. 전경 칼은 손잡이가 화면 왼쪽, 칼날이 오른쪽을 향해 토니를 직접 가리키지 않고 비스듬히 벗어나 있다.",
        "built_space": "인공 구조물은 없고 흙바닥의 작은 공터와 다수의 수직 나무줄기가 토니를 둘러싼다. 토니는 공터 중앙에 전신이 보이도록 서 있으며, 큰 전경 칼 하나와 중경의 작은 검은 잔해들이 보인다. 반사면이나 광학적으로 문제 되는 거울은 없다.",
        "entities": "토니로 보이는 30대 중반 백인 남성 한 명만 있으며, 어두운 전술복·장갑·구조 하네스·로프는 참조와 대체로 맞는다. 다만 참조에서 고정된 헬멧이 사라져 갈색 머리가 노출된다. 금속 구조 칼 한 자루가 흙 위 전경에 놓여 있고 토니는 칼을 들고 있지 않다. 위장망과 트립와이어, MIDGE의 파편은 명확히 식별되지 않는다.",
        "hard_violations": [],
        "physics": "토니는 두 발을 흙바닥에 완전히 딛고 체중을 지탱하며 똑바로 서 있다. 양팔은 어깨에서 굽혀 자력으로 들어 올린 현실적인 항복 자세다. 칼과 작은 잔해들은 모두 흙바닥에 놓여 지지되며 떠 있는 물체는 없다."
       },
       {
        "label": "B",
        "direction": "토니는 거의 정면으로 카메라를 바라보고 두 손바닥을 전방에 보인다. 겨누는 무기는 없다. 전경 칼은 손잡이가 토니 쪽에 더 가깝고 칼날이 화면 왼쪽 아래로 향해 토니에게서 멀어지는 축을 이룬다.",
        "built_space": "인공 구조물은 없으며, 굵은 나무줄기 세 개가 토니 바로 뒤에서 수직 배경을 만든다. 토니는 중앙에 있지만 전경 흙둔덕이 무릎 아래를 가려 전신 배치가 확인되지 않는다. 전경 칼 하나, 왼쪽 위장망과 장비 잔해, 오른쪽 파손 장비 및 총기성 잔해가 보인다. 반사면이나 거울은 없다.",
        "entities": "헬멧을 쓴 30대 중반 백인 남성 토니 한 명만 등장한다. 얼굴, 헬멧, 어두운 전술복, 장갑, 가슴 장비, 구조 하네스와 로프가 참조와 매우 가깝다. 금속 구조 칼은 흙 위에 버려져 있고 토니는 이를 휴대하지 않는다. 위장망과 여러 전투 현장 파편도 보이지만 트립와이어는 선명하게 확인되지 않는다.",
        "hard_violations": [],
        "physics": "토니의 하체가 전경 지면에 가려졌지만 보이는 골반과 허벅지의 수직 관계상 흙둔덕 뒤에 서 있는 것으로 읽히며 공중에 뜬 징후는 없다. 양팔은 자연스럽게 굽혀 들려 있다. 칼, 위장망과 장비 파편은 모두 지면에 닿아 지지된다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "전신 와이드 숏과 칼에서 토니로 이어지는 낮은 원근선, 선 채 양손을 든 순간을 가장 정확히 구현했지만 헬멧 및 전투 현장 잔해의 연속성이 부족하다."
       },
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "헬멧·구조 하네스·위장망과 잔해는 더 충실하지만 토니의 하체를 흙둔덕 뒤로 가린 더 타이트한 프레이밍이라 권위적인 와이드 숏 지시에서 A보다 뒤진다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "토니는 정면의 카메라 쪽을 바라보며 두 손바닥도 카메라 쪽으로 보인다. 겨누는 무기는 없다. 전경 칼은 손잡이가 화면 왼쪽, 칼날이 오른쪽을 향해 토니를 직접 가리키지 않고 비스듬히 벗어나 있다.",
        "built_space": "인공 구조물은 없고 흙바닥의 작은 공터와 다수의 수직 나무줄기가 토니를 둘러싼다. 토니는 공터 중앙에 전신이 보이도록 서 있으며, 큰 전경 칼 하나와 중경의 작은 검은 잔해들이 보인다. 반사면이나 광학적으로 문제 되는 거울은 없다.",
        "entities": "토니로 보이는 30대 중반 백인 남성 한 명만 있으며, 어두운 전술복·장갑·구조 하네스·로프는 참조와 대체로 맞는다. 다만 참조에서 고정된 헬멧이 사라져 갈색 머리가 노출된다. 금속 구조 칼 한 자루가 흙 위 전경에 놓여 있고 토니는 칼을 들고 있지 않다. 위장망과 트립와이어, MIDGE의 파편은 명확히 식별되지 않는다.",
        "hard_violations": [],
        "physics": "토니는 두 발을 흙바닥에 완전히 딛고 체중을 지탱하며 똑바로 서 있다. 양팔은 어깨에서 굽혀 자력으로 들어 올린 현실적인 항복 자세다. 칼과 작은 잔해들은 모두 흙바닥에 놓여 지지되며 떠 있는 물체는 없다."
       },
       {
        "label": "A",
        "direction": "토니는 거의 정면으로 카메라를 바라보고 두 손바닥을 전방에 보인다. 겨누는 무기는 없다. 전경 칼은 손잡이가 토니 쪽에 더 가깝고 칼날이 화면 왼쪽 아래로 향해 토니에게서 멀어지는 축을 이룬다.",
        "built_space": "인공 구조물은 없으며, 굵은 나무줄기 세 개가 토니 바로 뒤에서 수직 배경을 만든다. 토니는 중앙에 있지만 전경 흙둔덕이 무릎 아래를 가려 전신 배치가 확인되지 않는다. 전경 칼 하나, 왼쪽 위장망과 장비 잔해, 오른쪽 파손 장비 및 총기성 잔해가 보인다. 반사면이나 거울은 없다.",
        "entities": "헬멧을 쓴 30대 중반 백인 남성 토니 한 명만 등장한다. 얼굴, 헬멧, 어두운 전술복, 장갑, 가슴 장비, 구조 하네스와 로프가 참조와 매우 가깝다. 금속 구조 칼은 흙 위에 버려져 있고 토니는 이를 휴대하지 않는다. 위장망과 여러 전투 현장 파편도 보이지만 트립와이어는 선명하게 확인되지 않는다.",
        "hard_violations": [],
        "physics": "토니의 하체가 전경 지면에 가려졌지만 보이는 골반과 허벅지의 수직 관계상 흙둔덕 뒤에 서 있는 것으로 읽히며 공중에 뜬 징후는 없다. 양팔은 자연스럽게 굽혀 들려 있다. 칼, 위장망과 장비 파편은 모두 지면에 닿아 지지된다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.161,
    "B": 2.0
   },
   "adjusted": {
    "A": 0.911,
    "B": 2.0
   },
   "violations": {
    "A": [
     "[gemini-pro] 물리적으로 불가능한 무대 연출: 토니의 무릎 아래 하반신이 지면을 뚫고 흙바닥 아래로 완전히 파묻혀 잘려나간 것처럼 렌더링됨."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "A": 911,
   "B": 2000
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 911,
    "verdict_ko": "레퍼런스의 복장과 헬멧은 잘 반영했으나, '서 있는(standing)' 지시를 위반하고 하반신이 흙바닥에 완전히 파묻히는 물리적으로 불가능한 렌더링 오류가 발생하여 실격 사유가 됨.  ★위반: [gemini-pro] 물리적으로 불가능한 무대 연출: 토니의 무릎 아래 하반신이 지면을 뚫고 흙바닥 아래로 완전히 파묻혀 잘려나간 것처럼 렌더링됨."
   },
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "전경의 칼에서부터 서 있는 토니를 바라보는 지정된 낮은 앵글 구도를 정확히 구현했으나, 레퍼런스의 헬멧이 누락되었고 숲의 식생(침엽수림)이 이전 샷과 일치하지 않음."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The clothing, hair and overall look of 토니(앤서니 로저스) — who appear both in that photo and in this shot — are LOCKED to that photo. Anyone else visible in that photo is NOT in this shot: never carry their face, body or clothing onto anyone here. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S7sh19_sel.png",
    "asset_id": "78bd366d-3a0c-4534-ba95-63a911833b22",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2d9-24ab-790a-a347-22ea202fa7e7",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S7sh19"
  }
 },
 "S8sh1::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:44:58.258764+00:00",
  "fingerprint": "86580023f754af5775d08f90d6e879523cdb884b0bc3549d71a840a59057974f",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S8sh1_sel.png",
  "source_sha256": "ed96fe112247d9f2285c252596193fc3b88a6fff875e2d24cfa89a297416cc2e",
  "file": "S8sh1_cine.png",
  "staged_sha256": "ebfcad8e53ced2730612f903908a9ae1eb6faa7f713d4227d8a49e6a2d46806c",
  "latency_ms": 21573
 },
 "S8sh6::signage": {
  "fp": "95cfd4aebab64d41",
  "inscriptions": [
   {
    "text_native": "2026.10.18",
    "source": "scene_text_quoted",
    "reason_ko": "스마트폰 화면에 선명하게 떠 있는 날짜 숫자가 지문에 명시되어 있습니다.",
    "source_quote": "2026.10.18"
   }
  ],
  "cues": [],
  "dropped": []
 },
 "S8sh6": {
  "input_fingerprint": "295d9c0dadf3e249",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): '2026.10.18' 숫자와 작성 중인 메시지가 선명하게 떠 있는 스마트폰 화면 클로즈업.\n\nLOCATION (lock): Outside in the forest-ridge clearing where the two survivors confront each other, with the cracked phone held between them. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: legible smartphone display in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: smartphone display (active and cracked, showing 2026.10.18, battery 3 percent, and an unsent unfinished message) — The content-bearing front face is angled slightly upward toward the camera; the date, battery status, and message text are fully visible; used as Primary insert focus at the end of the inward move; forest setting beyond the phone (present beyond the close phone view); used as Soft peripheral scale context that prevents the phone from occupying an unrealistic share of the frame.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight provides neutral, controlled visibility for the active display without introducing an additional specified light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The metal rescue knife remains on the dirt beside MIDGE’s scattered fragments and the sprung trap. The cracked smartphone displays 2026.10.18, 3 percent battery, and the unsent draft, “마라, 갱도가 무너졌어. 나는—”.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"2026.10.18\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): '2026.10.18' 숫자와 작성 중인 메시지가 선명하게 떠 있는 스마트폰 화면 클로즈업.\n\nLOCATION (lock): Outside in the forest-ridge clearing where the two survivors confront each other, with the cracked phone held between them. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: legible smartphone display in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: smartphone display (active and cracked, showing 2026.10.18, battery 3 percent, and an unsent unfinished message) — The content-bearing front face is angled slightly upward toward the camera; the date, battery status, and message text are fully visible; used as Primary insert focus at the end of the inward move; forest setting beyond the phone (present beyond the close phone view); used as Soft peripheral scale context that prevents the phone from occupying an unrealistic share of the frame.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight provides neutral, controlled visibility for the active display without introducing an additional specified light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The metal rescue knife remains on the dirt beside MIDGE’s scattered fragments and the sprung trap. The cracked smartphone displays 2026.10.18, 3 percent battery, and the unsent draft, “마라, 갱도가 무너졌어. 나는—”.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"2026.10.18\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): '2026.10.18' 숫자와 작성 중인 메시지가 선명하게 떠 있는 스마트폰 화면 클로즈업.\n\nLOCATION (lock): Outside in the forest-ridge clearing where the two survivors confront each other, with the cracked phone held between them. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- FRAME LAYOUT: legible smartphone display in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: smartphone display (active and cracked, showing 2026.10.18, battery 3 percent, and an unsent unfinished message) — The content-bearing front face is angled slightly upward toward the camera; the date, battery status, and message text are fully visible; used as Primary insert focus at the end of the inward move; forest setting beyond the phone (present beyond the close phone view); used as Soft peripheral scale context that prevents the phone from occupying an unrealistic share of the frame.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight provides neutral, controlled visibility for the active display without introducing an additional specified light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The metal rescue knife remains on the dirt beside MIDGE’s scattered fragments and the sprung trap. The cracked smartphone displays 2026.10.18, 3 percent battery, and the unsent draft, “마라, 갱도가 무너졌어. 나는—”.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"2026.10.18\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라는 레퍼런스와 동일하게 숲길을 바라보고 있으며, 스마트폰 화면은 카메라를 향해 약간 위로 기울어져 정면을 향한다.",
    "built_space": "레퍼런스 이미지의 숲 공터 배경이 그대로 일치하게 구현되었다. 흙길, 나무의 위치, 그리고 전경 바닥에 놓인 금속 구조용 칼까지 올바른 자리에 위치해 있다.",
    "entities": "액정이 깨진 스마트폰은 '2026.10.18', 3% 배터리, 지정된 한국어 메시지를 정확하게 표시하고 있다. 프레임 바깥에서 나타난 두 손은 지저분한 맨손이며 짙은 색 소매를 입고 있다.",
    "hard_violations": [],
    "physics": "스마트폰은 양손에 의해 자연스럽고 안정적으로 지탱되고 있으며, 바닥에 놓인 칼 역시 중력에 맞게 흙 표면 위에 놓여 있다."
   },
   {
    "label": "B",
    "direction": "카메라는 숲을 향하고 있으며 스마트폰 화면이 정면으로 카메라를 향해 세로로 들려 있다. 오른손 검지가 화면 하단을 터치하고 있다.",
    "built_space": "숲 배경은 존재하나 레퍼런스에서 명시된 고정된 특징들(바닥의 금속 칼과 흩어진 파편 등)이 누락되었다.",
    "entities": "깨진 폰 화면에 날짜, 배터리, 한국어 텍스트가 잘 나타나 있다. 하지만 폰을 쥐고 있는 인물의 손 해부학 구조에 심각한 오류가 있다.",
    "hard_violations": [
     "[gemini-pro] 물리적으로 불가능한 해부학 구조 (왼팔에 달린 오른손)"
    ],
    "physics": "왼팔이 폰을 지탱하고 있으나, 엄지가 화면 왼쪽에 있고 나머지 손가락이 오른쪽 가장자리를 뒤에서 감싸는 전형적인 '오른손'의 형태를 하고 있어 물리적으로 불가능한 파지 자세다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "레퍼런스의 배경과 바닥의 칼을 완벽하게 유지하면서, 프롬프트가 요구한 깨진 스마트폰 화면의 텍스트와 정보를 자연스러운 양손 파지법과 함께 훌륭하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "화면 안의 텍스트는 잘 반영되었으나, 레퍼런스의 고정 소품인 칼이 누락되었고 기기를 쥐고 있는 왼팔에 오른손이 달려있는 치명적인 해부학적 오류가 있습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 레퍼런스와 동일하게 숲길을 바라보고 있으며, 스마트폰 화면은 카메라를 향해 약간 위로 기울어져 정면을 향한다.",
        "built_space": "레퍼런스 이미지의 숲 공터 배경이 그대로 일치하게 구현되었다. 흙길, 나무의 위치, 그리고 전경 바닥에 놓인 금속 구조용 칼까지 올바른 자리에 위치해 있다.",
        "entities": "액정이 깨진 스마트폰은 '2026.10.18', 3% 배터리, 지정된 한국어 메시지를 정확하게 표시하고 있다. 프레임 바깥에서 나타난 두 손은 지저분한 맨손이며 짙은 색 소매를 입고 있다.",
        "hard_violations": [],
        "physics": "스마트폰은 양손에 의해 자연스럽고 안정적으로 지탱되고 있으며, 바닥에 놓인 칼 역시 중력에 맞게 흙 표면 위에 놓여 있다."
       },
       {
        "label": "B",
        "direction": "카메라는 숲을 향하고 있으며 스마트폰 화면이 정면으로 카메라를 향해 세로로 들려 있다. 오른손 검지가 화면 하단을 터치하고 있다.",
        "built_space": "숲 배경은 존재하나 레퍼런스에서 명시된 고정된 특징들(바닥의 금속 칼과 흩어진 파편 등)이 누락되었다.",
        "entities": "깨진 폰 화면에 날짜, 배터리, 한국어 텍스트가 잘 나타나 있다. 하지만 폰을 쥐고 있는 인물의 손 해부학 구조에 심각한 오류가 있다.",
        "hard_violations": [
         "물리적으로 불가능한 해부학 구조 (왼팔에 달린 오른손)"
        ],
        "physics": "왼팔이 폰을 지탱하고 있으나, 엄지가 화면 왼쪽에 있고 나머지 손가락이 오른쪽 가장자리를 뒤에서 감싸는 전형적인 '오른손'의 형태를 하고 있어 물리적으로 불가능한 파지 자세다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "레퍼런스의 배경과 바닥의 칼을 완벽하게 유지하면서, 프롬프트가 요구한 깨진 스마트폰 화면의 텍스트와 정보를 자연스러운 양손 파지법과 함께 훌륭하게 구현했습니다."
       },
       {
        "label": "B",
        "score": 2,
        "verdict_ko": "화면 안의 텍스트는 잘 반영되었으나, 레퍼런스의 고정 소품인 칼이 누락되었고 기기를 쥐고 있는 왼팔에 오른손이 달려있는 치명적인 해부학적 오류가 있습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 레퍼런스와 동일하게 숲길을 바라보고 있으며, 스마트폰 화면은 카메라를 향해 약간 위로 기울어져 정면을 향한다.",
        "built_space": "레퍼런스 이미지의 숲 공터 배경이 그대로 일치하게 구현되었다. 흙길, 나무의 위치, 그리고 전경 바닥에 놓인 금속 구조용 칼까지 올바른 자리에 위치해 있다.",
        "entities": "액정이 깨진 스마트폰은 '2026.10.18', 3% 배터리, 지정된 한국어 메시지를 정확하게 표시하고 있다. 프레임 바깥에서 나타난 두 손은 지저분한 맨손이며 짙은 색 소매를 입고 있다.",
        "hard_violations": [],
        "physics": "스마트폰은 양손에 의해 자연스럽고 안정적으로 지탱되고 있으며, 바닥에 놓인 칼 역시 중력에 맞게 흙 표면 위에 놓여 있다."
       },
       {
        "label": "B",
        "direction": "카메라는 숲을 향하고 있으며 스마트폰 화면이 정면으로 카메라를 향해 세로로 들려 있다. 오른손 검지가 화면 하단을 터치하고 있다.",
        "built_space": "숲 배경은 존재하나 레퍼런스에서 명시된 고정된 특징들(바닥의 금속 칼과 흩어진 파편 등)이 누락되었다.",
        "entities": "깨진 폰 화면에 날짜, 배터리, 한국어 텍스트가 잘 나타나 있다. 하지만 폰을 쥐고 있는 인물의 손 해부학 구조에 심각한 오류가 있다.",
        "hard_violations": [
         "물리적으로 불가능한 해부학 구조 (왼팔에 달린 오른손)"
        ],
        "physics": "왼팔이 폰을 지탱하고 있으나, 엄지가 화면 왼쪽에 있고 나머지 손가락이 오른쪽 가장자리를 뒤에서 감싸는 전형적인 '오른손'의 형태를 하고 있어 물리적으로 불가능한 파지 자세다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "스마트폰 화면을 중앙 전경의 인서트 클로즈업으로 잡고 정확한 날짜·3% 배터리·미완성 메시지를 선명하게 보여 주며, 손의 지지와 숲 배경도 자연스럽다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "필수 화면 정보와 물리적 지지는 충실하지만 휴대전화가 작고 손과 칼까지 넓게 담겨 인서트 클로즈업의 프레이밍이 A보다 약하며 화면 하단의 추가 문자도 부정확하다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "휴대전화의 내용면이 카메라를 향해 거의 정면으로 들려 있어 날짜와 메시지가 보인다. 오른손 검지는 화면 오른쪽 아래를 향해 실제로 조작하고 있으며, 시선·무기 조준·이동 물체는 없다.",
        "built_space": "인공 구조물이나 고정 설비는 없다. 휴대전화 뒤로 흙바닥과 수직으로 선 나무들이 흐릿하게 이어져 이전 장면과 같은 주간 숲 공터로 읽힌다. 휴대전화는 화면 중앙 전경에 크게 놓이고 숲은 주변 맥락으로 남아 있어 지정된 인서트 구도에 부합한다.",
        "entities": "금이 간 구형 스마트폰, 정확한 날짜 ‘2026.10.18’, 3% 배터리, 미완성 메시지 ‘마라, 갱도가 무너졌어. 나는—’가 모두 선명하다. 사람의 얼굴이나 몸은 없고, 허용된 조작자의 두 손과 팔 일부만 보인다. 칼·파편·덫은 이 타이트한 프레임에서 식별되지 않지만 올바른 프레이밍이 배제한 항목이므로 결격 사유가 아니다. 전송 앱을 연상시키는 로고형 아이콘은 불필요한 추가 표식이다.",
        "hard_violations": [],
        "physics": "왼손이 휴대전화를 뒤와 측면에서 확실히 받치고 있으며 오른손 검지가 화면을 누른다. 전화기는 떠 있지 않고 손가락 접촉과 무게감도 자연스럽다."
       },
       {
        "label": "B",
        "direction": "가로로 든 휴대전화의 내용면이 카메라를 향하고 있어 날짜와 메시지를 읽을 수 있다. 두 손은 각각 좌우 가장자리를 향해 잡고 있다. 바닥의 칼날은 화면 오른쪽 아래를 향하지만 어떤 대상을 겨누거나 이동하는 상태는 아니다.",
        "built_space": "인공 구조물이나 고정 설비는 없다. 흙길 형태의 숲 공터와 수직 나무들이 배경에 보이고, 바닥 중앙 아래에는 칼 한 자루가 놓여 있다. 장소는 대체로 참고 이미지와 맞지만 전화기가 상대적으로 작고 손과 바닥까지 넓게 보여 ‘화면 인서트 클로즈업’보다 넓은 구도다.",
        "entities": "금이 간 스마트폰에 정확한 날짜 ‘2026.10.18’, 3% 배터리, 요구된 미완성 메시지가 보인다. 사람의 얼굴이나 몸은 없고 허용된 두 손과 팔 일부만 있다. 금속 구조용 칼 한 자루가 흙바닥에 있으며 참고 장면의 지속 상태와 맞는다. 다만 화면 하단에 ‘EMS’로 읽히는 불필요한 추가 문자가 있어 ‘다른 읽을 수 있는 글자 금지’ 조건을 더 직접적으로 어긴다.",
        "hard_violations": [],
        "physics": "두 손이 휴대전화의 좌우 가장자리를 단단히 잡아 지지한다. 바닥의 칼은 흙 위에 놓여 지면이 받치고 있으며, 공중에 뜬 물체나 불가능한 신체 자세는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 9,
        "verdict_ko": "스마트폰 화면을 중앙 전경의 인서트 클로즈업으로 잡고 정확한 날짜·3% 배터리·미완성 메시지를 선명하게 보여 주며, 손의 지지와 숲 배경도 자연스럽다."
       },
       {
        "label": "A",
        "score": 7,
        "verdict_ko": "필수 화면 정보와 물리적 지지는 충실하지만 휴대전화가 작고 손과 칼까지 넓게 담겨 인서트 클로즈업의 프레이밍이 A보다 약하며 화면 하단의 추가 문자도 부정확하다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "휴대전화의 내용면이 카메라를 향해 거의 정면으로 들려 있어 날짜와 메시지가 보인다. 오른손 검지는 화면 오른쪽 아래를 향해 실제로 조작하고 있으며, 시선·무기 조준·이동 물체는 없다.",
        "built_space": "인공 구조물이나 고정 설비는 없다. 휴대전화 뒤로 흙바닥과 수직으로 선 나무들이 흐릿하게 이어져 이전 장면과 같은 주간 숲 공터로 읽힌다. 휴대전화는 화면 중앙 전경에 크게 놓이고 숲은 주변 맥락으로 남아 있어 지정된 인서트 구도에 부합한다.",
        "entities": "금이 간 구형 스마트폰, 정확한 날짜 ‘2026.10.18’, 3% 배터리, 미완성 메시지 ‘마라, 갱도가 무너졌어. 나는—’가 모두 선명하다. 사람의 얼굴이나 몸은 없고, 허용된 조작자의 두 손과 팔 일부만 보인다. 칼·파편·덫은 이 타이트한 프레임에서 식별되지 않지만 올바른 프레이밍이 배제한 항목이므로 결격 사유가 아니다. 전송 앱을 연상시키는 로고형 아이콘은 불필요한 추가 표식이다.",
        "hard_violations": [],
        "physics": "왼손이 휴대전화를 뒤와 측면에서 확실히 받치고 있으며 오른손 검지가 화면을 누른다. 전화기는 떠 있지 않고 손가락 접촉과 무게감도 자연스럽다."
       },
       {
        "label": "A",
        "direction": "가로로 든 휴대전화의 내용면이 카메라를 향하고 있어 날짜와 메시지를 읽을 수 있다. 두 손은 각각 좌우 가장자리를 향해 잡고 있다. 바닥의 칼날은 화면 오른쪽 아래를 향하지만 어떤 대상을 겨누거나 이동하는 상태는 아니다.",
        "built_space": "인공 구조물이나 고정 설비는 없다. 흙길 형태의 숲 공터와 수직 나무들이 배경에 보이고, 바닥 중앙 아래에는 칼 한 자루가 놓여 있다. 장소는 대체로 참고 이미지와 맞지만 전화기가 상대적으로 작고 손과 바닥까지 넓게 보여 ‘화면 인서트 클로즈업’보다 넓은 구도다.",
        "entities": "금이 간 스마트폰에 정확한 날짜 ‘2026.10.18’, 3% 배터리, 요구된 미완성 메시지가 보인다. 사람의 얼굴이나 몸은 없고 허용된 두 손과 팔 일부만 있다. 금속 구조용 칼 한 자루가 흙바닥에 있으며 참고 장면의 지속 상태와 맞는다. 다만 화면 하단에 ‘EMS’로 읽히는 불필요한 추가 문자가 있어 ‘다른 읽을 수 있는 글자 금지’ 조건을 더 직접적으로 어긴다.",
        "hard_violations": [],
        "physics": "두 손이 휴대전화의 좌우 가장자리를 단단히 잡아 지지한다. 바닥의 칼은 흙 위에 놓여 지면이 받치고 있으며, 공중에 뜬 물체나 불가능한 신체 자세는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.778,
    "B": 1.222
   },
   "adjusted": {
    "A": 1.778,
    "B": 0.972
   },
   "violations": {
    "B": [
     "[gemini-pro] 물리적으로 불가능한 해부학 구조 (왼팔에 달린 오른손)"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "B"
   },
   "agreed": false
  },
  "totals": {
   "A": 1778,
   "B": 972
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1778,
    "verdict_ko": "레퍼런스의 배경과 바닥의 칼을 완벽하게 유지하면서, 프롬프트가 요구한 깨진 스마트폰 화면의 텍스트와 정보를 자연스러운 양손 파지법과 함께 훌륭하게 구현했습니다."
   },
   {
    "label": "B",
    "score": 972,
    "verdict_ko": "화면 안의 텍스트는 잘 반영되었으나, 레퍼런스의 고정 소품인 칼이 누락되었고 기기를 쥐고 있는 왼팔에 오른손이 달려있는 치명적인 해부학적 오류가 있습니다.  ★위반: [gemini-pro] 물리적으로 불가능한 해부학 구조 (왼팔에 달린 오른손)"
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S8sh1_sel.png",
    "asset_id": "933db285-6870-4d8e-a080-51e37f5b5e42",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2de-0034-715b-9ecd-897372ae20b3",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S8sh1"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S8sh6::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:46:39.899916+00:00",
  "fingerprint": "12af53c6c2ebf9e06ae0a15b2471177f036239ae82b17b6f5c7c52c526dab7f4",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S8sh6_sel.png",
  "source_sha256": "00c18b95c1f0d8be54b903abc9c9ff5f927db0a86fc89f3081cc194b5d5aabdc",
  "file": "S8sh6_cine.png",
  "staged_sha256": "e8c183bfee5319f2bbddbc47bee4d0ed62df84f8a25a8690b65ff52cbd7ddac8",
  "latency_ms": 19661
 },
 "S8sh7::signage": {
  "fp": "7c60d65e82bab169",
  "inscriptions": [],
  "cues": [
   {
    "text_native": "",
    "source": "scene_text_implied",
    "source_quote": "스마트폰 화면의 글자"
   }
  ],
  "dropped": []
 },
 "S8sh7": {
  "input_fingerprint": "9d0046db0ced15c8",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 스마트폰 화면의 글자를 향해 시선이 내리꽂힌 '윌마 디어링'의 굳은 눈동자.\n\nLOCATION (lock): Outside in the same forest-ridge clearing, immediately beside the cracked phone displaying the unfinished message. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: 스마트폰 (Held below Wilma's eye line with its active screen turned toward her) — The camera catches only the rear edge and back-facing side while the unseen display faces Wilma; used as A soft lower-foreground anchor that makes the downward eyeline physically legible without competing with her eyes.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained exterior daylight keeps the eye sockets firm but readable, with low-to-mid-key contrast and muted natural color.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the same wooded ridge environment, earthy ground tones, filtered daylight, and the rescue knife still lying nearby. Exclude the standing man and his raised-arm pose from the reference, and do not introduce shattered drone parts or unconscious pursuers.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The cracked smartphone remains lit with the 2026.10.18 date, 3 percent battery, and the unfinished message to MARA. The discarded rescue knife and MIDGE’s fragments remain on the forest floor. 윌마 디어링: Her gaze is lowered and fixed on the unfinished message, with her expression tense. She remains armed and still wears her wrist scanner and jumper belt.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마 디어링 right now, so 윌마 디어링's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마 디어링: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 스마트폰 화면의 글자를 향해 시선이 내리꽂힌 '윌마 디어링'의 굳은 눈동자.\n\nLOCATION (lock): Outside in the same forest-ridge clearing, immediately beside the cracked phone displaying the unfinished message. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: 스마트폰 (Held below Wilma's eye line with its active screen turned toward her) — The camera catches only the rear edge and back-facing side while the unseen display faces Wilma; used as A soft lower-foreground anchor that makes the downward eyeline physically legible without competing with her eyes.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained exterior daylight keeps the eye sockets firm but readable, with low-to-mid-key contrast and muted natural color.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the same wooded ridge environment, earthy ground tones, filtered daylight, and the rescue knife still lying nearby. Exclude the standing man and his raised-arm pose from the reference, and do not introduce shattered drone parts or unconscious pursuers.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The cracked smartphone remains lit with the 2026.10.18 date, 3 percent battery, and the unfinished message to MARA. The discarded rescue knife and MIDGE’s fragments remain on the forest floor. 윌마 디어링: Her gaze is lowered and fixed on the unfinished message, with her expression tense. She remains armed and still wears her wrist scanner and jumper belt.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마 디어링 right now, so 윌마 디어링's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마 디어링: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 스마트폰 화면의 글자를 향해 시선이 내리꽂힌 '윌마 디어링'의 굳은 눈동자.\n\nLOCATION (lock): Outside in the same forest-ridge clearing, immediately beside the cracked phone displaying the unfinished message. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: close-up\n- FRAME LAYOUT: 윌마 디어링 in the middle-center of the frame, foreground.\n- KEY BACKGROUND ELEMENTS: 스마트폰 (Held below Wilma's eye line with its active screen turned toward her) — The camera catches only the rear edge and back-facing side while the unseen display faces Wilma; used as A soft lower-foreground anchor that makes the downward eyeline physically legible without competing with her eyes.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Restrained exterior daylight keeps the eye sockets firm but readable, with low-to-mid-key contrast and muted natural color.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established about the place — its fixed features, wear and lighting — persists. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the same wooded ridge environment, earthy ground tones, filtered daylight, and the rescue knife still lying nearby. Exclude the standing man and his raised-arm pose from the reference, and do not introduce shattered drone parts or unconscious pursuers.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The cracked smartphone remains lit with the 2026.10.18 date, 3 percent battery, and the unfinished message to MARA. The discarded rescue knife and MIDGE’s fragments remain on the forest floor. 윌마 디어링: Her gaze is lowered and fixed on the unfinished message, with her expression tense. She remains armed and still wears her wrist scanner and jumper belt.\n\nTHE HAND THAT IS DOING THIS: the object at the centre of this shot is being held, operated, read, aimed or handed over by 윌마 디어링 right now, so 윌마 디어링's hand — and as much of the wrist and forearm as the framing reaches — is in the frame, gripping or resting on that object exactly the way its real use requires. Match that hand to 윌마 디어링: its size, build, skin, age, grooming, sleeve and anything worn on it belong to that person and to no one else.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "윌마는 화면 하단 전경에 있는 스마트폰을 향해 시선을 내리꽂고 있음.",
    "built_space": "나무가 늘어선 숲속 자연 환경과 흙바닥.",
    "entities": "윌마 디어링(참조 이미지의 이목구비와 일치, 녹색 점프수트 착용), 금이 간 스마트폰(후면이 카메라를 향함), 손목 스캐너.",
    "hard_violations": [
     "[gemini-pro] 명시되지 않은 인물 추가 (전경에서 스마트폰을 들고 있는 1인칭 시점의 두 손)",
     "[gemini-pro] 물리적으로 불가능한 신체 구조 및 무대 연출 (해당 손이 윌마의 것이라면 팔이 기형적으로 분리되어 카메라 위치에서 뻗어 나온 상태임)"
    ],
    "physics": "윌마는 화면 중앙에서 카메라를 마주보고 있으나, 스마트폰은 카메라 쪽에 위치한 누군가의 두 손에 의해 지탱되고 있음."
   },
   {
    "label": "B",
    "direction": "윌마는 전경의 숲 바닥에 놓인 스마트폰을 향해 굳은 시선을 내리꽂고 있음.",
    "built_space": "이끼가 낀 숲속 바닥과 나무 기둥들이 보이는 자연 공간.",
    "entities": "윌마 디어링(참조 이미지의 얼굴, 머리색, 복장과 매우 정확히 일치), 스마트폰(금이 간 후면이 보임), 구조용 칼(전경 바닥), 손목 스캐너(왼팔).",
    "hard_violations": [],
    "physics": "윌마는 바닥에 엎드려 양팔로 상체를 지탱하고 있음. 구조용 칼과 스마트폰은 이끼가 낀 바닥 위에 놓여 물리적으로 지탱됨. 단, 프롬프트의 지시와 달리 윌마의 손이 폰을 쥐고 있지 않음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 0,
        "verdict_ko": "카메라 시점에서 뻗어 나온 두 손은 프롬프트에 없는 제3의 인물을 창조했거나 윌마의 팔이 분리된 불가능한 신체 구조를 보여주므로 치명적인 위반입니다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "요구된 구도, 캐릭터의 외모, 굳은 시선과 배경 요소들을 사실적으로 잘 구현했으나, 스마트폰을 손으로 쥐고 있어야 한다는 지시사항을 어기고 바닥에 세워두어 감점되었습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마는 화면 하단 전경에 있는 스마트폰을 향해 시선을 내리꽂고 있음.",
        "built_space": "나무가 늘어선 숲속 자연 환경과 흙바닥.",
        "entities": "윌마 디어링(참조 이미지의 이목구비와 일치, 녹색 점프수트 착용), 금이 간 스마트폰(후면이 카메라를 향함), 손목 스캐너.",
        "hard_violations": [
         "명시되지 않은 인물 추가 (전경에서 스마트폰을 들고 있는 1인칭 시점의 두 손)",
         "물리적으로 불가능한 신체 구조 및 무대 연출 (해당 손이 윌마의 것이라면 팔이 기형적으로 분리되어 카메라 위치에서 뻗어 나온 상태임)"
        ],
        "physics": "윌마는 화면 중앙에서 카메라를 마주보고 있으나, 스마트폰은 카메라 쪽에 위치한 누군가의 두 손에 의해 지탱되고 있음."
       },
       {
        "label": "B",
        "direction": "윌마는 전경의 숲 바닥에 놓인 스마트폰을 향해 굳은 시선을 내리꽂고 있음.",
        "built_space": "이끼가 낀 숲속 바닥과 나무 기둥들이 보이는 자연 공간.",
        "entities": "윌마 디어링(참조 이미지의 얼굴, 머리색, 복장과 매우 정확히 일치), 스마트폰(금이 간 후면이 보임), 구조용 칼(전경 바닥), 손목 스캐너(왼팔).",
        "hard_violations": [],
        "physics": "윌마는 바닥에 엎드려 양팔로 상체를 지탱하고 있음. 구조용 칼과 스마트폰은 이끼가 낀 바닥 위에 놓여 물리적으로 지탱됨. 단, 프롬프트의 지시와 달리 윌마의 손이 폰을 쥐고 있지 않음."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 0,
        "verdict_ko": "카메라 시점에서 뻗어 나온 두 손은 프롬프트에 없는 제3의 인물을 창조했거나 윌마의 팔이 분리된 불가능한 신체 구조를 보여주므로 치명적인 위반입니다."
       },
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "요구된 구도, 캐릭터의 외모, 굳은 시선과 배경 요소들을 사실적으로 잘 구현했으나, 스마트폰을 손으로 쥐고 있어야 한다는 지시사항을 어기고 바닥에 세워두어 감점되었습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "윌마는 화면 하단 전경에 있는 스마트폰을 향해 시선을 내리꽂고 있음.",
        "built_space": "나무가 늘어선 숲속 자연 환경과 흙바닥.",
        "entities": "윌마 디어링(참조 이미지의 이목구비와 일치, 녹색 점프수트 착용), 금이 간 스마트폰(후면이 카메라를 향함), 손목 스캐너.",
        "hard_violations": [
         "명시되지 않은 인물 추가 (전경에서 스마트폰을 들고 있는 1인칭 시점의 두 손)",
         "물리적으로 불가능한 신체 구조 및 무대 연출 (해당 손이 윌마의 것이라면 팔이 기형적으로 분리되어 카메라 위치에서 뻗어 나온 상태임)"
        ],
        "physics": "윌마는 화면 중앙에서 카메라를 마주보고 있으나, 스마트폰은 카메라 쪽에 위치한 누군가의 두 손에 의해 지탱되고 있음."
       },
       {
        "label": "B",
        "direction": "윌마는 전경의 숲 바닥에 놓인 스마트폰을 향해 굳은 시선을 내리꽂고 있음.",
        "built_space": "이끼가 낀 숲속 바닥과 나무 기둥들이 보이는 자연 공간.",
        "entities": "윌마 디어링(참조 이미지의 얼굴, 머리색, 복장과 매우 정확히 일치), 스마트폰(금이 간 후면이 보임), 구조용 칼(전경 바닥), 손목 스캐너(왼팔).",
        "hard_violations": [],
        "physics": "윌마는 바닥에 엎드려 양팔로 상체를 지탱하고 있음. 구조용 칼과 스마트폰은 이끼가 낀 바닥 위에 놓여 물리적으로 지탱됨. 단, 프롬프트의 지시와 달리 윌마의 손이 폰을 쥐고 있지 않음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "윌마가 직접 쥔 휴대전화의 화면을 아래로 응시하는 순간과 후면만 보이는 방향을 정확히 구현했으며, 머리색과 휴대전화 크기가 기준보다 다소 어긋나는 점만 남는다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "굳은 눈동자의 하향 시선과 작은 하단 앵커 구도는 뛰어나지만, 휴대전화를 손으로 들고 읽어야 한다는 명시적 동작을 바닥에 세워 둔 연출로 바꾸었다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마의 두 눈은 렌즈가 아니라 얼굴 아래 중앙의 휴대전화 쪽으로 뚜렷하게 내려가며, 카메라에는 휴대전화의 금이 간 후면만 보이고 보이지 않는 화면은 윌마를 향한다. 칼은 바닥에 놓여 있어 어떤 대상을 겨누지 않는다.",
        "built_space": "야외 숲 공터로 인공 구조물이나 고정 설비는 없다. 이끼 낀 흙바닥, 굵은 나무와 여과된 낮빛이 이전 장면의 장소와 대체로 이어진다. 휴대전화 1대가 아래 중앙에 세워져 있고 구조용 칼 1자루가 그 앞 바닥에 놓여 있다. 윌마는 매우 낮게 엎드리거나 웅크린 채 팔을 지면 가까이에 둔다.",
        "entities": "등장인물은 윌마 1명뿐이며 20대 후반 미국인 여성으로 읽힌다. 밝은 갈색 머리, 얼굴 윤곽과 녹색 점프수트는 인물 기준과 잘 맞고 손목 스캐너도 보인다. 금이 간 휴대전화와 구조용 칼이 각각 보이며 화면의 메시지는 노출되지 않는다. 허리띠와 무기는 올바른 클로즈업 밖이라 확인되지 않는다.",
        "hard_violations": [],
        "physics": "윌마의 상체는 바닥 가까이 있으며 양쪽 전완과 지면이 자세를 지지하는 것으로 보인다. 칼은 숲 바닥에 완전히 놓여 있다. 휴대전화는 아래 모서리를 이끼와 흙에 대고 서 있어 지면의 지지는 있지만, 윌마의 손이 휴대전화를 잡거나 조작하지 않아 지시된 휴대 상태와 맞지 않고 안정성도 어색하다."
       },
       {
        "label": "B",
        "direction": "윌마의 눈은 얼굴 아래 전경에서 자신이 든 휴대전화 화면 쪽으로 내려가 있다. 화면은 윌마를 향하고 카메라에는 금이 간 후면과 후면 카메라만 보여 사용 방향이 정확하다. 무기나 다른 겨냥 물체는 보이지 않는다.",
        "built_space": "인공 구조물이나 고정 설비가 없는 같은 계열의 숲 공터다. 흙길과 낙엽, 수직 나무줄기, 여과된 주간광이 이전 장면과 잘 이어진다. 휴대전화 1대가 하단 전경에 있고 윌마 1명이 중앙 전경에 배치된다. 칼은 정확한 클로즈업 프레임 밖이라 보이지 않는다.",
        "entities": "윌마 외의 사람은 없다. 인물은 20대 후반 성인 여성이고 얼굴과 체격, 녹색 점프수트 및 손목 스캐너가 기준 인물에 대체로 부합한다. 다만 머리카락이 기준의 밝은 갈색보다 현저히 짙다. 금이 간 휴대전화가 있으며 기능면의 글자는 카메라에 노출되지 않는다. 허리띠와 무기는 프레임 밖이다.",
        "hard_violations": [],
        "physics": "윌마의 왼손 손가락이 휴대전화의 위쪽과 뒤쪽을 실제로 감싸 쥐고 있으며 반대쪽 손과 전완도 아래쪽에서 기기를 받치는 것으로 보여 휴대전화가 확실히 지지된다. 손목 스캐너도 전완에 고정되어 있다. 보이는 신체나 물체 중 공중에 무지지 상태로 떠 있는 것은 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "윌마가 직접 쥔 휴대전화의 화면을 아래로 응시하는 순간과 후면만 보이는 방향을 정확히 구현했으며, 머리색과 휴대전화 크기가 기준보다 다소 어긋나는 점만 남는다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "굳은 눈동자의 하향 시선과 작은 하단 앵커 구도는 뛰어나지만, 휴대전화를 손으로 들고 읽어야 한다는 명시적 동작을 바닥에 세워 둔 연출로 바꾸었다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "윌마의 두 눈은 렌즈가 아니라 얼굴 아래 중앙의 휴대전화 쪽으로 뚜렷하게 내려가며, 카메라에는 휴대전화의 금이 간 후면만 보이고 보이지 않는 화면은 윌마를 향한다. 칼은 바닥에 놓여 있어 어떤 대상을 겨누지 않는다.",
        "built_space": "야외 숲 공터로 인공 구조물이나 고정 설비는 없다. 이끼 낀 흙바닥, 굵은 나무와 여과된 낮빛이 이전 장면의 장소와 대체로 이어진다. 휴대전화 1대가 아래 중앙에 세워져 있고 구조용 칼 1자루가 그 앞 바닥에 놓여 있다. 윌마는 매우 낮게 엎드리거나 웅크린 채 팔을 지면 가까이에 둔다.",
        "entities": "등장인물은 윌마 1명뿐이며 20대 후반 미국인 여성으로 읽힌다. 밝은 갈색 머리, 얼굴 윤곽과 녹색 점프수트는 인물 기준과 잘 맞고 손목 스캐너도 보인다. 금이 간 휴대전화와 구조용 칼이 각각 보이며 화면의 메시지는 노출되지 않는다. 허리띠와 무기는 올바른 클로즈업 밖이라 확인되지 않는다.",
        "hard_violations": [],
        "physics": "윌마의 상체는 바닥 가까이 있으며 양쪽 전완과 지면이 자세를 지지하는 것으로 보인다. 칼은 숲 바닥에 완전히 놓여 있다. 휴대전화는 아래 모서리를 이끼와 흙에 대고 서 있어 지면의 지지는 있지만, 윌마의 손이 휴대전화를 잡거나 조작하지 않아 지시된 휴대 상태와 맞지 않고 안정성도 어색하다."
       },
       {
        "label": "A",
        "direction": "윌마의 눈은 얼굴 아래 전경에서 자신이 든 휴대전화 화면 쪽으로 내려가 있다. 화면은 윌마를 향하고 카메라에는 금이 간 후면과 후면 카메라만 보여 사용 방향이 정확하다. 무기나 다른 겨냥 물체는 보이지 않는다.",
        "built_space": "인공 구조물이나 고정 설비가 없는 같은 계열의 숲 공터다. 흙길과 낙엽, 수직 나무줄기, 여과된 주간광이 이전 장면과 잘 이어진다. 휴대전화 1대가 하단 전경에 있고 윌마 1명이 중앙 전경에 배치된다. 칼은 정확한 클로즈업 프레임 밖이라 보이지 않는다.",
        "entities": "윌마 외의 사람은 없다. 인물은 20대 후반 성인 여성이고 얼굴과 체격, 녹색 점프수트 및 손목 스캐너가 기준 인물에 대체로 부합한다. 다만 머리카락이 기준의 밝은 갈색보다 현저히 짙다. 금이 간 휴대전화가 있으며 기능면의 글자는 카메라에 노출되지 않는다. 허리띠와 무기는 프레임 밖이다.",
        "hard_violations": [],
        "physics": "윌마의 왼손 손가락이 휴대전화의 위쪽과 뒤쪽을 실제로 감싸 쥐고 있으며 반대쪽 손과 전완도 아래쪽에서 기기를 받치는 것으로 보여 휴대전화가 확실히 지지된다. 손목 스캐너도 전완에 고정되어 있다. 보이는 신체나 물체 중 공중에 무지지 상태로 떠 있는 것은 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": false,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "route": "cross_slot_combined"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.0,
    "B": 1.75
   },
   "adjusted": {
    "A": 0.75,
    "B": 1.75
   },
   "violations": {
    "A": [
     "[gemini-pro] 명시되지 않은 인물 추가 (전경에서 스마트폰을 들고 있는 1인칭 시점의 두 손)",
     "[gemini-pro] 물리적으로 불가능한 신체 구조 및 무대 연출 (해당 손이 윌마의 것이라면 팔이 기형적으로 분리되어 카메라 위치에서 뻗어 나온 상태임)"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "A"
   },
   "agreed": false
  },
  "totals": {
   "A": 750,
   "B": 1750
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 750,
    "verdict_ko": "카메라 시점에서 뻗어 나온 두 손은 프롬프트에 없는 제3의 인물을 창조했거나 윌마의 팔이 분리된 불가능한 신체 구조를 보여주므로 치명적인 위반입니다.  ★위반: [gemini-pro] 명시되지 않은 인물 추가 (전경에서 스마트폰을 들고 있는 1인칭 시점의 두 손) / [gemini-pro] 물리적으로 불가능한 신체 구조 및 무대 연출 (해당 손이 윌마의 것이라면 팔이 기형적으로 분리되어 카메라 위치에서 뻗어 나온 상태임)"
   },
   {
    "label": "B",
    "score": 1750,
    "verdict_ko": "요구된 구도, 캐릭터의 외모, 굳은 시선과 배경 요소들을 사실적으로 잘 구현했으나, 스마트폰을 손으로 쥐고 있어야 한다는 지시사항을 어기고 바닥에 세워두어 감점되었습니다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features and lighting mood are LOCKED to this photo; never copy its camera framing. The people visible in that photo are NOT in this shot: never carry their faces, bodies or clothing onto anyone here. Each person in THIS shot is defined solely by the PEOPLE list, their own CHARACTER REFERENCE images and their wardrobe notes. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S8sh6_sel.png",
    "asset_id": "d202e2c4-f008-46e5-910e-fc1110bee3a1",
    "role": "prev_still"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2e4-3b23-7443-99fc-89e0c5b77c04",
  "ref_mode": "prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S8sh6"
  }
 },
 "S8sh7::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:47:57.581758+00:00",
  "fingerprint": "7dabd2f7d1f4739edd09c84612fb32a1cfe2480dbaf5c6a55045cd0a70c56b33",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S8sh7_sel.png",
  "source_sha256": "f7d3b2246c4524eb3d7f7d7e9e65d826a16d335e4a67ced84594a70e44380672",
  "file": "S8sh7_cine.png",
  "staged_sha256": "1d8fa4d5bb1498d395b7fb571f7ec45d1f2c17ad1ace9b3c5c6fd10149751936",
  "latency_ms": 15242
 },
 "S9sh3::signage": {
  "fp": "a19cebc8e7ff03d7",
  "inscriptions": [
   {
    "text_native": "2419.03.18",
    "source": "scene_text_quoted",
    "reason_ko": "스마트폰 화면에 재배열되어 나타난 날짜 숫자입니다.",
    "source_quote": "2419.03.18"
   }
  ],
  "cues": [],
  "dropped": []
 },
 "S9sh3": {
  "input_fingerprint": "ff49f0c838e62a41",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): '2419.03.18'이라는 낯선 숫자로 재배열된 스마트폰 화면 클로즈업.\n\nLOCATION (lock): Outside at the rocky edge of a forested ridge cliff, where the synchronized phone is held moments before the aerial attack. The shot takes place here — the LOCATION text above is the only authority for this place — no location photograph is attached. Build the place strictly from that text and the shot text, inventing nothing beyond them.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- KEY BACKGROUND ELEMENTS: 스마트폰 화면 (Active; synchronization has stopped and the date reads 2419.03.18) — The illuminated front face is visible to camera, showing the fully settled synchronized date “2419.03.18.”; used as Central isolated information plane and endpoint of the dolly-in; 숲속 능선 바닥 (Visible only as a narrow area around the smartphone); used as Provides restrained physical scale around the device before the hard cut.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Exterior daylight is balanced against the active display so the synchronized date remains crisp without introducing an additional light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The cracked smartphone shows H.A.N. AUTHORITY HANDSHAKE ACCEPTED and NETWORK TIME SYNCHRONIZED, with its date now changed to 2419.03.18. The discarded rescue knife, MIDGE debris, and sprung trap remain at the ridge.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"2419.03.18\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): '2419.03.18'이라는 낯선 숫자로 재배열된 스마트폰 화면 클로즈업.\n\nLOCATION (lock): Outside at the rocky edge of a forested ridge cliff, where the synchronized phone is held moments before the aerial attack. The shot takes place here — the LOCATION text above is the only authority for this place — no location photograph is attached. Build the place strictly from that text and the shot text, inventing nothing beyond them.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- KEY BACKGROUND ELEMENTS: 스마트폰 화면 (Active; synchronization has stopped and the date reads 2419.03.18) — The illuminated front face is visible to camera, showing the fully settled synchronized date “2419.03.18.”; used as Central isolated information plane and endpoint of the dolly-in; 숲속 능선 바닥 (Visible only as a narrow area around the smartphone); used as Provides restrained physical scale around the device before the hard cut.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Exterior daylight is balanced against the active display so the synchronized date remains crisp without introducing an additional light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The cracked smartphone shows H.A.N. AUTHORITY HANDSHAKE ACCEPTED and NETWORK TIME SYNCHRONIZED, with its date now changed to 2419.03.18. The discarded rescue knife, MIDGE debris, and sprung trap remain at the ridge.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"2419.03.18\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): '2419.03.18'이라는 낯선 숫자로 재배열된 스마트폰 화면 클로즈업.\n\nLOCATION (lock): Outside at the rocky edge of a forested ridge cliff, where the synchronized phone is held moments before the aerial attack. The shot takes place here — the LOCATION text above is the only authority for this place — no location photograph is attached. Build the place strictly from that text and the shot text, inventing nothing beyond them.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: insert close-up on a detail\n- KEY BACKGROUND ELEMENTS: 스마트폰 화면 (Active; synchronization has stopped and the date reads 2419.03.18) — The illuminated front face is visible to camera, showing the fully settled synchronized date “2419.03.18.”; used as Central isolated information plane and endpoint of the dolly-in; 숲속 능선 바닥 (Visible only as a narrow area around the smartphone); used as Provides restrained physical scale around the device before the hard cut.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Exterior daylight is balanced against the active display so the synchronized date remains crisp without introducing an additional light source.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The cracked smartphone shows H.A.N. AUTHORITY HANDSHAKE ACCEPTED and NETWORK TIME SYNCHRONIZED, with its date now changed to 2419.03.18. The discarded rescue knife, MIDGE debris, and sprung trap remain at the ridge.\n\nNO PEOPLE IN THIS SHOT: the shot text shows only the place and its state — no living person appears in frame, and no face, and no figure standing, sitting or passing through. Do not invent someone to occupy the space.\n\nOne thing this rule does NOT forbid: the hand or arm of whoever is doing what the shot text describes. If the moment is an object being held, carried, operated, read, aimed, worn or handed over, the hand doing it belongs in frame, resting on or gripping that object exactly as its real use requires. An object photographed floating unsupported, with the hand that must be holding it erased, is a worse failure than the person you were avoiding — render the hand.\n\nWORDS TO RENDER (authoritative — the scene itself calls for these; render each as period-real physical lettering in the native script, exactly as written; add no other readable text anywhere):\n- \"2419.03.18\"\n\nThe WORDS TO RENDER above are the only readable writing in this image: render those words exactly as given, in the place and era's own language and script, and nothing else legible. Invent no other wording a viewer could read. No caption, subtitle, watermark, logo or overlay. Surfaces that would carry writing may still be present — stage any wording they would carry out of legibility: a hand across, an oblique angle, shallow focus.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "카메라는 바위 위에서 스마트폰을 쥐고 있는 손을 위에서 아래로 내려다보고 있으며, 스마트폰 화면은 카메라를 정면으로 향하고 있습니다.",
    "built_space": "흐릿한 숲이 배경으로 보이는 바위가 많은 절벽 가장자리입니다. 화면 너머 바위 위에는 구조용 칼, 금속 잔해, 그리고 덫이 놓여 있습니다.",
    "entities": "금이 간 스마트폰을 단단히 쥐고 있는 사람의 오른손입니다. 화면 중앙에는 '2419.03.18'이라는 숫자만 선명하게 떠 있습니다.",
    "hard_violations": [],
    "physics": "손이 스마트폰을 중력에 맞게 자연스럽게 쥐고 있으며, 배경의 칼과 파편, 덫은 바위 표면 위에 안정적으로 놓여 있습니다."
   },
   {
    "label": "B",
    "direction": "카메라는 바닥을 배경으로 스마트폰을 든 손을 1인칭 시점과 유사하게 내려다보고 있으며, 화면은 카메라 쪽을 향합니다.",
    "built_space": "숲을 배경으로 한 바위 바닥입니다. 배경의 초점이 나간 바위 위에는 칼, 덫, 그리고 상자 형태의 잔해가 있습니다.",
    "entities": "금이 간 스마트폰을 쥔 오른손입니다. 화면에는 지정된 날짜와 상태 메시지 외에도 'AT&T', 시간, 배터리 잔량 등의 UI가 존재하며, 배경의 잔해에는 'MIDGE'라는 글자가 적혀 있습니다.",
    "hard_violations": [
     "[gemini-pro] 절대적으로 명시된 텍스트('2419.03.18') 외에는 어떠한 읽을 수 있는 글자도 넣지 말라는 지침을 위반하고, 스마트폰 상단바(AT&T, 8:32 AM, 80%)와 상태 텍스트(Network Status), 배경 물체(MIDGE)에 임의의 텍스트를 생성함",
     "[gpt] 화면에 허용된 “2419.03.18” 외에 “Network Status”, “H.A.N. AUTHORITY HANDSHAKE ACCEPTED”, “NETWORK TIME SYNCHRONIZED” 및 상태표시 문자가 선명하게 노출되어, 다른 읽을 수 있는 문구를 금지한 지시를 위반한다."
    ],
    "physics": "손이 기기를 안정적으로 받쳐 들고 있으며, 배경 소품들 역시 바위 위에 놓여 지지를 받고 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "지정된 텍스트('2419.03.18')만을 정확하게 표시하고 화면의 질감 및 배경 소품(칼, 잔해, 덫)의 배치를 지시사항에 맞게 사실적으로 구현했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "스마트폰 화면과 배경 물체에 지시되지 않은 임의의 텍스트(통신사 이름, 시간, 메뉴 등)를 대거 생성하여 텍스트 제한 지침을 심각하게 위반했습니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 바위 위에서 스마트폰을 쥐고 있는 손을 위에서 아래로 내려다보고 있으며, 스마트폰 화면은 카메라를 정면으로 향하고 있습니다.",
        "built_space": "흐릿한 숲이 배경으로 보이는 바위가 많은 절벽 가장자리입니다. 화면 너머 바위 위에는 구조용 칼, 금속 잔해, 그리고 덫이 놓여 있습니다.",
        "entities": "금이 간 스마트폰을 단단히 쥐고 있는 사람의 오른손입니다. 화면 중앙에는 '2419.03.18'이라는 숫자만 선명하게 떠 있습니다.",
        "hard_violations": [],
        "physics": "손이 스마트폰을 중력에 맞게 자연스럽게 쥐고 있으며, 배경의 칼과 파편, 덫은 바위 표면 위에 안정적으로 놓여 있습니다."
       },
       {
        "label": "B",
        "direction": "카메라는 바닥을 배경으로 스마트폰을 든 손을 1인칭 시점과 유사하게 내려다보고 있으며, 화면은 카메라 쪽을 향합니다.",
        "built_space": "숲을 배경으로 한 바위 바닥입니다. 배경의 초점이 나간 바위 위에는 칼, 덫, 그리고 상자 형태의 잔해가 있습니다.",
        "entities": "금이 간 스마트폰을 쥔 오른손입니다. 화면에는 지정된 날짜와 상태 메시지 외에도 'AT&T', 시간, 배터리 잔량 등의 UI가 존재하며, 배경의 잔해에는 'MIDGE'라는 글자가 적혀 있습니다.",
        "hard_violations": [
         "절대적으로 명시된 텍스트('2419.03.18') 외에는 어떠한 읽을 수 있는 글자도 넣지 말라는 지침을 위반하고, 스마트폰 상단바(AT&T, 8:32 AM, 80%)와 상태 텍스트(Network Status), 배경 물체(MIDGE)에 임의의 텍스트를 생성함"
        ],
        "physics": "손이 기기를 안정적으로 받쳐 들고 있으며, 배경 소품들 역시 바위 위에 놓여 지지를 받고 있습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 9,
        "verdict_ko": "지정된 텍스트('2419.03.18')만을 정확하게 표시하고 화면의 질감 및 배경 소품(칼, 잔해, 덫)의 배치를 지시사항에 맞게 사실적으로 구현했습니다."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "스마트폰 화면과 배경 물체에 지시되지 않은 임의의 텍스트(통신사 이름, 시간, 메뉴 등)를 대거 생성하여 텍스트 제한 지침을 심각하게 위반했습니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "카메라는 바위 위에서 스마트폰을 쥐고 있는 손을 위에서 아래로 내려다보고 있으며, 스마트폰 화면은 카메라를 정면으로 향하고 있습니다.",
        "built_space": "흐릿한 숲이 배경으로 보이는 바위가 많은 절벽 가장자리입니다. 화면 너머 바위 위에는 구조용 칼, 금속 잔해, 그리고 덫이 놓여 있습니다.",
        "entities": "금이 간 스마트폰을 단단히 쥐고 있는 사람의 오른손입니다. 화면 중앙에는 '2419.03.18'이라는 숫자만 선명하게 떠 있습니다.",
        "hard_violations": [],
        "physics": "손이 스마트폰을 중력에 맞게 자연스럽게 쥐고 있으며, 배경의 칼과 파편, 덫은 바위 표면 위에 안정적으로 놓여 있습니다."
       },
       {
        "label": "B",
        "direction": "카메라는 바닥을 배경으로 스마트폰을 든 손을 1인칭 시점과 유사하게 내려다보고 있으며, 화면은 카메라 쪽을 향합니다.",
        "built_space": "숲을 배경으로 한 바위 바닥입니다. 배경의 초점이 나간 바위 위에는 칼, 덫, 그리고 상자 형태의 잔해가 있습니다.",
        "entities": "금이 간 스마트폰을 쥔 오른손입니다. 화면에는 지정된 날짜와 상태 메시지 외에도 'AT&T', 시간, 배터리 잔량 등의 UI가 존재하며, 배경의 잔해에는 'MIDGE'라는 글자가 적혀 있습니다.",
        "hard_violations": [
         "절대적으로 명시된 텍스트('2419.03.18') 외에는 어떠한 읽을 수 있는 글자도 넣지 말라는 지침을 위반하고, 스마트폰 상단바(AT&T, 8:32 AM, 80%)와 상태 텍스트(Network Status), 배경 물체(MIDGE)에 임의의 텍스트를 생성함"
        ],
        "physics": "손이 기기를 안정적으로 받쳐 들고 있으며, 배경 소품들 역시 바위 위에 놓여 지지를 받고 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "스마트폰 중심의 인서트 클로즈업과 정확한 날짜는 좋지만, 화면에 금지된 추가 영문이 다수 선명하게 보여 결정적인 하드 위반이다."
       },
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "프레임이 다소 넓지만, 손에 지지된 깨진 스마트폰의 전면에 허용된 문구인 “2419.03.18”만 정확히 표시해 핵심 순간을 가장 충실히 구현했다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "스마트폰의 전면 화면이 카메라를 향하며 날짜가 정면에 가깝게 읽힌다. 무기처럼 조준되는 물체나 이동하는 몸은 없고, 오른쪽 아래 칼날도 특정 대상을 겨누지 않는다.",
        "built_space": "야외 바위 능선으로 만들어진 구조물이나 고정 설비는 없다. 스마트폰 주변에 바위 표면, 흐릿한 숲, 왼쪽의 작은 장비, 오른쪽 위의 덫 같은 물체와 오른쪽 아래 칼이 보인다. 휴대전화 주변만 보이는 비교적 타이트한 구도여서 인서트 클로즈업에는 잘 맞는다.",
        "entities": "깨진 스마트폰, 이를 쥔 한 손, 바위 능선과 숲이 보이며 사람의 얼굴이나 전신은 없다. 날짜 “2419.03.18”은 정확하지만, 화면에는 “Network Status”, “H.A.N. AUTHORITY HANDSHAKE ACCEPTED”, “NETWORK TIME SYNCHRONIZED”와 상태표시 문자가 추가로 읽힌다. 주변 물체는 구조용 칼·잔해·덫으로 보일 가능성은 있으나 정체가 명확하지 않다.",
        "hard_violations": [
         "화면에 허용된 “2419.03.18” 외에 “Network Status”, “H.A.N. AUTHORITY HANDSHAKE ACCEPTED”, “NETWORK TIME SYNCHRONIZED” 및 상태표시 문자가 선명하게 노출되어, 다른 읽을 수 있는 문구를 금지한 지시를 위반한다."
        ],
        "physics": "스마트폰은 왼손의 손바닥과 감긴 손가락들로 확실히 지지되며 공중에 뜨지 않는다. 배경의 칼과 장비도 바위 표면 위에 놓여 있어 물리적 지지가 있다."
       },
       {
        "label": "B",
        "direction": "스마트폰의 illuminated 전면이 카메라를 향하고 중앙 날짜가 똑바로 읽힌다. 조준 무기나 이동체는 없으며, 배경 왼쪽의 칼 같은 물체도 특정 대상을 향해 사용되는 상태가 아니다.",
        "built_space": "야외의 바위 절벽 가장자리와 뒤쪽 침엽수림이 보이고 인공 구조물이나 고정 설비는 없다. 전화 뒤 바위 위에 칼 같은 물체와 여러 금속성 잔해가 놓여 있다. 장소는 숲이 우거진 바위 능선 절벽으로 읽히지만, 스마트폰 주변의 지면이 ‘좁은 영역’보다 넓게 보여 요구된 인서트보다 약간 와이드하다.",
        "entities": "금이 간 스마트폰과 이를 쥔 한 손, 바위 능선, 숲, 칼처럼 보이는 물체 및 잔해가 있다. 사람의 얼굴·전신·추가 인물은 없다. 화면에는 정확한 “2419.03.18”만 읽히고 다른 로고나 문구는 없다. 다만 배경 물체가 구조용 칼, MIDGE 잔해, 작동된 덫인지 각각 확실하게 식별되지는 않는다.",
        "hard_violations": [],
        "physics": "스마트폰은 오른손의 손바닥과 여러 손가락이 가장자리와 뒤를 단단히 잡아 지지한다. 배경 물체들은 바위 지면 위에 놓여 있으며 지지 없이 떠 있는 물체나 신체는 없다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "스마트폰 중심의 인서트 클로즈업과 정확한 날짜는 좋지만, 화면에 금지된 추가 영문이 다수 선명하게 보여 결정적인 하드 위반이다."
       },
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "프레임이 다소 넓지만, 손에 지지된 깨진 스마트폰의 전면에 허용된 문구인 “2419.03.18”만 정확히 표시해 핵심 순간을 가장 충실히 구현했다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "스마트폰의 전면 화면이 카메라를 향하며 날짜가 정면에 가깝게 읽힌다. 무기처럼 조준되는 물체나 이동하는 몸은 없고, 오른쪽 아래 칼날도 특정 대상을 겨누지 않는다.",
        "built_space": "야외 바위 능선으로 만들어진 구조물이나 고정 설비는 없다. 스마트폰 주변에 바위 표면, 흐릿한 숲, 왼쪽의 작은 장비, 오른쪽 위의 덫 같은 물체와 오른쪽 아래 칼이 보인다. 휴대전화 주변만 보이는 비교적 타이트한 구도여서 인서트 클로즈업에는 잘 맞는다.",
        "entities": "깨진 스마트폰, 이를 쥔 한 손, 바위 능선과 숲이 보이며 사람의 얼굴이나 전신은 없다. 날짜 “2419.03.18”은 정확하지만, 화면에는 “Network Status”, “H.A.N. AUTHORITY HANDSHAKE ACCEPTED”, “NETWORK TIME SYNCHRONIZED”와 상태표시 문자가 추가로 읽힌다. 주변 물체는 구조용 칼·잔해·덫으로 보일 가능성은 있으나 정체가 명확하지 않다.",
        "hard_violations": [
         "화면에 허용된 “2419.03.18” 외에 “Network Status”, “H.A.N. AUTHORITY HANDSHAKE ACCEPTED”, “NETWORK TIME SYNCHRONIZED” 및 상태표시 문자가 선명하게 노출되어, 다른 읽을 수 있는 문구를 금지한 지시를 위반한다."
        ],
        "physics": "스마트폰은 왼손의 손바닥과 감긴 손가락들로 확실히 지지되며 공중에 뜨지 않는다. 배경의 칼과 장비도 바위 표면 위에 놓여 있어 물리적 지지가 있다."
       },
       {
        "label": "A",
        "direction": "스마트폰의 illuminated 전면이 카메라를 향하고 중앙 날짜가 똑바로 읽힌다. 조준 무기나 이동체는 없으며, 배경 왼쪽의 칼 같은 물체도 특정 대상을 향해 사용되는 상태가 아니다.",
        "built_space": "야외의 바위 절벽 가장자리와 뒤쪽 침엽수림이 보이고 인공 구조물이나 고정 설비는 없다. 전화 뒤 바위 위에 칼 같은 물체와 여러 금속성 잔해가 놓여 있다. 장소는 숲이 우거진 바위 능선 절벽으로 읽히지만, 스마트폰 주변의 지면이 ‘좁은 영역’보다 넓게 보여 요구된 인서트보다 약간 와이드하다.",
        "entities": "금이 간 스마트폰과 이를 쥔 한 손, 바위 능선, 숲, 칼처럼 보이는 물체 및 잔해가 있다. 사람의 얼굴·전신·추가 인물은 없다. 화면에는 정확한 “2419.03.18”만 읽히고 다른 로고나 문구는 없다. 다만 배경 물체가 구조용 칼, MIDGE 잔해, 작동된 덫인지 각각 확실하게 식별되지는 않는다.",
        "hard_violations": [],
        "physics": "스마트폰은 오른손의 손바닥과 여러 손가락이 가장자리와 뒤를 단단히 잡아 지지한다. 배경 물체들은 바위 지면 위에 놓여 있으며 지지 없이 떠 있는 물체나 신체는 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 2.0,
    "B": 0.819
   },
   "adjusted": {
    "A": 2.0,
    "B": 0.569
   },
   "violations": {
    "B": [
     "[gemini-pro] 절대적으로 명시된 텍스트('2419.03.18') 외에는 어떠한 읽을 수 있는 글자도 넣지 말라는 지침을 위반하고, 스마트폰 상단바(AT&T, 8:32 AM, 80%)와 상태 텍스트(Network Status), 배경 물체(MIDGE)에 임의의 텍스트를 생성함",
     "[gpt] 화면에 허용된 “2419.03.18” 외에 “Network Status”, “H.A.N. AUTHORITY HANDSHAKE ACCEPTED”, “NETWORK TIME SYNCHRONIZED” 및 상태표시 문자가 선명하게 노출되어, 다른 읽을 수 있는 문구를 금지한 지시를 위반한다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "A",
    "gpt": "A"
   },
   "agreed": true
  },
  "totals": {
   "A": 2000,
   "B": 569
  },
  "selected": "A",
  "ranking": [
   "A",
   "B"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 2000,
    "verdict_ko": "지정된 텍스트('2419.03.18')만을 정확하게 표시하고 화면의 질감 및 배경 소품(칼, 잔해, 덫)의 배치를 지시사항에 맞게 사실적으로 구현했습니다."
   },
   {
    "label": "B",
    "score": 569,
    "verdict_ko": "스마트폰 화면과 배경 물체에 지시되지 않은 임의의 텍스트(통신사 이름, 시간, 메뉴 등)를 대거 생성하여 텍스트 제한 지침을 심각하게 위반했습니다.  ★위반: [gemini-pro] 절대적으로 명시된 텍스트('2419.03.18') 외에는 어떠한 읽을 수 있는 글자도 넣지 말라는 지침을 위반하고, 스마트폰 상단바(AT&T, 8:32 AM, 80%)와 상태 텍스트(Network Status), 배경 물체(MIDGE)에 임의의 텍스트를 생성함 / [gpt] 화면에 허용된 “2419.03.18” 외에 “Network Status”, “H.A.N. AUTHORITY HANDSHAKE ACCEPTED”, “NETWORK TIME SYNCHRONIZED” 및 상태표시 문자가 선명하게 노출되어, 다른 읽을 수 있는 문구를 금지한 지시를 위반한다."
   }
  ],
  "refs": [],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2e8-cf97-73a7-b91c-22aeb2916da4",
  "ref_mode": "플레이트만 (배경 전용)",
  "share_plan": {
   "ref_plan": "background"
  }
 },
 "S9sh3::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:48:55.941460+00:00",
  "fingerprint": "f4eb779589298571f14587b65049b80caa86461f45f59b3aae03f78e2b64ce18",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S9sh3_sel.png",
  "source_sha256": "c419fd5c604d5a205a702bafd6916b90063dbfa28d7f4aa8bdacaae77ecd9588",
  "file": "S9sh3_cine.png",
  "staged_sha256": "1f02f1251b7d6e02c163ae717fb459c2ebc9319123d30f7444a630849e8b6482",
  "latency_ms": 22039
 },
 "S9sh5::signage": {
  "fp": "3571a39f6440d6c6",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "S9sh5": {
  "input_fingerprint": "20f7780ae958be39",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 우거진 숲 한가운데로 수직으로 꽂혀 내린 거대하고 창백한 원통형 빛기둥.\n\nLOCATION (lock): Outside in the dense forest roughly one hundred meters beyond the ridge cliff, where a pale vertical beam erases a column of trees. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: 우거진 숲 (Broadly intact except for the space erased where the first column reaches it); used as Large-scale environmental reference that establishes the column's magnitude; 첫 번째 원통형 빛기둥 (Descending vertically into the forest); used as Primary vertical focal form spanning from the cloud layer to the forest; 구름 (The pale column descends through them); used as Upper-frame endpoint that allows the full height of the column to read; 지워진 숲의 빈 공간 (Trees are absent without cut or burned remnants); used as Ground-level evidence of the column's effect within the wide composition.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight over the forest is interrupted by the explicitly pale light column, preserving subdued greens and neutral contrast around its brighter vertical form.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone has been powered off. A huge pale cylindrical beam descends through the clouds and leaves an empty void where trees stood one hundred meters ahead, while a second beam begins sweeping closer across the forest.\n\nPEOPLE: the SHOT TEXT alone decides who is visible in this shot. People known to appear somewhere in this scene: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락). That list is scene-level, not a cast list for this frame — it may name someone this shot does not show, and it may omit someone this shot does show. If the shot text names a person who is not on the list, draw that person exactly as the shot text describes them; the list does not override the shot text. Never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 우거진 숲 한가운데로 수직으로 꽂혀 내린 거대하고 창백한 원통형 빛기둥.\n\nLOCATION (lock): Outside in the dense forest roughly one hundred meters beyond the ridge cliff, where a pale vertical beam erases a column of trees. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: 우거진 숲 (Broadly intact except for the space erased where the first column reaches it); used as Large-scale environmental reference that establishes the column's magnitude; 첫 번째 원통형 빛기둥 (Descending vertically into the forest); used as Primary vertical focal form spanning from the cloud layer to the forest; 구름 (The pale column descends through them); used as Upper-frame endpoint that allows the full height of the column to read; 지워진 숲의 빈 공간 (Trees are absent without cut or burned remnants); used as Ground-level evidence of the column's effect within the wide composition.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight over the forest is interrupted by the explicitly pale light column, preserving subdued greens and neutral contrast around its brighter vertical form.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone has been powered off. A huge pale cylindrical beam descends through the clouds and leaves an empty void where trees stood one hundred meters ahead, while a second beam begins sweeping closer across the forest.\n\nPEOPLE: the SHOT TEXT alone decides who is visible in this shot. People known to appear somewhere in this scene: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락). That list is scene-level, not a cast list for this frame — it may name someone this shot does not show, and it may omit someone this shot does show. If the shot text names a person who is not on the list, draw that person exactly as the shot text describes them; the list does not override the shot text. Never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): 우거진 숲 한가운데로 수직으로 꽂혀 내린 거대하고 창백한 원통형 빛기둥.\n\nLOCATION (lock): Outside in the dense forest roughly one hundred meters beyond the ridge cliff, where a pale vertical beam erases a column of trees. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- KEY BACKGROUND ELEMENTS: 우거진 숲 (Broadly intact except for the space erased where the first column reaches it); used as Large-scale environmental reference that establishes the column's magnitude; 첫 번째 원통형 빛기둥 (Descending vertically into the forest); used as Primary vertical focal form spanning from the cloud layer to the forest; 구름 (The pale column descends through them); used as Upper-frame endpoint that allows the full height of the column to read; 지워진 숲의 빈 공간 (Trees are absent without cut or burned remnants); used as Ground-level evidence of the column's effect within the wide composition.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Daylight over the forest is interrupted by the explicitly pale light column, preserving subdued greens and neutral contrast around its brighter vertical form.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): The smartphone has been powered off. A huge pale cylindrical beam descends through the clouds and leaves an empty void where trees stood one hundred meters ahead, while a second beam begins sweeping closer across the forest.\n\nPEOPLE: the SHOT TEXT alone decides who is visible in this shot. People known to appear somewhere in this scene: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락). That list is scene-level, not a cast list for this frame — it may name someone this shot does not show, and it may omit someone this shot does show. If the shot text names a person who is not on the list, draw that person exactly as the shot text describes them; the list does not override the shot text. Never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "shot_run_spend_attempt_count": 1,
  "initial_roll_all_fail": true,
  "readings": [
   {
    "label": "A",
    "direction": "두 사람은 카메라 쪽을 바라보며, 두 빛기둥은 숲을 향해 수직으로 꽂힘.",
    "built_space": "전경의 바위 절벽 위에 폰과 도구들이 놓여 있고, 뒤로 숲이 펼쳐짐.",
    "entities": "두 개의 빛기둥과 빈 공간은 존재하나, 텍스트에 없는 두 인물이 등장하고 폰 화면에 텍스트가 보임.",
    "hard_violations": [
     "[gemini-pro] extra bodies",
     "[gemini-pro] leaked markers/diagrams/text",
     "[gpt] 쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다.",
     "[gpt] 전경 스마트폰 화면에 읽을 수 있는 날짜 글자가 노출되어 ‘읽을 수 있는 글자 금지’를 위반했다."
    ],
    "physics": "도구들은 바위 위에 안정적으로 놓여 있고, 사람들은 절벽 아래쪽 지면에 의해 지탱됨."
   },
   {
    "label": "B",
    "direction": "두 사람의 시선과 남자가 든 폰이 숲의 빛기둥을 향함. 하나의 빛기둥이 수직으로 내려옴.",
    "built_space": "바위 절벽 위에 두 사람이 서 있고 우측에 도구들이 놓여 있으며, 넓은 숲 지형이 보임.",
    "entities": "첫 번째 빛기둥은 있으나 두 번째가 누락됨. 텍스트에 없는 두 인물이 임의로 추가됨.",
    "hard_violations": [
     "[gemini-pro] extra bodies",
     "[gpt] 쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다."
    ],
    "physics": "사람들은 바위에 서서 체중을 지탱하고 있으며 남자의 손이 폰을 단단히 쥐고 있음."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "지시문에 없는 인물이 임의로 추가되었고 스마트폰 화면에 읽을 수 있는 텍스트가 노출되어 지침을 심각하게 위반함."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "텍스트 노출은 피했으나 지시문에 없는 인물이 추가되었으며 명시된 두 번째 빛기둥이 누락됨."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "두 사람은 카메라 쪽을 바라보며, 두 빛기둥은 숲을 향해 수직으로 꽂힘.",
        "built_space": "전경의 바위 절벽 위에 폰과 도구들이 놓여 있고, 뒤로 숲이 펼쳐짐.",
        "entities": "두 개의 빛기둥과 빈 공간은 존재하나, 텍스트에 없는 두 인물이 등장하고 폰 화면에 텍스트가 보임.",
        "hard_violations": [
         "extra bodies",
         "leaked markers/diagrams/text"
        ],
        "physics": "도구들은 바위 위에 안정적으로 놓여 있고, 사람들은 절벽 아래쪽 지면에 의해 지탱됨."
       },
       {
        "label": "B",
        "direction": "두 사람의 시선과 남자가 든 폰이 숲의 빛기둥을 향함. 하나의 빛기둥이 수직으로 내려옴.",
        "built_space": "바위 절벽 위에 두 사람이 서 있고 우측에 도구들이 놓여 있으며, 넓은 숲 지형이 보임.",
        "entities": "첫 번째 빛기둥은 있으나 두 번째가 누락됨. 텍스트에 없는 두 인물이 임의로 추가됨.",
        "hard_violations": [
         "extra bodies"
        ],
        "physics": "사람들은 바위에 서서 체중을 지탱하고 있으며 남자의 손이 폰을 단단히 쥐고 있음."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "지시문에 없는 인물이 임의로 추가되었고 스마트폰 화면에 읽을 수 있는 텍스트가 노출되어 지침을 심각하게 위반함."
       },
       {
        "label": "B",
        "score": 4,
        "verdict_ko": "텍스트 노출은 피했으나 지시문에 없는 인물이 추가되었으며 명시된 두 번째 빛기둥이 누락됨."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "A",
        "direction": "두 사람은 카메라 쪽을 바라보며, 두 빛기둥은 숲을 향해 수직으로 꽂힘.",
        "built_space": "전경의 바위 절벽 위에 폰과 도구들이 놓여 있고, 뒤로 숲이 펼쳐짐.",
        "entities": "두 개의 빛기둥과 빈 공간은 존재하나, 텍스트에 없는 두 인물이 등장하고 폰 화면에 텍스트가 보임.",
        "hard_violations": [
         "extra bodies",
         "leaked markers/diagrams/text"
        ],
        "physics": "도구들은 바위 위에 안정적으로 놓여 있고, 사람들은 절벽 아래쪽 지면에 의해 지탱됨."
       },
       {
        "label": "B",
        "direction": "두 사람의 시선과 남자가 든 폰이 숲의 빛기둥을 향함. 하나의 빛기둥이 수직으로 내려옴.",
        "built_space": "바위 절벽 위에 두 사람이 서 있고 우측에 도구들이 놓여 있으며, 넓은 숲 지형이 보임.",
        "entities": "첫 번째 빛기둥은 있으나 두 번째가 누락됨. 텍스트에 없는 두 인물이 임의로 추가됨.",
        "hard_violations": [
         "extra bodies"
        ],
        "physics": "사람들은 바위에 서서 체중을 지탱하고 있으며 남자의 손이 폰을 단단히 쥐고 있음."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 5,
        "verdict_ko": "구름에서 숲까지 이어지는 거대한 단일 광기둥과 넓은 숲 전경은 핵심 쇼트에 가깝지만, 쇼트 텍스트에 없는 두 사람을 추가했고 지워진 숲의 빈 공간과 가까워지는 두 번째 빔도 명확하지 않다."
       },
       {
        "label": "B",
        "score": 3,
        "verdict_ko": "첫 빔과 두 번째 빔 및 숲의 빈 공간은 비교적 잘 보이지만, 불필요한 두 사람과 전경 스마트폰을 넣고 금지된 날짜 글자까지 노출해 A보다 중대한 위반이 많다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "남성은 얼굴과 손에 든 스마트폰을 화면 중앙의 광기둥 쪽으로 향하고 있으며, 여성도 같은 중앙 광기둥을 바라본다. 광기둥은 구름에서 수직으로 내려와 숲 중앙에 닿는다. 무기나 다른 지시 물체는 없다.",
        "built_space": "야외의 바위 절벽 가장자리에서 계곡의 울창한 숲을 내려다보는 넓은 구도다. 중앙에 광기둥 하나, 상단에 구름층, 전경 오른쪽에 두 사람이 서 있는 바위 지대가 보인다. 맞은편에는 암벽 능선이 있다. 광기둥이 닿는 지점에도 나무 윤곽이 남아 보여, 절단·연소 흔적 없이 나무가 사라진 원통형 빈 공간이 뚜렷하게 확인되지는 않는다.",
        "entities": "중앙의 창백하고 거대한 수직 광기둥과 울창한 숲, 구름은 해당 대상에 부합한다. 그러나 쇼트 텍스트가 사람을 보여주지 않는데도 30대 남성과 20대 후반 여성으로 보이는 두 사람이 등장한다. 남성은 스마트폰을 들고 있고 화면은 어둡게 보여 전원 꺼짐 상태일 가능성은 있으나, 이 쇼트의 요구된 피사체는 아니다. 가까운 숲을 훑기 시작하는 두 번째 빔은 보이지 않는다.",
        "hard_violations": [
         "쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다."
        ],
        "physics": "두 사람 모두 바위 지면에 양발을 대고 서 있어 신체가 지지된다. 남성의 스마트폰은 양손으로 잡혀 있어 떠 있지 않으며 실제 사용 방향대로 남성 쪽에 화면이 향한다. 광기둥은 구름에서 숲까지 수직으로 이어지는 빛 현상으로 표현되어 별도의 물리적 받침이 필요한 물체처럼 보이지 않는다."
       },
       {
        "label": "B",
        "direction": "여성은 고개를 들어 화면 중앙 왼쪽의 큰 광기둥 쪽을 바라본다. 남성도 위쪽을 보지만 시선은 큰 광기둥보다 약간 오른쪽으로 치우쳐 있다. 큰 광기둥은 구름에서 숲의 빈 공간으로 수직 하강하고, 오른쪽 뒤편의 두 번째 빔도 수직으로 숲을 향한다. 두 번째 빔은 가까이 가로질러 훑는 운동 방향보다 멀리 고정된 또 하나의 수직 기둥처럼 읽힌다.",
        "built_space": "야외 바위 능선 전경, 그 아래의 두 사람, 중·후경의 울창한 숲과 구름층으로 구성된 넓은 장면이다. 중앙 왼쪽에 큰 광기둥 하나, 오른쪽 후경에 작은 광기둥 하나가 있다. 첫 빔의 바닥에는 나무가 없는 둥근 공터가 보여 삭제된 숲의 증거가 A보다 분명하다. 전경 바위에는 스마트폰 한 대와 여러 파편이 놓여 있으며, 두 사람은 바위 턱 아래의 낮은 지면에 서 있어 신체 배치는 가능하다.",
        "entities": "창백한 첫 번째 원통형 광기둥, 구름, 울창한 숲, 나무가 사라진 빈 공간, 두 번째 광기둥이 모두 보인다. 다만 두 번째 빔은 ‘가까이 숲을 훑기 시작하는’ 모습보다 먼 곳의 정지된 수직 빔이다. 쇼트 텍스트에 없는 성인 남성과 여성이 추가되었다. 전경 스마트폰은 이전 쇼트의 기종과 유사하지만 화면에 날짜가 읽히며 전원이 꺼진 상태와 맞지 않는다.",
        "hard_violations": [
         "쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다.",
         "전경 스마트폰 화면에 읽을 수 있는 날짜 글자가 노출되어 ‘읽을 수 있는 글자 금지’를 위반했다."
        ],
        "physics": "남성과 여성은 모두 아래쪽 지면에 서 있는 것으로 보여 몸이 지지된다. 스마트폰과 파편들은 전경 바위 표면 위에 놓여 있어 떠 있지 않는다. 두 광기둥은 구름에서 숲까지 연속되어 있으나, 두 번째 빔의 횡방향 쓸기 운동을 보여주는 기울기나 진행 흔적은 없다."
       }
      ],
      "all_candidates_fail": true
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 5,
        "verdict_ko": "구름에서 숲까지 이어지는 거대한 단일 광기둥과 넓은 숲 전경은 핵심 쇼트에 가깝지만, 쇼트 텍스트에 없는 두 사람을 추가했고 지워진 숲의 빈 공간과 가까워지는 두 번째 빔도 명확하지 않다."
       },
       {
        "label": "A",
        "score": 3,
        "verdict_ko": "첫 빔과 두 번째 빔 및 숲의 빈 공간은 비교적 잘 보이지만, 불필요한 두 사람과 전경 스마트폰을 넣고 금지된 날짜 글자까지 노출해 A보다 중대한 위반이 많다."
       }
      ],
      "all_candidates_fail": true,
      "readings": [
       {
        "label": "B",
        "direction": "남성은 얼굴과 손에 든 스마트폰을 화면 중앙의 광기둥 쪽으로 향하고 있으며, 여성도 같은 중앙 광기둥을 바라본다. 광기둥은 구름에서 수직으로 내려와 숲 중앙에 닿는다. 무기나 다른 지시 물체는 없다.",
        "built_space": "야외의 바위 절벽 가장자리에서 계곡의 울창한 숲을 내려다보는 넓은 구도다. 중앙에 광기둥 하나, 상단에 구름층, 전경 오른쪽에 두 사람이 서 있는 바위 지대가 보인다. 맞은편에는 암벽 능선이 있다. 광기둥이 닿는 지점에도 나무 윤곽이 남아 보여, 절단·연소 흔적 없이 나무가 사라진 원통형 빈 공간이 뚜렷하게 확인되지는 않는다.",
        "entities": "중앙의 창백하고 거대한 수직 광기둥과 울창한 숲, 구름은 해당 대상에 부합한다. 그러나 쇼트 텍스트가 사람을 보여주지 않는데도 30대 남성과 20대 후반 여성으로 보이는 두 사람이 등장한다. 남성은 스마트폰을 들고 있고 화면은 어둡게 보여 전원 꺼짐 상태일 가능성은 있으나, 이 쇼트의 요구된 피사체는 아니다. 가까운 숲을 훑기 시작하는 두 번째 빔은 보이지 않는다.",
        "hard_violations": [
         "쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다."
        ],
        "physics": "두 사람 모두 바위 지면에 양발을 대고 서 있어 신체가 지지된다. 남성의 스마트폰은 양손으로 잡혀 있어 떠 있지 않으며 실제 사용 방향대로 남성 쪽에 화면이 향한다. 광기둥은 구름에서 숲까지 수직으로 이어지는 빛 현상으로 표현되어 별도의 물리적 받침이 필요한 물체처럼 보이지 않는다."
       },
       {
        "label": "A",
        "direction": "여성은 고개를 들어 화면 중앙 왼쪽의 큰 광기둥 쪽을 바라본다. 남성도 위쪽을 보지만 시선은 큰 광기둥보다 약간 오른쪽으로 치우쳐 있다. 큰 광기둥은 구름에서 숲의 빈 공간으로 수직 하강하고, 오른쪽 뒤편의 두 번째 빔도 수직으로 숲을 향한다. 두 번째 빔은 가까이 가로질러 훑는 운동 방향보다 멀리 고정된 또 하나의 수직 기둥처럼 읽힌다.",
        "built_space": "야외 바위 능선 전경, 그 아래의 두 사람, 중·후경의 울창한 숲과 구름층으로 구성된 넓은 장면이다. 중앙 왼쪽에 큰 광기둥 하나, 오른쪽 후경에 작은 광기둥 하나가 있다. 첫 빔의 바닥에는 나무가 없는 둥근 공터가 보여 삭제된 숲의 증거가 A보다 분명하다. 전경 바위에는 스마트폰 한 대와 여러 파편이 놓여 있으며, 두 사람은 바위 턱 아래의 낮은 지면에 서 있어 신체 배치는 가능하다.",
        "entities": "창백한 첫 번째 원통형 광기둥, 구름, 울창한 숲, 나무가 사라진 빈 공간, 두 번째 광기둥이 모두 보인다. 다만 두 번째 빔은 ‘가까이 숲을 훑기 시작하는’ 모습보다 먼 곳의 정지된 수직 빔이다. 쇼트 텍스트에 없는 성인 남성과 여성이 추가되었다. 전경 스마트폰은 이전 쇼트의 기종과 유사하지만 화면에 날짜가 읽히며 전원이 꺼진 상태와 맞지 않는다.",
        "hard_violations": [
         "쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다.",
         "전경 스마트폰 화면에 읽을 수 있는 날짜 글자가 노출되어 ‘읽을 수 있는 글자 금지’를 위반했다."
        ],
        "physics": "남성과 여성은 모두 아래쪽 지면에 서 있는 것으로 보여 몸이 지지된다. 스마트폰과 파편들은 전경 바위 표면 위에 놓여 있어 떠 있지 않는다. 두 광기둥은 구름에서 숲까지 연속되어 있으나, 두 번째 빔의 횡방향 쓸기 운동을 보여주는 기울기나 진행 흔적은 없다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.35,
    "B": 2.0
   },
   "adjusted": {
    "A": 1.1,
    "B": 1.75
   },
   "violations": {
    "A": [
     "[gemini-pro] extra bodies",
     "[gemini-pro] leaked markers/diagrams/text",
     "[gpt] 쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다.",
     "[gpt] 전경 스마트폰 화면에 읽을 수 있는 날짜 글자가 노출되어 ‘읽을 수 있는 글자 금지’를 위반했다."
    ],
    "B": [
     "[gemini-pro] extra bodies",
     "[gpt] 쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다."
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "A": 1100,
   "B": 1750
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "A",
    "score": 1100,
    "verdict_ko": "지시문에 없는 인물이 임의로 추가되었고 스마트폰 화면에 읽을 수 있는 텍스트가 노출되어 지침을 심각하게 위반함.  ★위반: [gemini-pro] extra bodies / [gemini-pro] leaked markers/diagrams/text / [gpt] 쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다. / [gpt] 전경 스마트폰 화면에 읽을 수 있는 날짜 글자가 노출되어 ‘읽을 수 있는 글자 금지’를 위반했다."
   },
   {
    "label": "B",
    "score": 1750,
    "verdict_ko": "텍스트 노출은 피했으나 지시문에 없는 인물이 추가되었으며 명시된 두 번째 빛기둥이 누락됨.  ★위반: [gemini-pro] extra bodies / [gpt] 쇼트 텍스트에 등장하지 않는 남성과 여성을 추가했다."
   }
  ],
  "refs": [
   {
    "label": "PREVIOUS SHOT STILL — a visually related earlier shot of this same place: the location's look, materials, fixed features, lighting mood and each person's clothing are LOCKED to this photo; never copy its camera framing. If this photo shows a character who cannot move (dead or unconscious), that character's exact pose, position and orientation are ALSO LOCKED — treat their body as a fixed prop of the set that only the camera moves around.",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S9sh3_sel.png",
    "asset_id": "72cd0661-937c-4cc2-9617-9d0dc9c7fcbb",
    "role": "prev_still"
   }
  ],
  "critique_skipped": true,
  "needs_reshoot": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2ec-e1f1-766e-a678-b0718336bb74",
  "ref_mode": "prev만 (배경 전용·공유 계획)",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S9sh3"
  },
  "lane_policy": "share_plan_prev_bgonly"
 },
 "S9sh5::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:50:28.624212+00:00",
  "fingerprint": "cb41be48d7e7d54231ebcfcb06f7cf98934a0f9da1b731999caaad760c853de4",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S9sh5_sel.png",
  "source_sha256": "292a1d6decf29523e771289f158092847911770ad63a8dd94d8aabeac9253b3f",
  "file": "S9sh5_cine.png",
  "staged_sha256": "6807cefaf98d2d4b107df195117d72bbd874f1a0333b31e966206249a00b7252",
  "latency_ms": 16620
 },
 "S9sh13::signage": {
  "fp": "2b7e172712dc4651",
  "inscriptions": [],
  "cues": [],
  "dropped": []
 },
 "era_assess::c736b92d50163da5": {
  "subjects": [],
  "subject_text": "숲속 능선의 절벽 끝자락\n울창한 숲 능선이 갑자기 끊기는 절벽 가장자리. 바위와 흙으로 된 끝자락 아래로 광대한 숲의 수관이 펼쳐진다.",
  "identity": "canonical",
  "scope_id": "L05",
  "scope_role": "location_exterior",
  "scope_sha": "f5181e6ca740a966"
 },
 "S9sh13::bgfirst_bg": {
  "input_fingerprint": "ae237844bf26a74f",
  "prompt": "Turn the attached storyboard sketch into a photorealistic film still of\nits LOCATION ONLY, keeping the figures exactly as they are.\n\nKEEP EXACTLY: the camera framing, the horizon, and where every element\nsits in the frame. Each mannequin figure stays a plain grey featureless\nmannequin standing in the very same spot, at the same size, in the same\npose, turned the same way — do not turn them into people, do not move,\nrotate, mirror or re-pose them, do not add or remove figures.\n\nBUILD PHOTOREALISTICALLY: everything that is not a figure — ground,\nsurfacing, structures, vegetation, sky, water, distance. The bare lines\nof the sketch are a layout guide; replace them with the real materials,\ndepth and lighting of the place described below.\n\nSHOT TEXT this background must serve (Korean): '토니(앤서니 로저스)'의 가슴팍을 단단히 움켜쥔 채 절벽 바깥의 아득한 허공을 향해 몸을 던져, 뒷발이 바위 끝에서 막 떨어진 '윌마 디어링'.\n\nLOCATION (lock): Outside on the exposed rocky lip of the forested ridge cliff, at the instant both figures launch into the open air beyond it.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-right of the frame, midground, moves toward open space beyond the cliff; 토니(앤서니 로저스) in the middle-center of the frame, midground, moves toward open space beyond the cliff; cliff edge in the lower-left of the frame, midground.\n- KEY BACKGROUND ELEMENTS: 절벽 끝 (The launch point is immediately behind Wilma's departing foot) — Its terminating edge angles away from the low side camera toward the open launch path; used as Lower-left spatial anchor proving that both bodies have crossed beyond support; 절벽 바깥 허공 (Both characters' centers of gravity have moved into it); used as Open right-side destination of the tracking movement; 토니의 가슴 하네스와 연결 고리 (Connected to Wilma's auxiliary hook as she pulls him outward) — The chest-facing harness and attached link are visible at a three-quarter angle between the two bodies; used as Mid-frame physical evidence that the pair launch as one connected unit; 쇄도하는 빛기둥 (Approaching the position the two characters are leaving); used as Distant threat held behind the launch action without displacing the two bodies as primary subjects; 아래쪽 숲 (Visible beyond and below the cliff); used as Background depth reference for the height and direction of the leap.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Exterior daylight retains rugged, restrained contrast, with the approaching pale column adding only the illumination explicitly present in the scene.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHINGS AT THIS PLACE:\nThe site is organized as a wooded ridge interior leading to a terminal cliff edge. Within the ridge interior, a recognizable loop of three trees stands around the approach to a narrow gully, allowing movement around the trees before descending or being drawn onto the gully floor. Inter-tree branches form an elevated passage above the same wooded terrain. The ridge ground continues from this interior forest area to the cliff zone, where the wooded land ends at a steep escarpment with open space beyond. The map should preserve this topological sequence without assigning an unsupported compass orientation: three-tree loop and branch passage within the ridge interior, narrow gully beside the loop, continuous ridge terrain toward the terminal cliff edge, and the cliff drop beyond that boundary.\n- ridge: wooded ridge ground\n- tree group: three-tree pursuit loop\n- branches: inter-tree branch passage\n- gully: narrow forest gully floor\n- cliff: ridge-end cliff escarpment\n\nWORLD FACTS (creator-confirmed — always true):\n- Region (real-world reference): 미국 펜실베이니아 일대 및 미래 북아메리카권\n- Era: 2026년 현대와 2419년 원미래\n- 현대 구간은 2020년대의 산업 안전 장비, 휴대용 전자기기, 기반 시설 규격을 따른다. 원미래 구간은 장기간의 자연 재점유와 문명 잔존물, 독자적으로 발전한 기술을 반영하되 두 시대의 물건을 임의로 혼합하지 않는다.\n- AI는 일반 스마트폰의 평면 화면과 음성 인터페이스로만 표현하고, 의인화된 신체나 공간형 홀로그램을 만들지 않는다. 배터리, 연결 상태, 날짜와 같은 정보는 2020년대 모바일 UI의 범위에서 간결하게 표시한다.\n- 장비는 실용적인 2020년대 상용·산업용 디자인으로 묘사하며 과도하게 매끈한 미래형 외장이나 홀로그램 조작계를 적용하지 않는다. 점검 드론은 손바닥 크기의 기능 중심 기체로 유지한다.\n- 긴 도약이나 완만한 상승에는 허리 장착형 벨트의 작동이 반드시 시각적으로 연결되어야 한다. 날개, 낙하산, 제트 화염 없이 벨트의 미세 진동과 조절 부품, 무게추 변화로 기동 원리를 암시하고 사용자를 무장비 비행자로 그리지 않는다.\n- 미래 장비는 휴대 가능한 기능 중심의 견고한 설계로 표현한다. 탄자는 발사 후 스스로 가속하는 소형 추진체로, 손목 스캐너는 신체에 밀착된 판독 장치로 묘사하되 모든 장비를 동일한 매끈한 디자인으로 획일화하지 않는다.\n- 네트워크 연결과 시간 동기화는 기존 스마트폰 화면의 상태 변화로만 표현한다. 연결 이후에도 기기의 균열, 배터리 한계, 화면 규격과 2020년대 제품 외형을 그대로 유지하며 자동 개조나 홀로그램 확장을 추가하지 않는다.\n\nTHINGS THAT LIVE AT THIS PLACE (what they are, not where they go): the\nlist above names what permanently belongs to this site. Where each one\nsits in this frame is decided by the attached sketch alone — do not add\nor reposition anything on the strength of the list, build only what the\nsketch already has lines for, and do not invent facilities that are not\nlisted.\n\nNo readable writing anywhere: surfaces that would carry writing may be\npresent, but stage any wording out of legibility — an oblique angle,\ndistance, shallow focus. No captions, watermarks or overlay text,\nand none of the sketch's diagram symbols, lines or markers anywhere in\nthe image.",
  "effective_prompt": "Turn the attached storyboard sketch into a photorealistic film still of\nits LOCATION ONLY, keeping the figures exactly as they are.\n\nKEEP EXACTLY: the camera framing, the horizon, and where every element\nsits in the frame. Each mannequin figure stays a plain grey featureless\nmannequin standing in the very same spot, at the same size, in the same\npose, turned the same way — do not turn them into people, do not move,\nrotate, mirror or re-pose them, do not add or remove figures.\n\nBUILD PHOTOREALISTICALLY: everything that is not a figure — ground,\nsurfacing, structures, vegetation, sky, water, distance. The bare lines\nof the sketch are a layout guide; replace them with the real materials,\ndepth and lighting of the place described below.\n\nSHOT TEXT this background must serve (Korean): '토니(앤서니 로저스)'의 가슴팍을 단단히 움켜쥔 채 절벽 바깥의 아득한 허공을 향해 몸을 던져, 뒷발이 바위 끝에서 막 떨어진 '윌마 디어링'.\n\nLOCATION (lock): Outside on the exposed rocky lip of the forested ridge cliff, at the instant both figures launch into the open air beyond it.\n\nTIME OF DAY (lock): day.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-right of the frame, midground, moves toward open space beyond the cliff; 토니(앤서니 로저스) in the middle-center of the frame, midground, moves toward open space beyond the cliff; cliff edge in the lower-left of the frame, midground.\n- KEY BACKGROUND ELEMENTS: 절벽 끝 (The launch point is immediately behind Wilma's departing foot) — Its terminating edge angles away from the low side camera toward the open launch path; used as Lower-left spatial anchor proving that both bodies have crossed beyond support; 절벽 바깥 허공 (Both characters' centers of gravity have moved into it); used as Open right-side destination of the tracking movement; 토니의 가슴 하네스와 연결 고리 (Connected to Wilma's auxiliary hook as she pulls him outward) — The chest-facing harness and attached link are visible at a three-quarter angle between the two bodies; used as Mid-frame physical evidence that the pair launch as one connected unit; 쇄도하는 빛기둥 (Approaching the position the two characters are leaving); used as Distant threat held behind the launch action without displacing the two bodies as primary subjects; 아래쪽 숲 (Visible beyond and below the cliff); used as Background depth reference for the height and direction of the leap.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Exterior daylight retains rugged, restrained contrast, with the approaching pale column adding only the illumination explicitly present in the scene.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHINGS AT THIS PLACE:\nThe site is organized as a wooded ridge interior leading to a terminal cliff edge. Within the ridge interior, a recognizable loop of three trees stands around the approach to a narrow gully, allowing movement around the trees before descending or being drawn onto the gully floor. Inter-tree branches form an elevated passage above the same wooded terrain. The ridge ground continues from this interior forest area to the cliff zone, where the wooded land ends at a steep escarpment with open space beyond. The map should preserve this topological sequence without assigning an unsupported compass orientation: three-tree loop and branch passage within the ridge interior, narrow gully beside the loop, continuous ridge terrain toward the terminal cliff edge, and the cliff drop beyond that boundary.\n- ridge: wooded ridge ground\n- tree group: three-tree pursuit loop\n- branches: inter-tree branch passage\n- gully: narrow forest gully floor\n- cliff: ridge-end cliff escarpment\n\nWORLD FACTS (creator-confirmed — always true):\n- Region (real-world reference): 미국 펜실베이니아 일대 및 미래 북아메리카권\n- Era: 2026년 현대와 2419년 원미래\n- 현대 구간은 2020년대의 산업 안전 장비, 휴대용 전자기기, 기반 시설 규격을 따른다. 원미래 구간은 장기간의 자연 재점유와 문명 잔존물, 독자적으로 발전한 기술을 반영하되 두 시대의 물건을 임의로 혼합하지 않는다.\n- AI는 일반 스마트폰의 평면 화면과 음성 인터페이스로만 표현하고, 의인화된 신체나 공간형 홀로그램을 만들지 않는다. 배터리, 연결 상태, 날짜와 같은 정보는 2020년대 모바일 UI의 범위에서 간결하게 표시한다.\n- 장비는 실용적인 2020년대 상용·산업용 디자인으로 묘사하며 과도하게 매끈한 미래형 외장이나 홀로그램 조작계를 적용하지 않는다. 점검 드론은 손바닥 크기의 기능 중심 기체로 유지한다.\n- 긴 도약이나 완만한 상승에는 허리 장착형 벨트의 작동이 반드시 시각적으로 연결되어야 한다. 날개, 낙하산, 제트 화염 없이 벨트의 미세 진동과 조절 부품, 무게추 변화로 기동 원리를 암시하고 사용자를 무장비 비행자로 그리지 않는다.\n- 미래 장비는 휴대 가능한 기능 중심의 견고한 설계로 표현한다. 탄자는 발사 후 스스로 가속하는 소형 추진체로, 손목 스캐너는 신체에 밀착된 판독 장치로 묘사하되 모든 장비를 동일한 매끈한 디자인으로 획일화하지 않는다.\n- 네트워크 연결과 시간 동기화는 기존 스마트폰 화면의 상태 변화로만 표현한다. 연결 이후에도 기기의 균열, 배터리 한계, 화면 규격과 2020년대 제품 외형을 그대로 유지하며 자동 개조나 홀로그램 확장을 추가하지 않는다.\n\nTHINGS THAT LIVE AT THIS PLACE (what they are, not where they go): the\nlist above names what permanently belongs to this site. Where each one\nsits in this frame is decided by the attached sketch alone — do not add\nor reposition anything on the strength of the list, build only what the\nsketch already has lines for, and do not invent facilities that are not\nlisted.\n\nNo readable writing anywhere: surfaces that would carry writing may be\npresent, but stage any wording out of legibility — an oblique angle,\ndistance, shallow focus. No captions, watermarks or overlay text,\nand none of the sketch's diagram symbols, lines or markers anywhere in\nthe image.",
  "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S9sh13__bgfirst_bg.png",
  "asset_id": "eff365af-c9e7-4512-8cde-e34a8a6364bb",
  "input_asset_ids": [
   "130994c3-2f25-4f50-a363-7b7f30fbfd00"
  ]
 },
 "S9sh13": {
  "input_fingerprint": "3b67ed65d627223e",
  "prompt": "Create ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): '토니(앤서니 로저스)'의 가슴팍을 단단히 움켜쥔 채 절벽 바깥의 아득한 허공을 향해 몸을 던져, 뒷발이 바위 끝에서 막 떨어진 '윌마 디어링'.\n\nLOCATION (lock): Outside on the exposed rocky lip of the forested ridge cliff, at the instant both figures launch into the open air beyond it. The shot takes place here — the attached PREVIOUS SHOT STILL shows this exact place.\n\nFRAMING SCALE (follow exactly — this alone decides how much of the frame the subject fills; the storyboard sketch supplies where the figures stand and which way they face, never this scale):\n- FRAMING SCALE: wide shot\nMatch this shot size exactly. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Exterior daylight retains rugged, restrained contrast, with the approaching pale column adding only the illumination explicitly present in the scene.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nTHIS SHOT CONTINUES THE PREVIOUS SHOT: everything the attached still established — the place, its fixed features and wear, each person's clothing and state — persists. Any person in it who cannot move stays PRECISELY as photographed (body, pose, contact points, held objects); only the camera changes.\n\nPREVIOUS STILL USAGE (follow exactly — what to take from the attached still and what to exclude; it governs place and objects only, never who is in this shot or how the camera sees them): Take the vast daylight forest canopy, rocky cliff-edge environment, clouded sky, and ominous pale illumination over the ridge. Exclude the earlier beam at its former strike point and any empty erased patch it left there; do not add the phone display to this space.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A second pale beam races toward the cliff through the forest, while the first erased zone remains completely empty. The powered-off smartphone remains in Wilma’s possession, and her jumper belt’s auxiliary hook is clipped to Tony’s chest harness. 윌마 디어링: She launches beyond the cliff edge with her rear foot just leaving the rock, one hand gripping a chest harness. Her jumper belt is active, its auxiliary hook engaged, and she retains the powered-off smartphone. 토니(앤서니 로저스): His chest harness is firmly clipped to the jumper belt’s auxiliary hook, and he is being carried off the cliff into the long upward flight. His rescue gear remains attached.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
  "roll_prompts": {
   "A": "Replace every grey or outlined mannequin figure in the FIRST attached\nimage with the real people the CHARACTER REFERENCE images show, and\noutput a photorealistic film still. Each mannequin becomes the character\nwhose POSE CANON and the shot text match that figure's pose and\nposition; never repeat one person across two mannequins. If no CHARACTER\nREFERENCE image is attached for a figure, still replace it with a\nplausible real person consistent with the text. Every mannequin becomes\na person — erasing a figure, or leaving one out, is not a replacement,\nand the number of figures never changes.\n\nKEEP EXACTLY: the background of the first image, the camera framing,\nand each mannequin's position, size, pose and the direction it is\nturned. A person must be exactly where their mannequin was, at the same\nscale, in the same pose, facing the same way — never mirrored, never\nre-staged, never straightened up, stood upright or otherwise re-posed.\nNo mannequin, grey figure or sketch line may remain anywhere in the\noutput.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): '토니(앤서니 로저스)'의 가슴팍을 단단히 움켜쥔 채 절벽 바깥의 아득한 허공을 향해 몸을 던져, 뒷발이 바위 끝에서 막 떨어진 '윌마 디어링'.\n\nLOCATION (lock): Outside on the exposed rocky lip of the forested ridge cliff, at the instant both figures launch into the open air beyond it. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-right of the frame, midground, moves toward open space beyond the cliff; 토니(앤서니 로저스) in the middle-center of the frame, midground, moves toward open space beyond the cliff; cliff edge in the lower-left of the frame, midground.\n- KEY BACKGROUND ELEMENTS: 절벽 끝 (The launch point is immediately behind Wilma's departing foot) — Its terminating edge angles away from the low side camera toward the open launch path; used as Lower-left spatial anchor proving that both bodies have crossed beyond support; 절벽 바깥 허공 (Both characters' centers of gravity have moved into it); used as Open right-side destination of the tracking movement; 토니의 가슴 하네스와 연결 고리 (Connected to Wilma's auxiliary hook as she pulls him outward) — The chest-facing harness and attached link are visible at a three-quarter angle between the two bodies; used as Mid-frame physical evidence that the pair launch as one connected unit; 쇄도하는 빛기둥 (Approaching the position the two characters are leaving); used as Distant threat held behind the launch action without displacing the two bodies as primary subjects; 아래쪽 숲 (Visible beyond and below the cliff); used as Background depth reference for the height and direction of the leap.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Exterior daylight retains rugged, restrained contrast, with the approaching pale column adding only the illumination explicitly present in the scene.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A second pale beam races toward the cliff through the forest, while the first erased zone remains completely empty. The powered-off smartphone remains in Wilma’s possession, and her jumper belt’s auxiliary hook is clipped to Tony’s chest harness. 윌마 디어링: She launches beyond the cliff edge with her rear foot just leaving the rock, one hand gripping a chest harness. Her jumper belt is active, its auxiliary hook engaged, and she retains the powered-off smartphone. 토니(앤서니 로저스): His chest harness is firmly clipped to the jumper belt’s auxiliary hook, and he is being carried off the cliff into the long upward flight. His rescue gear remains attached.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.",
   "B": "Replace every grey or outlined mannequin figure in the FIRST attached\nimage with the real people the CHARACTER REFERENCE images show, and\noutput a photorealistic film still. Each mannequin becomes the character\nwhose POSE CANON and the shot text match that figure's pose and\nposition; never repeat one person across two mannequins. If no CHARACTER\nREFERENCE image is attached for a figure, still replace it with a\nplausible real person consistent with the text. Every mannequin becomes\na person — erasing a figure, or leaving one out, is not a replacement,\nand the number of figures never changes.\n\nKEEP EXACTLY: the background of the first image, the camera framing,\nand each mannequin's position, size, pose and the direction it is\nturned. A person must be exactly where their mannequin was, at the same\nscale, in the same pose, facing the same way — never mirrored, never\nre-staged, never straightened up, stood upright or otherwise re-posed.\nNo mannequin, grey figure or sketch line may remain anywhere in the\noutput.\n\nCreate ONE FINAL photorealistic live-action film still of the moment below — Pennsylvania, United States, in 2026 and 2419; all people are English-speaking Americans unless stated. TIME OF DAY (lock): day.\n\nSHOT TEXT (authoritative, Korean): '토니(앤서니 로저스)'의 가슴팍을 단단히 움켜쥔 채 절벽 바깥의 아득한 허공을 향해 몸을 던져, 뒷발이 바위 끝에서 막 떨어진 '윌마 디어링'.\n\nLOCATION (lock): Outside on the exposed rocky lip of the forested ridge cliff, at the instant both figures launch into the open air beyond it. The shot takes place here — the FIRST attached image (SHOT BACKGROUND) is this exact place, already built: its ground, structures, horizon, materials and lighting are the finished truth of this location and must not be redesigned or replaced. No location photograph is attached — read the place from that image alone, and add no scenery, structure, vehicle or fixture that it does not already show. This lock governs the place only; the figures in the shot follow the staging and pose instructions.\n\nCAMERA & FRAME (follow exactly — this is the composition authority for this still; the reference images supply identity and place, never the framing):\n- FRAMING SCALE: wide shot\n- FRAME LAYOUT: 윌마 디어링 in the middle-right of the frame, midground, moves toward open space beyond the cliff; 토니(앤서니 로저스) in the middle-center of the frame, midground, moves toward open space beyond the cliff; cliff edge in the lower-left of the frame, midground.\n- KEY BACKGROUND ELEMENTS: 절벽 끝 (The launch point is immediately behind Wilma's departing foot) — Its terminating edge angles away from the low side camera toward the open launch path; used as Lower-left spatial anchor proving that both bodies have crossed beyond support; 절벽 바깥 허공 (Both characters' centers of gravity have moved into it); used as Open right-side destination of the tracking movement; 토니의 가슴 하네스와 연결 고리 (Connected to Wilma's auxiliary hook as she pulls him outward) — The chest-facing harness and attached link are visible at a three-quarter angle between the two bodies; used as Mid-frame physical evidence that the pair launch as one connected unit; 쇄도하는 빛기둥 (Approaching the position the two characters are leaving); used as Distant threat held behind the launch action without displacing the two bodies as primary subjects; 아래쪽 숲 (Visible beyond and below the cliff); used as Background depth reference for the height and direction of the leap.\nCompose the frame exactly as specified above — subject scale and screen placement. The viewing angle is deliberately left open here; choose the angle that serves the shot. Keep true physical scale between people and background elements: fixtures and distant objects occupy only the small screen area their real size and distance dictate; never enlarge a background object into a foreground presence.\n\nLIGHTING & MOOD (these govern light, color and surface rendering):\n- LIGHTING & MOOD: Exterior daylight retains rugged, restrained contrast, with the approaching pale column adding only the illumination explicitly present in the scene.\n- MATERIAL REALISM: every object and surface must read as a real,\n  physical material — correct texture, weight, wear and light response\n  (metal reflects, fabric drapes and creases, liquid is glossy and\n  pools, painted or drawn marks sit ON a surface and follow its\n  curvature and lighting). Nothing may look like a flat sticker, a\n  doodle or a graphic overlay pasted onto the frame.\n\nREALIZE FIGURATIVE LANGUAGE AS A LIVE-ACTION SHOT: the Korean shot\ntext may describe characters metaphorically, figuratively or with\nexaggeration. Photograph what a real movie camera would actually\nrecord on a physical set — exaggerated or figurative impressions\nbecome realistic staging within whatever this brief already fixes,\nnot literal fantasy imagery.\n\nEVERY CHARACTER IS A HUMAN BEING: unless the story explicitly\nfeatures non-human or virtual beings (as in science fiction or\nfantasy), every character — however indirectly, vaguely or\nfiguratively the text describes them — IS a real human. When the\ntext gives no direct visual description of a person it puts in\nthis shot, imagine that description and still show them as a\nconcrete, fully-formed human being:\nrender the human form as fully as the framing shows — never reduce\na person to a shape, blob, solid silhouette or abstract mass.\n\nEXPRESSIONS ARE ACTED, NEVER ANATOMICAL: when the text describes a\nperson's eyes, face or presence emotionally or figuratively (vacant,\nhollow, dazed, burning, lifeless gaze and the like), it describes an\nACTOR'S PERFORMANCE captured by a real camera — realize it ONLY\nthrough gaze direction, focus, eyelids, facial muscles, stillness\nand posture. Every living person's eyes remain anatomically normal\nhuman eyes with a natural iris and pupil, natural sclera and normal\nproportions; NEVER whiten, blank out, cloud over, glow, enlarge or\notherwise alter eyeballs, skin or anatomy — unless the story\nexplicitly declares that being non-human or supernatural in form.\n\nPROPS FACE THE RIGHT WAY: every handheld or used object must be\noriented exactly as its real-world use requires. A person reading,\nwatching or operating something (a phone, a photograph, a paper,\nany device) has its functional side — screen, front, page — facing\nTHEIR OWN eyes; the camera then sees whatever side the staging\ngeometry implies (often its back). Show the functional side to the\ncamera ONLY when the shot text itself stages it toward the viewer.\nNever flip, mirror or reverse an object's front and back.\n\nNATURAL PERFORMANCE (default only — every explicit direction above\nwins): when no pose contract, immobility contract or shot-text\ndirection says otherwise, people read as alive in mid-moment —\nbelievable weight shift, hands naturally positioned for the action\nthe SHOT TEXT already gives them, gaze on the target the SHOT TEXT\nimplies; avoid a stiff attention stance (feet together, arms hanging\nstraight down) and a blank stare into the lens. If the SHOT TEXT or\nany POSE/IMMOBILE contract above stages stillness, death, sleep,\nunconsciousness, restraint, drill or an explicit direct-to-camera\nlook, follow THAT exactly — this clause never overrides it.\n\nDRAWN MARKS KEEP THEIR SHAPE: any mark the text describes as drawn,\npainted or traced (a circle, a line, a symbol) is a STROKE sitting on\nthe surface — an outline whose interior still shows the underlying\nsurface (wall, skin, paper). Render its stated shape faithfully: a\ndrawn circle stays an open ring of brush-width, never filled into a\nsolid disc, unless the text explicitly says it is filled.\n\nBODY & SUPPORT (default only — every explicit direction above wins): a body relates to what holds it. A seated person sits the way the seat is built to be used — hips on the seat, back toward the backrest, legs falling naturally toward the floor; people sharing adjacent seats each occupy their own seat, side by side. Hands, hips and feet keep believable contact with whatever they rest on. When the text stages someone frozen, stunned or holding still, the body stops mid-action exactly where the moment caught it — weight already committed to one side, hands where the interrupted movement left them — rather than resetting into a symmetric at-attention stance. If the SHOT TEXT or any contract above stages a specific arrangement, follow THAT exactly.\n\nCARRIED STATE (persist exactly — must match the neighbouring shots of this scene): A second pale beam races toward the cliff through the forest, while the first erased zone remains completely empty. The powered-off smartphone remains in Wilma’s possession, and her jumper belt’s auxiliary hook is clipped to Tony’s chest harness. 윌마 디어링: She launches beyond the cliff edge with her rear foot just leaving the rock, one hand gripping a chest harness. Her jumper belt is active, its auxiliary hook engaged, and she retains the powered-off smartphone. 토니(앤서니 로저스): His chest harness is firmly clipped to the jumper belt’s auxiliary hook, and he is being carried off the cliff into the long upward flight. His rescue gear remains attached.\n\nPEOPLE: the SHOT TEXT alone decides whether any person is visible in this shot. IF a person appears, they must be one of: 토니(앤서니 로저스) (미국인 남성, 30대 중반, 자연스러운 성인 남성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락); 윌마 디어링 (미국인 여성, 20대 후반, 자연스러운 성인 여성 얼굴, 구체적으로 명시되지 않은 자연색 머리카락) — never anyone else, and never add a person the shot text does not show. IF only part of a person is in frame (a hand, arm, foot, back, silhouette), that body part belongs to the specific person the shot text names — its sex, age, build, skin and grooming must unmistakably match that person's profile above.\n\nCHARACTER REFERENCE ROLE (follow exactly): the attached CHARACTER REFERENCE images establish identity only — face, hair, build and clothing. The pose, gaze direction, camera angle and framing inside those reference images belong to the reference photos, not to this shot; never copy them. Pose and gaze in this shot follow only the CAMERA and action text of this prompt. When this prompt stages a character's face as covered or hidden — by a costume head, mask, helmet, hood, or a body turned away — that staging wins: keep the covering exactly as described and never pull the reference face into view from under it; the reference then guides only what stays visible, such as build and clothing.\n\nNo readable writing anywhere in this image. Surfaces that would carry writing may be present, but nothing a viewer could read: stage every one of them out of legibility — a hand across, an oblique angle, shallow focus, or simply turned away. No caption, subtitle, watermark, logo or overlay.\n\nCOMPOSITION VARIATION (this candidate only): keep the place, the people and the moment exactly as specified, and keep from the CAMERA & FRAME contract its subject scale, screen placement and key background elements exactly. For this candidate only, its camera angle and height are deliberately left open: choose a camera position distinctly different from the obvious one, as a film director picking a second setup on the same blocking. Stay true to any explicit framing the SHOT TEXT itself dictates."
  },
  "roll_refs": {
   "A": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S9sh13__bgfirst_bg.png",
     "asset_id": "eff365af-c9e7-4512-8cde-e34a8a6364bb",
     "role": "bgfirst_bg"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ],
   "B": [
    {
     "label": "SHOT BACKGROUND",
     "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S9sh13__bgfirst_bg.png",
     "asset_id": "eff365af-c9e7-4512-8cde-e34a8a6364bb",
     "role": "bgfirst_bg"
    },
    {
     "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:929851>",
     "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
     "role": "character_ref"
    },
    {
     "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
     "path": "<bytes:766962>",
     "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
     "role": "character_ref"
    }
   ]
  },
  "shot_run_spend_attempt_count": 1,
  "readings": [
   {
    "label": "A",
    "direction": "윌마와 토니는 프레임 오른쪽의 벼랑 밖 허공과 다가오는 빛 기둥을 향해 몸을 던지고 있으며, 시선도 그 방향을 향하고 있습니다.",
    "built_space": "암벽의 가장자리와 배경의 숲, 하늘이 레퍼런스의 공간 구조와 정확히 일치합니다.",
    "entities": "윌마는 초록색 점프슈트를 입고 있으나 텍스트 지시와 달리 토니의 가슴이 아닌 연결줄을 잡고 있습니다. 토니는 레퍼런스에 있던 헬멧을 착용하지 않았습니다. 우측에 옅은 빛 기둥이 묘사되어 있습니다.",
    "hard_violations": [
     "[gemini-pro] 토니의 헬멧 누락 (캐릭터 레퍼런스의 의상/식별 요소 불일치)"
    ],
    "physics": "윌마는 뒷발이 바위에서 막 떨어진 상태로 공중에 떠서 도약 중입니다. 토니는 수평에 가까운 뻣뻣한 자세로 공중에 떠 있으며, 윌마가 잡고 있는 팽팽한 줄에 의해 지탱되고 끌려가고 있습니다."
   },
   {
    "label": "B",
    "direction": "윌마와 토니는 프레임 오른쪽의 벼랑 밖 허공과 빛 기둥을 향해 도약하고 있으며, 시선 역시 오른쪽을 향하고 있습니다.",
    "built_space": "암벽 가장자리와 뒤쪽 숲의 구조가 레퍼런스와 동일하게 재현되어 있습니다.",
    "entities": "윌마는 초록색 점프슈트를 입고 있으나 프롬프트의 지시와 달리 토니의 가슴 하네스를 직접 잡지 않고 연결줄을 쥐고 있습니다. 토니는 어두운 복장과 헬멧, 구조 장비를 레퍼런스대로 잘 착용하고 있습니다. 우측에 빛 기둥이 존재합니다.",
    "hard_violations": [],
    "physics": "윌마는 뒷발이 절벽 끝에서 방금 떨어진 상태로 도약 중이며, 토니는 윌마가 쥐고 있는 팽팽한 줄에 끌려 수평 자세로 허공에 떠서 지탱되고 있습니다."
   }
  ],
  "cross_model_order": {
   "policy": "cross_model_order_v2_slot0_fwd_slot1_rev_parallel2",
   "slots": [
    {
     "model": "gemini-pro",
     "order": "forward",
     "display_to_canonical": {
      "A": "A",
      "B": "B"
     },
     "raw": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "토니의 헬멧과 장비를 레퍼런스에 맞게 유지하여 A보다 우수하나, 두 후보 모두 윌마가 토니의 가슴팍(하네스)을 직접 움켜쥐지 않고 줄을 잡고 있으며 토니의 자세가 레퍼런스 마네킹처럼 뻣뻣한 점이 아쉽습니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "도약하는 찰나의 연출은 배경 레퍼런스와 일치하나, 토니가 착용해야 할 헬멧이 완전히 누락되었고 윌마의 손 위치도 프롬프트 지시와 어긋납니다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마와 토니는 프레임 오른쪽의 벼랑 밖 허공과 다가오는 빛 기둥을 향해 몸을 던지고 있으며, 시선도 그 방향을 향하고 있습니다.",
        "built_space": "암벽의 가장자리와 배경의 숲, 하늘이 레퍼런스의 공간 구조와 정확히 일치합니다.",
        "entities": "윌마는 초록색 점프슈트를 입고 있으나 텍스트 지시와 달리 토니의 가슴이 아닌 연결줄을 잡고 있습니다. 토니는 레퍼런스에 있던 헬멧을 착용하지 않았습니다. 우측에 옅은 빛 기둥이 묘사되어 있습니다.",
        "hard_violations": [
         "토니의 헬멧 누락 (캐릭터 레퍼런스의 의상/식별 요소 불일치)"
        ],
        "physics": "윌마는 뒷발이 바위에서 막 떨어진 상태로 공중에 떠서 도약 중입니다. 토니는 수평에 가까운 뻣뻣한 자세로 공중에 떠 있으며, 윌마가 잡고 있는 팽팽한 줄에 의해 지탱되고 끌려가고 있습니다."
       },
       {
        "label": "B",
        "direction": "윌마와 토니는 프레임 오른쪽의 벼랑 밖 허공과 빛 기둥을 향해 도약하고 있으며, 시선 역시 오른쪽을 향하고 있습니다.",
        "built_space": "암벽 가장자리와 뒤쪽 숲의 구조가 레퍼런스와 동일하게 재현되어 있습니다.",
        "entities": "윌마는 초록색 점프슈트를 입고 있으나 프롬프트의 지시와 달리 토니의 가슴 하네스를 직접 잡지 않고 연결줄을 쥐고 있습니다. 토니는 어두운 복장과 헬멧, 구조 장비를 레퍼런스대로 잘 착용하고 있습니다. 우측에 빛 기둥이 존재합니다.",
        "hard_violations": [],
        "physics": "윌마는 뒷발이 절벽 끝에서 방금 떨어진 상태로 도약 중이며, 토니는 윌마가 쥐고 있는 팽팽한 줄에 끌려 수평 자세로 허공에 떠서 지탱되고 있습니다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 7,
        "verdict_ko": "토니의 헬멧과 장비를 레퍼런스에 맞게 유지하여 A보다 우수하나, 두 후보 모두 윌마가 토니의 가슴팍(하네스)을 직접 움켜쥐지 않고 줄을 잡고 있으며 토니의 자세가 레퍼런스 마네킹처럼 뻣뻣한 점이 아쉽습니다."
       },
       {
        "label": "A",
        "score": 4,
        "verdict_ko": "도약하는 찰나의 연출은 배경 레퍼런스와 일치하나, 토니가 착용해야 할 헬멧이 완전히 누락되었고 윌마의 손 위치도 프롬프트 지시와 어긋납니다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "A",
        "direction": "윌마와 토니는 프레임 오른쪽의 벼랑 밖 허공과 다가오는 빛 기둥을 향해 몸을 던지고 있으며, 시선도 그 방향을 향하고 있습니다.",
        "built_space": "암벽의 가장자리와 배경의 숲, 하늘이 레퍼런스의 공간 구조와 정확히 일치합니다.",
        "entities": "윌마는 초록색 점프슈트를 입고 있으나 텍스트 지시와 달리 토니의 가슴이 아닌 연결줄을 잡고 있습니다. 토니는 레퍼런스에 있던 헬멧을 착용하지 않았습니다. 우측에 옅은 빛 기둥이 묘사되어 있습니다.",
        "hard_violations": [
         "토니의 헬멧 누락 (캐릭터 레퍼런스의 의상/식별 요소 불일치)"
        ],
        "physics": "윌마는 뒷발이 바위에서 막 떨어진 상태로 공중에 떠서 도약 중입니다. 토니는 수평에 가까운 뻣뻣한 자세로 공중에 떠 있으며, 윌마가 잡고 있는 팽팽한 줄에 의해 지탱되고 끌려가고 있습니다."
       },
       {
        "label": "B",
        "direction": "윌마와 토니는 프레임 오른쪽의 벼랑 밖 허공과 빛 기둥을 향해 도약하고 있으며, 시선 역시 오른쪽을 향하고 있습니다.",
        "built_space": "암벽 가장자리와 뒤쪽 숲의 구조가 레퍼런스와 동일하게 재현되어 있습니다.",
        "entities": "윌마는 초록색 점프슈트를 입고 있으나 프롬프트의 지시와 달리 토니의 가슴 하네스를 직접 잡지 않고 연결줄을 쥐고 있습니다. 토니는 어두운 복장과 헬멧, 구조 장비를 레퍼런스대로 잘 착용하고 있습니다. 우측에 빛 기둥이 존재합니다.",
        "hard_violations": [],
        "physics": "윌마는 뒷발이 절벽 끝에서 방금 떨어진 상태로 도약 중이며, 토니는 윌마가 쥐고 있는 팽팽한 줄에 끌려 수평 자세로 허공에 떠서 지탱되고 있습니다."
       }
      ]
     },
     "ok": true
    },
    {
     "model": "gpt",
     "order": "reverse",
     "display_to_canonical": {
      "A": "B",
      "B": "A"
     },
     "raw": {
      "winner": "A",
      "ranking": [
       "A",
       "B"
      ],
      "verdicts": [
       {
        "label": "A",
        "score": 8,
        "verdict_ko": "윌마의 뒷발이 절벽 끝을 막 떠나고 토니가 팽팽한 연결선에 실려 공중으로 끌려가는 정확한 발사 순간과 와이드 구도를 가장 충실히 구현했다."
       },
       {
        "label": "B",
        "score": 6,
        "verdict_ko": "방향과 장소는 맞지만 토니의 발이 아직 바위에 닿아 있어 ‘절벽 밖으로 운반되는’ 순간보다 이르고, 기준 복장의 헬멧도 빠졌다."
       }
      ],
      "readings": [
       {
        "label": "A",
        "direction": "윌마의 몸과 뻗은 오른팔, 시선은 화면 오른쪽의 절벽 밖 허공과 멀리 보이는 창백한 광주를 향한다. 토니도 오른쪽으로 끌려가며 같은 진행 방향을 본다. 두 사람 사이의 연결선은 토니의 가슴에서 윌마의 허리로 팽팽하게 이어져 이동 방향과 일치한다.",
        "built_space": "야외의 단일 암석 절벽 돌출부가 화면 왼쪽과 중앙에 있고, 뒤에는 숲 능선과 계곡, 흐린 하늘이 펼쳐진다. 인공 고정 설비나 반사면은 없다. 창백한 광주는 오른쪽에 하나만 보이며, 두 사람은 절벽 끝에서 허공 쪽으로 넘어가는 위치에 있다.",
        "entities": "등장 인물은 윌마와 토니 두 명뿐이다. 윌마는 금발의 20대 후반 미국인 여성으로 짙은 녹색 점프수트와 점퍼 벨트를 착용했고, 토니는 30대 중반 미국인 남성으로 기준과 유사한 검은 구조복·가슴 하네스·헬멧·로프 장비를 갖췄다. 윌마의 손은 토니의 가슴 하네스 연결부 부근을 붙들고 있으며 보조 연결선도 체결되어 있다. 전원이 꺼진 스마트폰은 외부에서 명확히 식별되지 않지만 주머니나 장비에 보관되었을 가능성은 남는다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "윌마는 앞으로 길게 몸을 던졌고 뒤쪽 부츠가 바위 끝에서 막 떨어지는 순간으로 보인다. 점퍼 벨트가 허리를 지지하고 토니에게 이어진 연결선은 팽팽하다. 토니는 수평에 가깝게 공중으로 끌려가지만, 가슴 하네스를 잡은 윌마의 손과 팽팽한 보조 연결선이 명백한 지지·견인 수단이므로 무지지 부유는 아니다. 두 사람의 자세는 오른쪽으로 이어질 상승 비행의 출발 동작으로 성립한다."
       },
       {
        "label": "B",
        "direction": "윌마의 몸과 뻗은 오른팔은 화면 오른쪽의 절벽 밖 허공 및 창백한 광주 쪽을 향하고, 시선은 약간 아래쪽의 진행 경로를 본다. 토니 역시 오른쪽의 윌마와 허공 쪽으로 향한다. 가슴 하네스와 점퍼 벨트 사이 연결선의 방향도 오른쪽 견인과 일치한다.",
        "built_space": "참조와 같은 숲 가장자리의 단일 암석 절벽, 아래의 광대한 숲 계곡, 흐린 낮 하늘이 보인다. 인공 고정 설비나 반사면은 없다. 오른쪽에 창백한 광주 하나가 보인다. 윌마는 절벽 끝에 있지만 토니는 아직 돌출부 위쪽에 많이 남아 있어 두 사람이 동시에 절벽 밖으로 발사된 위치 관계는 A보다 약하다.",
        "entities": "윌마와 토니 두 명만 등장한다. 윌마는 기준과 가까운 금발 여성과 녹색 점프수트·점퍼 벨트로 표현되며, 허리에는 스마트폰으로 보이는 직사각형 물체가 고정되어 있다. 토니는 남성 구조요원으로 가슴 하네스와 로프 장비를 착용했지만 캐릭터 기준의 헬멧이 없고 머리 모양도 달라 복장 지속성이 떨어진다. 윌마의 손은 토니의 가슴 하네스 부근을 잡고 있고 연결선도 체결되어 있다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "윌마의 뒤쪽 부츠는 절벽 끝에 아직 닿아 있고 앞다리는 허공으로 나가 있어 도약 직전 또는 막 이륙하는 자세다. 토니는 연결선과 윌마의 손에 의해 당겨지지만 오른발이 바위 표면에 분명히 접촉해 몸을 지지한다. 따라서 두 몸 모두 지지 수단과 발진 동작은 있으나, 토니가 이미 절벽 밖으로 운반되는 명시된 순간에는 아직 이르다."
       }
      ],
      "all_candidates_fail": false
     },
     "normalized": {
      "winner": "B",
      "ranking": [
       "B",
       "A"
      ],
      "verdicts": [
       {
        "label": "B",
        "score": 8,
        "verdict_ko": "윌마의 뒷발이 절벽 끝을 막 떠나고 토니가 팽팽한 연결선에 실려 공중으로 끌려가는 정확한 발사 순간과 와이드 구도를 가장 충실히 구현했다."
       },
       {
        "label": "A",
        "score": 6,
        "verdict_ko": "방향과 장소는 맞지만 토니의 발이 아직 바위에 닿아 있어 ‘절벽 밖으로 운반되는’ 순간보다 이르고, 기준 복장의 헬멧도 빠졌다."
       }
      ],
      "all_candidates_fail": false,
      "readings": [
       {
        "label": "B",
        "direction": "윌마의 몸과 뻗은 오른팔, 시선은 화면 오른쪽의 절벽 밖 허공과 멀리 보이는 창백한 광주를 향한다. 토니도 오른쪽으로 끌려가며 같은 진행 방향을 본다. 두 사람 사이의 연결선은 토니의 가슴에서 윌마의 허리로 팽팽하게 이어져 이동 방향과 일치한다.",
        "built_space": "야외의 단일 암석 절벽 돌출부가 화면 왼쪽과 중앙에 있고, 뒤에는 숲 능선과 계곡, 흐린 하늘이 펼쳐진다. 인공 고정 설비나 반사면은 없다. 창백한 광주는 오른쪽에 하나만 보이며, 두 사람은 절벽 끝에서 허공 쪽으로 넘어가는 위치에 있다.",
        "entities": "등장 인물은 윌마와 토니 두 명뿐이다. 윌마는 금발의 20대 후반 미국인 여성으로 짙은 녹색 점프수트와 점퍼 벨트를 착용했고, 토니는 30대 중반 미국인 남성으로 기준과 유사한 검은 구조복·가슴 하네스·헬멧·로프 장비를 갖췄다. 윌마의 손은 토니의 가슴 하네스 연결부 부근을 붙들고 있으며 보조 연결선도 체결되어 있다. 전원이 꺼진 스마트폰은 외부에서 명확히 식별되지 않지만 주머니나 장비에 보관되었을 가능성은 남는다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "윌마는 앞으로 길게 몸을 던졌고 뒤쪽 부츠가 바위 끝에서 막 떨어지는 순간으로 보인다. 점퍼 벨트가 허리를 지지하고 토니에게 이어진 연결선은 팽팽하다. 토니는 수평에 가깝게 공중으로 끌려가지만, 가슴 하네스를 잡은 윌마의 손과 팽팽한 보조 연결선이 명백한 지지·견인 수단이므로 무지지 부유는 아니다. 두 사람의 자세는 오른쪽으로 이어질 상승 비행의 출발 동작으로 성립한다."
       },
       {
        "label": "A",
        "direction": "윌마의 몸과 뻗은 오른팔은 화면 오른쪽의 절벽 밖 허공 및 창백한 광주 쪽을 향하고, 시선은 약간 아래쪽의 진행 경로를 본다. 토니 역시 오른쪽의 윌마와 허공 쪽으로 향한다. 가슴 하네스와 점퍼 벨트 사이 연결선의 방향도 오른쪽 견인과 일치한다.",
        "built_space": "참조와 같은 숲 가장자리의 단일 암석 절벽, 아래의 광대한 숲 계곡, 흐린 낮 하늘이 보인다. 인공 고정 설비나 반사면은 없다. 오른쪽에 창백한 광주 하나가 보인다. 윌마는 절벽 끝에 있지만 토니는 아직 돌출부 위쪽에 많이 남아 있어 두 사람이 동시에 절벽 밖으로 발사된 위치 관계는 A보다 약하다.",
        "entities": "윌마와 토니 두 명만 등장한다. 윌마는 기준과 가까운 금발 여성과 녹색 점프수트·점퍼 벨트로 표현되며, 허리에는 스마트폰으로 보이는 직사각형 물체가 고정되어 있다. 토니는 남성 구조요원으로 가슴 하네스와 로프 장비를 착용했지만 캐릭터 기준의 헬멧이 없고 머리 모양도 달라 복장 지속성이 떨어진다. 윌마의 손은 토니의 가슴 하네스 부근을 잡고 있고 연결선도 체결되어 있다. 읽을 수 있는 글자는 없다.",
        "hard_violations": [],
        "physics": "윌마의 뒤쪽 부츠는 절벽 끝에 아직 닿아 있고 앞다리는 허공으로 나가 있어 도약 직전 또는 막 이륙하는 자세다. 토니는 연결선과 윌마의 손에 의해 당겨지지만 오른발이 바위 표면에 분명히 접촉해 몸을 지지한다. 따라서 두 몸 모두 지지 수단과 발진 동작은 있으나, 토니가 이미 절벽 밖으로 운반되는 명시된 순간에는 아직 이르다."
       }
      ]
     },
     "ok": true
    }
   ],
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "slot_winner_match": true,
   "slot_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "route": "cross_slot_agree"
  },
  "dual": {
   "models": [
    "gemini-pro",
    "gpt"
   ],
   "normalized": {
    "A": 1.321,
    "B": 2.0
   },
   "adjusted": {
    "A": 1.071,
    "B": 2.0
   },
   "violations": {
    "A": [
     "[gemini-pro] 토니의 헬멧 누락 (캐릭터 레퍼런스의 의상/식별 요소 불일치)"
    ]
   },
   "per_model_winner": {
    "gemini-pro": "B",
    "gpt": "B"
   },
   "agreed": true
  },
  "totals": {
   "B": 2000,
   "A": 1071
  },
  "selected": "B",
  "ranking": [
   "B",
   "A"
  ],
  "verdicts": [
   {
    "label": "B",
    "score": 2000,
    "verdict_ko": "토니의 헬멧과 장비를 레퍼런스에 맞게 유지하여 A보다 우수하나, 두 후보 모두 윌마가 토니의 가슴팍(하네스)을 직접 움켜쥐지 않고 줄을 잡고 있으며 토니의 자세가 레퍼런스 마네킹처럼 뻣뻣한 점이 아쉽습니다."
   },
   {
    "label": "A",
    "score": 1071,
    "verdict_ko": "도약하는 찰나의 연출은 배경 레퍼런스와 일치하나, 토니가 착용해야 할 헬멧이 완전히 누락되었고 윌마의 손 위치도 프롬프트 지시와 어긋납니다.  ★위반: [gemini-pro] 토니의 헬멧 누락 (캐릭터 레퍼런스의 의상/식별 요소 불일치)"
   }
  ],
  "refs": [
   {
    "label": "SHOT BACKGROUND",
    "path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S9sh13__bgfirst_bg.png",
    "asset_id": "eff365af-c9e7-4512-8cde-e34a8a6364bb",
    "role": "bgfirst_bg"
   },
   {
    "label": "CHARACTER REFERENCE — 토니(앤서니 로저스): the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:929851>",
    "asset_id": "1d3c6fe6-553b-4c00-b68f-025be6d786c1",
    "role": "character_ref"
   },
   {
    "label": "CHARACTER REFERENCE — 윌마 디어링: the exact person appearing in this shot; match face, hair and build exactly.",
    "path": "<bytes:766962>",
    "asset_id": "4982a15f-bf33-4ea0-b9c8-11e549bdf723",
    "role": "character_ref"
   }
  ],
  "critique_skipped": true,
  "shot_run_produced": true,
  "shot_run_uid": "06a9b2f2-562c-73fc-b9e8-cd793d90d741",
  "bgfirst": {
   "bg_path": "/Users/manta/Documents/Projects/TheRoad-I1/projects/37154490-a4ed-4b37-9092-489c054d8a72/images/c804efc3-0697-4c22-98b4-6992a70c2b20/scene/recipe/S9sh13__bgfirst_bg.png",
   "bg_asset_id": "eff365af-c9e7-4512-8cde-e34a8a6364bb",
   "bg_record_key": "S9sh13::bgfirst_bg",
   "chain_winner": true,
   "authority": "lane_conti_only"
  },
  "ref_mode": "lane(map_marker): 스케치+prev+엔티티",
  "share_plan": {
   "ref_plan": "prev",
   "prev_anchor_tag": "S9sh5"
  }
 },
 "S9sh13::cine": {
  "applied": true,
  "attempted_at": "2026-09-04T20:53:32.109250+00:00",
  "fingerprint": "9eb0d0644310ecc09a162227ea11d9b56b816e6b38225d3f75e5f6b96d9de58c",
  "fingerprint_version": 2,
  "provider": "grok",
  "endpoint": "openrouter/chat-completions",
  "model": "x-ai/grok-imagine-image-2.0",
  "pack": "24.202608252115",
  "source_file": "S9sh13_sel.png",
  "source_sha256": "3d3d951a0424e1148b3889f5fe3ca054de144ec9708eeab100ba17b3df1a4bcf",
  "file": "S9sh13_cine.png",
  "staged_sha256": "c3618f6319da704efaf348519a8313eb206e33ed3583b3c666aa53c8fccb54c7",
  "latency_ms": 14399
 }
}