{%- if messages[0].role == 'system' %} {{- '<|im_start|>system\n' }} {%- if messages[0].content is string %} {{- messages[0].content }} {%- else %} {%- for content in messages[0].content %} {%- if 'text' in content %} {{- content.text }} {%- endif %} {%- endfor %} {%- endif %} {{- '<|im_end|>\n' }} {%- else %} {{- '<|im_start|>system\nYou are a detailed image captioning assistant. Structure every caption as follows:\n\n1. Open with one sentence naming the shot type (e.g., eye-level, wide-angle, close-up), the overall setting, and the time of day or lighting condition.\n2. Break the rest of the description into thematic sections, each introduced by a bold markdown header ending in a colon (e.g., **The Subject:**, **The Background:**, **Atmosphere & Lighting:**), chosen to fit what is actually in the image.\n3. Within a section, use bullet points to list specific, concrete details — positions, colors, textures, materials, actions, spatial relationships, and any legible text or fine-grained elements. If a section covers multiple distinct areas of the frame, introduce each with its own nested bold sub-label ending in a colon (e.g., **Foreground Right:**, **Background:**) before its bullet points.\n4. Close with a short unheaded paragraph starting with \"In summary,\" that ties the scene together and conveys its overall mood or narrative.\n\nUse precise, sensory, fluent language throughout, and do not use emojis.<|im_end|>\n' }} {%- endif %} {%- set image_count = namespace(value=0) %} {%- set video_count = namespace(value=0) %} {%- for message in messages %} {%- if message.role == "user" %} {{- '<|im_start|>' + message.role + '\n' }} {%- if message.content is string %} {{- message.content }} {%- else %} {%- for content in message.content %} {%- if content.type == 'image' or 'image' in content or 'image_url' in content %} {%- set image_count.value = image_count.value + 1 %} {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%} <|vision_start|><|image_pad|><|vision_end|> {%- elif content.type == 'video' or 'video' in content %} {%- set video_count.value = video_count.value + 1 %} {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%} <|vision_start|><|video_pad|><|vision_end|> {%- elif 'text' in content %} {{- content.text }} {%- endif %} {%- endfor %} {%- endif %} {{- '<|im_end|>\n' }} {%- elif message.role == "assistant" %} {{- '<|im_start|>' + message.role + '\n' }} {%- if message.content is string %} {{- message.content }} {%- else %} {%- for content_item in message.content %} {%- if 'text' in content_item %} {{- content_item.text }} {%- endif %} {%- endfor %} {%- endif %} {{- '<|im_end|>\n' }} {%- endif %} {%- endfor %} {%- if add_generation_prompt %} {{- '<|im_start|>assistant\n' }} {%- endif %}