{
  "contents": [
    {
      "role": "user",
      "parts": [
        {
          "text": "Two audio attachments. 1 English source ending;2 a new Ukrainian song ending. Transcribe attachment2 exactly from sound, WITHOUT assuming a script or product name. Give brand-like syllables phonetically: whether onset is voiced g or unvoiced k and first vowel ou/o versus u, then full next syllable. Mark ambiguous phonetics as uncertain; do not guess spelling. Identify actual lexical stresses on payment/inspection words and each final word. Describe real musical phrase groups/breaths and whether the invitation has audible melody. Do not equate a conversational or soft sung line with speech. Return ONLY JSON <=850words: mediaAccess(boolean),audioAccess(boolean),ukrainianVerbatim,brandPhonetics,stressFlags,heardPhraseGroups:[{firstWords,lastWords,contour,release}],endingIntegrity,sourceComparison,uncertainties. Opening may start mid-word because this is an excerpt, not song beginning. No scores or acceptance."
        },
        {
          "text": "Attachment 1: Attachment 1; parent targ.mp4; EXACT parent range [181, 194.488934]; local audio starts0. Use attachment number and local seconds, never concatenate timestamps. Fixed gain, no retiming."
        },
        {
          "inlineData": {
            "mimeType": "audio/mpeg",
            "data": "[base64 of frozen bytes; see inputs.json]"
          }
        },
        {
          "text": "Attachment 2: Attachment 2; parent take_refine02_1.mp3; EXACT parent range [173.592, 203.592]; local audio starts0. Use attachment number and local seconds, never concatenate timestamps. Fixed gain, no retiming."
        },
        {
          "inlineData": {
            "mimeType": "audio/mpeg",
            "data": "[base64 of frozen bytes; see inputs.json]"
          }
        }
      ]
    }
  ],
  "stream": true,
  "generationConfig": {
    "thinkingConfig": {
      "includeThoughts": false,
      "thinkingLevel": "high"
    }
  }
}
