{
  "contents": [
    {
      "role": "user",
      "parts": [
        {
          "text": "Two actual audio clips:1 own English source passage,2 a new Ukrainian candidate passage. Transcribe2 exactly without assuming the words, then mark the actually HEARD stresses. Evaluate smoothness: any cut consonant, vowel blur, voice/room/pitch discontinuity, beat jump or obvious edit? Distinguish melodic height from lexical stress. Do not assume a problem exists, nor accept it merely because intelligible. Return ONLY JSON <=800words: mediaAccess(boolean),audioAccess(boolean),ukrainianVerbatim,heardWordStresses,phoneticIntegrity,musicalContinuity,voiceContinuity,concreteDefects,uncertainties. Use words as anchors; source/source-product details must not leak into Ukrainian transcription. No scores or claim same real person."
        },
        {
          "text": "Attachment 1: Attachment 1; parent targ.mp4; EXACT parent range [2.6, 11]; local audio starts0. Use attachment number and local seconds, never concatenate timestamps. Fixed gain, no retiming."
        },
        {
          "inlineData": {
            "mimeType": "audio/mpeg",
            "data": "[base64 of frozen bytes; see inputs.json]"
          }
        },
        {
          "text": "Attachment 2: Attachment 2; parent gopure-targ-uk-master.wav; EXACT parent range [3, 10]; local audio starts0. Use attachment number and local seconds, never concatenate timestamps. Fixed gain, no retiming."
        },
        {
          "inlineData": {
            "mimeType": "audio/mpeg",
            "data": "[base64 of frozen bytes; see inputs.json]"
          }
        }
      ]
    }
  ],
  "stream": true,
  "generationConfig": {
    "thinkingConfig": {
      "includeThoughts": false,
      "thinkingLevel": "high"
    }
  }
}
