Skip to content

Live API with text modality is not working #800

Description

@torquelab

Environment details

  • Programming language: Python
  • OS: MacOS 15.4.1
  • Language runtime version: 3.12
  • Package version: 1.14.0

Steps to reproduce

We are currently testing the Live API and so for that I took the https://github.com/google-gemini/cookbook/blob/main/quickstarts/Get_started_LiveAPI.py code and updated it to print the text response from the model by adding the following code -

async def receive_text(self):
        full_response = []
        async for response in self.session.receive():
            print("Received response chunk")
            if text := response.text:
                full_response.append(text)
                print(f"Received response chunk: {text}")
            else:
                print("No text in response chunk")

Then I modified the run() method logic to print the text response instead of playing the audio -

                send_text_task = tg.create_task(self.send_text())
                tg.create_task(self.send_realtime())
                # tg.create_task(self.listen_audio())
                # if self.video_mode == "camera":
                #     tg.create_task(self.get_frames())
                if self.video_mode == "screen":
                    tg.create_task(self.get_screen())

                # tg.create_task(self.receive_audio())
                # tg.create_task(self.play_audio())
                tg.create_task(self.receive_text())

Now when I am trying to run this code, it is asking for the next input once i type something and press enter but no response from the model is being printed. I can see on the console that it is posting screenshots continuously to the model but not sure why the model is not returning any response.

I wrote another program where the model returns immediately if i use only session.send_realtime_input(text='abc') but using this after posting few screenshots makes model to return no response.

Here is the updated python script -

# -*- coding: utf-8 -*-
# Copyright 2025 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

"""
## Setup

To install the dependencies for this script, run:

pip install google-genai opencv-python pyaudio pillow mss


Before running this script, ensure the `GOOGLE_API_KEY` environment
variable is set to the api-key you obtained from Google AI Studio.

Important: **Use headphones**. This script uses the system default audio
input and output, which often won't include echo cancellation. So to prevent
the model from interrupting itself it is important that you use headphones. 

## Run

To run the script:

python Get_started_LiveAPI.py


The script takes a video-mode flag `--mode`, this can be "camera", "screen", or "none".
The default is "camera". To share your screen run:

python Get_started_LiveAPI.py --mode screen

"""

import asyncio
import base64
import io
import os
import sys
import traceback

import cv2
# import pyaudio
import PIL.Image
import mss

import argparse

from google import genai

if sys.version_info < (3, 11, 0):
    import taskgroup, exceptiongroup

    asyncio.TaskGroup = taskgroup.TaskGroup
    asyncio.ExceptionGroup = exceptiongroup.ExceptionGroup

# FORMAT = pyaudio.paInt16
# CHANNELS = 1
# SEND_SAMPLE_RATE = 16000
# RECEIVE_SAMPLE_RATE = 24000
# CHUNK_SIZE = 1024

MODEL = "models/gemini-2.0-flash-live-001"

DEFAULT_MODE = "screen"

client = genai.Client(http_options={"api_version": "v1beta"})

CONFIG = {"response_modalities": ["TEXT"]}

#  pya = pyaudio.PyAudio()


class AudioLoop:
    def __init__(self, video_mode=DEFAULT_MODE):
        self.video_mode = video_mode

        self.audio_in_queue = None
        self.out_queue = None

        self.session = None

        self.send_text_task = None
        self.receive_audio_task = None
        self.play_audio_task = None

    async def send_text(self):
        while True:
            text = await asyncio.to_thread(
                input,
                "message > ",
            )
            if text.lower() == "q":
                break
            await self.session.send(input=text or ".", end_of_turn=False)

    # def _get_frame(self, cap):
    #     # Read the frameq
    #     ret, frame = cap.read()
    #     # Check if the frame was read successfully
    #     if not ret:
    #         return None
    #     # Fix: Convert BGR to RGB color space
    #     # OpenCV captures in BGR but PIL expects RGB format
    #     # This prevents the blue tint in the video feed
    #     frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
    #     img = PIL.Image.fromarray(frame_rgb)  # Now using RGB frame
    #     img.thumbnail([1024, 1024])

    #     image_io = io.BytesIO()
    #     img.save(image_io, format="jpeg")
    #     image_io.seek(0)

    #     mime_type = "image/jpeg"
    #     image_bytes = image_io.read()
    #     return {"mime_type": mime_type, "data": base64.b64encode(image_bytes).decode()}

    # async def get_frames(self):
    #     # This takes about a second, and will block the whole program
    #     # causing the audio pipeline to overflow if you don't to_thread it.
    #     cap = await asyncio.to_thread(
    #         cv2.VideoCapture, 0
    #     )  # 0 represents the default camera

    #     while True:
    #         frame = await asyncio.to_thread(self._get_frame, cap)
    #         if frame is None:
    #             break

    #         await asyncio.sleep(1.0)

    #         await self.out_queue.put(frame)

    #     # Release the VideoCapture object
    #     cap.release()

    def _get_screen(self):
        sct = mss.mss()
        monitor = sct.monitors[0]

        i = sct.grab(monitor)

        mime_type = "image/jpeg"
        image_bytes = mss.tools.to_png(i.rgb, i.size)
        img = PIL.Image.open(io.BytesIO(image_bytes))

        image_io = io.BytesIO()
        img.save(image_io, format="jpeg")
        image_io.seek(0)

        image_bytes = image_io.read()
        return {"mime_type": mime_type, "data": base64.b64encode(image_bytes).decode()}

    async def get_screen(self):

        while True:
            frame = await asyncio.to_thread(self._get_screen)
            if frame is None:
                break

            await asyncio.sleep(15.0)

            await self.out_queue.put(frame)

    async def send_realtime(self):
        while True:
            msg = await self.out_queue.get()
            await self.session.send(input=msg)
            print("Sent realtime input")

    # async def listen_audio(self):
    #     mic_info = pya.get_default_input_device_info()
    #     self.audio_stream = await asyncio.to_thread(
    #         pya.open,
    #         format=FORMAT,
    #         channels=CHANNELS,
    #         rate=SEND_SAMPLE_RATE,
    #         input=True,
    #         input_device_index=mic_info["index"],
    #         frames_per_buffer=CHUNK_SIZE,
    #     )
    #     if __debug__:
    #         kwargs = {"exception_on_overflow": False}
    #     else:
    #         kwargs = {}
    #     while True:
    #         data = await asyncio.to_thread(self.audio_stream.read, CHUNK_SIZE, **kwargs)
    #         await self.out_queue.put({"data": data, "mime_type": "audio/pcm"})

    # async def receive_audio(self):
    #     "Background task to reads from the websocket and write pcm chunks to the output queue"
    #     while True:
    #         turn = self.session.receive()
    #         async for response in turn:
    #             if data := response.data:
    #                 self.audio_in_queue.put_nowait(data)
    #                 continue
    #             if text := response.text:
    #                 print(text, end="")

    #         # If you interrupt the model, it sends a turn_complete.
    #         # For interruptions to work, we need to stop playback.
    #         # So empty out the audio queue because it may have loaded
    #         # much more audio than has played yet.
    #         while not self.audio_in_queue.empty():
    #             self.audio_in_queue.get_nowait()

    async def receive_text(self):
        full_response = []
        async for response in self.session.receive():
            print("Received response chunk")
            if text := response.text:
                full_response.append(text)
                print(f"Received response chunk: {text}")
            else:
                print("No text in response chunk")

    # async def play_audio(self):
    #     stream = await asyncio.to_thread(
    #         pya.open,
    #         format=FORMAT,
    #         channels=CHANNELS,
    #         rate=RECEIVE_SAMPLE_RATE,
    #         output=True,
    #     )
    #     while True:
    #         bytestream = await self.audio_in_queue.get()
    #         await asyncio.to_thread(stream.write, bytestream)

    async def run(self):
        try:
            async with (
                client.aio.live.connect(model=MODEL, config=CONFIG) as session,
                asyncio.TaskGroup() as tg,
            ):
                self.session = session

                self.audio_in_queue = asyncio.Queue()
                self.out_queue = asyncio.Queue(maxsize=5)

                send_text_task = tg.create_task(self.send_text())
                tg.create_task(self.send_realtime())
                # tg.create_task(self.listen_audio())
                # if self.video_mode == "camera":
                #     tg.create_task(self.get_frames())
                if self.video_mode == "screen":
                    tg.create_task(self.get_screen())

                # tg.create_task(self.receive_audio())
                # tg.create_task(self.play_audio())
                tg.create_task(self.receive_text())

                await send_text_task
                raise asyncio.CancelledError("User requested exit")

        except asyncio.CancelledError:
            pass
        except ExceptionGroup as EG:
            self.audio_stream.close()
            traceback.print_exception(EG)


if __name__ == "__main__":
    # parser = argparse.ArgumentParser()
    # parser.add_argument(
    #     "--mode",
    #     type=str,
    #     default=DEFAULT_MODE,
    #     help="pixels to stream from",
    #     choices=["camera", "screen", "none"],
    # )
    # args = parser.parse_args()
    # main = AudioLoop(video_mode=args.mode)
    main = AudioLoop()
    asyncio.run(main.run())

Sample logs

$ uv run test-sample-code.py 
message > analyze stream
/Users/torque/CascadeProjects/project1/CascadeProjects/windsurf-project/test-live-api/test-sample-code.py:109: DeprecationWarning: The `session.send` method is deprecated and will be removed in a future version (not before Q3 2025).
Please use one of the more specific methods: `send_client_content`, `send_realtime_input`, or `send_tool_response` instead.
  await self.session.send(input=text or ".", end_of_turn=False)
message > /Users/torque/CascadeProjects/project1/CascadeProjects/windsurf-project/test-live-api/test-sample-code.py:182: DeprecationWarning: The `session.send` method is deprecated and will be removed in a future version (not before Q3 2025).
Please use one of the more specific methods: `send_client_content`, `send_realtime_input`, or `send_tool_response` instead.
  await self.session.send(input=msg)
Sent realtime input
Sent realtime input
Sent realtime input

Metadata

Metadata

Assignees

Labels

priority: p2Moderately-important priority. Fix may not be included in next release.status:awaiting user responsetype: bugError or flaw in code with unintended results or allowing sub-optimal usage patterns.

Type

No type

Projects

No projects

Milestone

No milestone

Relationships

None yet

Development

No branches or pull requests

Issue actions