Environment details
- Programming language: Python
- OS: MacOS 15.4.1
- Language runtime version: 3.12
- Package version: 1.14.0
Steps to reproduce
We are currently testing the Live API and so for that I took the https://github.com/google-gemini/cookbook/blob/main/quickstarts/Get_started_LiveAPI.py code and updated it to print the text response from the model by adding the following code -
async def receive_text(self):
full_response = []
async for response in self.session.receive():
print("Received response chunk")
if text := response.text:
full_response.append(text)
print(f"Received response chunk: {text}")
else:
print("No text in response chunk")
Then I modified the run() method logic to print the text response instead of playing the audio -
send_text_task = tg.create_task(self.send_text())
tg.create_task(self.send_realtime())
# tg.create_task(self.listen_audio())
# if self.video_mode == "camera":
# tg.create_task(self.get_frames())
if self.video_mode == "screen":
tg.create_task(self.get_screen())
# tg.create_task(self.receive_audio())
# tg.create_task(self.play_audio())
tg.create_task(self.receive_text())
Now when I am trying to run this code, it is asking for the next input once i type something and press enter but no response from the model is being printed. I can see on the console that it is posting screenshots continuously to the model but not sure why the model is not returning any response.
I wrote another program where the model returns immediately if i use only session.send_realtime_input(text='abc') but using this after posting few screenshots makes model to return no response.
Here is the updated python script -
# -*- coding: utf-8 -*-
# Copyright 2025 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
"""
## Setup
To install the dependencies for this script, run:
pip install google-genai opencv-python pyaudio pillow mss
Before running this script, ensure the `GOOGLE_API_KEY` environment
variable is set to the api-key you obtained from Google AI Studio.
Important: **Use headphones**. This script uses the system default audio
input and output, which often won't include echo cancellation. So to prevent
the model from interrupting itself it is important that you use headphones.
## Run
To run the script:
python Get_started_LiveAPI.py
The script takes a video-mode flag `--mode`, this can be "camera", "screen", or "none".
The default is "camera". To share your screen run:
python Get_started_LiveAPI.py --mode screen
"""
import asyncio
import base64
import io
import os
import sys
import traceback
import cv2
# import pyaudio
import PIL.Image
import mss
import argparse
from google import genai
if sys.version_info < (3, 11, 0):
import taskgroup, exceptiongroup
asyncio.TaskGroup = taskgroup.TaskGroup
asyncio.ExceptionGroup = exceptiongroup.ExceptionGroup
# FORMAT = pyaudio.paInt16
# CHANNELS = 1
# SEND_SAMPLE_RATE = 16000
# RECEIVE_SAMPLE_RATE = 24000
# CHUNK_SIZE = 1024
MODEL = "models/gemini-2.0-flash-live-001"
DEFAULT_MODE = "screen"
client = genai.Client(http_options={"api_version": "v1beta"})
CONFIG = {"response_modalities": ["TEXT"]}
# pya = pyaudio.PyAudio()
class AudioLoop:
def __init__(self, video_mode=DEFAULT_MODE):
self.video_mode = video_mode
self.audio_in_queue = None
self.out_queue = None
self.session = None
self.send_text_task = None
self.receive_audio_task = None
self.play_audio_task = None
async def send_text(self):
while True:
text = await asyncio.to_thread(
input,
"message > ",
)
if text.lower() == "q":
break
await self.session.send(input=text or ".", end_of_turn=False)
# def _get_frame(self, cap):
# # Read the frameq
# ret, frame = cap.read()
# # Check if the frame was read successfully
# if not ret:
# return None
# # Fix: Convert BGR to RGB color space
# # OpenCV captures in BGR but PIL expects RGB format
# # This prevents the blue tint in the video feed
# frame_rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
# img = PIL.Image.fromarray(frame_rgb) # Now using RGB frame
# img.thumbnail([1024, 1024])
# image_io = io.BytesIO()
# img.save(image_io, format="jpeg")
# image_io.seek(0)
# mime_type = "image/jpeg"
# image_bytes = image_io.read()
# return {"mime_type": mime_type, "data": base64.b64encode(image_bytes).decode()}
# async def get_frames(self):
# # This takes about a second, and will block the whole program
# # causing the audio pipeline to overflow if you don't to_thread it.
# cap = await asyncio.to_thread(
# cv2.VideoCapture, 0
# ) # 0 represents the default camera
# while True:
# frame = await asyncio.to_thread(self._get_frame, cap)
# if frame is None:
# break
# await asyncio.sleep(1.0)
# await self.out_queue.put(frame)
# # Release the VideoCapture object
# cap.release()
def _get_screen(self):
sct = mss.mss()
monitor = sct.monitors[0]
i = sct.grab(monitor)
mime_type = "image/jpeg"
image_bytes = mss.tools.to_png(i.rgb, i.size)
img = PIL.Image.open(io.BytesIO(image_bytes))
image_io = io.BytesIO()
img.save(image_io, format="jpeg")
image_io.seek(0)
image_bytes = image_io.read()
return {"mime_type": mime_type, "data": base64.b64encode(image_bytes).decode()}
async def get_screen(self):
while True:
frame = await asyncio.to_thread(self._get_screen)
if frame is None:
break
await asyncio.sleep(15.0)
await self.out_queue.put(frame)
async def send_realtime(self):
while True:
msg = await self.out_queue.get()
await self.session.send(input=msg)
print("Sent realtime input")
# async def listen_audio(self):
# mic_info = pya.get_default_input_device_info()
# self.audio_stream = await asyncio.to_thread(
# pya.open,
# format=FORMAT,
# channels=CHANNELS,
# rate=SEND_SAMPLE_RATE,
# input=True,
# input_device_index=mic_info["index"],
# frames_per_buffer=CHUNK_SIZE,
# )
# if __debug__:
# kwargs = {"exception_on_overflow": False}
# else:
# kwargs = {}
# while True:
# data = await asyncio.to_thread(self.audio_stream.read, CHUNK_SIZE, **kwargs)
# await self.out_queue.put({"data": data, "mime_type": "audio/pcm"})
# async def receive_audio(self):
# "Background task to reads from the websocket and write pcm chunks to the output queue"
# while True:
# turn = self.session.receive()
# async for response in turn:
# if data := response.data:
# self.audio_in_queue.put_nowait(data)
# continue
# if text := response.text:
# print(text, end="")
# # If you interrupt the model, it sends a turn_complete.
# # For interruptions to work, we need to stop playback.
# # So empty out the audio queue because it may have loaded
# # much more audio than has played yet.
# while not self.audio_in_queue.empty():
# self.audio_in_queue.get_nowait()
async def receive_text(self):
full_response = []
async for response in self.session.receive():
print("Received response chunk")
if text := response.text:
full_response.append(text)
print(f"Received response chunk: {text}")
else:
print("No text in response chunk")
# async def play_audio(self):
# stream = await asyncio.to_thread(
# pya.open,
# format=FORMAT,
# channels=CHANNELS,
# rate=RECEIVE_SAMPLE_RATE,
# output=True,
# )
# while True:
# bytestream = await self.audio_in_queue.get()
# await asyncio.to_thread(stream.write, bytestream)
async def run(self):
try:
async with (
client.aio.live.connect(model=MODEL, config=CONFIG) as session,
asyncio.TaskGroup() as tg,
):
self.session = session
self.audio_in_queue = asyncio.Queue()
self.out_queue = asyncio.Queue(maxsize=5)
send_text_task = tg.create_task(self.send_text())
tg.create_task(self.send_realtime())
# tg.create_task(self.listen_audio())
# if self.video_mode == "camera":
# tg.create_task(self.get_frames())
if self.video_mode == "screen":
tg.create_task(self.get_screen())
# tg.create_task(self.receive_audio())
# tg.create_task(self.play_audio())
tg.create_task(self.receive_text())
await send_text_task
raise asyncio.CancelledError("User requested exit")
except asyncio.CancelledError:
pass
except ExceptionGroup as EG:
self.audio_stream.close()
traceback.print_exception(EG)
if __name__ == "__main__":
# parser = argparse.ArgumentParser()
# parser.add_argument(
# "--mode",
# type=str,
# default=DEFAULT_MODE,
# help="pixels to stream from",
# choices=["camera", "screen", "none"],
# )
# args = parser.parse_args()
# main = AudioLoop(video_mode=args.mode)
main = AudioLoop()
asyncio.run(main.run())
Sample logs
$ uv run test-sample-code.py
message > analyze stream
/Users/torque/CascadeProjects/project1/CascadeProjects/windsurf-project/test-live-api/test-sample-code.py:109: DeprecationWarning: The `session.send` method is deprecated and will be removed in a future version (not before Q3 2025).
Please use one of the more specific methods: `send_client_content`, `send_realtime_input`, or `send_tool_response` instead.
await self.session.send(input=text or ".", end_of_turn=False)
message > /Users/torque/CascadeProjects/project1/CascadeProjects/windsurf-project/test-live-api/test-sample-code.py:182: DeprecationWarning: The `session.send` method is deprecated and will be removed in a future version (not before Q3 2025).
Please use one of the more specific methods: `send_client_content`, `send_realtime_input`, or `send_tool_response` instead.
await self.session.send(input=msg)
Sent realtime input
Sent realtime input
Sent realtime input
Environment details
Steps to reproduce
We are currently testing the Live API and so for that I took the https://github.com/google-gemini/cookbook/blob/main/quickstarts/Get_started_LiveAPI.py code and updated it to print the text response from the model by adding the following code -
Then I modified the run() method logic to print the text response instead of playing the audio -
Now when I am trying to run this code, it is asking for the next input once i type something and press enter but no response from the model is being printed. I can see on the console that it is posting screenshots continuously to the model but not sure why the model is not returning any response.
I wrote another program where the model returns immediately if i use only
session.send_realtime_input(text='abc')but using this after posting few screenshots makes model to return no response.Here is the updated python script -
pip install google-genai opencv-python pyaudio pillow mss
python Get_started_LiveAPI.py
python Get_started_LiveAPI.py --mode screen
Sample logs