Merge pull request #1296 from pipecat-ai/aleix/google-always-send-text-with-audio
GoogleLLMService: always send text with audio
This commit is contained in:
@@ -82,6 +82,9 @@ stt = DeepgramSTTService(..., live_options=LiveOptions(model="nova-2-general"))
|
|||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
|
- Fixed a `GoogleLLMService` that was causing an exception when sending inline
|
||||||
|
audio in some cases.
|
||||||
|
|
||||||
- Fixed an `AudioContextWordTTSService` issue that would cause an `EndFrame` to
|
- Fixed an `AudioContextWordTTSService` issue that would cause an `EndFrame` to
|
||||||
disconnect from the TTS service before audio from all the contexts was
|
disconnect from the TTS service before audio from all the contexts was
|
||||||
received. This affected services like Cartesia and Rime.
|
received. This affected services like Cartesia and Rime.
|
||||||
|
|||||||
@@ -389,7 +389,7 @@ class AudioAccumulator(FrameProcessor):
|
|||||||
)
|
)
|
||||||
self._user_speaking = False
|
self._user_speaking = False
|
||||||
context = GoogleLLMContext()
|
context = GoogleLLMContext()
|
||||||
context.add_audio_frames_message(text="Audio follows", audio_frames=self._audio_frames)
|
context.add_audio_frames_message(audio_frames=self._audio_frames)
|
||||||
await self.push_frame(OpenAILLMContextFrame(context=context))
|
await self.push_frame(OpenAILLMContextFrame(context=context))
|
||||||
elif isinstance(frame, InputAudioRawFrame):
|
elif isinstance(frame, InputAudioRawFrame):
|
||||||
# Append the audio frame to our buffer. Treat the buffer as a ring buffer, dropping the oldest
|
# Append the audio frame to our buffer. Treat the buffer as a ring buffer, dropping the oldest
|
||||||
|
|||||||
@@ -7,20 +7,15 @@ import argparse
|
|||||||
import asyncio
|
import asyncio
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
from dataclasses import dataclass
|
|
||||||
from typing import Optional
|
from typing import Optional
|
||||||
|
|
||||||
import google.ai.generativelanguage as glm
|
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from loguru import logger
|
from loguru import logger
|
||||||
|
|
||||||
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
||||||
from pipecat.frames.frames import (
|
from pipecat.frames.frames import (
|
||||||
BotStoppedSpeakingFrame,
|
|
||||||
EndTaskFrame,
|
EndTaskFrame,
|
||||||
Frame,
|
|
||||||
InputAudioRawFrame,
|
InputAudioRawFrame,
|
||||||
SystemFrame,
|
|
||||||
TranscriptionFrame,
|
TranscriptionFrame,
|
||||||
UserStartedSpeakingFrame,
|
UserStartedSpeakingFrame,
|
||||||
UserStoppedSpeakingFrame,
|
UserStoppedSpeakingFrame,
|
||||||
@@ -28,11 +23,11 @@ from pipecat.frames.frames import (
|
|||||||
from pipecat.pipeline.pipeline import Pipeline
|
from pipecat.pipeline.pipeline import Pipeline
|
||||||
from pipecat.pipeline.runner import PipelineRunner
|
from pipecat.pipeline.runner import PipelineRunner
|
||||||
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
||||||
from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContextFrame
|
|
||||||
from pipecat.processors.frame_processor import FrameDirection, FrameProcessor
|
from pipecat.processors.frame_processor import FrameDirection, FrameProcessor
|
||||||
from pipecat.services.ai_services import LLMService
|
from pipecat.services.ai_services import LLMService
|
||||||
from pipecat.services.elevenlabs import ElevenLabsTTSService
|
from pipecat.services.elevenlabs import ElevenLabsTTSService
|
||||||
from pipecat.services.google import GoogleLLMContext, GoogleLLMService
|
from pipecat.services.google import GoogleLLMService
|
||||||
|
from pipecat.services.google.google import GoogleLLMContext
|
||||||
from pipecat.transports.services.daily import DailyDialinSettings, DailyParams, DailyTransport
|
from pipecat.transports.services.daily import DailyDialinSettings, DailyParams, DailyTransport
|
||||||
|
|
||||||
load_dotenv(override=True)
|
load_dotenv(override=True)
|
||||||
@@ -240,7 +235,7 @@ If it sounds like a human (saying hello, asking questions, etc.), call the funct
|
|||||||
DO NOT say anything until you've determined if this is a voicemail or human."""
|
DO NOT say anything until you've determined if this is a voicemail or human."""
|
||||||
|
|
||||||
llm = GoogleLLMService(
|
llm = GoogleLLMService(
|
||||||
model="models/gemini-2.0-flash-lite-preview-02-05",
|
model="models/gemini-2.0-flash-lite",
|
||||||
api_key=os.getenv("GOOGLE_API_KEY"),
|
api_key=os.getenv("GOOGLE_API_KEY"),
|
||||||
system_instruction=system_instruction,
|
system_instruction=system_instruction,
|
||||||
tools=tools,
|
tools=tools,
|
||||||
|
|||||||
@@ -22,9 +22,6 @@ classifiers = [
|
|||||||
dependencies = [
|
dependencies = [
|
||||||
"aiohttp~=3.11.11",
|
"aiohttp~=3.11.11",
|
||||||
"audioop-lts~=0.2.1; python_version>='3.13'",
|
"audioop-lts~=0.2.1; python_version>='3.13'",
|
||||||
# We need an older version of `httpx` that doesn't remove the deprecated
|
|
||||||
# `proxies` argument. This is necessary for Azure and Anthropic clients.
|
|
||||||
"httpx~=0.27.2",
|
|
||||||
"loguru~=0.7.3",
|
"loguru~=0.7.3",
|
||||||
"Markdown~=3.7",
|
"Markdown~=3.7",
|
||||||
"numpy~=1.26.4",
|
"numpy~=1.26.4",
|
||||||
@@ -42,7 +39,7 @@ Source = "https://github.com/pipecat-ai/pipecat"
|
|||||||
Website = "https://pipecat.ai"
|
Website = "https://pipecat.ai"
|
||||||
|
|
||||||
[project.optional-dependencies]
|
[project.optional-dependencies]
|
||||||
anthropic = [ "anthropic~=0.45.2" ]
|
anthropic = [ "anthropic~=0.47.2" ]
|
||||||
assemblyai = [ "assemblyai~=0.36.0" ]
|
assemblyai = [ "assemblyai~=0.36.0" ]
|
||||||
aws = [ "boto3~=1.35.99" ]
|
aws = [ "boto3~=1.35.99" ]
|
||||||
azure = [ "azure-cognitiveservices-speech~=1.42.0"]
|
azure = [ "azure-cognitiveservices-speech~=1.42.0"]
|
||||||
@@ -56,7 +53,7 @@ elevenlabs = [ "websockets~=13.1" ]
|
|||||||
fal = [ "fal-client~=0.5.6" ]
|
fal = [ "fal-client~=0.5.6" ]
|
||||||
fish = [ "ormsgpack~=1.7.0", "websockets~=13.1" ]
|
fish = [ "ormsgpack~=1.7.0", "websockets~=13.1" ]
|
||||||
gladia = [ "websockets~=13.1" ]
|
gladia = [ "websockets~=13.1" ]
|
||||||
google = [ "google-cloud-speech~=2.31.0", "google-cloud-texttospeech~=2.25.0", "google-genai~=1.2.0", "google-generativeai~=0.8.4" ]
|
google = [ "google-cloud-speech~=2.31.0", "google-cloud-texttospeech~=2.25.0", "google-genai~=1.3.0", "google-generativeai~=0.8.4" ]
|
||||||
grok = []
|
grok = []
|
||||||
groq = []
|
groq = []
|
||||||
gstreamer = [ "pygobject~=3.50.0" ]
|
gstreamer = [ "pygobject~=3.50.0" ]
|
||||||
|
|||||||
@@ -722,7 +722,9 @@ class GoogleLLMContext(OpenAILLMContext):
|
|||||||
|
|
||||||
self.add_message(glm.Content(role="user", parts=parts))
|
self.add_message(glm.Content(role="user", parts=parts))
|
||||||
|
|
||||||
def add_audio_frames_message(self, *, audio_frames: list[AudioRawFrame], text: str = None):
|
def add_audio_frames_message(
|
||||||
|
self, *, audio_frames: list[AudioRawFrame], text: str = "Audio follows"
|
||||||
|
):
|
||||||
if not audio_frames:
|
if not audio_frames:
|
||||||
return
|
return
|
||||||
|
|
||||||
@@ -731,8 +733,9 @@ class GoogleLLMContext(OpenAILLMContext):
|
|||||||
|
|
||||||
parts = []
|
parts = []
|
||||||
data = b"".join(frame.audio for frame in audio_frames)
|
data = b"".join(frame.audio for frame in audio_frames)
|
||||||
if text:
|
# NOTE(aleix): According to the docs only text or inline_data should be needed.
|
||||||
parts.append(glm.Part(text=text))
|
# (see https://cloud.google.com/vertex-ai/generative-ai/docs/model-reference/inference)
|
||||||
|
parts.append(glm.Part(text=text))
|
||||||
parts.append(
|
parts.append(
|
||||||
glm.Part(
|
glm.Part(
|
||||||
inline_data=glm.Blob(
|
inline_data=glm.Blob(
|
||||||
|
|||||||
Reference in New Issue
Block a user