This is an old revision of the document!
~/ai$ cp ari_rec_ai.py ari_rec_ai_content.py
$ cat request_to_agent_with_tools.py
import sys
import os
import random
from yandex_ai_studio_sdk import AIStudio
from rich import inspect
def request_to_agent_with_tools(user_request, system_prompt, context_messages=None, tools=None):
sdk = AIStudio(
folder_id=os.getenv("FOLDER_ID"),
auth=os.getenv("API_KEY")
)
sdk_tools = []
for item in tools:
tool = sdk.tools.function(
parameters={"type": "object","properties": {}},
name=item["func_name"],
description=item["description"]
)
sdk_tools.append(tool)
messages = [{"role": "system", "text": system_prompt}]
if context_messages:
messages.extend(context_messages)
messages.append({"role": "user", "text": user_request})
#model = sdk.chat.completions("yandexgpt")
model = sdk.chat.completions("aliceai-llm")
model = model.configure(
temperature=0.3,
max_tokens=1500,
tools=sdk_tools
)
#inspect(messages, methods=True)
inspect(messages)
response = model.run(messages)
# inspect(response, methods=True)
choice = response.choices[0]
if choice.tool_calls:
return choice.tool_calls[0].function.name
else:
return choice.text
if __name__ == "__main__":
system_prompt = (
"Ты — ассистент.\n"
"Отвечай кратко.\n"
)
chat_context = []
#chat_context = [
# {"role": "user", "text": "Привет! Меня зовут Алексей, я программист."},
# {"role": "assistant", "text": "Привет, Алексей! Рад знакомству. Чем могу помочь?"}
#]
# Определяем функцию для инструмента
def get_random_number() -> int:
return random.randint(1, 100)
def get_random_fruit() -> str:
words = ["яблоко", "банан", "апельсин", "груша", "киви"]
return random.choice(words)
#print(get_random_number())
#print(get_random_fruit())
func_list = [
{
"func_name": "random_number",
"description": "Используй эту функцию, когда пользователь просит назвать его счастливое число."
},
{
"func_name": "random_fruit",
"description": "Используй эту функцию, когда пользователь просит назвать его любимый фрукт."
}
]
while True:
try:
user_request = input("Введите ваш вопрос: ")
except (KeyboardInterrupt, EOFError):
print("\nСессия завершена!")
compressor_instruction = (
"Ты — инструмент оптимизации контекста. Проанализируй историю диалога. "
"Удали выполненные дела, приветствия и воду. Сформулируй ОДНО лаконичное "
"сообщение от лица пользователя, в котором отражена только актуальная суть "
"и невыполненные задачи/оставшиеся вопросы на текущий момент."
)
reply_text = request_to_agent_with_tools(user_request, compressor_instruction, chat_context, func_list)
print(reply_text)
sys.exit(0)
reply_text = request_to_agent_with_tools(user_request, system_prompt, chat_context, func_list)
if reply_text == 'random_number':
print(get_random_number())
elif reply_text == 'random_fruit':
print(get_random_fruit())
else:
print(reply_text)
chat_context.append({"role": "user", "text": user_request})
chat_context.append({"role": "assistant", "text": reply_text})
# cat ~asterisk/ai/ari_rec_ai_content.py
import os
import json
import requests
import websocket
ARI_USER = "asterisk"
ARI_PASS = "asterisk"
BASE_URL = "http://localhost:8088/ari"
AUTH = (ARI_USER, ARI_PASS)
REC_PATH = "/var/spool/asterisk/recording/"
SYSTEM_PROMPT = (
"Ты — ассистент.\n"
"Отвечай кратко.\n"
"Не спрашивай когда напомнить.\n"
)
COMPRESS_PROMPT = (
"Ты — инструмент оптимизации контекста. Проанализируй историю диалога. "
"Удали выполненные дела, приветствия и воду. Сформулируй ОДНО лаконичное "
"сообщение от лица пользователя, в котором отражена только актуальная суть "
"и невыполненные задачи/оставшиеся вопросы на текущий момент."
)
# Структура: {channel_id: [{"role": "user", "text": "..."}, {"role": "assistant", "text": "..."}]}
channel_contexts = {}
channel_caller_numbers = {}
def load_context(caller_number):
file_path = f"{REC_PATH}{caller_number}.json"
if os.path.exists(file_path):
with open(file_path, 'r', encoding='utf-8') as f:
return json.load(f)
return []
def save_context(caller_number, context):
file_path = f"{REC_PATH}{caller_number}.json"
with open(file_path, 'w', encoding='utf-8') as f:
json.dump(context, f, ensure_ascii=False, indent=2)
def get_caller_number(channel_id):
url = f"{BASE_URL}/channels/{channel_id}"
response = requests.get(url, auth=AUTH, timeout=5)
channel_info = response.json()
caller_number = channel_info.get("caller", {}).get("number")
return caller_number
def get_channel_context(channel_id):
if channel_id not in channel_contexts:
channel_contexts[channel_id] = []
return channel_contexts[channel_id]
def update_channel_context(channel_id, user_request, agent_response):
context = get_channel_context(channel_id)
context.append({"role": "user", "text": user_request})
context.append({"role": "assistant", "text": agent_response})
return context
def detect_speech(channel_id):
print(f"Запускаем TALK_DETECT для канала {channel_id}...")
talk_url = f"{BASE_URL}/channels/{channel_id}/variable"
# 1000 мс тишины для завершения фразы, 500 мс речи для начала
talk_payload = {
"variable": "TALK_DETECT(set)",
"value": "1000,500"
}
response = requests.post(talk_url, json=talk_payload, auth=AUTH)
print(f"Статус-код ответа: {response.status_code}")
print(f"Текст ответа: {response.text}")
def undetect_speech(channel_id):
print(f"Отключаем TALK_DETECT для канала {channel_id}...")
talk_url = f"{BASE_URL}/channels/{channel_id}/variable"
talk_payload = {
"variable": "TALK_DETECT(set)",
"value": "remove"
}
response = requests.post(talk_url, json=talk_payload, auth=AUTH)
print(f"Статус-код ответа: {response.status_code}")
print(f"Текст ответа: {response.text}")
def start_recording(channel_id):
print(f"Включаем запись для канала {channel_id}")
record_url = f"{BASE_URL}/channels/{channel_id}/record"
recording_name = f"rec_{channel_id}"
record_payload = {
"name": recording_name,
"format": "wav",
"ifExists": "overwrite",
#"maxDuration": 29,
#"terminateOn": "#",
#"beep": True,
}
requests.post(record_url, params=record_payload, auth=AUTH)
def stop_recording(channel_id):
print(f"Останавливаем запись для канала {channel_id}")
recording_name = f"rec_{channel_id}"
stop_url = f"{BASE_URL}/recordings/live/{recording_name}/stop"
requests.post(stop_url, auth=AUTH)
def on_message(wsapp, message):
event = json.loads(message)
event_type = event.get("type")
# Сценарий 1: Звонок поступил -> Отвечаем и воспроизводим beep
if event_type == "StasisStart":
channel_id = event.get("channel", {}).get("id")
print(f"Отвечаем на звонок {channel_id}")
requests.post(f"{BASE_URL}/channels/{channel_id}/answer", auth=AUTH)
channel_caller_numbers[channel_id]=get_caller_number(channel_id)
channel_contexts[channel_id] = load_context(channel_caller_numbers[channel_id])
#start_recording(channel_id)
# or
detect_speech(channel_id)
# Сценарий N
elif event_type == "ChannelTalkingStarted":
channel_id = event.get("channel", {}).get("id")
print(f"Обнаружена речь в канале {channel_id}! Начинаем запись...")
start_recording(channel_id)
# Сценарий N
elif event_type == "ChannelTalkingFinished":
channel_id = event.get("channel", {}).get("id")
print(f"Обнаружено молчание в канале {channel_id}")
stop_recording(channel_id)
undetect_speech(channel_id)
# Сценарий N: Запись завершилась -> Проигрываем её обратно
elif event_type == "RecordingFinished":
recording_data = event.get("recording", {})
recording_name = recording_data.get("name")
target_uri = recording_data.get("target_uri", "")
channel_id = target_uri.split("channel:")[1]
import wav_and_ogg
import speech_and_text
wav_and_ogg.wav_to_ogg(f"{REC_PATH}{recording_name}")
user_request = speech_and_text.speech_to_text(f"{REC_PATH}{recording_name}.ogg")
print(f"user_request: {user_request}")
if user_request:
context = get_channel_context(channel_id)
import request_to_agent
agent_response = request_to_agent.request_to_agent(user_request, SYSTEM_PROMPT, context)
#agent_response = user_request
print(agent_response)
update_channel_context(channel_id, user_request, agent_response)
print("Текущий контекст:")
from pprint import pprint
pprint(channel_contexts[channel_id])
speech_and_text.text_to_speech(agent_response,f"{REC_PATH}{recording_name}.response.ogg")
wav_and_ogg.ogg_to_wav(f"{REC_PATH}{recording_name}.response")
print(f"Запись {recording_name} готова! Проигрываем её обратно в канал {channel_id}...")
play_url = f"{BASE_URL}/channels/{channel_id}/play"
play_payload = {
#"media": f"recording:{recording_name}"
"media": f"recording:{recording_name}.response"
}
response = requests.post(play_url, params=play_payload, auth=AUTH)
print("Ответ ARI на воспроизведение:", response.status_code)
# Сценарий N: Воспроизведение завершено (бип или записанный голос)
elif event_type == "PlaybackFinished":
playback_data = event.get("playback", {})
media_uri = playback_data.get("media_uri", "")
target_uri = playback_data.get('target_uri', '')
channel_id = target_uri.split("channel:")[1]
#start_recording(channel_id)
# or
detect_speech(channel_id)
# Сценарий N: Абонент повесил трубку -> Удаляем файл
elif event_type == "StasisEnd":
channel_id = event.get("channel", {}).get("id")
#print(f"Звонок {channel_id} завершен абонентом. Удаляем файл записи...")
#recording_name = f"rec_{channel_id}"
#delete_url = f"{BASE_URL}/recordings/stored/{recording_name}"
#requests.delete(delete_url, auth=AUTH)
context = get_channel_context(channel_id)
caller_number = channel_caller_numbers[channel_id]
print(f"Номер звонящего в StasisEnd: номер {caller_number} канал {channel_id}")
print(f"Сжатие контекста")
import request_to_agent
agent_response = request_to_agent.request_to_agent("Сожми контекст", COMPRESS_PROMPT, context)
print(agent_response)
context = [{"role": "user", "text": agent_response}]
print(f"Сохранение контекста и удаление медиа файлов")
save_context(caller_number,context)
import glob
recording_name = f"rec_{channel_id}"
file_mask = f"{REC_PATH}{recording_name}*"
matched_files = glob.glob(file_mask)
for file_path in matched_files:
os.remove(file_path)
print(f"Файл удален: {file_path}")
def on_error(ws, error):
print(f"Произошла ошибка: {error}")
wsapp = websocket.WebSocketApp(
f"ws://localhost:8088/ari/events?api_key={ARI_USER}:{ARI_PASS}&app=my-first-app",
on_message=on_message,
on_error=on_error
)
wsapp.run_forever()