User Tools

Site Tools


использование_ai-агентов

This is an old revision of the document!


Использование AI-агентов

Реклама

  • Сегодня, наверное, нет пользователя сети Интернет, который бы не “общался” (хотя бы через чат в поисковой системе) с большими языковыми моделями (LLM), или, как это часто называют, искусственным интеллектом (AI)
  • Невероятные результаты, которые они показывают в качестве справочных и систем обработки данных, вызывают желание “поручить” им самостоятельное решение каких либо задач
  • Для этого используются AI агенты, и если Вы пока не работали с ними, наш вебинар для Вас

Техническое задание

  • Написать своего первого AI агента на python
  • Реализовать MVP с использованием AI и Asterisk по такому описанию: перед походом в магазин, звоним (можно несколько раз) на некий номер и просим запомнить, что хотим купить, в магазине, время от времени, звоним на тот же номер, говорим, что купили и просим напомнить, что еще осталось
  • При разработке, в свою очередь, активно использовать LLM

Запись вебинара

  • Ютуб:
  • Вк:
  • Рутуб:
  • Тэги:

Шаг 1. Что у нас есть, для начала

Черновик

~/ai$ cp ari_rec_ai.py ari_rec_ai_content.py
# cat ~asterisk/ai/ari_rec_ai_content.py
import os
import json
import requests
import websocket

ARI_USER = "asterisk"
ARI_PASS = "asterisk"
BASE_URL = "http://localhost:8088/ari"
AUTH = (ARI_USER, ARI_PASS)

REC_PATH = "/var/spool/asterisk/recording/"

SYSTEM_PROMPT = (
    "Ты — ассистент.\n"
    "Отвечай кратко.\n"
    "Не спрашивай когда напомнить.\n"
)

COMPRESS_PROMPT = (
    "Ты — инструмент оптимизации контекста. Проанализируй историю диалога. "
    "Удали выполненные дела, приветствия и воду. Сформулируй ОДНО лаконичное "
    "сообщение от лица пользователя, в котором отражена только актуальная суть "
    "и невыполненные задачи/оставшиеся вопросы на текущий момент."
)

# Структура: {channel_id: [{"role": "user", "text": "..."}, {"role": "assistant", "text": "..."}]}
channel_contexts = {}

channel_caller_numbers = {}

def load_context(caller_number):
    file_path = f"{REC_PATH}{caller_number}.json"
    if os.path.exists(file_path):
        with open(file_path, 'r', encoding='utf-8') as f:
            return json.load(f)
    return []

def save_context(caller_number, context):
    file_path = f"{REC_PATH}{caller_number}.json"
    with open(file_path, 'w', encoding='utf-8') as f:
        json.dump(context, f, ensure_ascii=False, indent=2)

def get_caller_number(channel_id):
    url = f"{BASE_URL}/channels/{channel_id}"
    response = requests.get(url, auth=AUTH, timeout=5)
    channel_info = response.json()
    caller_number = channel_info.get("caller", {}).get("number")
    return caller_number

def get_channel_context(channel_id):
    if channel_id not in channel_contexts:
        channel_contexts[channel_id] = []
    return channel_contexts[channel_id]

def update_channel_context(channel_id, user_request, agent_response):
    context = get_channel_context(channel_id)
    context.append({"role": "user", "text": user_request})
    context.append({"role": "assistant", "text": agent_response})
    return context

def detect_speech(channel_id):
    print(f"Запускаем TALK_DETECT для канала {channel_id}...")
    talk_url = f"{BASE_URL}/channels/{channel_id}/variable"

    # 1000 мс тишины для завершения фразы, 500 мс речи для начала
    talk_payload = {
        "variable": "TALK_DETECT(set)",
        "value": "1000,500"
    }

    response = requests.post(talk_url, json=talk_payload, auth=AUTH)
    print(f"Статус-код ответа: {response.status_code}")
    print(f"Текст ответа: {response.text}")

def undetect_speech(channel_id):
    print(f"Отключаем TALK_DETECT для канала {channel_id}...")
    talk_url = f"{BASE_URL}/channels/{channel_id}/variable"

    talk_payload = {
        "variable": "TALK_DETECT(set)",
        "value": "remove"
    }

    response = requests.post(talk_url, json=talk_payload, auth=AUTH)
    print(f"Статус-код ответа: {response.status_code}")
    print(f"Текст ответа: {response.text}")

def start_recording(channel_id):
    print(f"Включаем запись для канала {channel_id}")
    record_url = f"{BASE_URL}/channels/{channel_id}/record"
    recording_name = f"rec_{channel_id}"
    record_payload = {
        "name": recording_name,
        "format": "wav",
        "ifExists": "overwrite",
        #"maxDuration": 29,
        #"terminateOn": "#",
        #"beep": True,
    }
    requests.post(record_url, params=record_payload, auth=AUTH)

def stop_recording(channel_id):
    print(f"Останавливаем запись для канала {channel_id}")
    recording_name = f"rec_{channel_id}"
    stop_url = f"{BASE_URL}/recordings/live/{recording_name}/stop"
    requests.post(stop_url, auth=AUTH)

def on_message(wsapp, message):
    event = json.loads(message)
    event_type = event.get("type")

    # Сценарий 1: Звонок поступил -> Отвечаем и воспроизводим beep
    if event_type == "StasisStart":
        channel_id = event.get("channel", {}).get("id")
        print(f"Отвечаем на звонок {channel_id}")
        requests.post(f"{BASE_URL}/channels/{channel_id}/answer", auth=AUTH)

        channel_caller_numbers[channel_id]=get_caller_number(channel_id)
        channel_contexts[channel_id] = load_context(channel_caller_numbers[channel_id])

        #start_recording(channel_id)
        # or
        detect_speech(channel_id)

    # Сценарий N
    elif event_type == "ChannelTalkingStarted":
        channel_id = event.get("channel", {}).get("id")
        print(f"Обнаружена речь в канале {channel_id}! Начинаем запись...")
        start_recording(channel_id)

    # Сценарий N
    elif event_type == "ChannelTalkingFinished":
        channel_id = event.get("channel", {}).get("id")
        print(f"Обнаружено молчание в канале {channel_id}")
        stop_recording(channel_id)
        undetect_speech(channel_id)

    # Сценарий N: Запись завершилась -> Проигрываем её обратно
    elif event_type == "RecordingFinished":
        recording_data = event.get("recording", {})
        recording_name = recording_data.get("name")
        target_uri = recording_data.get("target_uri", "")
        channel_id = target_uri.split("channel:")[1]

        import wav_and_ogg
        import speech_and_text
        wav_and_ogg.wav_to_ogg(f"{REC_PATH}{recording_name}")

        user_request = speech_and_text.speech_to_text(f"{REC_PATH}{recording_name}.ogg")
        print(f"user_request: {user_request}")

        if user_request:
            context = get_channel_context(channel_id)

            import request_to_agent
            agent_response = request_to_agent.request_to_agent(user_request, SYSTEM_PROMPT, context)
            #agent_response = user_request

            print(agent_response)

            update_channel_context(channel_id, user_request, agent_response)

            print("Текущий контекст:")
            from pprint import pprint
            pprint(channel_contexts[channel_id])

            speech_and_text.text_to_speech(agent_response,f"{REC_PATH}{recording_name}.response.ogg")
            wav_and_ogg.ogg_to_wav(f"{REC_PATH}{recording_name}.response")

            print(f"Запись {recording_name} готова! Проигрываем её обратно в канал {channel_id}...")
            play_url = f"{BASE_URL}/channels/{channel_id}/play"
            play_payload = {
                #"media": f"recording:{recording_name}"
                "media": f"recording:{recording_name}.response"
            }
            response = requests.post(play_url, params=play_payload, auth=AUTH)
            print("Ответ ARI на воспроизведение:", response.status_code)

    # Сценарий N: Воспроизведение завершено (бип или записанный голос)
    elif event_type == "PlaybackFinished":
        playback_data = event.get("playback", {})
        media_uri = playback_data.get("media_uri", "")
        target_uri = playback_data.get('target_uri', '')
        channel_id = target_uri.split("channel:")[1]

        #start_recording(channel_id)
        # or
        detect_speech(channel_id)

    # Сценарий N: Абонент повесил трубку -> Удаляем файл
    elif event_type == "StasisEnd":
        channel_id = event.get("channel", {}).get("id")

        #print(f"Звонок {channel_id} завершен абонентом. Удаляем файл записи...")
        #recording_name = f"rec_{channel_id}"
        #delete_url = f"{BASE_URL}/recordings/stored/{recording_name}"
        #requests.delete(delete_url, auth=AUTH)

        context = get_channel_context(channel_id)
        caller_number = channel_caller_numbers[channel_id]
        print(f"Номер звонящего в StasisEnd: номер {caller_number} канал {channel_id}")

        print(f"Сжатие контекста")

        import request_to_agent
        agent_response = request_to_agent.request_to_agent("Сожми контекст", COMPRESS_PROMPT, context)
        print(agent_response)
        context = [{"role": "user", "text": agent_response}]

        print(f"Сохранение контекста и удаление медиа файлов")

        save_context(caller_number,context)

        import glob
        recording_name = f"rec_{channel_id}"
        file_mask = f"{REC_PATH}{recording_name}*"
        matched_files = glob.glob(file_mask)
        for file_path in matched_files:
            os.remove(file_path)
            print(f"Файл удален: {file_path}")

def on_error(ws, error):
    print(f"Произошла ошибка: {error}")

wsapp = websocket.WebSocketApp(
    f"ws://localhost:8088/ari/events?api_key={ARI_USER}:{ARI_PASS}&app=my-first-app",
    on_message=on_message,
    on_error=on_error
)
wsapp.run_forever()
использование_ai-агентов.1784261220.txt.gz · Last modified: 2026/07/17 07:07 by val