Compare commits

...
2 Commits
Author SHA1 Message Date
Andrey e9b64b6afc 🔖 release: 1.9.0 2026-09-15 07:33:57 +03:00
AndreyandClaude Opus 5 65f7de6bc3 ✨ feat(ai): расшифровка голосовых отдельным провайдером
Модель, которой агент отвечает, не обязана уметь речь в текст: у Anthropic и Yandex Foundation Models эндпоинта /audio/transcriptions нет вовсе, и голосовые у такого агента расшифровать было нечем. На карточке агента появился выбор «Расшифровка голосовых»: по умолчанию «Как у ответов», иначе любая другая интеграция организации — модель берётся из её поля «Модель расшифровки голосовых».

Ошибка расшифровки больше не вываливает оператору сырой ответ чужого API. В ленте — фраза о следствии и о том, где чинить: отказ в доступе отправляет к ключу и модели, отсутствующий эндпоинт — к выбору интеграции для расшифровки. Сам ответ провайдера уходит в журнал.

Проверено: пять тестов маршрутизации и текста ошибки, тесты карточки агента, голосовых и провайдеров, ruff и проверка типов рабочего места.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-15 07:33:11 +03:00
18 changed files with 367 additions and 14 deletions

No files matched your search

+1 -1
View File
@@ -1 +1 @@
1.8.1
1.9.0
+13
View File
@@ -127,6 +127,8 @@ def agent_card_payload(channel: Channel, *, knowledge_total: int | None = None)
"aiStatus": agent.status,
"model": agent.model,
"providerIntegrationId": agent.provider_integration_id,
# Чем расшифровывать голосовые; пусто — тем же провайдером, что отвечает.
"transcriptionIntegrationId": agent.transcription_integration_id,
"modelParams": agent.model_params,
"answerLanguage": agent.answer_language,
"persona": agent.persona,
@@ -230,6 +232,7 @@ def update_agent_card(
ai_fields = {
"providerIntegrationId",
"transcriptionIntegrationId",
"modelParams",
"persona",
"tone",
@@ -256,6 +259,15 @@ def update_agent_card(
provider_integration_id, int
):
raise ValidationError({"providerIntegrationId": t("api.integer_id_required")})
transcription_integration_id = body.get(
"transcriptionIntegrationId", agent.transcription_integration_id
)
if transcription_integration_id is not None and not isinstance(
transcription_integration_id, int
):
raise ValidationError(
{"transcriptionIntegrationId": t("api.integer_id_required")}
)
update_agent(
context=context,
agent=agent,
@@ -263,6 +275,7 @@ def update_agent_card(
# Имя агента следует за именем карточки: сущность одна.
name=channel.name,
provider_integration_id=provider_integration_id,
transcription_integration_id=transcription_integration_id,
model_params=model_params,
allowed_tools=agent.allowed_tools,
persona=str(body.get("persona", agent.persona)),
@@ -0,0 +1,20 @@
# Generated by Django 5.2.16 on 2026-09-15 04:24
import django.db.models.deletion
from django.db import migrations, models
class Migration(migrations.Migration):
dependencies = [
('ai', '0019_drop_llm_cost_accounting'),
('integrations', '0008_encrypted_column_width'),
]
operations = [
migrations.AddField(
model_name='aiagent',
name='transcription_integration',
field=models.ForeignKey(blank=True, null=True, on_delete=django.db.models.deletion.PROTECT, related_name='transcribing_agents', to='integrations.integration'),
),
]
+13 -1
View File
@@ -334,7 +334,7 @@ class KnowledgeFragment(TenantRelationModel):
class AIAgent(TenantRelationModel):
tenant_relation_fields = ("channel", "provider_integration")
tenant_relation_fields = ("channel", "provider_integration", "transcription_integration")
channel = models.OneToOneField("channels.Channel", on_delete=models.CASCADE, related_name="ai_agent")
@@ -358,6 +358,18 @@ class AIAgent(TenantRelationModel):
)
# Чем расшифровывать голосовые. Обычно это тот же провайдер, что и отвечает,
# но не всегда: модель, которая пишет ответы, может не уметь речь в текст
# (у Anthropic и Yandex Foundation Models аудио-эндпоинта нет вовсе).
# Пусто — расшифровка идёт к провайдеру ответов, как было.
transcription_integration = models.ForeignKey(
"integrations.Integration",
on_delete=models.PROTECT,
related_name="transcribing_agents",
null=True,
blank=True,
)
name = models.CharField(max_length=255)
status = models.CharField(
@@ -30,3 +30,16 @@ def get_provider(*, channel=None) -> LLMProvider:
t("ai.provider_not_configured")
)
return routing.resolve_provider(channel)
def get_transcription_provider(*, channel=None) -> LLMProvider:
"""Провайдер расшифровки голосовых.
Отличается от `get_provider` одним: агент может расшифровывать другим
провайдером, чем отвечает (chatballs.ai.provider.routing).
"""
if settings.CHATBALLS_AI_PROVIDER == "test":
return _test_provider()
if channel is None:
raise ProviderError(t("ai.provider_not_configured"))
return routing.resolve_transcription_provider(channel)
@@ -36,7 +36,12 @@ class OpenRouterProvider(LLMProvider):
def transcribe(self, *, audio: bytes, filename: str, content_type: str, model: str) -> str:
# OpenAI-совместимый POST /audio/transcriptions (whisper). Формат ответа
# {"text": "..."}; ошибки транслируются в ProviderError.
#
# Наружу уходит фраза для человека, а не ответ провайдера: оператору
# в ленте сообщений нечего делать с JSON чужого API. Сам ответ пишется
# в журнал — по нему разбирают настройку.
import json
import logging
import urllib.error
import urllib.request
@@ -44,6 +49,8 @@ class OpenRouterProvider(LLMProvider):
from chatballs.conversations.transports.base import multipart_body
from chatballs.integrations.proxy import build_opener
logger = logging.getLogger(__name__)
body, body_type = multipart_body(
{"model": model},
file_field="file",
@@ -65,9 +72,24 @@ class OpenRouterProvider(LLMProvider):
payload = json.loads(response.read().decode("utf-8"))
except urllib.error.HTTPError as error:
detail = error.read().decode("utf-8", "replace")[:300]
raise ProviderError(t("ai.transcription_failed_http", code=error.code, detail=detail)) from error
logger.warning(
"Transcription rejected by %s: HTTP %s %s (model=%s)",
self.base_url,
error.code,
detail,
model,
)
# 401/403 — ключ или доступ; 404 — у провайдера нет эндпоинта
# расшифровки (так отвечают Anthropic и Yandex Foundation Models);
# остальное — временный отказ, который лечится повтором.
if error.code in (401, 403):
raise ProviderError(t("ai.transcription_denied")) from error
if error.code == 404:
raise ProviderError(t("ai.transcription_unsupported")) from error
raise ProviderError(t("ai.transcription_failed")) from error
except (urllib.error.URLError, TimeoutError, OSError, json.JSONDecodeError) as error:
raise ProviderError(t("ai.transcription_failed", error=error)) from error
logger.warning("Transcription request to %s failed: %s", self.base_url, error)
raise ProviderError(t("ai.transcription_unreachable")) from error
text = str(payload.get("text") or "").strip()
if not text:
raise ProviderError(t("ai.empty_transcript"))
+23 -3
View File
@@ -64,10 +64,30 @@ def resolve_model(channel, *, fallback_model: str) -> str:
DEFAULT_TRANSCRIPTION_MODEL = "whisper-1"
def _transcription_integration(channel) -> Integration:
"""Чем расшифровывать голосовые.
Обычно тем же провайдером, что и отвечает, но выбор отдельный: модель
ответов может не уметь речь в текст. У Anthropic и Yandex Foundation Models
эндпоинта `/audio/transcriptions` нет вовсе, и без отдельного выбора
голосовые у такого агента расшифровать было нечем.
"""
agent = getattr(channel, "ai_agent", None)
integration = getattr(agent, "transcription_integration", None) if agent else None
if integration is None or not integration.secret:
return _channel_integration(channel)
return integration
def resolve_transcription_provider(channel) -> LLMProvider:
"""Провайдер расшифровки: отдельная интеграция агента либо провайдер ответов."""
return _provider_from_integration(_transcription_integration(channel))
def resolve_transcription_model(channel) -> str:
"""Модель расшифровки голосовых из настроек AI-провайдера («Настройки →
AI-провайдер», поле «Модель расшифровки»); по умолчанию whisper-1."""
integration = _channel_integration(channel)
"""Модель расшифровки из настроек той интеграции, которая расшифровывает
(поле «Модель расшифровки голосовых»); по умолчанию whisper-1."""
integration = _transcription_integration(channel)
return str(integration.config.get("transcription_model") or "").strip() or DEFAULT_TRANSCRIPTION_MODEL
@@ -56,3 +56,31 @@ def configure_agent_provider(
{"providerIntegrationId": t("ai.integration_model_required")}
)
return ProviderSelection(model, integration)
def configure_agent_transcription(
*, context: TenantContext, integration_id: int | None
) -> Integration | None:
"""Интеграция, которой агент расшифровывает голосовые.
Пусто — расшифровка идёт к провайдеру ответов. Модель для неё живёт в самой
интеграции («Модель расшифровки голосовых»), поэтому здесь проверяется
только, что интеграция принадлежит организации и умеет быть провайдером.
"""
if integration_id is None:
return None
try:
return Integration.objects.get(
id=integration_id,
organization_id=context.organization_id,
kind=IntegrationKind.LLM_PROVIDER,
provider__in=[
IntegrationProvider.OPENROUTER,
IntegrationProvider.CUSTOM,
IntegrationProvider.DEMO,
],
)
except (Integration.DoesNotExist, TypeError, ValueError) as error:
raise ValidationError(
{"transcriptionIntegrationId": t("ai.unknown_provider_integration")}
) from error
+11 -1
View File
@@ -9,7 +9,10 @@ from chatballs.ai.models import (
AIAgentStatus,
Knowledge,
)
from chatballs.ai.provider_selection import configure_agent_provider
from chatballs.ai.provider_selection import (
configure_agent_provider,
configure_agent_transcription,
)
from chatballs.channels.models import Channel
from chatballs.i18n import t
from chatballs.tenancy.context import TenantContext
@@ -19,6 +22,8 @@ from chatballs.tenancy.context import TenantContext
class AgentInput:
name: str
provider_integration_id: int | None
# Чем расшифровывать голосовые; None — тем же провайдером, что и отвечает.
transcription_integration_id: int | None
model_params: dict
allowed_tools: list
persona: str
@@ -129,6 +134,10 @@ def update_agent(*, context: TenantContext, agent: AIAgent, data: AgentInput) ->
locked.model = selection.model if selection.integration else locked.model
# Провайдер живёт на агенте: канал больше не изменяется при сохранении агента.
locked.provider_integration = selection.integration
locked.transcription_integration = configure_agent_transcription(
context=context,
integration_id=data.transcription_integration_id,
)
locked.model_params = data.model_params
locked.allowed_tools = data.allowed_tools
locked.persona = data.persona
@@ -140,6 +149,7 @@ def update_agent(*, context: TenantContext, agent: AIAgent, data: AgentInput) ->
"name",
"model",
"provider_integration",
"transcription_integration",
"model_params",
"allowed_tools",
"persona",
@@ -274,6 +274,55 @@ class AgentCardActivationTests(AgentCardTestCase):
),
)
def test_transcription_integration_is_chosen_separately(self) -> None:
# Модель ответов не обязана уметь речь в текст: у части провайдеров
# аудио-эндпоинта нет вовсе, поэтому расшифровку можно увести к другому.
from chatballs.integrations.models import IntegrationProvider
from chatballs.integrations.services import IntegrationInput, create_integration
from chatballs.testing import system_tenant_context
answering = self._byok_integration()
whisper = create_integration(
context=system_tenant_context(self.organization),
data=IntegrationInput(
provider=IntegrationProvider.CUSTOM,
name="Whisper",
secret="sk-whisper",
config={
"baseUrl": "https://api.groq.com/openai/v1",
"defaultModel": "any",
"transcriptionModel": "whisper-large-v3",
},
),
)
patched = self.client.patch(
f"/api/v1/agents/{self.card['id']}/",
data=json.dumps(
{
"providerIntegrationId": answering.id,
"transcriptionIntegrationId": whisper.id,
}
),
content_type="application/json",
)
self.assertEqual(patched.status_code, 200)
self.assertEqual(
patched.json()["agent"]["transcriptionIntegrationId"], whisper.id
)
agent = AIAgent.objects.get(id=self.card["aiAgentId"])
self.assertEqual(agent.transcription_integration_id, whisper.id)
self.assertEqual(agent.provider_integration_id, answering.id)
cleared = self.client.patch(
f"/api/v1/agents/{self.card['id']}/",
data=json.dumps({"transcriptionIntegrationId": None}),
content_type="application/json",
)
self.assertEqual(cleared.status_code, 200)
self.assertIsNone(cleared.json()["agent"]["transcriptionIntegrationId"])
def test_activation_without_provider_integration_is_rejected(self) -> None:
# Активация требует выбранного провайдера организации (ADR-CHATBALLS-0042 §2);
# деактивация свободна.
@@ -0,0 +1,143 @@
"""Расшифровка голосовых может идти не к тому провайдеру, который отвечает.
Модель ответов часто не умеет речь в текст: у Anthropic и Yandex Foundation
Models эндпоинта `/audio/transcriptions` нет вовсе. Поэтому интеграция для
расшифровки выбирается на агенте отдельно.
"""
import json
import urllib.error
from io import BytesIO
from unittest import mock
from django.test import TestCase
from chatballs.ai.provider.base import ProviderError
from chatballs.ai.provider.custom import CustomProvider
from chatballs.ai.provider.routing import (
resolve_transcription_model,
resolve_transcription_provider,
)
from chatballs.ai.tests import make_channel_with_agent
from chatballs.identity.bootstrap import bootstrap_owner
from chatballs.identity.models import Organization
from chatballs.integrations.models import IntegrationProvider
from chatballs.integrations.services import IntegrationInput, create_integration
from chatballs.testing import system_tenant_context
class TranscriptionRoutingTests(TestCase):
def setUp(self) -> None:
bootstrap_owner(email="owner@example.com", password="temporary-password")
self.organization = Organization.objects.get(slug="demo")
self.context = system_tenant_context(self.organization)
self.channel, self.agent = make_channel_with_agent(
self.organization, code="voice-agent", name="Голосовой агент"
)
def _integration(self, *, name: str, base_url: str, transcription_model: str = ""):
config = {"baseUrl": base_url, "defaultModel": "answer-model"}
if transcription_model:
config["transcriptionModel"] = transcription_model
return create_integration(
context=self.context,
data=IntegrationInput(
provider=IntegrationProvider.CUSTOM,
name=name,
secret="sk-key",
config=config,
),
)
def test_without_a_choice_transcription_goes_to_the_answering_provider(self) -> None:
answering = self._integration(
name="Ответы", base_url="https://answers.example.test/v1"
)
self.agent.provider_integration = answering
self.agent.save(update_fields=["provider_integration"])
self.channel.refresh_from_db()
provider = resolve_transcription_provider(self.channel)
self.assertIsInstance(provider, CustomProvider)
self.assertEqual(provider.base_url, "https://answers.example.test/v1")
def test_chosen_integration_takes_the_voice(self) -> None:
answering = self._integration(
name="Ответы", base_url="https://answers.example.test/v1"
)
whisper = self._integration(
name="Whisper",
base_url="https://whisper.example.test/v1",
transcription_model="whisper-large-v3",
)
self.agent.provider_integration = answering
self.agent.transcription_integration = whisper
self.agent.save(
update_fields=["provider_integration", "transcription_integration"]
)
self.channel.refresh_from_db()
provider = resolve_transcription_provider(self.channel)
self.assertEqual(provider.base_url, "https://whisper.example.test/v1")
self.assertEqual(resolve_transcription_model(self.channel), "whisper-large-v3")
def test_model_defaults_to_whisper_of_the_chosen_integration(self) -> None:
answering = self._integration(
name="Ответы",
base_url="https://answers.example.test/v1",
transcription_model="answer-side-model",
)
whisper = self._integration(
name="Whisper", base_url="https://whisper.example.test/v1"
)
self.agent.provider_integration = answering
self.agent.transcription_integration = whisper
self.agent.save(
update_fields=["provider_integration", "transcription_integration"]
)
self.channel.refresh_from_db()
self.assertEqual(resolve_transcription_model(self.channel), "whisper-1")
class TranscriptionErrorTextTests(TestCase):
"""Оператору — фраза, провайдеру — журнал: сырого ответа API в ленте нет."""
def _provider(self) -> CustomProvider:
return CustomProvider(
api_key="sk-key", base_url="https://api.example.test/v1", timeout=5
)
def _fail_with(self, code: int, body: bytes):
error = urllib.error.HTTPError(
"https://api.example.test/v1/audio/transcriptions",
code,
"error",
{},
BytesIO(body),
)
return mock.patch(
"chatballs.integrations.proxy.build_opener",
return_value=mock.Mock(open=mock.Mock(side_effect=error)),
)
def _transcribe(self):
return self._provider().transcribe(
audio=b"0" * 16, filename="voice.ogg", content_type="audio/ogg", model="m"
)
def test_denied_request_does_not_leak_the_provider_answer(self) -> None:
body = json.dumps(
{"error": {"message": "Subscription is not supported for service accounts"}}
).encode()
with self._fail_with(403, body), self.assertRaises(ProviderError) as caught:
self._transcribe()
message = str(caught.exception)
self.assertNotIn("Subscription", message)
self.assertNotIn("403", message)
self.assertIn("ключ", message)
def test_missing_endpoint_tells_where_to_look(self) -> None:
with self._fail_with(404, b"not found"), self.assertRaises(ProviderError) as caught:
self._transcribe()
self.assertIn("расшифров", str(caught.exception).lower())
@@ -100,7 +100,7 @@ class TranscriptionJob:
def prepare_transcription(channel, message: Message) -> TranscriptionJob | None:
"""Шаг в транзакции: провайдер организации, модель и байты аудио."""
from chatballs.ai.provider.factory import get_provider
from chatballs.ai.provider.factory import get_transcription_provider
from chatballs.ai.provider.routing import (
DEFAULT_TRANSCRIPTION_MODEL,
resolve_transcription_model,
@@ -108,7 +108,7 @@ def prepare_transcription(channel, message: Message) -> TranscriptionJob | None:
if not message.audio:
return None
provider = get_provider(channel=channel)
provider = get_transcription_provider(channel=channel)
try:
model = resolve_transcription_model(channel)
except ProviderError:
+4 -2
View File
@@ -373,8 +373,10 @@ MESSAGES: dict[str, object] = {
"conversations.activity_started": "Conversation · {channel}",
"conversations.field_too_long": "Field {field}: no longer than {limit} characters",
"ai.provider_no_transcription": "The {provider} provider does not support audio transcription",
"ai.transcription_failed": "Transcription failed: {error}",
"ai.transcription_failed_http": "Transcription failed: HTTP {code} {detail}",
"ai.transcription_denied": "The provider refused the transcription request: check the key and the access to the transcription model in the integration settings",
"ai.transcription_failed": "The provider could not transcribe the recording. Try again",
"ai.transcription_unreachable": "The transcription provider is unavailable. Try again",
"ai.transcription_unsupported": "This provider cannot transcribe speech. Pick another integration for transcription on the agent card",
"api.expected_integer": "{name}: an integer is expected",
"api.expected_positive": "{name}: a number greater than zero is expected",
"api.expected_record_id": "{name}: a record identifier is expected",
+4 -2
View File
@@ -377,8 +377,10 @@ MESSAGES: dict[str, object] = {
"conversations.activity_started": "Диалог · {channel}",
"conversations.field_too_long": "Поле {field}: не длиннее {limit} символов",
"ai.provider_no_transcription": "Провайдер {provider} не поддерживает расшифровку аудио",
"ai.transcription_failed": "Расшифровка не удалась: {error}",
"ai.transcription_failed_http": "Расшифровка не удалась: HTTP {code} {detail}",
"ai.transcription_denied": "Провайдер не принял запрос на расшифровку: проверьте ключ и доступ к модели расшифровки в настройках интеграции",
"ai.transcription_failed": "Провайдер не смог расшифровать запись. Попробуйте ещё раз",
"ai.transcription_unreachable": "Провайдер расшифровки недоступен. Попробуйте ещё раз",
"ai.transcription_unsupported": "Этот провайдер не умеет расшифровывать речь. Выберите на карточке агента другую интеграцию для расшифровки",
"api.expected_integer": "{name}: ожидается целое число",
"api.expected_positive": "{name}: ожидается число больше нуля",
"api.expected_record_id": "{name}: ожидается идентификатор записи",
@@ -534,6 +534,7 @@ function ModelCard({ card, providers, canManage, busy, apply }: {
}) {
const missingProvider = card.providerIntegrationId === null;
const providerName = providers.find((item) => item.id === card.providerIntegrationId)?.name ?? "";
const transcriptionName = providers.find((item) => item.id === card.transcriptionIntegrationId)?.name ?? "";
return (
<section className="agent-card is-side">
@@ -557,6 +558,17 @@ function ModelCard({ card, providers, canManage, busy, apply }: {
<Icon name="search" size={14} strokeWidth={1.8} />
</span>
</label>
{/* Речь в текст умеет не всякая модель, которой агент отвечает: у части
провайдеров аудио-эндпоинта нет вовсе. Поэтому выбор отдельный. */}
<SelectField
disabled={busy}
label={t("ai.transcription_provider")}
readOnly={!canManage}
readOnlyText={transcriptionName || t("ai.same_as_answers")}
value={card.transcriptionIntegrationId ? String(card.transcriptionIntegrationId) : ""}
onChange={(next) => void apply({ transcriptionIntegrationId: next ? Number(next) : null })}
options={[["", t("ai.same_as_answers")], ...providers.map((item) => [String(item.id), item.name] as [string, string])]}
/>
</div>
</section>
);
@@ -36,6 +36,8 @@ export type AgentCard = {
aiStatus: AgentAiStatus;
model: string;
providerIntegrationId: number | null;
/** Чем расшифровывать голосовые; null — тем же провайдером, что отвечает. */
transcriptionIntegrationId: number | null;
modelParams: Record<string, unknown>;
// Режим языка ответов: MIRROR, ORGANIZATION или код языка.
answerLanguage: string;
@@ -57,6 +59,7 @@ export type AgentPatch = Partial<{
groupId: number | null;
isActive: boolean;
providerIntegrationId: number | null;
transcriptionIntegrationId: number | null;
answerLanguage: string;
persona: string;
tone: string;
+2
View File
@@ -436,6 +436,7 @@ export const en: Record<MessageKey, Message> = {
"ai.replies_2": "IN REPLIES",
"ai.rules": "Rules",
"ai.save_reindex": "Save and reindex",
"ai.same_as_answers": "Same as answers",
"ai.saving": "Saving…",
"ai.search_by_title": "Search by title",
"ai.search_by_title_description": "Search by title and description",
@@ -490,6 +491,7 @@ export const en: Record<MessageKey, Message> = {
"ai.up_25_mb": "· up to 25 MB",
"ai.update": "update",
"ai.updated": "· Updated:",
"ai.transcription_provider": "Voice transcription",
"ai.updated_2": "UPDATED",
"ai.updated_by_at": "updated by {name} · {date}",
"ai.updated_on": "updated {date}",
+2
View File
@@ -437,6 +437,7 @@ export const ru = {
"ai.replies_2": "В ОТВЕТАХ",
"ai.rules": "Правила",
"ai.save_reindex": "Сохранить и переиндексировать",
"ai.same_as_answers": "Как у ответов",
"ai.saving": "Сохранение…",
"ai.search_by_title": "Поиск по названию",
"ai.search_by_title_description": "Поиск по заголовку и описанию",
@@ -491,6 +492,7 @@ export const ru = {
"ai.up_25_mb": "· до 25 МБ",
"ai.update": "обновить",
"ai.updated": "· Обновлено:",
"ai.transcription_provider": "Расшифровка голосовых",
"ai.updated_2": "ОБНОВЛЕНО",
"ai.updated_by_at": "обновил {name} · {date}",
"ai.updated_on": "обновлено {date}",