feat: audio transcribing
This commit is contained in:
@@ -7,6 +7,7 @@ BOT_TOKEN=123456:replace-me
|
|||||||
API_BASE_URL=https://api.openai.com/v1
|
API_BASE_URL=https://api.openai.com/v1
|
||||||
API_TOKEN=sk-replace-me
|
API_TOKEN=sk-replace-me
|
||||||
MODEL=gpt-4o-mini
|
MODEL=gpt-4o-mini
|
||||||
|
TRANSCRIPTION_MODEL=whisper-1
|
||||||
|
|
||||||
# Optional proxy used for both Telegram API and OpenAI-compatible API.
|
# Optional proxy used for both Telegram API and OpenAI-compatible API.
|
||||||
# Supports HTTP and SOCKS URLs when aiohttp-socks is installed.
|
# Supports HTTP and SOCKS URLs when aiohttp-socks is installed.
|
||||||
|
|||||||
12
README.md
12
README.md
@@ -1,6 +1,6 @@
|
|||||||
# Telegram OpenAI-Compatible Bot
|
# Telegram OpenAI-Compatible Bot
|
||||||
|
|
||||||
Minimal aiogram bot that forwards whitelisted Telegram users' text messages to an OpenAI-compatible chat completions API.
|
Minimal aiogram bot that forwards whitelisted Telegram users' text and voice messages to an OpenAI-compatible API.
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
@@ -11,6 +11,7 @@ Minimal aiogram bot that forwards whitelisted Telegram users' text messages to a
|
|||||||
- Persistent per-user context in SQLite
|
- Persistent per-user context in SQLite
|
||||||
- `/new` command to reset the current user's context
|
- `/new` command to reset the current user's context
|
||||||
- Per-user message batching before sending to the model
|
- Per-user message batching before sending to the model
|
||||||
|
- Voice message transcription through a separately configured transcription model
|
||||||
- Model-controlled multi-message replies with `<break>` tags
|
- Model-controlled multi-message replies with `<break>` tags
|
||||||
- Configurable model and system prompt
|
- Configurable model and system prompt
|
||||||
- API timeouts, retries, response limits, and safe user-facing fallbacks
|
- API timeouts, retries, response limits, and safe user-facing fallbacks
|
||||||
@@ -37,6 +38,7 @@ BOT_TOKEN=123456:telegram-token
|
|||||||
API_BASE_URL=https://api.openai.com/v1
|
API_BASE_URL=https://api.openai.com/v1
|
||||||
API_TOKEN=sk-your-api-token
|
API_TOKEN=sk-your-api-token
|
||||||
MODEL=gpt-4o-mini
|
MODEL=gpt-4o-mini
|
||||||
|
TRANSCRIPTION_MODEL=whisper-1
|
||||||
WHITELIST_USER_IDS=123456789
|
WHITELIST_USER_IDS=123456789
|
||||||
DATABASE_PATH=data/bot.sqlite3
|
DATABASE_PATH=data/bot.sqlite3
|
||||||
MAX_CONTEXT_MESSAGES=20
|
MAX_CONTEXT_MESSAGES=20
|
||||||
@@ -45,6 +47,8 @@ MESSAGE_BATCH_DELAY_SECONDS=3
|
|||||||
|
|
||||||
`API_BASE_URL` must be the API root. Do not include `/chat/completions`; the OpenAI SDK appends that path automatically.
|
`API_BASE_URL` must be the API root. Do not include `/chat/completions`; the OpenAI SDK appends that path automatically.
|
||||||
|
|
||||||
|
`TRANSCRIPTION_MODEL` is used for Telegram voice message transcription before the transcript is sent to `MODEL`.
|
||||||
|
|
||||||
For OpenRouter, use:
|
For OpenRouter, use:
|
||||||
|
|
||||||
```dotenv
|
```dotenv
|
||||||
@@ -75,6 +79,12 @@ The bot stores successful user/assistant turns in SQLite at `DATABASE_PATH`. Eac
|
|||||||
|
|
||||||
Send `/new` to delete your stored messages and start a fresh conversation.
|
Send `/new` to delete your stored messages and start a fresh conversation.
|
||||||
|
|
||||||
|
## Voice Messages
|
||||||
|
|
||||||
|
Telegram voice notes are downloaded, transcribed with `TRANSCRIPTION_MODEL`, then the transcript is passed through the same batching and chat-completions flow as text messages.
|
||||||
|
|
||||||
|
Telegram does not expose a bot API to mark a voice note as listened in the user interface, so that step is not supported.
|
||||||
|
|
||||||
## Message Batching
|
## Message Batching
|
||||||
|
|
||||||
`MESSAGE_BATCH_DELAY_SECONDS` delays the API request after each incoming Telegram message. If the same user sends more messages during that delay, the timer restarts and the messages are sent to the model together as separate entries:
|
`MESSAGE_BATCH_DELAY_SECONDS` delays the API request after each incoming Telegram message. If the same user sends more messages during that delay, the timer restarts and the messages are sent to the model together as separate entries:
|
||||||
|
|||||||
100
bot.py
100
bot.py
@@ -1,4 +1,5 @@
|
|||||||
import asyncio
|
import asyncio
|
||||||
|
import io
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
@@ -31,6 +32,7 @@ class Settings:
|
|||||||
api_base_url: str
|
api_base_url: str
|
||||||
api_token: str
|
api_token: str
|
||||||
model: str
|
model: str
|
||||||
|
transcription_model: str
|
||||||
telegram_proxy: str | None
|
telegram_proxy: str | None
|
||||||
whitelist_user_ids: frozenset[int]
|
whitelist_user_ids: frozenset[int]
|
||||||
system_prompt: str
|
system_prompt: str
|
||||||
@@ -123,6 +125,7 @@ def load_settings() -> Settings:
|
|||||||
api_base_url=_api_base_url_env(),
|
api_base_url=_api_base_url_env(),
|
||||||
api_token=_required_env("API_TOKEN"),
|
api_token=_required_env("API_TOKEN"),
|
||||||
model=_required_env("MODEL"),
|
model=_required_env("MODEL"),
|
||||||
|
transcription_model=_required_env("TRANSCRIPTION_MODEL"),
|
||||||
telegram_proxy=os.getenv("TELEGRAM_PROXY", "").strip() or None,
|
telegram_proxy=os.getenv("TELEGRAM_PROXY", "").strip() or None,
|
||||||
whitelist_user_ids=_whitelist_env(),
|
whitelist_user_ids=_whitelist_env(),
|
||||||
system_prompt=_load_system_prompt(),
|
system_prompt=_load_system_prompt(),
|
||||||
@@ -238,6 +241,10 @@ def combine_user_messages(messages: list[str]) -> str:
|
|||||||
return "\n".join(f"Message {index}: {text}" for index, text in enumerate(messages, start=1))
|
return "\n".join(f"Message {index}: {text}" for index, text in enumerate(messages, start=1))
|
||||||
|
|
||||||
|
|
||||||
|
def voice_transcript_text(transcript: str) -> str:
|
||||||
|
return f"[Voice message transcript]\n{transcript.strip()}"
|
||||||
|
|
||||||
|
|
||||||
async def keep_typing(bot: Bot, chat_id: int, stop: asyncio.Event) -> None:
|
async def keep_typing(bot: Bot, chat_id: int, stop: asyncio.Event) -> None:
|
||||||
while not stop.is_set():
|
while not stop.is_set():
|
||||||
try:
|
try:
|
||||||
@@ -280,6 +287,31 @@ async def stream_model_reply(user_text: str, context: list[dict[str, str]]) -> s
|
|||||||
return "".join(chunks).strip()
|
return "".join(chunks).strip()
|
||||||
|
|
||||||
|
|
||||||
|
async def download_voice_message(bot: Bot, message: Message) -> tuple[bytes, str]:
|
||||||
|
if not message.voice:
|
||||||
|
raise ValueError("Message does not contain a voice attachment")
|
||||||
|
|
||||||
|
file = await bot.get_file(message.voice.file_id)
|
||||||
|
if not file.file_path:
|
||||||
|
raise RuntimeError("Telegram did not return a file path for the voice message")
|
||||||
|
|
||||||
|
buffer = io.BytesIO()
|
||||||
|
await bot.download_file(file.file_path, destination=buffer)
|
||||||
|
return buffer.getvalue(), Path(file.file_path).name
|
||||||
|
|
||||||
|
|
||||||
|
async def transcribe_voice_message(bot: Bot, message: Message) -> str:
|
||||||
|
voice_bytes, filename = await download_voice_message(bot, message)
|
||||||
|
transcription = await client.audio.transcriptions.create(
|
||||||
|
model=settings.transcription_model,
|
||||||
|
file=(filename, voice_bytes, "audio/ogg"),
|
||||||
|
)
|
||||||
|
transcript = transcription.text.strip()
|
||||||
|
if not transcript:
|
||||||
|
raise RuntimeError("Voice message transcription was empty")
|
||||||
|
return transcript
|
||||||
|
|
||||||
|
|
||||||
async def safe_answer(message: Message, text: str) -> None:
|
async def safe_answer(message: Message, text: str) -> None:
|
||||||
try:
|
try:
|
||||||
await message.answer(text)
|
await message.answer(text)
|
||||||
@@ -371,12 +403,27 @@ async def process_pending_messages(message: Message, bot: Bot, user_id: int) ->
|
|||||||
raise
|
raise
|
||||||
|
|
||||||
|
|
||||||
|
async def queue_user_message(message: Message, bot: Bot, user_text: str) -> None:
|
||||||
|
if not message.from_user:
|
||||||
|
await message.answer("Cannot identify Telegram user.")
|
||||||
|
return
|
||||||
|
|
||||||
|
async with user_lock(message.from_user.id):
|
||||||
|
pending_user_messages.setdefault(message.from_user.id, []).append(user_text)
|
||||||
|
task = pending_tasks.get(message.from_user.id)
|
||||||
|
if task:
|
||||||
|
task.cancel()
|
||||||
|
pending_tasks[message.from_user.id] = asyncio.create_task(
|
||||||
|
process_pending_messages(message, bot, message.from_user.id)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@router.message(CommandStart())
|
@router.message(CommandStart())
|
||||||
async def start(message: Message) -> None:
|
async def start(message: Message) -> None:
|
||||||
if not is_allowed(message):
|
if not is_allowed(message):
|
||||||
await message.answer("Access denied.")
|
await message.answer("Access denied.")
|
||||||
return
|
return
|
||||||
await message.answer("Send a message and I will forward it to the configured model.")
|
await message.answer("Send a text or voice message and I will forward it to the configured model.")
|
||||||
|
|
||||||
|
|
||||||
@router.message(Command("new"))
|
@router.message(Command("new"))
|
||||||
@@ -407,14 +454,51 @@ async def handle_text(message: Message, bot: Bot) -> None:
|
|||||||
await message.answer("Only text messages are supported.")
|
await message.answer("Only text messages are supported.")
|
||||||
return
|
return
|
||||||
|
|
||||||
async with user_lock(message.from_user.id):
|
await queue_user_message(message, bot, message.text)
|
||||||
pending_user_messages.setdefault(message.from_user.id, []).append(message.text)
|
|
||||||
task = pending_tasks.get(message.from_user.id)
|
|
||||||
if task:
|
@router.message(F.voice)
|
||||||
task.cancel()
|
async def handle_voice(message: Message, bot: Bot) -> None:
|
||||||
pending_tasks[message.from_user.id] = asyncio.create_task(
|
if not is_allowed(message):
|
||||||
process_pending_messages(message, bot, message.from_user.id)
|
await message.answer("Access denied.")
|
||||||
|
return
|
||||||
|
|
||||||
|
if not message.voice:
|
||||||
|
await message.answer("Only Telegram voice messages are supported.")
|
||||||
|
return
|
||||||
|
|
||||||
|
logger.info("Transcribing voice message for user_id=%s", message.from_user.id)
|
||||||
|
|
||||||
|
try:
|
||||||
|
transcript = await transcribe_voice_message(bot, message)
|
||||||
|
except RateLimitError:
|
||||||
|
logger.warning("Transcription API rate limit exceeded", exc_info=True)
|
||||||
|
await message.answer("The transcription model provider is rate-limiting requests. Please try again later.")
|
||||||
|
return
|
||||||
|
except (APITimeoutError, APIConnectionError):
|
||||||
|
logger.warning("Transcription API connection problem", exc_info=True)
|
||||||
|
await message.answer("The transcription model provider is temporarily unreachable. Please try again later.")
|
||||||
|
return
|
||||||
|
except NotFoundError:
|
||||||
|
logger.exception("Transcription endpoint or model was not found")
|
||||||
|
await message.answer(
|
||||||
|
"The transcription model provider returned 404. Check API_BASE_URL and TRANSCRIPTION_MODEL configuration."
|
||||||
)
|
)
|
||||||
|
return
|
||||||
|
except APIError:
|
||||||
|
logger.exception("Transcription API returned an error")
|
||||||
|
await message.answer("The transcription model provider returned an error. Please try again later.")
|
||||||
|
return
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Unexpected voice transcription error")
|
||||||
|
await message.answer("Unexpected error while transcribing the voice message. Please try again later.")
|
||||||
|
return
|
||||||
|
|
||||||
|
transcript_message = voice_transcript_text(transcript)
|
||||||
|
logger.info("Voice message transcribed for user_id=%s", message.from_user.id)
|
||||||
|
|
||||||
|
# Telegram bots cannot mark a voice note as listened in the client UI.
|
||||||
|
await queue_user_message(message, bot, transcript_message)
|
||||||
|
|
||||||
|
|
||||||
async def main() -> None:
|
async def main() -> None:
|
||||||
|
|||||||
Reference in New Issue
Block a user