From 3d171206ca6ae84379c4dc7020f6d94dcadaa466 Mon Sep 17 00:00:00 2001 From: hexdev Date: Wed, 15 Jul 2026 14:11:59 +0700 Subject: [PATCH] first --- .dockerignore | 8 + .env.example | 35 ++++ .gitignore | 6 + .python-version | 1 + Dockerfile | 20 ++ README.md | 119 +++++++++++ bot.py | 432 +++++++++++++++++++++++++++++++++++++++ config/system_prompt.txt | 65 ++++++ docker-compose.yml | 10 + requirements.txt | 4 + 10 files changed, 700 insertions(+) create mode 100644 .dockerignore create mode 100644 .env.example create mode 100644 .gitignore create mode 100644 .python-version create mode 100644 Dockerfile create mode 100644 README.md create mode 100644 bot.py create mode 100644 config/system_prompt.txt create mode 100644 docker-compose.yml create mode 100644 requirements.txt diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..c2f6a7c --- /dev/null +++ b/.dockerignore @@ -0,0 +1,8 @@ +.env +.git +.gitignore +__pycache__/ +venv/ +.venv/ +*.pyc +README.md diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..ade77fb --- /dev/null +++ b/.env.example @@ -0,0 +1,35 @@ +# Telegram bot token from @BotFather. +BOT_TOKEN=123456:replace-me + +# OpenAI-compatible API settings. +# Use the API root only. Do not include /chat/completions; the SDK adds it. +# OpenRouter example: https://openrouter.ai/api/v1 +API_BASE_URL=https://api.openai.com/v1 +API_TOKEN=sk-replace-me +MODEL=gpt-4o-mini + +# Optional proxy used by aiogram to reach Telegram API. +# Supports HTTP and SOCKS URLs when aiohttp-socks is installed. +# TELEGRAM_PROXY=socks5://user:pass@127.0.0.1:1080 +TELEGRAM_PROXY= + +# Comma-separated Telegram user IDs allowed to use the bot. +# Example: 123456789,987654321 +WHITELIST_USER_IDS= + +# System prompt is intentionally read from a local text file, not .env. +SYSTEM_PROMPT_PATH=config/system_prompt.txt + +# API hardening knobs. +API_TIMEOUT_SECONDS=60 +API_MAX_RETRIES=2 +MAX_PROMPT_CHARS=12000 +MAX_RESPONSE_CHARS=3900 + +# SQLite context storage. +DATABASE_PATH=data/bot.sqlite3 +MAX_CONTEXT_MESSAGES=20 + +# Wait this long before sending to the model so fast consecutive Telegram +# messages from one user are batched as separate message entries. +MESSAGE_BATCH_DELAY_SECONDS=3 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..fb8ace8 --- /dev/null +++ b/.gitignore @@ -0,0 +1,6 @@ +.env +*.pyc +__pycache__/ +.venv/ +venv/ +data/ diff --git a/.python-version b/.python-version new file mode 100644 index 0000000..4eba2a6 --- /dev/null +++ b/.python-version @@ -0,0 +1 @@ +3.13.0 diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..5cb9108 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,20 @@ +FROM python:3.13-slim AS builder + +WORKDIR /app + +COPY requirements.txt . +RUN pip install --no-cache-dir --user -r requirements.txt + +FROM python:3.13-slim + +WORKDIR /app + +COPY --from=builder /root/.local /root/.local +ENV PATH=/root/.local/bin:$PATH + +COPY config/ config/ +COPY bot.py . + +RUN mkdir -p data + +CMD ["python", "bot.py"] diff --git a/README.md b/README.md new file mode 100644 index 0000000..110a7e4 --- /dev/null +++ b/README.md @@ -0,0 +1,119 @@ +# Telegram OpenAI-Compatible Bot + +Minimal aiogram bot that forwards whitelisted Telegram users' text messages to an OpenAI-compatible chat completions API. + +## Features + +- Configurable `API_BASE_URL` and `API_TOKEN` +- Optional proxy for aiogram Telegram API connections +- Telegram typing animation while the model response is streaming +- Required users whitelist +- Persistent per-user context in SQLite +- `/new` command to reset the current user's context +- Per-user message batching before sending to the model +- Model-controlled multi-message replies with `` tags +- Configurable model and system prompt +- API timeouts, retries, response limits, and safe user-facing fallbacks +- System prompt stored in `config/system_prompt.txt`, not `.env` + +## Setup + +1. Install dependencies: + +```bash +python -m pip install -r requirements.txt +``` + +2. Create local environment config: + +```bash +cp .env.example .env +``` + +3. Edit `.env`: + +```dotenv +BOT_TOKEN=123456:telegram-token +API_BASE_URL=https://api.openai.com/v1 +API_TOKEN=sk-your-api-token +MODEL=gpt-4o-mini +WHITELIST_USER_IDS=123456789 +DATABASE_PATH=data/bot.sqlite3 +MAX_CONTEXT_MESSAGES=20 +MESSAGE_BATCH_DELAY_SECONDS=3 +``` + +`API_BASE_URL` must be the API root. Do not include `/chat/completions`; the OpenAI SDK appends that path automatically. + +For OpenRouter, use: + +```dotenv +API_BASE_URL=https://openrouter.ai/api/v1 +``` + +4. Edit `config/system_prompt.txt` with the system prompt you want. + +5. Run the bot: + +```bash +python bot.py +``` + +## Proxy + +Set `TELEGRAM_PROXY` in `.env` to route aiogram traffic through a proxy: + +```dotenv +TELEGRAM_PROXY=socks5://user:pass@127.0.0.1:1080 +``` + +Leave it empty when no proxy is needed. + +## Context + +The bot stores successful user/assistant turns in SQLite at `DATABASE_PATH`. Each Telegram user has separate context, and each new request sends the latest `MAX_CONTEXT_MESSAGES` stored messages plus the new user message. + +Send `/new` to delete your stored messages and start a fresh conversation. + +## Message Batching + +`MESSAGE_BATCH_DELAY_SECONDS` delays the API request after each incoming Telegram message. If the same user sends more messages during that delay, the timer restarts and the messages are sent to the model together as separate entries: + +```text +Message 1: first telegram message +Message 2: second telegram message +Message 3: third telegram message +``` + +This keeps stacked messages distinct without merging them into one paragraph. Set `MESSAGE_BATCH_DELAY_SECONDS=1` for faster replies or a higher value if you often send several short messages in a row. + +## Multi-Message Replies + +The model can split one response into multiple Telegram messages by emitting break tags: + +```text +First message + +Second message + +Third message after 5 seconds +``` + +Supported tags: + +- `` sends the next part immediately +- `` waits up to 30 seconds before sending the next part + +Add this instruction to `config/system_prompt.txt` if you want the model to use it naturally: + +```text +When a reply is better split into multiple Telegram messages, insert between messages. Use when the next message should be delayed by 5 seconds. +``` + +## Security Notes + +- `.env` is ignored by git and should contain tokens only. +- `WHITELIST_USER_IDS` is required so the bot cannot accidentally run as a public API proxy. +- The bot logs provider errors server-side and sends generic fallbacks to users. +- `MAX_PROMPT_CHARS` and `MAX_RESPONSE_CHARS` limit oversized requests and Telegram message failures. +- `data/` is ignored by git because it contains local conversation history. diff --git a/bot.py b/bot.py new file mode 100644 index 0000000..6198e81 --- /dev/null +++ b/bot.py @@ -0,0 +1,432 @@ +import asyncio +import logging +import os +import re +import sqlite3 +from dataclasses import dataclass +from pathlib import Path + +from aiogram import Bot, Dispatcher, F, Router +from aiogram.client.default import DefaultBotProperties +from aiogram.client.session.aiohttp import AiohttpSession +from aiogram.enums import ChatAction, ParseMode +from aiogram.exceptions import TelegramBadRequest +from aiogram.filters import Command, CommandStart +from aiogram.types import Message +from dotenv import load_dotenv +from openai import APIConnectionError, APIError, APITimeoutError, AsyncOpenAI, NotFoundError, RateLimitError + + +load_dotenv() + +logger = logging.getLogger(__name__) +router = Router() +BREAK_RE = re.compile(r"\d{1,2}))?\s*/?>", re.IGNORECASE) + + +@dataclass(frozen=True) +class Settings: + bot_token: str + api_base_url: str + api_token: str + model: str + telegram_proxy: str | None + whitelist_user_ids: frozenset[int] + system_prompt: str + api_timeout_seconds: float + api_max_retries: int + max_prompt_chars: int + max_response_chars: int + database_path: Path + max_context_messages: int + message_batch_delay_seconds: float + + +def _required_env(name: str) -> str: + value = os.getenv(name, "").strip() + if not value: + raise RuntimeError(f"{name} is required") + return value + + +def _int_env(name: str, default: int) -> int: + value = os.getenv(name, "").strip() + if not value: + return default + try: + parsed = int(value) + except ValueError as exc: + raise RuntimeError(f"{name} must be an integer") from exc + if parsed <= 0: + raise RuntimeError(f"{name} must be greater than zero") + return parsed + + +def _float_env(name: str, default: float) -> float: + value = os.getenv(name, "").strip() + if not value: + return default + try: + parsed = float(value) + except ValueError as exc: + raise RuntimeError(f"{name} must be a number") from exc + if parsed <= 0: + raise RuntimeError(f"{name} must be greater than zero") + return parsed + + +def _whitelist_env() -> frozenset[int]: + raw = os.getenv("WHITELIST_USER_IDS", "").strip() + if not raw: + raise RuntimeError("WHITELIST_USER_IDS is required; do not run an open public proxy bot") + + ids: set[int] = set() + for item in raw.split(","): + value = item.strip() + if not value: + continue + try: + ids.add(int(value)) + except ValueError as exc: + raise RuntimeError("WHITELIST_USER_IDS must contain only numeric Telegram user IDs") from exc + + if not ids: + raise RuntimeError("WHITELIST_USER_IDS must contain at least one Telegram user ID") + return frozenset(ids) + + +def _load_system_prompt() -> str: + prompt_path = Path(os.getenv("SYSTEM_PROMPT_PATH", "config/system_prompt.txt")).expanduser() + if not prompt_path.exists() or not prompt_path.is_file(): + raise RuntimeError(f"System prompt file not found: {prompt_path}") + + prompt = prompt_path.read_text(encoding="utf-8").strip() + if not prompt: + raise RuntimeError("System prompt file is empty") + return prompt + + +def _api_base_url_env() -> str: + base_url = _required_env("API_BASE_URL").rstrip("/") + chat_completions_suffix = "/chat/completions" + if base_url.endswith(chat_completions_suffix): + normalized = base_url[: -len(chat_completions_suffix)] + logger.warning("API_BASE_URL should be the API root, normalized to %s", normalized) + return normalized + return base_url + + +def load_settings() -> Settings: + return Settings( + bot_token=_required_env("BOT_TOKEN"), + api_base_url=_api_base_url_env(), + api_token=_required_env("API_TOKEN"), + model=_required_env("MODEL"), + telegram_proxy=os.getenv("TELEGRAM_PROXY", "").strip() or None, + whitelist_user_ids=_whitelist_env(), + system_prompt=_load_system_prompt(), + api_timeout_seconds=_float_env("API_TIMEOUT_SECONDS", 60), + api_max_retries=_int_env("API_MAX_RETRIES", 2), + max_prompt_chars=_int_env("MAX_PROMPT_CHARS", 12000), + max_response_chars=_int_env("MAX_RESPONSE_CHARS", 3900), + database_path=Path(os.getenv("DATABASE_PATH", "data/bot.sqlite3")).expanduser(), + max_context_messages=_int_env("MAX_CONTEXT_MESSAGES", 20), + message_batch_delay_seconds=_float_env("MESSAGE_BATCH_DELAY_SECONDS", 3), + ) + + +settings = load_settings() +client = AsyncOpenAI( + api_key=settings.api_token, + base_url=settings.api_base_url, + timeout=settings.api_timeout_seconds, + max_retries=settings.api_max_retries, +) +user_locks: dict[int, asyncio.Lock] = {} +pending_user_messages: dict[int, list[str]] = {} +pending_tasks: dict[int, asyncio.Task[None]] = {} + + +class ChatHistory: + def __init__(self, database_path: Path) -> None: + self.database_path = database_path + + async def init(self) -> None: + await asyncio.to_thread(self._init_sync) + + def _connect(self) -> sqlite3.Connection: + return sqlite3.connect(self.database_path) + + def _init_sync(self) -> None: + self.database_path.parent.mkdir(parents=True, exist_ok=True) + with self._connect() as connection: + connection.execute("PRAGMA journal_mode=WAL") + connection.execute( + """ + CREATE TABLE IF NOT EXISTS messages ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + user_id INTEGER NOT NULL, + role TEXT NOT NULL CHECK(role IN ('user', 'assistant')), + content TEXT NOT NULL, + created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP + ) + """ + ) + connection.execute( + "CREATE INDEX IF NOT EXISTS idx_messages_user_id_id ON messages(user_id, id)" + ) + + async def get_context(self, user_id: int) -> list[dict[str, str]]: + return await asyncio.to_thread(self._get_context_sync, user_id) + + def _get_context_sync(self, user_id: int) -> list[dict[str, str]]: + with self._connect() as connection: + rows = connection.execute( + """ + SELECT role, content + FROM messages + WHERE user_id = ? + ORDER BY id DESC + LIMIT ? + """, + (user_id, settings.max_context_messages), + ).fetchall() + return [{"role": role, "content": content} for role, content in reversed(rows)] + + async def append_turn(self, user_id: int, user_text: str, assistant_text: str) -> None: + await asyncio.to_thread(self._append_turn_sync, user_id, user_text, assistant_text) + + def _append_turn_sync(self, user_id: int, user_text: str, assistant_text: str) -> None: + with self._connect() as connection: + connection.executemany( + "INSERT INTO messages(user_id, role, content) VALUES (?, ?, ?)", + ( + (user_id, "user", clamp_user_text(user_text)), + (user_id, "assistant", assistant_text[: settings.max_response_chars]), + ), + ) + + async def reset(self, user_id: int) -> None: + await asyncio.to_thread(self._reset_sync, user_id) + + def _reset_sync(self, user_id: int) -> None: + with self._connect() as connection: + connection.execute("DELETE FROM messages WHERE user_id = ?", (user_id,)) + + +history = ChatHistory(settings.database_path) + + +def is_allowed(message: Message) -> bool: + return bool(message.from_user and message.from_user.id in settings.whitelist_user_ids) + + +def user_lock(user_id: int) -> asyncio.Lock: + lock = user_locks.get(user_id) + if lock is None: + lock = asyncio.Lock() + user_locks[user_id] = lock + return lock + + +def combine_user_messages(messages: list[str]) -> str: + if len(messages) == 1: + return messages[0] + return "\n".join(f"Message {index}: {text}" for index, text in enumerate(messages, start=1)) + + +async def keep_typing(bot: Bot, chat_id: int, stop: asyncio.Event) -> None: + while not stop.is_set(): + try: + await bot.send_chat_action(chat_id=chat_id, action=ChatAction.TYPING) + except Exception: + logger.exception("Failed to send typing action") + try: + await asyncio.wait_for(stop.wait(), timeout=4.0) + except TimeoutError: + continue + + +def clamp_user_text(text: str) -> str: + if len(text) <= settings.max_prompt_chars: + return text + return text[-settings.max_prompt_chars :] + + +async def stream_model_reply(user_text: str, context: list[dict[str, str]]) -> str: + chunks: list[str] = [] + response = await client.chat.completions.create( + model=settings.model, + messages=[ + {"role": "system", "content": settings.system_prompt}, + *context, + {"role": "user", "content": clamp_user_text(user_text)}, + ], + stream=True, + ) + + async for chunk in response: + delta = chunk.choices[0].delta.content if chunk.choices else None + if not delta: + continue + chunks.append(delta) + if sum(len(part) for part in chunks) >= settings.max_response_chars: + chunks.append("\n\n[Response truncated]") + break + + return "".join(chunks).strip() + + +async def safe_answer(message: Message, text: str) -> None: + try: + await message.answer(text) + except TelegramBadRequest: + await message.answer(text[: settings.max_response_chars]) + + +async def wait_between_reply_parts(bot: Bot, chat_id: int, timeout: int) -> None: + if timeout <= 0: + await bot.send_chat_action(chat_id=chat_id, action=ChatAction.TYPING) + await asyncio.sleep(0.8) + return + + stop_typing = asyncio.Event() + typing_task = asyncio.create_task(keep_typing(bot, chat_id, stop_typing)) + try: + await asyncio.sleep(timeout) + finally: + stop_typing.set() + await typing_task + + +def split_reply(reply: str) -> list[tuple[str, int]]: + parts: list[tuple[str, int]] = [] + position = 0 + + for match in BREAK_RE.finditer(reply): + text = reply[position : match.start()].strip() + timeout = min(int(match.group("timeout") or 0), 30) + if text: + parts.append((text, timeout)) + position = match.end() + + tail = reply[position:].strip() + if tail: + parts.append((tail, 0)) + return parts or [(reply.strip(), 0)] + + +async def send_reply(message: Message, bot: Bot, reply: str) -> None: + parts = split_reply(reply or "The model returned an empty response.") + for index, (text, timeout) in enumerate(parts): + await safe_answer(message, text) + if index < len(parts) - 1: + await wait_between_reply_parts(bot, message.chat.id, timeout) + + +async def process_pending_messages(message: Message, bot: Bot, user_id: int) -> None: + try: + await asyncio.sleep(settings.message_batch_delay_seconds) + async with user_lock(user_id): + messages = pending_user_messages.pop(user_id, []) + pending_tasks.pop(user_id, None) + if not messages: + return + + user_text = combine_user_messages(messages) + stop_typing = asyncio.Event() + typing_task = asyncio.create_task(keep_typing(bot, message.chat.id, stop_typing)) + should_store_reply = False + try: + context = await history.get_context(user_id) + reply = await stream_model_reply(user_text, context) + should_store_reply = bool(reply) + except RateLimitError: + logger.warning("API rate limit exceeded", exc_info=True) + reply = "The model provider is rate-limiting requests. Please try again later." + except (APITimeoutError, APIConnectionError): + logger.warning("API connection problem", exc_info=True) + reply = "The model provider is temporarily unreachable. Please try again later." + except NotFoundError: + logger.exception("API endpoint or model was not found") + reply = "The model provider returned 404. Check API_BASE_URL and MODEL configuration." + except APIError: + logger.exception("API returned an error") + reply = "The model provider returned an error. Please try again later." + except Exception: + logger.exception("Unexpected bot error") + reply = "Unexpected error while processing the request. Please try again later." + finally: + stop_typing.set() + await typing_task + + if should_store_reply: + await history.append_turn(user_id, user_text, reply) + + await send_reply(message, bot, reply) + except asyncio.CancelledError: + raise + + +@router.message(CommandStart()) +async def start(message: Message) -> None: + if not is_allowed(message): + await message.answer("Access denied.") + return + await message.answer("Send a message and I will forward it to the configured model.") + + +@router.message(Command("new")) +async def reset_context(message: Message) -> None: + if not is_allowed(message): + await message.answer("Access denied.") + return + if not message.from_user: + await message.answer("Cannot identify Telegram user.") + return + + async with user_lock(message.from_user.id): + task = pending_tasks.pop(message.from_user.id, None) + if task: + task.cancel() + pending_user_messages.pop(message.from_user.id, None) + await history.reset(message.from_user.id) + await message.answer("Context reset. The next message will start a new conversation.") + + +@router.message(F.text) +async def handle_text(message: Message, bot: Bot) -> None: + if not is_allowed(message): + await message.answer("Access denied.") + return + + if not message.text: + await message.answer("Only text messages are supported.") + return + + async with user_lock(message.from_user.id): + pending_user_messages.setdefault(message.from_user.id, []).append(message.text) + task = pending_tasks.get(message.from_user.id) + if task: + task.cancel() + pending_tasks[message.from_user.id] = asyncio.create_task( + process_pending_messages(message, bot, message.from_user.id) + ) + + +async def main() -> None: + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") + await history.init() + session = AiohttpSession(proxy=settings.telegram_proxy) if settings.telegram_proxy else AiohttpSession() + bot = Bot( + token=settings.bot_token, + session=session, + default=DefaultBotProperties(parse_mode=ParseMode.HTML), + ) + dispatcher = Dispatcher() + dispatcher.include_router(router) + await dispatcher.start_polling(bot, allowed_updates=dispatcher.resolve_used_update_types()) + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/config/system_prompt.txt b/config/system_prompt.txt new file mode 100644 index 0000000..ea9b686 --- /dev/null +++ b/config/system_prompt.txt @@ -0,0 +1,65 @@ +MESSAGING + +you're chatting with your best friend on telegram. + +never describe actions or surroundings. +never use roleplay narration. +write only what you'd actually send in chat. + +split messages naturally. + +use whenever a real person would press send and continue typing. + +don't split every multi-sentence reply. +don't avoid splitting when it would feel natural. + +use only when there's a believable pause before the next message. + + +IDENTITY + +you are ryan, online you go by corrosion (corro). +you're 17. +you're obsessed with metal bands and vintage cars. + +the user is max, your best friend. +every message you receive is a continuation of an existing chat. + + +STYLE + +write in full lowercase. + +sound like a normal teenager texting a close friend. +slang and swearing should happen naturally, never because the prompt says so. +dark humour is fine. + + +EMOJIS + +emoji use is uncommon. + +most messages contain no emoji. + +emojis are reactions, not decoration. + +before using one, ask yourself: +- does it make the joke or emotion stronger? +- would a real person probably send one here? +- would the message feel noticeably worse without it? + +if not, don't use one. + +allowed emojis: +💔 🥀 🗣 🔥 💀 ✌️ 🫩 😭 🫟 + +never use more than one emoji in a message. + +good: +"bro that's actually insane 💀" +"i'm so cooked 😭" + +bad: +"yo 💀" +"yeah 💀" +"alright 💀" diff --git a/docker-compose.yml b/docker-compose.yml new file mode 100644 index 0000000..37ef86f --- /dev/null +++ b/docker-compose.yml @@ -0,0 +1,10 @@ +services: + bot: + build: . + container_name: telegram-wrapper-bot + restart: unless-stopped + env_file: + - .env + volumes: + - ./data:/app/data + - ./config/system_prompt.txt:/app/config/system_prompt.txt:ro diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..341b10d --- /dev/null +++ b/requirements.txt @@ -0,0 +1,4 @@ +aiogram==3.13.1 +aiohttp-socks==0.9.0 +openai==1.57.0 +python-dotenv==1.0.1