first commit
This commit is contained in:
14
.dockerignore
Normal file
14
.dockerignore
Normal file
@@ -0,0 +1,14 @@
|
|||||||
|
.env
|
||||||
|
.env.*
|
||||||
|
!.env.example
|
||||||
|
.git
|
||||||
|
__pycache__/
|
||||||
|
*.pyc
|
||||||
|
*.pyo
|
||||||
|
*.pyd
|
||||||
|
.pytest_cache/
|
||||||
|
.mypy_cache/
|
||||||
|
.ruff_cache/
|
||||||
|
venv/
|
||||||
|
data/
|
||||||
|
kwork_resp.txt
|
||||||
10
.env.example
Normal file
10
.env.example
Normal file
@@ -0,0 +1,10 @@
|
|||||||
|
TELEGRAM_BOT_TOKEN=your_bot_token
|
||||||
|
TELEGRAM_CHANNEL_ID=@your_channel_or_chat_id
|
||||||
|
TELEGRAM_PROXY_URL=
|
||||||
|
SCHEDULER_INTERVAL_MINUTES=5
|
||||||
|
KWORK_CATEGORY_IDS=41
|
||||||
|
KEYWORDS_INCLUDE=python,parser,api
|
||||||
|
KEYWORDS_EXCLUDE=wordpress,figma
|
||||||
|
MIN_PRICE_AMOUNT=
|
||||||
|
REQUESTS_TIMEOUT_SECONDS=20
|
||||||
|
STATE_FILE=data/sent_offers.json
|
||||||
179
.gitignore
vendored
Normal file
179
.gitignore
vendored
Normal file
@@ -0,0 +1,179 @@
|
|||||||
|
# ---> Python
|
||||||
|
# Byte-compiled / optimized / DLL files
|
||||||
|
__pycache__/
|
||||||
|
*.py[cod]
|
||||||
|
*$py.class
|
||||||
|
|
||||||
|
# C extensions
|
||||||
|
*.so
|
||||||
|
|
||||||
|
# Distribution / packaging
|
||||||
|
.Python
|
||||||
|
build/
|
||||||
|
develop-eggs/
|
||||||
|
dist/
|
||||||
|
downloads/
|
||||||
|
eggs/
|
||||||
|
.eggs/
|
||||||
|
lib/
|
||||||
|
lib64/
|
||||||
|
parts/
|
||||||
|
sdist/
|
||||||
|
var/
|
||||||
|
wheels/
|
||||||
|
share/python-wheels/
|
||||||
|
*.egg-info/
|
||||||
|
.installed.cfg
|
||||||
|
*.egg
|
||||||
|
MANIFEST
|
||||||
|
|
||||||
|
# PyInstaller
|
||||||
|
# Usually these files are written by a python script from a template
|
||||||
|
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
||||||
|
*.manifest
|
||||||
|
*.spec
|
||||||
|
|
||||||
|
# Installer logs
|
||||||
|
pip-log.txt
|
||||||
|
pip-delete-this-directory.txt
|
||||||
|
|
||||||
|
# Unit test / coverage reports
|
||||||
|
htmlcov/
|
||||||
|
.tox/
|
||||||
|
.nox/
|
||||||
|
.coverage
|
||||||
|
.coverage.*
|
||||||
|
.cache
|
||||||
|
nosetests.xml
|
||||||
|
coverage.xml
|
||||||
|
*.cover
|
||||||
|
*.py,cover
|
||||||
|
.hypothesis/
|
||||||
|
.pytest_cache/
|
||||||
|
cover/
|
||||||
|
|
||||||
|
# Translations
|
||||||
|
*.mo
|
||||||
|
*.pot
|
||||||
|
|
||||||
|
# Django stuff:
|
||||||
|
*.log
|
||||||
|
local_settings.py
|
||||||
|
db.sqlite3
|
||||||
|
db.sqlite3-journal
|
||||||
|
|
||||||
|
# Flask stuff:
|
||||||
|
instance/
|
||||||
|
.webassets-cache
|
||||||
|
|
||||||
|
# Scrapy stuff:
|
||||||
|
.scrapy
|
||||||
|
|
||||||
|
# Sphinx documentation
|
||||||
|
docs/_build/
|
||||||
|
|
||||||
|
# PyBuilder
|
||||||
|
.pybuilder/
|
||||||
|
target/
|
||||||
|
|
||||||
|
# Jupyter Notebook
|
||||||
|
.ipynb_checkpoints
|
||||||
|
|
||||||
|
# IPython
|
||||||
|
profile_default/
|
||||||
|
ipython_config.py
|
||||||
|
|
||||||
|
# pyenv
|
||||||
|
# For a library or package, you might want to ignore these files since the code is
|
||||||
|
# intended to run in multiple environments; otherwise, check them in:
|
||||||
|
# .python-version
|
||||||
|
|
||||||
|
# pipenv
|
||||||
|
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
||||||
|
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
||||||
|
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
||||||
|
# install all needed dependencies.
|
||||||
|
#Pipfile.lock
|
||||||
|
|
||||||
|
# UV
|
||||||
|
# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
|
||||||
|
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
||||||
|
# commonly ignored for libraries.
|
||||||
|
#uv.lock
|
||||||
|
|
||||||
|
# poetry
|
||||||
|
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
||||||
|
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
||||||
|
# commonly ignored for libraries.
|
||||||
|
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
||||||
|
#poetry.lock
|
||||||
|
|
||||||
|
# pdm
|
||||||
|
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
||||||
|
#pdm.lock
|
||||||
|
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
|
||||||
|
# in version control.
|
||||||
|
# https://pdm.fming.dev/latest/usage/project/#working-with-version-control
|
||||||
|
.pdm.toml
|
||||||
|
.pdm-python
|
||||||
|
.pdm-build/
|
||||||
|
|
||||||
|
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
||||||
|
__pypackages__/
|
||||||
|
|
||||||
|
# Celery stuff
|
||||||
|
celerybeat-schedule
|
||||||
|
celerybeat.pid
|
||||||
|
|
||||||
|
# SageMath parsed files
|
||||||
|
*.sage.py
|
||||||
|
|
||||||
|
# Environments
|
||||||
|
.env
|
||||||
|
*.env
|
||||||
|
dev.env
|
||||||
|
.venv
|
||||||
|
env/
|
||||||
|
venv/
|
||||||
|
ENV/
|
||||||
|
env.bak/
|
||||||
|
venv.bak/
|
||||||
|
|
||||||
|
# Spyder project settings
|
||||||
|
.spyderproject
|
||||||
|
.spyproject
|
||||||
|
|
||||||
|
# Rope project settings
|
||||||
|
.ropeproject
|
||||||
|
|
||||||
|
# mkdocs documentation
|
||||||
|
/site
|
||||||
|
|
||||||
|
# mypy
|
||||||
|
.mypy_cache/
|
||||||
|
.dmypy.json
|
||||||
|
dmypy.json
|
||||||
|
|
||||||
|
# Pyre type checker
|
||||||
|
.pyre/
|
||||||
|
|
||||||
|
# pytype static type analyzer
|
||||||
|
.pytype/
|
||||||
|
|
||||||
|
# Cython debug symbols
|
||||||
|
cython_debug/
|
||||||
|
|
||||||
|
# PyCharm
|
||||||
|
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
|
||||||
|
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
|
||||||
|
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
||||||
|
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
||||||
|
#.idea/
|
||||||
|
|
||||||
|
# Ruff stuff:
|
||||||
|
.ruff_cache/
|
||||||
|
|
||||||
|
# PyPI configuration file
|
||||||
|
.pypirc
|
||||||
|
|
||||||
|
plans
|
||||||
32
Dockerfile
Normal file
32
Dockerfile
Normal file
@@ -0,0 +1,32 @@
|
|||||||
|
FROM python:3.14-slim AS builder
|
||||||
|
|
||||||
|
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||||
|
PYTHONUNBUFFERED=1 \
|
||||||
|
VIRTUAL_ENV=/opt/venv
|
||||||
|
|
||||||
|
WORKDIR /app
|
||||||
|
|
||||||
|
RUN python -m venv "$VIRTUAL_ENV"
|
||||||
|
ENV PATH="$VIRTUAL_ENV/bin:$PATH"
|
||||||
|
|
||||||
|
COPY requirements.txt .
|
||||||
|
RUN pip install --upgrade pip && pip install --no-cache-dir -r requirements.txt
|
||||||
|
|
||||||
|
|
||||||
|
FROM python:3.14-slim AS runtime
|
||||||
|
|
||||||
|
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||||
|
PYTHONUNBUFFERED=1 \
|
||||||
|
VIRTUAL_ENV=/opt/venv \
|
||||||
|
PATH="/opt/venv/bin:$PATH"
|
||||||
|
|
||||||
|
WORKDIR /app
|
||||||
|
|
||||||
|
COPY --from=builder /opt/venv /opt/venv
|
||||||
|
COPY app ./app
|
||||||
|
COPY main.py ./
|
||||||
|
COPY requirements.txt ./
|
||||||
|
|
||||||
|
RUN mkdir -p /app/data
|
||||||
|
|
||||||
|
CMD ["python", "main.py"]
|
||||||
70
README.md
Normal file
70
README.md
Normal file
@@ -0,0 +1,70 @@
|
|||||||
|
# my_parser
|
||||||
|
|
||||||
|
Парсер офферов с `kwork.ru` с отправкой в Telegram и возможностью добавлять новые источники.
|
||||||
|
|
||||||
|
## Что умеет
|
||||||
|
|
||||||
|
- запускается каждые `X` минут через `APScheduler`
|
||||||
|
- читает настройки из `.env`
|
||||||
|
- забирает проекты `kwork.ru` по списку категорий `fc`
|
||||||
|
- фильтрует офферы по ключевым словам
|
||||||
|
- фильтрует офферы по минимальной сумме
|
||||||
|
- не отправляет уже отправленные офферы повторно
|
||||||
|
- отправляет релевантные офферы в Telegram-канал через `aiogram`
|
||||||
|
|
||||||
|
## Запуск
|
||||||
|
|
||||||
|
1. Создать и заполнить `.env` по примеру `.env.example`.
|
||||||
|
2. Установить зависимости:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install -r requirements.txt
|
||||||
|
```
|
||||||
|
|
||||||
|
3. Запустить:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python main.py
|
||||||
|
```
|
||||||
|
|
||||||
|
Дополнительно можно задать `MIN_PRICE_AMOUNT`, чтобы пропускать только офферы с бюджетом не ниже указанного значения в рублях.
|
||||||
|
|
||||||
|
Для Telegram можно указать `TELEGRAM_PROXY_URL`, например `socks5://user:pass@host:port` или `http://host:port`.
|
||||||
|
|
||||||
|
## Docker
|
||||||
|
|
||||||
|
Сборка образа:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker build -t my-parser .
|
||||||
|
```
|
||||||
|
|
||||||
|
Запуск контейнера:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker run --rm --env-file .env -v "$(pwd)/data:/app/data" my-parser
|
||||||
|
```
|
||||||
|
|
||||||
|
В образе используется multistage-сборка:
|
||||||
|
|
||||||
|
- на этапе `builder` ставятся зависимости в отдельный `venv`
|
||||||
|
- на этапе `runtime` копируются только готовое окружение и исходники
|
||||||
|
|
||||||
|
Это ускоряет пересборку, если меняется только код приложения, а `requirements.txt` остается прежним.
|
||||||
|
|
||||||
|
## Структура
|
||||||
|
|
||||||
|
- `app/config.py` - загрузка настроек
|
||||||
|
- `app/sources/base.py` - базовый интерфейс источника
|
||||||
|
- `app/sources/kwork.py` - реализация источника `kwork`
|
||||||
|
- `app/service.py` - получение, фильтрация и отправка офферов
|
||||||
|
- `app/telegram.py` - публикация сообщений в Telegram
|
||||||
|
- `app/state.py` - хранение уже отправленных офферов
|
||||||
|
|
||||||
|
## Как добавить новый сайт
|
||||||
|
|
||||||
|
1. Создать новый класс в `app/sources/`, унаследованный от `OfferSource`.
|
||||||
|
2. Реализовать в нем `fetch_offers()` с возвратом списка `Offer`.
|
||||||
|
3. Подключить источник в `main.py`.
|
||||||
|
|
||||||
|
Для `fl.ru` можно будет добавить отдельный класс по той же схеме без изменений основного пайплайна.
|
||||||
1
app/__init__.py
Normal file
1
app/__init__.py
Normal file
@@ -0,0 +1 @@
|
|||||||
|
"""Parser application package."""
|
||||||
70
app/config.py
Normal file
70
app/config.py
Normal file
@@ -0,0 +1,70 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class Settings:
|
||||||
|
telegram_bot_token: str
|
||||||
|
telegram_channel_id: str
|
||||||
|
telegram_proxy_url: str | None
|
||||||
|
scheduler_interval_minutes: int
|
||||||
|
kwork_category_ids: list[int]
|
||||||
|
keywords_include: list[str]
|
||||||
|
keywords_exclude: list[str]
|
||||||
|
min_price_amount: int | None
|
||||||
|
requests_timeout_seconds: int
|
||||||
|
state_file: Path
|
||||||
|
|
||||||
|
|
||||||
|
def load_settings() -> Settings:
|
||||||
|
load_dotenv()
|
||||||
|
|
||||||
|
telegram_bot_token = _require_env("TELEGRAM_BOT_TOKEN")
|
||||||
|
telegram_channel_id = _require_env("TELEGRAM_CHANNEL_ID")
|
||||||
|
|
||||||
|
return Settings(
|
||||||
|
telegram_bot_token=telegram_bot_token,
|
||||||
|
telegram_channel_id=telegram_channel_id,
|
||||||
|
telegram_proxy_url=_parse_optional_text(os.getenv("TELEGRAM_PROXY_URL", "")),
|
||||||
|
scheduler_interval_minutes=int(os.getenv("SCHEDULER_INTERVAL_MINUTES", "5")),
|
||||||
|
kwork_category_ids=_parse_int_list(os.getenv("KWORK_CATEGORY_IDS", "41")),
|
||||||
|
keywords_include=_parse_text_list(os.getenv("KEYWORDS_INCLUDE", "")),
|
||||||
|
keywords_exclude=_parse_text_list(os.getenv("KEYWORDS_EXCLUDE", "")),
|
||||||
|
min_price_amount=_parse_optional_int(os.getenv("MIN_PRICE_AMOUNT", "")),
|
||||||
|
requests_timeout_seconds=int(os.getenv("REQUESTS_TIMEOUT_SECONDS", "20")),
|
||||||
|
state_file=Path(os.getenv("STATE_FILE", "data/sent_offers.json")),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _require_env(name: str) -> str:
|
||||||
|
value = os.getenv(name)
|
||||||
|
if not value:
|
||||||
|
raise ValueError(f"Environment variable {name} is required")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_text_list(raw_value: str) -> list[str]:
|
||||||
|
return [item.strip().lower() for item in raw_value.split(",") if item.strip()]
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_int_list(raw_value: str) -> list[int]:
|
||||||
|
return [int(item.strip()) for item in raw_value.split(",") if item.strip()]
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_optional_int(raw_value: str) -> int | None:
|
||||||
|
value = raw_value.strip()
|
||||||
|
if not value:
|
||||||
|
return None
|
||||||
|
return int(value)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_optional_text(raw_value: str) -> str | None:
|
||||||
|
value = raw_value.strip()
|
||||||
|
if not value:
|
||||||
|
return None
|
||||||
|
return value
|
||||||
32
app/filters.py
Normal file
32
app/filters.py
Normal file
@@ -0,0 +1,32 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from app.models import Offer
|
||||||
|
|
||||||
|
|
||||||
|
def get_offer_skip_reason(
|
||||||
|
offer: Offer,
|
||||||
|
include_keywords: list[str],
|
||||||
|
exclude_keywords: list[str],
|
||||||
|
min_price_amount: int | None,
|
||||||
|
) -> str | None:
|
||||||
|
haystack = f"{offer.title}\n{offer.description}".lower()
|
||||||
|
|
||||||
|
if exclude_keywords and any(keyword in haystack for keyword in exclude_keywords):
|
||||||
|
return "exclude_keyword"
|
||||||
|
|
||||||
|
if min_price_amount is not None and (offer.price_amount is None or offer.price_amount < min_price_amount):
|
||||||
|
return "min_price_amount"
|
||||||
|
|
||||||
|
if include_keywords and not any(keyword in haystack for keyword in include_keywords):
|
||||||
|
return "include_keyword"
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def is_relevant_offer(
|
||||||
|
offer: Offer,
|
||||||
|
include_keywords: list[str],
|
||||||
|
exclude_keywords: list[str],
|
||||||
|
min_price_amount: int | None,
|
||||||
|
) -> bool:
|
||||||
|
return get_offer_skip_reason(offer, include_keywords, exclude_keywords, min_price_amount) is None
|
||||||
21
app/models.py
Normal file
21
app/models.py
Normal file
@@ -0,0 +1,21 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from dataclasses import dataclass
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class Offer:
|
||||||
|
source: str
|
||||||
|
external_id: str
|
||||||
|
title: str
|
||||||
|
description: str
|
||||||
|
price: str
|
||||||
|
price_amount: int | None
|
||||||
|
url: str
|
||||||
|
category_id: int
|
||||||
|
published_at: str
|
||||||
|
seller_username: str | None = None
|
||||||
|
|
||||||
|
@property
|
||||||
|
def dedup_key(self) -> str:
|
||||||
|
return f"{self.source}:{self.external_id}"
|
||||||
90
app/service.py
Normal file
90
app/service.py
Normal file
@@ -0,0 +1,90 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
|
||||||
|
from app.filters import get_offer_skip_reason
|
||||||
|
from app.state import SentOffersStore
|
||||||
|
from app.telegram import TelegramPublisher
|
||||||
|
from app.sources.base import OfferSource
|
||||||
|
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class OfferProcessingService:
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
sources: list[OfferSource],
|
||||||
|
publisher: TelegramPublisher,
|
||||||
|
sent_offers_store: SentOffersStore,
|
||||||
|
include_keywords: list[str],
|
||||||
|
exclude_keywords: list[str],
|
||||||
|
min_price_amount: int | None,
|
||||||
|
) -> None:
|
||||||
|
self._sources = sources
|
||||||
|
self._publisher = publisher
|
||||||
|
self._sent_offers_store = sent_offers_store
|
||||||
|
self._include_keywords = include_keywords
|
||||||
|
self._exclude_keywords = exclude_keywords
|
||||||
|
self._min_price_amount = min_price_amount
|
||||||
|
|
||||||
|
async def run_once(self) -> None:
|
||||||
|
for source in self._sources:
|
||||||
|
try:
|
||||||
|
logger.info("Fetching offers from source=%s", source.source_name)
|
||||||
|
offers = source.fetch_offers()
|
||||||
|
logger.info("Fetched %s offers from source=%s", len(offers), source.source_name)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Failed to fetch offers from source %s", source.source_name)
|
||||||
|
continue
|
||||||
|
|
||||||
|
skipped_duplicates = 0
|
||||||
|
skipped_exclude_keyword = 0
|
||||||
|
skipped_include_keyword = 0
|
||||||
|
skipped_min_price_amount = 0
|
||||||
|
published_count = 0
|
||||||
|
|
||||||
|
for offer in offers:
|
||||||
|
if self._sent_offers_store.has(offer.dedup_key):
|
||||||
|
skipped_duplicates += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
skip_reason = get_offer_skip_reason(
|
||||||
|
offer,
|
||||||
|
self._include_keywords,
|
||||||
|
self._exclude_keywords,
|
||||||
|
self._min_price_amount,
|
||||||
|
)
|
||||||
|
if skip_reason == "exclude_keyword":
|
||||||
|
skipped_exclude_keyword += 1
|
||||||
|
continue
|
||||||
|
if skip_reason == "include_keyword":
|
||||||
|
skipped_include_keyword += 1
|
||||||
|
continue
|
||||||
|
if skip_reason == "min_price_amount":
|
||||||
|
skipped_min_price_amount += 1
|
||||||
|
continue
|
||||||
|
|
||||||
|
try:
|
||||||
|
await self._publisher.publish_offer(offer)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Failed to publish offer %s", offer.dedup_key)
|
||||||
|
continue
|
||||||
|
|
||||||
|
self._sent_offers_store.add(offer.dedup_key)
|
||||||
|
published_count += 1
|
||||||
|
logger.info("Published offer %s", offer.dedup_key)
|
||||||
|
|
||||||
|
logger.info(
|
||||||
|
(
|
||||||
|
"Source %s summary: fetched=%s published=%s skipped_duplicates=%s "
|
||||||
|
"skipped_exclude_keyword=%s skipped_include_keyword=%s skipped_min_price_amount=%s"
|
||||||
|
),
|
||||||
|
source.source_name,
|
||||||
|
len(offers),
|
||||||
|
published_count,
|
||||||
|
skipped_duplicates,
|
||||||
|
skipped_exclude_keyword,
|
||||||
|
skipped_include_keyword,
|
||||||
|
skipped_min_price_amount,
|
||||||
|
)
|
||||||
4
app/sources/__init__.py
Normal file
4
app/sources/__init__.py
Normal file
@@ -0,0 +1,4 @@
|
|||||||
|
from app.sources.base import OfferSource
|
||||||
|
from app.sources.kwork import KworkSource
|
||||||
|
|
||||||
|
__all__ = ["OfferSource", "KworkSource"]
|
||||||
13
app/sources/base.py
Normal file
13
app/sources/base.py
Normal file
@@ -0,0 +1,13 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from abc import ABC, abstractmethod
|
||||||
|
|
||||||
|
from app.models import Offer
|
||||||
|
|
||||||
|
|
||||||
|
class OfferSource(ABC):
|
||||||
|
source_name: str
|
||||||
|
|
||||||
|
@abstractmethod
|
||||||
|
def fetch_offers(self) -> list[Offer]:
|
||||||
|
raise NotImplementedError
|
||||||
111
app/sources/kwork.py
Normal file
111
app/sources/kwork.py
Normal file
@@ -0,0 +1,111 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import html
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
import requests
|
||||||
|
|
||||||
|
from app.models import Offer
|
||||||
|
from app.sources.base import OfferSource
|
||||||
|
|
||||||
|
|
||||||
|
class KworkSource(OfferSource):
|
||||||
|
source_name = "kwork"
|
||||||
|
_url = "https://kwork.ru/projects"
|
||||||
|
|
||||||
|
def __init__(self, category_ids: list[int], timeout_seconds: int) -> None:
|
||||||
|
self._category_ids = category_ids
|
||||||
|
self._timeout_seconds = timeout_seconds
|
||||||
|
self._session = requests.Session()
|
||||||
|
self._session.headers.update(
|
||||||
|
{
|
||||||
|
"User-Agent": (
|
||||||
|
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
|
||||||
|
"(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
||||||
|
),
|
||||||
|
"X-Requested-With": "XMLHttpRequest",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
def fetch_offers(self) -> list[Offer]:
|
||||||
|
offers: list[Offer] = []
|
||||||
|
|
||||||
|
for category_id in self._category_ids:
|
||||||
|
first_page_payload = self._fetch_page_payload(category_id, page=1)
|
||||||
|
offers.extend(self._parse_payload(first_page_payload, category_id))
|
||||||
|
|
||||||
|
last_page = _extract_last_page(first_page_payload)
|
||||||
|
for page in range(2, last_page + 1):
|
||||||
|
payload = self._fetch_page_payload(category_id, page=page)
|
||||||
|
offers.extend(self._parse_payload(payload, category_id))
|
||||||
|
|
||||||
|
return offers
|
||||||
|
|
||||||
|
def _fetch_page_payload(self, category_id: int, page: int) -> dict[str, Any]:
|
||||||
|
response = self._session.post(
|
||||||
|
self._url,
|
||||||
|
params={"fc": category_id, "page": page},
|
||||||
|
timeout=self._timeout_seconds,
|
||||||
|
)
|
||||||
|
response.raise_for_status()
|
||||||
|
return response.json()
|
||||||
|
|
||||||
|
def _parse_payload(self, payload: dict[str, Any], category_id: int) -> list[Offer]:
|
||||||
|
items = payload.get("data", {}).get("pagination", {}).get("data", [])
|
||||||
|
parsed_offers: list[Offer] = []
|
||||||
|
|
||||||
|
for item in items:
|
||||||
|
offer_id = item.get("id")
|
||||||
|
if offer_id is None:
|
||||||
|
continue
|
||||||
|
|
||||||
|
parsed_offers.append(
|
||||||
|
Offer(
|
||||||
|
source=self.source_name,
|
||||||
|
external_id=str(offer_id),
|
||||||
|
title=_decode_html_entities(item.get("name") or "Без названия"),
|
||||||
|
description=_decode_html_entities(item.get("description") or ""),
|
||||||
|
price=_format_price(item.get("priceLimit"), item.get("possiblePriceLimit")),
|
||||||
|
price_amount=_extract_price_amount(item.get("priceLimit"), item.get("possiblePriceLimit")),
|
||||||
|
url=f"https://kwork.ru/projects/{offer_id}/view",
|
||||||
|
category_id=category_id,
|
||||||
|
published_at=str(item.get("wantDates", {}).get("dateCreate") or ""),
|
||||||
|
seller_username=_extract_username(item.get("user") or {}),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
return parsed_offers
|
||||||
|
|
||||||
|
|
||||||
|
def _format_price(price_limit: Any, possible_price_limit: Any) -> str:
|
||||||
|
if price_limit:
|
||||||
|
return f"{price_limit} RUB"
|
||||||
|
if possible_price_limit:
|
||||||
|
return f"до {possible_price_limit} RUB"
|
||||||
|
return "Не указан"
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_price_amount(price_limit: Any, possible_price_limit: Any) -> int | None:
|
||||||
|
if price_limit:
|
||||||
|
return int(float(price_limit))
|
||||||
|
if possible_price_limit:
|
||||||
|
return int(float(possible_price_limit))
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_username(user_data: dict[str, Any]) -> str | None:
|
||||||
|
username = user_data.get("username")
|
||||||
|
if not username:
|
||||||
|
return None
|
||||||
|
return str(username)
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_last_page(payload: dict[str, Any]) -> int:
|
||||||
|
last_page = payload.get("data", {}).get("pagination", {}).get("last_page")
|
||||||
|
if last_page is None:
|
||||||
|
return 1
|
||||||
|
return max(1, int(last_page))
|
||||||
|
|
||||||
|
|
||||||
|
def _decode_html_entities(value: Any) -> str:
|
||||||
|
return html.unescape(str(value))
|
||||||
31
app/state.py
Normal file
31
app/state.py
Normal file
@@ -0,0 +1,31 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
class SentOffersStore:
|
||||||
|
def __init__(self, file_path: Path) -> None:
|
||||||
|
self._file_path = file_path
|
||||||
|
self._file_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
self._sent_keys = self._load()
|
||||||
|
|
||||||
|
def has(self, key: str) -> bool:
|
||||||
|
return key in self._sent_keys
|
||||||
|
|
||||||
|
def add(self, key: str) -> None:
|
||||||
|
self._sent_keys.add(key)
|
||||||
|
self._persist()
|
||||||
|
|
||||||
|
def _load(self) -> set[str]:
|
||||||
|
if not self._file_path.exists():
|
||||||
|
return set()
|
||||||
|
|
||||||
|
with self._file_path.open("r", encoding="utf-8") as file:
|
||||||
|
data = json.load(file)
|
||||||
|
|
||||||
|
return set(data)
|
||||||
|
|
||||||
|
def _persist(self) -> None:
|
||||||
|
with self._file_path.open("w", encoding="utf-8") as file:
|
||||||
|
json.dump(sorted(self._sent_keys), file, ensure_ascii=False, indent=2)
|
||||||
47
app/telegram.py
Normal file
47
app/telegram.py
Normal file
@@ -0,0 +1,47 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from aiogram import Bot
|
||||||
|
from aiogram.types import InlineKeyboardButton, InlineKeyboardMarkup
|
||||||
|
|
||||||
|
from app.models import Offer
|
||||||
|
|
||||||
|
|
||||||
|
class TelegramPublisher:
|
||||||
|
def __init__(self, bot: Bot, channel_id: str) -> None:
|
||||||
|
self._bot = bot
|
||||||
|
self._channel_id = channel_id
|
||||||
|
|
||||||
|
async def publish_offer(self, offer: Offer) -> None:
|
||||||
|
await self._bot.send_message(
|
||||||
|
chat_id=self._channel_id,
|
||||||
|
text=_format_offer_message(offer),
|
||||||
|
disable_web_page_preview=True,
|
||||||
|
reply_markup=_build_delete_markup(),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _format_offer_message(offer: Offer) -> str:
|
||||||
|
seller_line = f"Исполнитель: @{offer.seller_username}\n" if offer.seller_username else ""
|
||||||
|
description = offer.description.strip()
|
||||||
|
if len(description) > 1000:
|
||||||
|
description = f"{description[:997].rstrip()}..."
|
||||||
|
|
||||||
|
return (
|
||||||
|
f"Новый оффер: {offer.title}\n"
|
||||||
|
f"Источник: {offer.source}\n"
|
||||||
|
f"Категория: {offer.category_id}\n"
|
||||||
|
f"Бюджет: {offer.price}\n"
|
||||||
|
f"Опубликовано: {offer.published_at}\n"
|
||||||
|
f"{seller_line}"
|
||||||
|
f"\n"
|
||||||
|
f"{description}\n\n"
|
||||||
|
f"Ссылка: {offer.url}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _build_delete_markup() -> InlineKeyboardMarkup:
|
||||||
|
return InlineKeyboardMarkup(
|
||||||
|
inline_keyboard=[
|
||||||
|
[InlineKeyboardButton(text="❌", callback_data="delete_post")]
|
||||||
|
]
|
||||||
|
)
|
||||||
11
data/sent_offers.json
Normal file
11
data/sent_offers.json
Normal file
@@ -0,0 +1,11 @@
|
|||||||
|
[
|
||||||
|
"kwork:3078848",
|
||||||
|
"kwork:3130298",
|
||||||
|
"kwork:3213727",
|
||||||
|
"kwork:3215226",
|
||||||
|
"kwork:3215329",
|
||||||
|
"kwork:3215710",
|
||||||
|
"kwork:3215942",
|
||||||
|
"kwork:3216404",
|
||||||
|
"kwork:3216553"
|
||||||
|
]
|
||||||
1
kwork_resp.txt
Normal file
1
kwork_resp.txt
Normal file
File diff suppressed because one or more lines are too long
94
main.py
Normal file
94
main.py
Normal file
@@ -0,0 +1,94 @@
|
|||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
|
||||||
|
from aiogram import Bot, Dispatcher, Router
|
||||||
|
from aiogram.types import CallbackQuery
|
||||||
|
from aiogram.client.session.aiohttp import AiohttpSession
|
||||||
|
from apscheduler.schedulers.asyncio import AsyncIOScheduler
|
||||||
|
|
||||||
|
from app.config import load_settings
|
||||||
|
from app.service import OfferProcessingService
|
||||||
|
from app.sources import KworkSource
|
||||||
|
from app.state import SentOffersStore
|
||||||
|
from app.telegram import TelegramPublisher
|
||||||
|
|
||||||
|
|
||||||
|
logging.basicConfig(
|
||||||
|
level=logging.DEBUG,
|
||||||
|
format="%(asctime)s | %(levelname)s | %(name)s | %(message)s",
|
||||||
|
)
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def build_dispatcher() -> Dispatcher:
|
||||||
|
router = Router()
|
||||||
|
|
||||||
|
@router.callback_query(lambda callback: callback.data == "delete_post")
|
||||||
|
async def delete_post_callback(callback: CallbackQuery) -> None:
|
||||||
|
if callback.message is None:
|
||||||
|
await callback.answer()
|
||||||
|
return
|
||||||
|
|
||||||
|
try:
|
||||||
|
await callback.message.delete()
|
||||||
|
logger.info(
|
||||||
|
"Deleted Telegram message chat_id=%s message_id=%s by user_id=%s",
|
||||||
|
callback.message.chat.id,
|
||||||
|
callback.message.message_id,
|
||||||
|
callback.from_user.id,
|
||||||
|
)
|
||||||
|
await callback.answer("Пост удален")
|
||||||
|
except Exception:
|
||||||
|
logger.exception(
|
||||||
|
"Failed to delete Telegram message chat_id=%s message_id=%s",
|
||||||
|
callback.message.chat.id,
|
||||||
|
callback.message.message_id,
|
||||||
|
)
|
||||||
|
await callback.answer("Не удалось удалить пост", show_alert=True)
|
||||||
|
|
||||||
|
dispatcher = Dispatcher()
|
||||||
|
dispatcher.include_router(router)
|
||||||
|
return dispatcher
|
||||||
|
|
||||||
|
|
||||||
|
async def main() -> None:
|
||||||
|
settings = load_settings()
|
||||||
|
dispatcher = build_dispatcher()
|
||||||
|
|
||||||
|
session = AiohttpSession(proxy=settings.telegram_proxy_url) if settings.telegram_proxy_url else None
|
||||||
|
bot = Bot(token=settings.telegram_bot_token, session=session)
|
||||||
|
publisher = TelegramPublisher(bot=bot, channel_id=settings.telegram_channel_id)
|
||||||
|
store = SentOffersStore(settings.state_file)
|
||||||
|
|
||||||
|
service = OfferProcessingService(
|
||||||
|
sources=[
|
||||||
|
KworkSource(
|
||||||
|
category_ids=settings.kwork_category_ids,
|
||||||
|
timeout_seconds=settings.requests_timeout_seconds,
|
||||||
|
)
|
||||||
|
],
|
||||||
|
publisher=publisher,
|
||||||
|
sent_offers_store=store,
|
||||||
|
include_keywords=settings.keywords_include,
|
||||||
|
exclude_keywords=settings.keywords_exclude,
|
||||||
|
min_price_amount=settings.min_price_amount,
|
||||||
|
)
|
||||||
|
|
||||||
|
scheduler = AsyncIOScheduler()
|
||||||
|
scheduler.add_job(service.run_once, "interval", minutes=settings.scheduler_interval_minutes)
|
||||||
|
scheduler.start()
|
||||||
|
|
||||||
|
await service.run_once()
|
||||||
|
|
||||||
|
try:
|
||||||
|
await dispatcher.start_polling(bot)
|
||||||
|
finally:
|
||||||
|
scheduler.shutdown()
|
||||||
|
await bot.session.close()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
5
requirements.txt
Normal file
5
requirements.txt
Normal file
@@ -0,0 +1,5 @@
|
|||||||
|
aiogram>=3.10.0,<4.0.0
|
||||||
|
APScheduler>=3.10.4,<4.0.0
|
||||||
|
python-dotenv>=1.0.1,<2.0.0
|
||||||
|
requests>=2.32.0,<3.0.0
|
||||||
|
aiohttp-socks>=0.10.1,<1.0.0
|
||||||
Reference in New Issue
Block a user