From 366d1d687c5fe2ff6b7993924b51736c1b7fc5f0 Mon Sep 17 00:00:00 2001 From: greendevilll Date: Sat, 6 Jun 2026 23:30:35 +0700 Subject: [PATCH] working at least --- .gitignore | 2 + README.md | 32 +++ tag_hunter.py | 505 +++++++++++++++++++++++++++++++++++++++ tests/test_tag_hunter.py | 122 ++++++++++ 4 files changed, 661 insertions(+) create mode 100644 .gitignore create mode 100644 README.md create mode 100644 tag_hunter.py create mode 100644 tests/test_tag_hunter.py diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..8480b72 --- /dev/null +++ b/.gitignore @@ -0,0 +1,2 @@ +venv +*.txt \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..2e0c670 --- /dev/null +++ b/README.md @@ -0,0 +1,32 @@ +# Tag Hunter + +Simple Python TUI/CLI utility for finding Telegram nicknames that: + +1. Do not appear to belong to a user, channel, or group on Telegram. +2. Do not show up as an auction/sale listing on Fragment. + +This MVP uses public web pages, not private Telegram credentials or Fragment tokens. +It also retries on `429` and transient `5xx` responses with exponential backoff. + +## Run + +```bash +python tag_hunter.py nicknames.txt +``` + +Or run without arguments and enter the file path interactively. + +## Input format + +One nickname per line. Leading `@` is optional. + +```text +alice +@bob +charlie_123 +``` + +## Notes + +- The checker is best-effort. Public pages can change, and some usernames may be reported as `UNKNOWN` if the response is ambiguous or the site rate-limits too aggressively. +- The estimated runtime shown at startup is approximate and depends on network latency and retries. diff --git a/tag_hunter.py b/tag_hunter.py new file mode 100644 index 0000000..ec2978b --- /dev/null +++ b/tag_hunter.py @@ -0,0 +1,505 @@ +"""Simple TUI utility for finding free Telegram nicknames. + +The checker is best-effort and uses public web pages from Telegram and +Fragment. It handles 429/5xx responses with retries and exponential backoff. +""" + +from __future__ import annotations + +import argparse +import concurrent.futures as futures +from dataclasses import dataclass +from html import unescape +from pathlib import Path +import random +import re +import threading +import time +from typing import Iterable +from urllib import error, parse, request + + +USERNAME_RE = re.compile(r"^[a-z0-9_]{1,32}$") + + +DEFAULT_HEADERS = { + "User-Agent": ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/125.0 Safari/537.36" + ), + "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", + "Accept-Language": "en-US,en;q=0.8", + "Connection": "close", +} + + +@dataclass(frozen=True) +class FetchResult: + url: str + status: int + headers: dict[str, str] + body: bytes + final_url: str + + +@dataclass(frozen=True) +class CheckResult: + username: str + telegram_state: str + fragment_state: str + telegram_reason: str + fragment_reason: str + + @property + def is_free(self) -> bool: + return self.telegram_state == "free" and self.fragment_state == "free" + + +class HostThrottle: + def __init__(self, min_interval: float = 0.25) -> None: + self.min_interval = min_interval + self._lock = threading.Lock() + self._next_allowed: dict[str, float] = {} + + def wait(self, host: str) -> None: + with self._lock: + now = time.monotonic() + ready_at = self._next_allowed.get(host, now) + delay = max(0.0, ready_at - now) + self._next_allowed[host] = max(ready_at, now) + self.min_interval + if delay > 0: + time.sleep(delay) + + +THROTTLE = HostThrottle() + + +class SafeRedirectHandler(request.HTTPRedirectHandler): + def redirect_request(self, req, fp, code, msg, headers, newurl): + parsed = parse.urlparse(newurl) + if parsed.scheme and parsed.scheme not in {"http", "https"}: + raise error.HTTPError(req.full_url, code, msg, headers, fp) + return super().redirect_request(req, fp, code, msg, headers, newurl) + + +OPENER = request.build_opener(SafeRedirectHandler()) + + +def fetch_url(url: str, timeout: float = 12.0, retries: int = 5) -> FetchResult: + host = parse.urlparse(url).netloc + backoff = 1.0 + last_exc: Exception | None = None + + for attempt in range(retries + 1): + THROTTLE.wait(host) + req = request.Request(url, headers=DEFAULT_HEADERS, method="GET") + try: + with OPENER.open(req, timeout=timeout) as resp: + body = resp.read() + headers = {k: v for k, v in resp.headers.items()} + return FetchResult( + url=url, + status=getattr(resp, "status", resp.getcode()), + headers=headers, + body=body, + final_url=resp.geturl(), + ) + except error.HTTPError as exc: + body = exc.read() if getattr(exc, "fp", None) else b"" + headers = {k: v for k, v in (exc.headers.items() if exc.headers else [])} + status = exc.code + + if status == 429 or status >= 500: + retry_after = headers.get("Retry-After") + sleep_for = _retry_sleep(retry_after, backoff) + if attempt < retries: + time.sleep(sleep_for) + backoff = min(backoff * 2, 30.0) + continue + + return FetchResult( + url=url, + status=status, + headers=headers, + body=body, + final_url=getattr(exc, "url", url) or url, + ) + except (error.URLError, TimeoutError, OSError) as exc: + last_exc = exc + if attempt < retries: + time.sleep(backoff + random.uniform(0.0, 0.3)) + backoff = min(backoff * 2, 30.0) + continue + break + + raise RuntimeError(f"Failed to fetch {url}: {last_exc}") + + +def _retry_sleep(retry_after: str | None, backoff: float) -> float: + if retry_after: + try: + return max(1.0, float(retry_after)) + random.uniform(0.0, 0.5) + except ValueError: + pass + return backoff + random.uniform(0.0, 0.5) + + +def load_usernames(path: Path) -> list[str]: + if not path.exists(): + raise FileNotFoundError(path) + + items: list[str] = [] + seen: set[str] = set() + for raw_line in path.read_text(encoding="utf-8-sig", errors="ignore").splitlines(): + line = raw_line.strip() + if not line or line.startswith("#"): + continue + username = normalize_username(line) + if username is None: + continue + if username in seen: + continue + seen.add(username) + items.append(username) + return items + + +def normalize_username(value: str) -> str | None: + value = value.strip().lstrip("@").lower() + if not value: + return None + if not USERNAME_RE.fullmatch(value): + return None + return value + + +def estimate_duration(entries: int, workers: int) -> str: + if entries <= 0: + return "0s" + workers = max(workers, 1) + estimated_seconds = entries * 1.6 / workers + return format_duration(estimated_seconds) + + +def format_duration(seconds: float) -> str: + total = max(0, int(round(seconds))) + minutes, secs = divmod(total, 60) + hours, minutes = divmod(minutes, 60) + if hours: + return f"{hours}h {minutes}m {secs}s" + if minutes: + return f"{minutes}m {secs}s" + return f"{secs}s" + + +def check_username(username: str, timeout: float = 12.0) -> CheckResult: + telegram_url = f"https://t.me/{parse.quote(username)}" + fragment_url = f"https://fragment.com/username/{parse.quote(username)}" + + telegram = fetch_url(telegram_url, timeout=timeout) + fragment = fetch_url(fragment_url, timeout=timeout) + + telegram_state, telegram_reason = classify_telegram(username, telegram) + fragment_state, fragment_reason = classify_fragment(username, fragment) + + return CheckResult( + username=username, + telegram_state=telegram_state, + fragment_state=fragment_state, + telegram_reason=telegram_reason, + fragment_reason=fragment_reason, + ) + + +def classify_telegram(username: str, response: FetchResult) -> tuple[str, str]: + text = decode_text(response.body) + normalized = collapse_spaces(text.lower()) + title = extract_tag_content(text, "title") + title_norm = collapse_spaces(title.lower()) if title else "" + og_description = extract_meta_content(text, "og:description") + og_description_norm = collapse_spaces(og_description.lower()) if og_description else "" + + free_markers = ( + "sorry, this page isn't available", + "sorry, this page is unavailable", + "this page is unavailable", + "page not found", + "username is unavailable", + "username not available", + "doesn't seem to exist", + "does not seem to exist", + ) + + if response.status in {404, 410}: + return "free", f"HTTP {response.status}" + + for marker in free_markers: + if marker in normalized: + return "free", marker + + if title_norm.startswith(f"telegram: contact @{username}"): + return "free", title_norm + + if title_norm.startswith(f"telegram: view @{username}"): + return "taken", title_norm + + if f"you can contact @{username} right away" in og_description_norm: + return "free", og_description_norm + + if f"you can view and join @{username} right away" in og_description_norm: + return "taken", og_description_norm + + if "view in telegram" in normalized or "preview channel" in normalized: + return "taken", "telegram view marker" + + if "send message" in normalized and f"contact @{username}" not in normalized: + return "taken", "telegram send message marker" + + if response.status == 200: + return "unknown", "telegram response is ambiguous" + + return "unknown", f"HTTP {response.status}" + + +def classify_fragment(username: str, response: FetchResult) -> tuple[str, str]: + text = decode_text(response.body) + normalized = collapse_spaces(text.lower()) + + row_html = extract_fragment_row(text, username) + row_normalized = collapse_spaces(row_html.lower()) if row_html else "" + + free_markers = ( + "currently not for sale", + "not for sale", + "coming soon", + "address unavailable", + ) + sale_markers = ( + "on auction", + "for sale", + "sold", + "place bid", + "buy now", + "make an offer", + "minimum bid", + "highest bid", + "auction ends in", + "resale", + ) + + if response.status in {404, 410}: + return "free", f"HTTP {response.status}" + + # Prefer the result row for the specific username when present. + search_space = row_normalized or normalized + + for marker in free_markers: + if marker in search_space: + return "free", marker + + if row_html and any(marker in row_normalized for marker in sale_markers): + return "taken", first_marker(row_normalized, sale_markers) or "sale marker" + + # Fallback for pages that don't expose a direct row snippet. + for marker in sale_markers: + if marker in normalized: + return "taken", marker + + if response.status == 200: + return "unknown", "no fragment listing detected" + + if response.status >= 500: + return "unknown", f"HTTP {response.status}" + + return "unknown", f"HTTP {response.status}" + + +def extract_fragment_row(text: str, username: str) -> str: + username_escaped = re.escape(username) + patterns = ( + rf"]*data-username=\"@{username_escaped}\"[^>]*>.*?", + rf"]*>.*?@{username_escaped}.*?", + ) + for pattern in patterns: + match = re.search(pattern, text, re.IGNORECASE | re.DOTALL) + if match: + return match.group(0) + return "" + + +def first_marker(text: str, markers: tuple[str, ...]) -> str | None: + for marker in markers: + if marker in text: + return marker + return None + + +def extract_tag_content(text: str, tag: str) -> str: + pattern = rf"<{tag}\b[^>]*>(.*?)" + match = re.search(pattern, text, re.IGNORECASE | re.DOTALL) + if not match: + return "" + return collapse_spaces(unescape(re.sub(r"<[^>]+>", " ", match.group(1)))) + + +def extract_meta_content(text: str, property_name: str) -> str: + pattern = ( + rf']*property=["\']{re.escape(property_name)}["\']' + rf'[^>]*content=["\'](.*?)["\']' + ) + match = re.search(pattern, text, re.IGNORECASE | re.DOTALL) + if not match: + return "" + return collapse_spaces(unescape(match.group(1))) + + +def decode_text(body: bytes) -> str: + for encoding in ("utf-8", "cp1251", "latin-1"): + try: + return unescape(body.decode(encoding)) + except UnicodeDecodeError: + continue + return unescape(body.decode("utf-8", errors="ignore")) + + +def collapse_spaces(value: str) -> str: + return re.sub(r"\s+", " ", value).strip() + + +def print_banner() -> None: + print("Tag Hunter") + print("Find Telegram nicknames that are not occupied and not listed on Fragment.") + print("This is a best-effort checker based on public pages and retry/backoff.") + print() + + +def prompt_path(default: str | None = None) -> Path: + suffix = f" [{default}]" if default else "" + raw = input(f"Path to .txt file{suffix}: ").strip() + if not raw and default: + raw = default + return Path(raw).expanduser().resolve() + + +def prompt_workers(default: int = 4) -> int: + raw = input(f"Workers [{default}]: ").strip() + if not raw: + return default + try: + workers = int(raw) + except ValueError: + return default + return max(1, min(workers, 16)) + + +def run_interactive(path: Path | None, workers: int) -> int: + print_banner() + + if path is None: + path = prompt_path() + if workers <= 0: + workers = prompt_workers() + + try: + usernames = load_usernames(path) + except FileNotFoundError: + print(f"File not found: {path}") + return 1 + + if not usernames: + print("No valid usernames found in the input file.") + return 1 + + print(f"Loaded {len(usernames)} usernames from {path}") + print(f"Parallel workers: {workers}") + print(f"Approximate time: {estimate_duration(len(usernames), workers)}") + print() + + input("Press Enter to start...") + + free: list[str] = [] + taken: list[str] = [] + unknown: list[str] = [] + lock = threading.Lock() + started = time.monotonic() + total = len(usernames) + + with futures.ThreadPoolExecutor(max_workers=workers) as pool: + future_map = {pool.submit(check_safe, username): username for username in usernames} + done = 0 + for future in futures.as_completed(future_map): + username = future_map[future] + done += 1 + try: + result = future.result() + except Exception as exc: # pragma: no cover - runtime safety + result = CheckResult( + username=username, + telegram_state="unknown", + fragment_state="unknown", + telegram_reason=str(exc), + fragment_reason=str(exc), + ) + + with lock: + if result.is_free: + free.append(result.username) + status = "FREE" + elif result.telegram_state == "taken" or result.fragment_state == "taken": + taken.append(result.username) + status = "TAKEN" + else: + unknown.append(result.username) + status = "UNKNOWN" + + elapsed = time.monotonic() - started + remaining = max(total - done, 0) + per_item = elapsed / done if done else 0.0 + eta = format_duration(per_item * remaining) + print( + f"[{done}/{total}] {username:<32} {status:<7} ETA {eta} " + f"(tg: {result.telegram_state}, fr: {result.fragment_state})" + ) + + print() + print(f"Free: {len(free)}") + print(f"Taken: {len(taken)}") + print(f"Unknown: {len(unknown)}") + if free: + print("\nFree nicknames:") + for username in free: + print(f"@{username}") + + return 0 + + +def check_safe(username: str) -> CheckResult: + try: + return check_username(username) + except Exception as exc: # pragma: no cover - runtime safety + return CheckResult( + username=username, + telegram_state="unknown", + fragment_state="unknown", + telegram_reason=str(exc), + fragment_reason=str(exc), + ) + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description="Find free Telegram nicknames.") + parser.add_argument("path", nargs="?", type=Path, help="Path to .txt file with nicknames") + parser.add_argument("--workers", type=int, default=4, help="Parallel workers (default: 4)") + return parser + + +def main(argv: Iterable[str] | None = None) -> int: + parser = build_parser() + args = parser.parse_args(list(argv) if argv is not None else None) + path = args.path.expanduser().resolve() if args.path else None + return run_interactive(path=path, workers=max(1, args.workers)) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_tag_hunter.py b/tests/test_tag_hunter.py new file mode 100644 index 0000000..c0dce9e --- /dev/null +++ b/tests/test_tag_hunter.py @@ -0,0 +1,122 @@ +import unittest +from pathlib import Path + +from tag_hunter import FetchResult, classify_fragment, classify_telegram + + +class FragmentClassifierTests(unittest.TestCase): + def test_prefers_not_for_sale_over_nav_auction_markers(self) -> None: + body = """ + + +
On auction For sale Sold
+ +
Not for sale
+ + + + + """.encode("utf-8") + result = FetchResult( + url="https://fragment.com/username/dsfhsjdfkfhshfkshfksj", + status=200, + headers={}, + body=body, + final_url="https://fragment.com/username/dsfhsjdfkfhshfkshfksj", + ) + + state, reason = classify_fragment("dsfhsjdfkfhshfkshfksj", result) + + self.assertEqual(state, "free") + self.assertIn("not for sale", reason) + + def test_taken_when_fragment_row_contains_sale_marker(self) -> None: + body = """ + + + +
On auction
+ + + + """.encode("utf-8") + result = FetchResult( + url="https://fragment.com/username/cocaine", + status=200, + headers={}, + body=body, + final_url="https://fragment.com/username/cocaine", + ) + + state, reason = classify_fragment("cocaine", result) + + self.assertEqual(state, "taken") + self.assertEqual(reason, "on auction") + + +class TelegramClassifierTests(unittest.TestCase): + def test_existing_profile_is_taken_from_view_markers(self) -> None: + body = Path(__file__).resolve().parents[1].joinpath("exist.txt").read_text(encoding="utf-8") + result = FetchResult( + url="https://t.me/dicksmack", + status=200, + headers={}, + body=body.encode("utf-8"), + final_url="https://t.me/dicksmack", + ) + + state, reason = classify_telegram("dicksmack", result) + + self.assertEqual(state, "taken") + self.assertTrue( + any(marker in reason for marker in ("view @dicksmack", "view and join @dicksmack", "preview channel")) + ) + + def test_nonexisting_profile_is_free_from_contact_markers(self) -> None: + body = Path(__file__).resolve().parents[1].joinpath("nonexist.txt").read_text(encoding="utf-8") + result = FetchResult( + url="https://t.me/sahdfjkhdfjsahdkf", + status=200, + headers={}, + body=body.encode("utf-8"), + final_url="https://t.me/sahdfjkhdfjsahdkf", + ) + + state, reason = classify_telegram("sahdfjkhdfjsahdkf", result) + + self.assertEqual(state, "free") + self.assertTrue( + any(marker in reason for marker in ("contact @sahdfjkhdfjsahdkf", "right away")) + ) + + def test_redirect_without_explicit_body_is_unknown(self) -> None: + result = FetchResult( + url="https://t.me/doesnotexist123", + status=200, + headers={"Location": "tg://resolve?domain=doesnotexist123"}, + body=b"Telegram", + final_url="https://t.me/doesnotexist123", + ) + + state, reason = classify_telegram("doesnotexist123", result) + + self.assertEqual(state, "unknown") + self.assertIn("ambiguous", reason) + + def test_explicit_unavailable_marker_is_free(self) -> None: + result = FetchResult( + url="https://t.me/doesnotexist123", + status=200, + headers={}, + body=b"Sorry, this page isn't available.", + final_url="https://t.me/doesnotexist123", + ) + + state, reason = classify_telegram("doesnotexist123", result) + + self.assertEqual(state, "free") + self.assertIn("sorry, this page isn't available", reason) + + +if __name__ == "__main__": + unittest.main()