506 lines
15 KiB
Python
506 lines
15 KiB
Python
"""Simple TUI utility for finding free Telegram nicknames.
|
|
|
|
The checker is best-effort and uses public web pages from Telegram and
|
|
Fragment. It handles 429/5xx responses with retries and exponential backoff.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import concurrent.futures as futures
|
|
from dataclasses import dataclass
|
|
from html import unescape
|
|
from pathlib import Path
|
|
import random
|
|
import re
|
|
import threading
|
|
import time
|
|
from typing import Iterable
|
|
from urllib import error, parse, request
|
|
|
|
|
|
USERNAME_RE = re.compile(r"^[a-z0-9_]{1,32}$")
|
|
|
|
|
|
DEFAULT_HEADERS = {
|
|
"User-Agent": (
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
|
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
|
"Chrome/125.0 Safari/537.36"
|
|
),
|
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|
"Accept-Language": "en-US,en;q=0.8",
|
|
"Connection": "close",
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class FetchResult:
|
|
url: str
|
|
status: int
|
|
headers: dict[str, str]
|
|
body: bytes
|
|
final_url: str
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class CheckResult:
|
|
username: str
|
|
telegram_state: str
|
|
fragment_state: str
|
|
telegram_reason: str
|
|
fragment_reason: str
|
|
|
|
@property
|
|
def is_free(self) -> bool:
|
|
return self.telegram_state == "free" and self.fragment_state == "free"
|
|
|
|
|
|
class HostThrottle:
|
|
def __init__(self, min_interval: float = 0.25) -> None:
|
|
self.min_interval = min_interval
|
|
self._lock = threading.Lock()
|
|
self._next_allowed: dict[str, float] = {}
|
|
|
|
def wait(self, host: str) -> None:
|
|
with self._lock:
|
|
now = time.monotonic()
|
|
ready_at = self._next_allowed.get(host, now)
|
|
delay = max(0.0, ready_at - now)
|
|
self._next_allowed[host] = max(ready_at, now) + self.min_interval
|
|
if delay > 0:
|
|
time.sleep(delay)
|
|
|
|
|
|
THROTTLE = HostThrottle()
|
|
|
|
|
|
class SafeRedirectHandler(request.HTTPRedirectHandler):
|
|
def redirect_request(self, req, fp, code, msg, headers, newurl):
|
|
parsed = parse.urlparse(newurl)
|
|
if parsed.scheme and parsed.scheme not in {"http", "https"}:
|
|
raise error.HTTPError(req.full_url, code, msg, headers, fp)
|
|
return super().redirect_request(req, fp, code, msg, headers, newurl)
|
|
|
|
|
|
OPENER = request.build_opener(SafeRedirectHandler())
|
|
|
|
|
|
def fetch_url(url: str, timeout: float = 12.0, retries: int = 5) -> FetchResult:
|
|
host = parse.urlparse(url).netloc
|
|
backoff = 1.0
|
|
last_exc: Exception | None = None
|
|
|
|
for attempt in range(retries + 1):
|
|
THROTTLE.wait(host)
|
|
req = request.Request(url, headers=DEFAULT_HEADERS, method="GET")
|
|
try:
|
|
with OPENER.open(req, timeout=timeout) as resp:
|
|
body = resp.read()
|
|
headers = {k: v for k, v in resp.headers.items()}
|
|
return FetchResult(
|
|
url=url,
|
|
status=getattr(resp, "status", resp.getcode()),
|
|
headers=headers,
|
|
body=body,
|
|
final_url=resp.geturl(),
|
|
)
|
|
except error.HTTPError as exc:
|
|
body = exc.read() if getattr(exc, "fp", None) else b""
|
|
headers = {k: v for k, v in (exc.headers.items() if exc.headers else [])}
|
|
status = exc.code
|
|
|
|
if status == 429 or status >= 500:
|
|
retry_after = headers.get("Retry-After")
|
|
sleep_for = _retry_sleep(retry_after, backoff)
|
|
if attempt < retries:
|
|
time.sleep(sleep_for)
|
|
backoff = min(backoff * 2, 30.0)
|
|
continue
|
|
|
|
return FetchResult(
|
|
url=url,
|
|
status=status,
|
|
headers=headers,
|
|
body=body,
|
|
final_url=getattr(exc, "url", url) or url,
|
|
)
|
|
except (error.URLError, TimeoutError, OSError) as exc:
|
|
last_exc = exc
|
|
if attempt < retries:
|
|
time.sleep(backoff + random.uniform(0.0, 0.3))
|
|
backoff = min(backoff * 2, 30.0)
|
|
continue
|
|
break
|
|
|
|
raise RuntimeError(f"Failed to fetch {url}: {last_exc}")
|
|
|
|
|
|
def _retry_sleep(retry_after: str | None, backoff: float) -> float:
|
|
if retry_after:
|
|
try:
|
|
return max(1.0, float(retry_after)) + random.uniform(0.0, 0.5)
|
|
except ValueError:
|
|
pass
|
|
return backoff + random.uniform(0.0, 0.5)
|
|
|
|
|
|
def load_usernames(path: Path) -> list[str]:
|
|
if not path.exists():
|
|
raise FileNotFoundError(path)
|
|
|
|
items: list[str] = []
|
|
seen: set[str] = set()
|
|
for raw_line in path.read_text(encoding="utf-8-sig", errors="ignore").splitlines():
|
|
line = raw_line.strip()
|
|
if not line or line.startswith("#"):
|
|
continue
|
|
username = normalize_username(line)
|
|
if username is None:
|
|
continue
|
|
if username in seen:
|
|
continue
|
|
seen.add(username)
|
|
items.append(username)
|
|
return items
|
|
|
|
|
|
def normalize_username(value: str) -> str | None:
|
|
value = value.strip().lstrip("@").lower()
|
|
if not value:
|
|
return None
|
|
if not USERNAME_RE.fullmatch(value):
|
|
return None
|
|
return value
|
|
|
|
|
|
def estimate_duration(entries: int, workers: int) -> str:
|
|
if entries <= 0:
|
|
return "0s"
|
|
workers = max(workers, 1)
|
|
estimated_seconds = entries * 1.6 / workers
|
|
return format_duration(estimated_seconds)
|
|
|
|
|
|
def format_duration(seconds: float) -> str:
|
|
total = max(0, int(round(seconds)))
|
|
minutes, secs = divmod(total, 60)
|
|
hours, minutes = divmod(minutes, 60)
|
|
if hours:
|
|
return f"{hours}h {minutes}m {secs}s"
|
|
if minutes:
|
|
return f"{minutes}m {secs}s"
|
|
return f"{secs}s"
|
|
|
|
|
|
def check_username(username: str, timeout: float = 12.0) -> CheckResult:
|
|
telegram_url = f"https://t.me/{parse.quote(username)}"
|
|
fragment_url = f"https://fragment.com/username/{parse.quote(username)}"
|
|
|
|
telegram = fetch_url(telegram_url, timeout=timeout)
|
|
fragment = fetch_url(fragment_url, timeout=timeout)
|
|
|
|
telegram_state, telegram_reason = classify_telegram(username, telegram)
|
|
fragment_state, fragment_reason = classify_fragment(username, fragment)
|
|
|
|
return CheckResult(
|
|
username=username,
|
|
telegram_state=telegram_state,
|
|
fragment_state=fragment_state,
|
|
telegram_reason=telegram_reason,
|
|
fragment_reason=fragment_reason,
|
|
)
|
|
|
|
|
|
def classify_telegram(username: str, response: FetchResult) -> tuple[str, str]:
|
|
text = decode_text(response.body)
|
|
normalized = collapse_spaces(text.lower())
|
|
title = extract_tag_content(text, "title")
|
|
title_norm = collapse_spaces(title.lower()) if title else ""
|
|
og_description = extract_meta_content(text, "og:description")
|
|
og_description_norm = collapse_spaces(og_description.lower()) if og_description else ""
|
|
|
|
free_markers = (
|
|
"sorry, this page isn't available",
|
|
"sorry, this page is unavailable",
|
|
"this page is unavailable",
|
|
"page not found",
|
|
"username is unavailable",
|
|
"username not available",
|
|
"doesn't seem to exist",
|
|
"does not seem to exist",
|
|
)
|
|
|
|
if response.status in {404, 410}:
|
|
return "free", f"HTTP {response.status}"
|
|
|
|
for marker in free_markers:
|
|
if marker in normalized:
|
|
return "free", marker
|
|
|
|
if title_norm.startswith(f"telegram: contact @{username}"):
|
|
return "free", title_norm
|
|
|
|
if title_norm.startswith(f"telegram: view @{username}"):
|
|
return "taken", title_norm
|
|
|
|
if f"you can contact @{username} right away" in og_description_norm:
|
|
return "free", og_description_norm
|
|
|
|
if f"you can view and join @{username} right away" in og_description_norm:
|
|
return "taken", og_description_norm
|
|
|
|
if "view in telegram" in normalized or "preview channel" in normalized:
|
|
return "taken", "telegram view marker"
|
|
|
|
if "send message" in normalized and f"contact @{username}" not in normalized:
|
|
return "taken", "telegram send message marker"
|
|
|
|
if response.status == 200:
|
|
return "unknown", "telegram response is ambiguous"
|
|
|
|
return "unknown", f"HTTP {response.status}"
|
|
|
|
|
|
def classify_fragment(username: str, response: FetchResult) -> tuple[str, str]:
|
|
text = decode_text(response.body)
|
|
normalized = collapse_spaces(text.lower())
|
|
|
|
row_html = extract_fragment_row(text, username)
|
|
row_normalized = collapse_spaces(row_html.lower()) if row_html else ""
|
|
|
|
free_markers = (
|
|
"currently not for sale",
|
|
"not for sale",
|
|
"coming soon",
|
|
"address unavailable",
|
|
)
|
|
sale_markers = (
|
|
"on auction",
|
|
"for sale",
|
|
"sold",
|
|
"place bid",
|
|
"buy now",
|
|
"make an offer",
|
|
"minimum bid",
|
|
"highest bid",
|
|
"auction ends in",
|
|
"resale",
|
|
)
|
|
|
|
if response.status in {404, 410}:
|
|
return "free", f"HTTP {response.status}"
|
|
|
|
# Prefer the result row for the specific username when present.
|
|
search_space = row_normalized or normalized
|
|
|
|
for marker in free_markers:
|
|
if marker in search_space:
|
|
return "free", marker
|
|
|
|
if row_html and any(marker in row_normalized for marker in sale_markers):
|
|
return "taken", first_marker(row_normalized, sale_markers) or "sale marker"
|
|
|
|
# Fallback for pages that don't expose a direct row snippet.
|
|
for marker in sale_markers:
|
|
if marker in normalized:
|
|
return "taken", marker
|
|
|
|
if response.status == 200:
|
|
return "unknown", "no fragment listing detected"
|
|
|
|
if response.status >= 500:
|
|
return "unknown", f"HTTP {response.status}"
|
|
|
|
return "unknown", f"HTTP {response.status}"
|
|
|
|
|
|
def extract_fragment_row(text: str, username: str) -> str:
|
|
username_escaped = re.escape(username)
|
|
patterns = (
|
|
rf"<tr[^>]*data-username=\"@{username_escaped}\"[^>]*>.*?</tr>",
|
|
rf"<tr[^>]*>.*?@{username_escaped}.*?</tr>",
|
|
)
|
|
for pattern in patterns:
|
|
match = re.search(pattern, text, re.IGNORECASE | re.DOTALL)
|
|
if match:
|
|
return match.group(0)
|
|
return ""
|
|
|
|
|
|
def first_marker(text: str, markers: tuple[str, ...]) -> str | None:
|
|
for marker in markers:
|
|
if marker in text:
|
|
return marker
|
|
return None
|
|
|
|
|
|
def extract_tag_content(text: str, tag: str) -> str:
|
|
pattern = rf"<{tag}\b[^>]*>(.*?)</{tag}>"
|
|
match = re.search(pattern, text, re.IGNORECASE | re.DOTALL)
|
|
if not match:
|
|
return ""
|
|
return collapse_spaces(unescape(re.sub(r"<[^>]+>", " ", match.group(1))))
|
|
|
|
|
|
def extract_meta_content(text: str, property_name: str) -> str:
|
|
pattern = (
|
|
rf'<meta\b[^>]*property=["\']{re.escape(property_name)}["\']'
|
|
rf'[^>]*content=["\'](.*?)["\']'
|
|
)
|
|
match = re.search(pattern, text, re.IGNORECASE | re.DOTALL)
|
|
if not match:
|
|
return ""
|
|
return collapse_spaces(unescape(match.group(1)))
|
|
|
|
|
|
def decode_text(body: bytes) -> str:
|
|
for encoding in ("utf-8", "cp1251", "latin-1"):
|
|
try:
|
|
return unescape(body.decode(encoding))
|
|
except UnicodeDecodeError:
|
|
continue
|
|
return unescape(body.decode("utf-8", errors="ignore"))
|
|
|
|
|
|
def collapse_spaces(value: str) -> str:
|
|
return re.sub(r"\s+", " ", value).strip()
|
|
|
|
|
|
def print_banner() -> None:
|
|
print("Tag Hunter")
|
|
print("Find Telegram nicknames that are not occupied and not listed on Fragment.")
|
|
print("This is a best-effort checker based on public pages and retry/backoff.")
|
|
print()
|
|
|
|
|
|
def prompt_path(default: str | None = None) -> Path:
|
|
suffix = f" [{default}]" if default else ""
|
|
raw = input(f"Path to .txt file{suffix}: ").strip()
|
|
if not raw and default:
|
|
raw = default
|
|
return Path(raw).expanduser().resolve()
|
|
|
|
|
|
def prompt_workers(default: int = 4) -> int:
|
|
raw = input(f"Workers [{default}]: ").strip()
|
|
if not raw:
|
|
return default
|
|
try:
|
|
workers = int(raw)
|
|
except ValueError:
|
|
return default
|
|
return max(1, min(workers, 16))
|
|
|
|
|
|
def run_interactive(path: Path | None, workers: int) -> int:
|
|
print_banner()
|
|
|
|
if path is None:
|
|
path = prompt_path()
|
|
if workers <= 0:
|
|
workers = prompt_workers()
|
|
|
|
try:
|
|
usernames = load_usernames(path)
|
|
except FileNotFoundError:
|
|
print(f"File not found: {path}")
|
|
return 1
|
|
|
|
if not usernames:
|
|
print("No valid usernames found in the input file.")
|
|
return 1
|
|
|
|
print(f"Loaded {len(usernames)} usernames from {path}")
|
|
print(f"Parallel workers: {workers}")
|
|
print(f"Approximate time: {estimate_duration(len(usernames), workers)}")
|
|
print()
|
|
|
|
input("Press Enter to start...")
|
|
|
|
free: list[str] = []
|
|
taken: list[str] = []
|
|
unknown: list[str] = []
|
|
lock = threading.Lock()
|
|
started = time.monotonic()
|
|
total = len(usernames)
|
|
|
|
with futures.ThreadPoolExecutor(max_workers=workers) as pool:
|
|
future_map = {pool.submit(check_safe, username): username for username in usernames}
|
|
done = 0
|
|
for future in futures.as_completed(future_map):
|
|
username = future_map[future]
|
|
done += 1
|
|
try:
|
|
result = future.result()
|
|
except Exception as exc: # pragma: no cover - runtime safety
|
|
result = CheckResult(
|
|
username=username,
|
|
telegram_state="unknown",
|
|
fragment_state="unknown",
|
|
telegram_reason=str(exc),
|
|
fragment_reason=str(exc),
|
|
)
|
|
|
|
with lock:
|
|
if result.is_free:
|
|
free.append(result.username)
|
|
status = "FREE"
|
|
elif result.telegram_state == "taken" or result.fragment_state == "taken":
|
|
taken.append(result.username)
|
|
status = "TAKEN"
|
|
else:
|
|
unknown.append(result.username)
|
|
status = "UNKNOWN"
|
|
|
|
elapsed = time.monotonic() - started
|
|
remaining = max(total - done, 0)
|
|
per_item = elapsed / done if done else 0.0
|
|
eta = format_duration(per_item * remaining)
|
|
print(
|
|
f"[{done}/{total}] {username:<32} {status:<7} ETA {eta} "
|
|
f"(tg: {result.telegram_state}, fr: {result.fragment_state})"
|
|
)
|
|
|
|
print()
|
|
print(f"Free: {len(free)}")
|
|
print(f"Taken: {len(taken)}")
|
|
print(f"Unknown: {len(unknown)}")
|
|
if free:
|
|
print("\nFree nicknames:")
|
|
for username in free:
|
|
print(f"@{username}")
|
|
|
|
return 0
|
|
|
|
|
|
def check_safe(username: str) -> CheckResult:
|
|
try:
|
|
return check_username(username)
|
|
except Exception as exc: # pragma: no cover - runtime safety
|
|
return CheckResult(
|
|
username=username,
|
|
telegram_state="unknown",
|
|
fragment_state="unknown",
|
|
telegram_reason=str(exc),
|
|
fragment_reason=str(exc),
|
|
)
|
|
|
|
|
|
def build_parser() -> argparse.ArgumentParser:
|
|
parser = argparse.ArgumentParser(description="Find free Telegram nicknames.")
|
|
parser.add_argument("path", nargs="?", type=Path, help="Path to .txt file with nicknames")
|
|
parser.add_argument("--workers", type=int, default=4, help="Parallel workers (default: 4)")
|
|
return parser
|
|
|
|
|
|
def main(argv: Iterable[str] | None = None) -> int:
|
|
parser = build_parser()
|
|
args = parser.parse_args(list(argv) if argv is not None else None)
|
|
path = args.path.expanduser().resolve() if args.path else None
|
|
return run_interactive(path=path, workers=max(1, args.workers))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|