working at least

This commit is contained in:
greendevilll
2026-06-06 23:30:35 +07:00
commit 366d1d687c
4 changed files with 661 additions and 0 deletions

505
tag_hunter.py Normal file
View File

@@ -0,0 +1,505 @@
"""Simple TUI utility for finding free Telegram nicknames.
The checker is best-effort and uses public web pages from Telegram and
Fragment. It handles 429/5xx responses with retries and exponential backoff.
"""
from __future__ import annotations
import argparse
import concurrent.futures as futures
from dataclasses import dataclass
from html import unescape
from pathlib import Path
import random
import re
import threading
import time
from typing import Iterable
from urllib import error, parse, request
USERNAME_RE = re.compile(r"^[a-z0-9_]{1,32}$")
DEFAULT_HEADERS = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/125.0 Safari/537.36"
),
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
"Accept-Language": "en-US,en;q=0.8",
"Connection": "close",
}
@dataclass(frozen=True)
class FetchResult:
url: str
status: int
headers: dict[str, str]
body: bytes
final_url: str
@dataclass(frozen=True)
class CheckResult:
username: str
telegram_state: str
fragment_state: str
telegram_reason: str
fragment_reason: str
@property
def is_free(self) -> bool:
return self.telegram_state == "free" and self.fragment_state == "free"
class HostThrottle:
def __init__(self, min_interval: float = 0.25) -> None:
self.min_interval = min_interval
self._lock = threading.Lock()
self._next_allowed: dict[str, float] = {}
def wait(self, host: str) -> None:
with self._lock:
now = time.monotonic()
ready_at = self._next_allowed.get(host, now)
delay = max(0.0, ready_at - now)
self._next_allowed[host] = max(ready_at, now) + self.min_interval
if delay > 0:
time.sleep(delay)
THROTTLE = HostThrottle()
class SafeRedirectHandler(request.HTTPRedirectHandler):
def redirect_request(self, req, fp, code, msg, headers, newurl):
parsed = parse.urlparse(newurl)
if parsed.scheme and parsed.scheme not in {"http", "https"}:
raise error.HTTPError(req.full_url, code, msg, headers, fp)
return super().redirect_request(req, fp, code, msg, headers, newurl)
OPENER = request.build_opener(SafeRedirectHandler())
def fetch_url(url: str, timeout: float = 12.0, retries: int = 5) -> FetchResult:
host = parse.urlparse(url).netloc
backoff = 1.0
last_exc: Exception | None = None
for attempt in range(retries + 1):
THROTTLE.wait(host)
req = request.Request(url, headers=DEFAULT_HEADERS, method="GET")
try:
with OPENER.open(req, timeout=timeout) as resp:
body = resp.read()
headers = {k: v for k, v in resp.headers.items()}
return FetchResult(
url=url,
status=getattr(resp, "status", resp.getcode()),
headers=headers,
body=body,
final_url=resp.geturl(),
)
except error.HTTPError as exc:
body = exc.read() if getattr(exc, "fp", None) else b""
headers = {k: v for k, v in (exc.headers.items() if exc.headers else [])}
status = exc.code
if status == 429 or status >= 500:
retry_after = headers.get("Retry-After")
sleep_for = _retry_sleep(retry_after, backoff)
if attempt < retries:
time.sleep(sleep_for)
backoff = min(backoff * 2, 30.0)
continue
return FetchResult(
url=url,
status=status,
headers=headers,
body=body,
final_url=getattr(exc, "url", url) or url,
)
except (error.URLError, TimeoutError, OSError) as exc:
last_exc = exc
if attempt < retries:
time.sleep(backoff + random.uniform(0.0, 0.3))
backoff = min(backoff * 2, 30.0)
continue
break
raise RuntimeError(f"Failed to fetch {url}: {last_exc}")
def _retry_sleep(retry_after: str | None, backoff: float) -> float:
if retry_after:
try:
return max(1.0, float(retry_after)) + random.uniform(0.0, 0.5)
except ValueError:
pass
return backoff + random.uniform(0.0, 0.5)
def load_usernames(path: Path) -> list[str]:
if not path.exists():
raise FileNotFoundError(path)
items: list[str] = []
seen: set[str] = set()
for raw_line in path.read_text(encoding="utf-8-sig", errors="ignore").splitlines():
line = raw_line.strip()
if not line or line.startswith("#"):
continue
username = normalize_username(line)
if username is None:
continue
if username in seen:
continue
seen.add(username)
items.append(username)
return items
def normalize_username(value: str) -> str | None:
value = value.strip().lstrip("@").lower()
if not value:
return None
if not USERNAME_RE.fullmatch(value):
return None
return value
def estimate_duration(entries: int, workers: int) -> str:
if entries <= 0:
return "0s"
workers = max(workers, 1)
estimated_seconds = entries * 1.6 / workers
return format_duration(estimated_seconds)
def format_duration(seconds: float) -> str:
total = max(0, int(round(seconds)))
minutes, secs = divmod(total, 60)
hours, minutes = divmod(minutes, 60)
if hours:
return f"{hours}h {minutes}m {secs}s"
if minutes:
return f"{minutes}m {secs}s"
return f"{secs}s"
def check_username(username: str, timeout: float = 12.0) -> CheckResult:
telegram_url = f"https://t.me/{parse.quote(username)}"
fragment_url = f"https://fragment.com/username/{parse.quote(username)}"
telegram = fetch_url(telegram_url, timeout=timeout)
fragment = fetch_url(fragment_url, timeout=timeout)
telegram_state, telegram_reason = classify_telegram(username, telegram)
fragment_state, fragment_reason = classify_fragment(username, fragment)
return CheckResult(
username=username,
telegram_state=telegram_state,
fragment_state=fragment_state,
telegram_reason=telegram_reason,
fragment_reason=fragment_reason,
)
def classify_telegram(username: str, response: FetchResult) -> tuple[str, str]:
text = decode_text(response.body)
normalized = collapse_spaces(text.lower())
title = extract_tag_content(text, "title")
title_norm = collapse_spaces(title.lower()) if title else ""
og_description = extract_meta_content(text, "og:description")
og_description_norm = collapse_spaces(og_description.lower()) if og_description else ""
free_markers = (
"sorry, this page isn't available",
"sorry, this page is unavailable",
"this page is unavailable",
"page not found",
"username is unavailable",
"username not available",
"doesn't seem to exist",
"does not seem to exist",
)
if response.status in {404, 410}:
return "free", f"HTTP {response.status}"
for marker in free_markers:
if marker in normalized:
return "free", marker
if title_norm.startswith(f"telegram: contact @{username}"):
return "free", title_norm
if title_norm.startswith(f"telegram: view @{username}"):
return "taken", title_norm
if f"you can contact @{username} right away" in og_description_norm:
return "free", og_description_norm
if f"you can view and join @{username} right away" in og_description_norm:
return "taken", og_description_norm
if "view in telegram" in normalized or "preview channel" in normalized:
return "taken", "telegram view marker"
if "send message" in normalized and f"contact @{username}" not in normalized:
return "taken", "telegram send message marker"
if response.status == 200:
return "unknown", "telegram response is ambiguous"
return "unknown", f"HTTP {response.status}"
def classify_fragment(username: str, response: FetchResult) -> tuple[str, str]:
text = decode_text(response.body)
normalized = collapse_spaces(text.lower())
row_html = extract_fragment_row(text, username)
row_normalized = collapse_spaces(row_html.lower()) if row_html else ""
free_markers = (
"currently not for sale",
"not for sale",
"coming soon",
"address unavailable",
)
sale_markers = (
"on auction",
"for sale",
"sold",
"place bid",
"buy now",
"make an offer",
"minimum bid",
"highest bid",
"auction ends in",
"resale",
)
if response.status in {404, 410}:
return "free", f"HTTP {response.status}"
# Prefer the result row for the specific username when present.
search_space = row_normalized or normalized
for marker in free_markers:
if marker in search_space:
return "free", marker
if row_html and any(marker in row_normalized for marker in sale_markers):
return "taken", first_marker(row_normalized, sale_markers) or "sale marker"
# Fallback for pages that don't expose a direct row snippet.
for marker in sale_markers:
if marker in normalized:
return "taken", marker
if response.status == 200:
return "unknown", "no fragment listing detected"
if response.status >= 500:
return "unknown", f"HTTP {response.status}"
return "unknown", f"HTTP {response.status}"
def extract_fragment_row(text: str, username: str) -> str:
username_escaped = re.escape(username)
patterns = (
rf"<tr[^>]*data-username=\"@{username_escaped}\"[^>]*>.*?</tr>",
rf"<tr[^>]*>.*?@{username_escaped}.*?</tr>",
)
for pattern in patterns:
match = re.search(pattern, text, re.IGNORECASE | re.DOTALL)
if match:
return match.group(0)
return ""
def first_marker(text: str, markers: tuple[str, ...]) -> str | None:
for marker in markers:
if marker in text:
return marker
return None
def extract_tag_content(text: str, tag: str) -> str:
pattern = rf"<{tag}\b[^>]*>(.*?)</{tag}>"
match = re.search(pattern, text, re.IGNORECASE | re.DOTALL)
if not match:
return ""
return collapse_spaces(unescape(re.sub(r"<[^>]+>", " ", match.group(1))))
def extract_meta_content(text: str, property_name: str) -> str:
pattern = (
rf'<meta\b[^>]*property=["\']{re.escape(property_name)}["\']'
rf'[^>]*content=["\'](.*?)["\']'
)
match = re.search(pattern, text, re.IGNORECASE | re.DOTALL)
if not match:
return ""
return collapse_spaces(unescape(match.group(1)))
def decode_text(body: bytes) -> str:
for encoding in ("utf-8", "cp1251", "latin-1"):
try:
return unescape(body.decode(encoding))
except UnicodeDecodeError:
continue
return unescape(body.decode("utf-8", errors="ignore"))
def collapse_spaces(value: str) -> str:
return re.sub(r"\s+", " ", value).strip()
def print_banner() -> None:
print("Tag Hunter")
print("Find Telegram nicknames that are not occupied and not listed on Fragment.")
print("This is a best-effort checker based on public pages and retry/backoff.")
print()
def prompt_path(default: str | None = None) -> Path:
suffix = f" [{default}]" if default else ""
raw = input(f"Path to .txt file{suffix}: ").strip()
if not raw and default:
raw = default
return Path(raw).expanduser().resolve()
def prompt_workers(default: int = 4) -> int:
raw = input(f"Workers [{default}]: ").strip()
if not raw:
return default
try:
workers = int(raw)
except ValueError:
return default
return max(1, min(workers, 16))
def run_interactive(path: Path | None, workers: int) -> int:
print_banner()
if path is None:
path = prompt_path()
if workers <= 0:
workers = prompt_workers()
try:
usernames = load_usernames(path)
except FileNotFoundError:
print(f"File not found: {path}")
return 1
if not usernames:
print("No valid usernames found in the input file.")
return 1
print(f"Loaded {len(usernames)} usernames from {path}")
print(f"Parallel workers: {workers}")
print(f"Approximate time: {estimate_duration(len(usernames), workers)}")
print()
input("Press Enter to start...")
free: list[str] = []
taken: list[str] = []
unknown: list[str] = []
lock = threading.Lock()
started = time.monotonic()
total = len(usernames)
with futures.ThreadPoolExecutor(max_workers=workers) as pool:
future_map = {pool.submit(check_safe, username): username for username in usernames}
done = 0
for future in futures.as_completed(future_map):
username = future_map[future]
done += 1
try:
result = future.result()
except Exception as exc: # pragma: no cover - runtime safety
result = CheckResult(
username=username,
telegram_state="unknown",
fragment_state="unknown",
telegram_reason=str(exc),
fragment_reason=str(exc),
)
with lock:
if result.is_free:
free.append(result.username)
status = "FREE"
elif result.telegram_state == "taken" or result.fragment_state == "taken":
taken.append(result.username)
status = "TAKEN"
else:
unknown.append(result.username)
status = "UNKNOWN"
elapsed = time.monotonic() - started
remaining = max(total - done, 0)
per_item = elapsed / done if done else 0.0
eta = format_duration(per_item * remaining)
print(
f"[{done}/{total}] {username:<32} {status:<7} ETA {eta} "
f"(tg: {result.telegram_state}, fr: {result.fragment_state})"
)
print()
print(f"Free: {len(free)}")
print(f"Taken: {len(taken)}")
print(f"Unknown: {len(unknown)}")
if free:
print("\nFree nicknames:")
for username in free:
print(f"@{username}")
return 0
def check_safe(username: str) -> CheckResult:
try:
return check_username(username)
except Exception as exc: # pragma: no cover - runtime safety
return CheckResult(
username=username,
telegram_state="unknown",
fragment_state="unknown",
telegram_reason=str(exc),
fragment_reason=str(exc),
)
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description="Find free Telegram nicknames.")
parser.add_argument("path", nargs="?", type=Path, help="Path to .txt file with nicknames")
parser.add_argument("--workers", type=int, default=4, help="Parallel workers (default: 4)")
return parser
def main(argv: Iterable[str] | None = None) -> int:
parser = build_parser()
args = parser.parse_args(list(argv) if argv is not None else None)
path = args.path.expanduser().resolve() if args.path else None
return run_interactive(path=path, workers=max(1, args.workers))
if __name__ == "__main__":
raise SystemExit(main())