Скрипт читає TXT зі списком адрес, перевіряє їх паралельно і зберігає CSV (адреса, статус, час, помилка). Нижче реальний прогін на 15 адресах.
check_sites.pyreport_example.csvlinks_example.txtREADME.md
| url | status | ms | error | final_url |
|---|---|---|---|---|
| https://example.com | 200 | 174 | https://example.com | |
| https://www.python.org | 200 | 180 | https://www.python.org | |
| https://github.com | 200 | 165 | https://github.com | |
| https://uk.wikipedia.org | 403 | 171 | HTTP 403 | https://uk.wikipedia.org |
| https://httpbin.org/status/404 | 404 | 964 | HTTP 404 | https://httpbin.org/status/404 |
| https://httpbin.org/status/500 | 500 | 1100 | HTTP 500 | https://httpbin.org/status/500 |
| https://httpbin.org/redirect/2 | 200 | 1594 | https://httpbin.org/get | |
| http://github.com | 200 | 250 | https://github.com/ | |
| https://httpbin.org/delay/15 | 8975 | timeout | ||
| https://this-domain-does-not-exist-12345.com | 27 | dns: домен не знайдено | ||
| https://expired.badssl.com | 368 | ssl: проблема із сертифікатом | ||
| not a valid url | 0 | некоректний URL | ||
| example.org | 200 | 232 | https://example.org | |
| https://www.google.com | 200 | 244 | https://www.google.com | |
| https://www.freelancehunt.com | 403 | 213 | HTTP 403 | https://www.freelancehunt.com |
# check_sites.py — перевірка доступності сайтів 1. Встановіть залежність: `pip install aiohttp` (Python 3.8+). 2. Список адрес — у TXT, по одному URL на рядок (рядки з `#` і порожні ігноруються; без `https://` схема додається сама). 3. Запуск: `python check_sites.py --input links_example.txt --output report.csv --workers 10 --timeout 10` 4. Звіт CSV: `url, status, response_time_ms, error, final_url` — редіректи проходяться автоматично, `final_url` показує, куди привели; недоступний сайт (timeout, DNS, SSL, 4xx/5xx) дає рядок з помилкою і не зупиняє решту. 5. `report_example.csv` — реальний прогін на 15 адресах (живі, 404, 500, редірект, timeout, DNS, SSL). Частина сайтів може віддавати 403 автоматичним запитам — це теж чесно потрапляє у звіт.
#!/usr/bin/env python3
"""Перевірка доступності сайтів зі списку (TXT) -> CSV.
Запуск:
python check_sites.py --input links.txt --output report.csv --workers 20 --timeout 10
Потрібно: Python 3.8+ та aiohttp (pip install aiohttp).
Колонки CSV: url, status, response_time_ms, error, final_url
"""
import argparse
import asyncio
import csv
import re
import sys
import time
import aiohttp
DEFAULT_UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/124.0 Safari/537.36")
def read_urls(path):
with open(path, encoding="utf-8-sig") as f:
lines = [l.strip() for l in f]
return [l for l in lines if l and not l.startswith("#")]
def normalize(url):
return url if "://" in url else "https://" + url # example.com -> https://example.com
def describe(exc):
"""Коротке людське пояснення помилки."""
if isinstance(exc, asyncio.TimeoutError):
return "timeout"
if isinstance(exc, aiohttp.TooManyRedirects):
return "забагато редіректів"
if isinstance(exc, aiohttp.InvalidURL):
return "некоректний URL"
if isinstance(exc, aiohttp.ClientConnectorCertificateError) or \
isinstance(exc, aiohttp.ClientSSLError):
return "ssl: проблема із сертифікатом"
if isinstance(exc, aiohttp.ClientConnectorDNSError):
return "dns: домен не знайдено"
if isinstance(exc, aiohttp.ClientConnectorError):
return "connection: %s" % str(exc)[:100]
return "%s: %s" % (type(exc).__name__, str(exc)[:100])
async def check(session, sem, url, timeout):
"""Один URL -> рядок звіту. Будь-який виняток стає рядком з помилкою."""
row = {"url": url, "status": "", "response_time_ms": "", "error": "", "final_url": ""}
async with sem:
start = time.perf_counter()
try:
full = normalize(url)
if re.search(r"\s", full):
raise aiohttp.InvalidURL(full)
# total=timeout: жорсткий ліміт на весь запит, разом з редіректами
async with session.get(full, allow_redirects=True, max_redirects=10,
timeout=aiohttp.ClientTimeout(total=timeout)) as r:
row["status"] = r.status
row["final_url"] = str(r.url)
if r.status >= 400:
row["error"] = "HTTP %d" % r.status
except Exception as e: # один збій не валить решту
row["error"] = describe(e)
row["response_time_ms"] = int((time.perf_counter() - start) * 1000)
return row
async def run(urls, workers, timeout, ua):
sem = asyncio.Semaphore(workers)
conn = aiohttp.TCPConnector(limit=0)
async with aiohttp.ClientSession(headers={"User-Agent": ua}, connector=conn) as session:
return await asyncio.gather(*(check(session, sem, u, timeout) for u in urls))
def main():
ap = argparse.ArgumentParser(description="Перевірка доступності сайтів")
ap.add_argument("--input", required=True, help="TXT: один URL на рядок")
ap.add_argument("--output", default="report.csv", help="CSV зі звітом")
ap.add_argument("--workers", type=int, default=10, help="одночасних запитів (10)")
ap.add_argument("--timeout", type=float, default=10, help="таймаут на сайт, сек (10)")
ap.add_argument("--user-agent", default=DEFAULT_UA, help="User-Agent (за замовчуванням браузерний)")
a = ap.parse_args()
try:
urls = read_urls(a.input)
except OSError as e:
sys.exit("Не вдалося прочитати файл: %s" % e)
if not urls:
sys.exit("У файлі немає URL")
rows = asyncio.run(run(urls, max(1, a.workers), a.timeout, a.user_agent)) # порядок як у файлі
cols = ["url", "status", "response_time_ms", "error", "final_url"]
with open(a.output, "w", newline="", encoding="utf-8-sig") as f:
w = csv.DictWriter(f, fieldnames=cols)
w.writeheader()
w.writerows(rows)
ok = sum(1 for r in rows if r["status"] != "" and r["status"] < 400)
print("Перевірено: %d, доступні: %d, з помилками: %d -> %s"
% (len(rows), ok, len(rows) - ok, a.output))
if __name__ == "__main__":
main()