Python · 3 分钟阅读
Python:检查 URL 是否可访问
目录
- 1. 标准库
urllib.request - 2.
requests:最常用的同步方案 - 3. 异步:
aiohttp - 4. 进阶:把“健康检查”做成一个可复用的类
- 5. 选型建议
- 6. 常见问题
- 7. 合规与安全
检查 URL 可用性有几种典型场景:单次探活、批量巡检、高并发压测。Python 有
urllib(标准库)、requests(同步事实标准)、httpx / aiohttp(异步)三档
可选。
⚠️ HTTP 200 不等于“业务可用”。如果你要监控真实业务,可能要:
- 检查状态码(200/204)
- 检查响应体关键词
- 解析 JSON 返回值中的业务字段
- 测量响应时间(>2s 视为劣化)
1. 标准库 urllib.request
import urllib.request
import urllib.error
from typing import Iterable
def check_url(url: str, timeout: float = 5.0) -> bool:
"""最朴素的可达性检查:能 connect + 拿到 2xx/3xx 即视为可用。"""
req = urllib.request.Request(
url,
headers={"User-Agent": "Mozilla/5.0 (compatible; health-check/1.0)"},
)
try:
with urllib.request.urlopen(req, timeout=timeout) as resp:
return 200 <= resp.status < 400
except urllib.error.HTTPError as e:
# 4xx/5xx 也算“能访问到服务器”
print(f"HTTP {e.code}: {url}")
return False
except urllib.error.URLError as e:
print(f"unreachable: {url} - {e.reason}")
return False
def check_urls(urls: Iterable[str], timeout: float = 5.0) -> dict[str, bool]:
return {u: check_url(u, timeout) for u in urls}
只用标准库是优点,但功能很基础:没有连接池、不能方便地控制重试、不能
写自定义 header 的 cookie。
2. requests:最常用的同步方案
pip install requests
from typing import Iterable
import requests
DEFAULT_HEADERS = {"User-Agent": "health-check/1.0"}
def check_url(url: str, session: requests.Session | None = None, timeout: float = 5.0) -> bool:
s = session or requests
try:
resp = s.get(url, headers=DEFAULT_HEADERS, timeout=timeout, allow_redirects=True)
return 200 <= resp.status_code < 400
except requests.RequestException as e:
print(f"fail: {url} - {e}")
return False
def check_urls(urls: Iterable[str], timeout: float = 5.0) -> dict[str, bool]:
"""用 Session 共享连接池,比逐个 get 快。"""
with requests.Session() as s:
s.headers.update(DEFAULT_HEADERS)
return {u: check_url(u, s, timeout) for u in urls}
要点:
- 用
Session复用 TCP 连接,比循环requests.get快很多。 timeout=必加,否则一个挂起的请求可能拖死整个程序。allow_redirects=True(默认)能正确处理 301/302。
2.1 加上重试(urllib3 内置)
from requests.adapters import HTTPAdapter
from urllib3.util.retry import Retry
def make_session(retries: int = 3, backoff: float = 0.5) -> requests.Session:
s = requests.Session()
retry = Retry(
total=retries,
backoff_factor=backoff,
status_forcelist=[500, 502, 503, 504],
allowed_methods=["GET", "HEAD"],
)
adapter = HTTPAdapter(max_retries=retry, pool_connections=20, pool_maxsize=20)
s.mount("https://", adapter)
s.mount("http://", adapter)
return s
3. 异步:aiohttp
pip install aiohttp
import asyncio
from typing import Iterable
import aiohttp
async def check_one(session: aiohttp.ClientSession, url: str, timeout: float = 5.0) -> bool:
try:
async with session.get(url, timeout=timeout) as resp:
return 200 <= resp.status < 400
except (aiohttp.ClientError, asyncio.TimeoutError) as e:
print(f"fail: {url} - {e}")
return False
async def check_urls(
urls: Iterable[str],
concurrency: int = 20,
timeout: float = 5.0,
) -> dict[str, bool]:
sem = asyncio.Semaphore(concurrency)
async with aiohttp.ClientSession(
headers={"User-Agent": "health-check/1.0"}
) as session:
async def task(u: str) -> tuple[str, bool]:
async with sem:
return u, await check_one(session, u, timeout)
results = await asyncio.gather(*(task(u) for u in urls))
return dict(results)
if __name__ == "__main__":
urls = ["https://www.python.org", "https://httpbin.org/status/200"]
print(asyncio.run(check_urls(urls)))
要点:
- 共享 一个
ClientSession+ 用Semaphore限制 并发数(避免被对方
拉黑)。 timeout=在ClientSession.get(...)那一层加。asyncio.gather让你同时跑成百上千个 URL。
4. 进阶:把“健康检查”做成一个可复用的类
import time
from dataclasses import dataclass
from typing import Callable
@dataclass
class HealthResult:
url: str
ok: bool
status: int | None
cost_ms: float
error: str | None = None
def health_check(
url: str,
session: requests.Session | None = None,
expected_status: tuple[int, ...] = (200, 204, 301, 302),
timeout: float = 5.0,
) -> HealthResult:
s = session or requests
start = time.perf_counter()
try:
resp = s.get(url, timeout=timeout, allow_redirects=False)
ok = resp.status_code in expected_status
return HealthResult(url, ok, resp.status_code, (time.perf_counter() - start) * 1000)
except requests.RequestException as e:
return HealthResult(url, False, None, (time.perf_counter() - start) * 1000, str(e))
再接入 Prometheus 指标 / 日志 / 告警即可。
5. 选型建议
| 场景 | 推荐 |
|---|---|
| 不想装第三方 | urllib.request |
| 单机脚本 / 中等并发(< 100) | requests + Session + 重试 |
| 高并发 / 与现有异步项目整合 | aiohttp 或 httpx.AsyncClient |
| 需要 HTTP/2 / 强类型 | httpx(同步/异步一套 API) |
6. 常见问题
- 「能 ping 通但 HTTP 失败」? 检查是否被 WAF / CDN 拦截;多发一次带
User-Agent的请求试试。 - HTTPS 证书报错? 升级
certifi;生产请不要全局verify=False。 - 中文 URL? 用
urllib.parse.quote()编码。 - 被反爬? 加
User-Agent、降低并发、加随机time.sleep(0.5~2)。 - 想顺便测 DNS 解析时间?
socket.getaddrinfo或dnspython。
7. 合规与安全
- 遵守
robots.txt和网站的服务条款。 - 不要高频请求 个人或小众站点。
- 不要把用户凭证写进 URL;用
Authorization头。 - 监控数据里不要带 cookie / token 明文。