1. urllib 是什么? #

2. 模块结构 #

# 导入请求发送模块
import urllib.request
# 导入 URL 解析与编码模块
import urllib.parse
# 导入网络异常处理模块
import urllib.error

3. 前置知识:HTTP 与 URL #

4. GET 请求 #

# 导入 urllib.request 模块
import urllib.request

# 发送 GET 请求,with 确保连接自动关闭
with urllib.request.urlopen('https://httpbin.org/get') as response:
    # 读取响应体并解码为 UTF-8 字符串
    html = response.read().decode('utf-8')
    # 打印 HTTP 状态码
    print(f"状态码:{response.status}")
    # 打印响应内容前 200 个字符
    print(html[:200])

5. 带查询参数的 GET #

# 导入请求和解析模块
import urllib.request
import urllib.parse

# 定义查询参数字典
params = {'q': 'Python', 'page': 1, 'limit': 10}
# 将字典编码为查询字符串
query = urllib.parse.urlencode(params)
# 拼接完整 URL
url = f'https://httpbin.org/get?{query}'

# 发送带参数的 GET 请求
with urllib.request.urlopen(url) as response:
    print(response.read().decode('utf-8'))

6. POST 请求(JSON) #

# 导入请求模块和 json 模块
import urllib.request
import json

# 将字典转为 JSON 字符串再编码为字节
data = json.dumps({'name': '张三', 'age': 25}).encode('utf-8')

# 构造 POST 请求对象
req = urllib.request.Request(
    'https://httpbin.org/post',
    data=data,
    headers={'Content-Type': 'application/json'},
    method='POST'
)

# 发送请求并打印响应
with urllib.request.urlopen(req) as response:
    print(response.read().decode('utf-8'))

7. 设置请求头与超时 #

# 导入请求和异常模块
import urllib.request
import urllib.error

# 定义自定义请求头
headers = {
    'User-Agent': 'MyApp/1.0',
    'Accept': 'application/json',
    'Authorization': 'Bearer your_token'
}
# 创建带请求头的 Request 对象
req = urllib.request.Request('https://httpbin.org/headers', headers=headers)

try:
    # 发送请求,设置 10 秒超时
    with urllib.request.urlopen(req, timeout=10) as response:
        print(response.read().decode('utf-8'))
except urllib.error.URLError as e:
    print(f"请求失败:{e.reason}")

8. urllib.parse:URL 处理 #

# 导入常用解析函数
from urllib.parse import urlparse, urlencode, quote, unquote, parse_qs

# 解析 URL 为各部分
result = urlparse('https://api.example.com/users?page=1')
# 打印协议、主机名、路径、查询参数
print(result.scheme, result.hostname, result.path, result.query)

# 将字典编码为查询字符串
print(urlencode({'q': 'Python', 'page': 1}))

# 编码特殊字符
encoded = quote('特殊<字符>测试')
print(encoded)
# 解码还原
print(unquote(encoded))

# 解析查询字符串为字典(值为列表)
print(parse_qs('name=张三&hobbies=编程&hobbies=读书'))

9. urllib.error:异常处理 #

# 导入请求和异常模块
import urllib.request
import urllib.error

try:
    # 请求会返回 404 的测试地址
    with urllib.request.urlopen('https://httpbin.org/status/404') as response:
        print(response.read())
except urllib.error.HTTPError as e:
    # 处理 HTTP 状态码错误
    print(f"HTTP 错误:{e.code} - {e.reason}")
except urllib.error.URLError as e:
    # 处理网络层错误
    print(f"URL 错误:{e.reason}")

10. 下载文件 #

# 导入请求模块和 os 模块
import urllib.request
import os

# 定义下载函数
def download_file(url, save_path):
    # 确保目标目录存在
    os.makedirs(os.path.dirname(save_path), exist_ok=True)
    try:
        # 下载文件到本地路径
        urllib.request.urlretrieve(url, save_path)
        print(f'下载成功:{save_path}')
    except Exception as e:
        print(f'下载失败:{e}')

# 下载文件
download_file('https://www.baidu.com', 'html/baidu.html')

11. 常见错误与注意事项 #

# 正确读取响应:先 read 再 decode
content = response.read().decode('utf-8')

# POST 数据必须是字节
data = json.dumps({'key': 'value'}).encode('utf-8')

# 始终设置超时,避免无限等待
urllib.request.urlopen(url, timeout=10)

11. 总结 #

11.1 速查 #

# 导入常用模块
import urllib.request
import urllib.parse
import urllib.error
import json

# GET 请求
with urllib.request.urlopen(url, timeout=10) as resp:
    text = resp.read().decode('utf-8')

# GET 带查询参数
url = f'{base}?{urllib.parse.urlencode(params)}'

# POST JSON
data = json.dumps(payload).encode('utf-8')
req = urllib.request.Request(url, data=data,
    headers={'Content-Type': 'application/json'}, method='POST')
with urllib.request.urlopen(req) as resp:
    result = json.loads(resp.read())

# 下载文件到本地
urllib.request.urlretrieve(url, 'local_file.zip')

13.2 最佳实践 #