Repository navigation
Expand file tree
/
Copy pathhttp_utils.py
More file actions
185 lines (143 loc) · 4.61 KB
/
Copy pathhttp_utils.py
File metadata and controls
185 lines (143 loc) · 4.61 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
# -*- coding: utf-8 -*-
"""
HTTP fetching and disk caching utilities.
"""
import gzip
import hashlib
import os
import struct
import urllib.error
import urllib.request
import zlib
from io import BytesIO
import config_data
URLOpener = urllib.request.build_opener()
URLOpener.addheaders = [
("Cache-Control", "no-transform"),
("User-Agent", config_data.http_user_agent),
]
def fetch_url(url, timeout=30, max_size=100 * 1024 * 1024):
"""
Fetch URL with timeout and size limit.
Args:
url: URL to fetch
timeout: Connection timeout in seconds
max_size: Maximum download size in bytes
Returns:
(content, base_href, headers, info) tuple
Raises:
urllib.error.URLError: On network errors
ValueError: If content exceeds max_size
"""
try:
httpcon = URLOpener.open(url, timeout=timeout)
content = httpcon.read(max_size)
extra = httpcon.read(1)
if extra:
httpcon.close()
raise ValueError(f"Content exceeds max size of {max_size} bytes")
httpcon.close()
except urllib.error.URLError as e:
print(f"Failed to fetch {url}")
if hasattr(e, "reason"):
print("Reason:", e.reason)
elif hasattr(e, "code"):
print("Error code:", e.code)
raise
base_href = httpcon.geturl()
headers = dict((k.lower(), v) for k, v in httpcon.headers.items())
info = httpcon.info()
if content and "gzip" in headers.get("content-encoding", ""):
try:
content = gzip.GzipFile(fileobj=BytesIO(content)).read()
except (EOFError, IOError, struct.error):
content = None
elif content and "deflate" in headers.get("content-encoding", ""):
try:
content = zlib.decompress(content)
except zlib.error:
try:
content = zlib.decompress(content, -15)
except zlib.error:
content = None
return (content, base_href, headers, info)
def fetch_and_decode_url(url, timeout=30, max_size=100 * 1024 * 1024):
"""
Fetch URL and extract mimetype information.
Args:
url: URL to fetch
timeout: Connection timeout in seconds
max_size: Maximum download size in bytes
Returns:
(content, base_href, headers, info, mimetype, subtype) tuple
"""
content, base_href, headers, info = fetch_url(url, timeout, max_size)
mimetype = headers.get("content-type", "application/octet-stream")
reststring = mimetype.split("/", 2)[1] if "/" in mimetype else ""
subtype = reststring.split(";", 2)[0]
return (content, base_href, headers, info, mimetype, subtype)
def get_filename_from_headers_or_url(headers, url):
"""
Extract filename from Content-Disposition header or URL.
Args:
headers: Dict of HTTP headers
url: Fallback URL to extract filename from
Returns:
Filename string
"""
if "content-disposition" in headers:
cd = headers["content-disposition"]
if "filename=" in cd:
start = cd.index("filename=") + 9
end = cd.find(";", start)
filename = cd[start:end] if end > 0 else cd[start:]
return filename.strip("\"'")
name = (
url.replace("http://", "")
.replace("https://", "")
.replace("/", " ")
.replace(".", " ")
)
return os.path.basename(name) if name else "download"
def get_cached(url):
"""
Get content from cache if available.
Args:
url: URL to check
Returns:
Cached content or None if not cached
"""
cache_key = hashlib.sha256(url.encode("utf-8")).hexdigest()
cache_path = os.path.join(config_data.cache_prefix, cache_key)
if os.path.isfile(cache_path):
with open(cache_path, "rb") as f:
return f.read()
return None
def store_cached(url, content):
"""
Store content in cache.
Args:
url: URL key
content: Content to cache
"""
os.makedirs(config_data.cache_prefix, exist_ok=True)
cache_key = hashlib.sha256(url.encode("utf-8")).hexdigest()
cache_path = os.path.join(config_data.cache_prefix, cache_key)
with open(cache_path, "wb") as f:
f.write(content)
def get_or_fetch(url, timeout=30):
"""
Get content from cache or fetch and cache.
Args:
url: URL to fetch
timeout: Connection timeout in seconds
Returns:
Content bytes
"""
cached = get_cached(url)
if cached:
return cached
content, _, _, _ = fetch_url(url, timeout=timeout)
if content:
store_cached(url, content)
return content