Skip to content

Commit db83bea

Browse files
refactor: execute get_url_content in sandbox.
1 parent 85684af commit db83bea

1 file changed

Lines changed: 24 additions & 192 deletions

File tree

‎apps/oss/serializers/file.py‎

Lines changed: 24 additions & 192 deletions
Original file line numberDiff line numberDiff line change
@@ -1,12 +1,7 @@
11
# coding=utf-8
2-
import base64
3-
import ipaddress
42
import re
5-
import socket
63
import urllib
7-
from urllib.parse import urlparse, urlunparse
84

9-
import requests
105
import uuid_utils.compat as uuid
116
from django.db.models import QuerySet
127
from django.http import HttpResponse
@@ -166,56 +161,6 @@ def delete(self):
166161
return True
167162

168163

169-
from requests.adapters import HTTPAdapter
170-
171-
172-
class SafeHTTPAdapter(HTTPAdapter):
173-
"""
174-
安全的 HTTP 适配器,防止 DNS 重绑定攻击
175-
在建立连接前验证目标 IP 地址
176-
"""
177-
178-
def send(self, request, **kwargs):
179-
# 解析 URL 获取主机名
180-
parsed_url = urlparse(request.url)
181-
host = parsed_url.hostname
182-
183-
if host:
184-
# 验证目标 IP 是否安全
185-
self._validate_host_ip(host)
186-
187-
return super().send(request, **kwargs)
188-
189-
def _validate_host_ip(self, host: str):
190-
"""验证主机解析的 IP 地址是否安全"""
191-
try:
192-
# 获取所有 IP 地址(包括 IPv4 和 IPv6)
193-
addr_infos = socket.getaddrinfo(host, None, socket.AF_UNSPEC, socket.SOCK_STREAM)
194-
195-
for addr_info in addr_infos:
196-
ip = addr_info[4][0]
197-
if self._is_unsafe_ip(ip):
198-
raise AppApiException(500, _('Access to internal IP addresses is blocked'))
199-
except AppApiException:
200-
raise
201-
except Exception as e:
202-
raise AppApiException(500, _('Failed to resolve host: {error}').format(error=str(e)))
203-
204-
def _is_unsafe_ip(self, ip: str) -> bool:
205-
"""检查 IP 地址是否属于不安全的范围"""
206-
try:
207-
ip_addr = ipaddress.ip_address(ip)
208-
return (
209-
ip_addr.is_private or
210-
ip_addr.is_loopback or
211-
ip_addr.is_reserved or
212-
ip_addr.is_link_local or
213-
ip_addr.is_multicast
214-
)
215-
except Exception:
216-
return True
217-
218-
219164
def get_url_content(url, application_id: str):
220165
application = Application.objects.filter(id=application_id).first()
221166
if application is None:
@@ -225,149 +170,36 @@ def get_url_content(url, application_id: str):
225170
file_limit = 50 * 1024 * 1024
226171
if application.file_upload_setting and application.file_upload_setting.get('fileLimit'):
227172
file_limit = application.file_upload_setting.get('fileLimit') * 1024 * 1024
228-
parsed = validate_url(url)
229-
230-
# 创建带有安全检查的 session
231-
session = requests.Session()
232-
safe_adapter = SafeHTTPAdapter()
233-
session.mount('http://', safe_adapter)
234-
session.mount('https://', safe_adapter)
235-
236173
try:
237-
response = session.get(
238-
url,
239-
timeout=3,
240-
allow_redirects=False
241-
)
242-
finally:
243-
session.close()
244-
245-
final_host = urlparse(response.url).hostname
246-
if is_private_ip(final_host):
247-
raise ValueError("Blocked unsafe redirect to internal host")
248-
# 判断文件大小
249-
if int(response.headers.get('Content-Length', 0)) > file_limit:
250-
raise AppApiException(500, _('File size exceeds limit'))
251-
# 返回状态码 响应内容大小 响应的contenttype 还有字节流
174+
from common.utils.tool_code import ToolExecutor
175+
response = ToolExecutor().exec_code(
176+
"""
177+
def get_url_content(url):
178+
import requests
179+
requests.packages.urllib3.disable_warnings()
180+
response = requests.get(url, verify=False, allow_redirects=False)
252181
content_type = response.headers.get('Content-Type', '')
253-
# 根据内容类型决定如何处理
254182
if 'text' in content_type or 'json' in content_type:
255183
content = response.text
256184
else:
257-
# 二进制内容使用Base64编码
185+
import base64
258186
content = base64.b64encode(response.content).decode('utf-8')
259-
260187
return {
261-
'status_code': response.status_code,
262-
'Content-Length': response.headers.get('Content-Length', 0),
263-
'Content-Type': content_type,
264-
'content': content,
188+
"status_code": response.status_code,
189+
"Content-Type": content_type,
190+
"Content-Length": response.headers.get('Content-Length', 0),
191+
"content": content,
265192
}
266-
267-
268-
def is_private_ip(host: str) -> bool:
269-
"""检测 IP 是否属于内网、环回、云 metadata 的危险地址"""
270-
try:
271-
ip = ipaddress.ip_address(socket.gethostbyname(host))
272-
return (
273-
ip.is_private or
274-
ip.is_loopback or
275-
ip.is_reserved or
276-
ip.is_link_local or
277-
ip.is_multicast
193+
""",
194+
{"url": url}
278195
)
279-
except Exception:
280-
return True
281-
282-
283-
def validate_and_normalize_url(url: str) -> str:
284-
"""
285-
严格验证并规范化 URL,防止 URL 解析绕过攻击
286-
287-
防御场景:
288-
- http://127.0.0.1:6666\@1.1.1.1/ (反斜杠绕过)
289-
- http://127.0.0.1:6666@1.1.1.1/ (认证信息混淆)
290-
- http://1.1.1.1#@127.0.0.1:6666/ (片段注入)
291-
"""
292-
if not url:
293-
raise ValueError("URL is required")
294-
295-
# 1. 拒绝包含危险字符的 URL
296-
dangerous_patterns = [
297-
r'\\', # 反斜杠
298-
r'\s', # 空白字符
299-
r'%00', # 空字节
300-
r'%0a', # 换行符
301-
r'%0d', # 回车符
302-
]
303-
304-
url_lower = url.lower()
305-
for pattern in dangerous_patterns:
306-
if re.search(pattern, url_lower):
307-
raise ValueError("URL contains dangerous characters")
308-
309-
# 2. 解析 URL
310-
parsed = urlparse(url)
311-
312-
# 3. 仅允许 http / https
313-
if parsed.scheme not in ("http", "https"):
314-
raise ValueError("Only http and https are allowed")
315-
316-
# 4. 提取主机名(从 netloc 中)
317-
netloc = parsed.netloc
318-
319-
# 5. 如果 netloc 中包含 @,说明有认证信息,需要特别处理
320-
if '@' in netloc:
321-
# 分离认证信息和主机
322-
auth_part, host_part = netloc.rsplit('@', 1)
323-
324-
# 检查认证部分是否包含危险的 IP 或端口信息
325-
# 攻击者可能在认证部分放置内网地址
326-
if ':' in auth_part or '.' in auth_part:
327-
raise ValueError("Authentication part contains suspicious content")
328-
329-
# 使用真实的主机部分
330-
actual_host = host_part.split(':')[0] if ':' in host_part else host_part
331-
else:
332-
# 没有认证信息,直接提取主机
333-
actual_host = parsed.hostname
334-
335-
# 6. 验证主机名不为空
336-
if not actual_host:
337-
raise ValueError("Invalid URL: missing hostname")
338-
339-
# 7. 验证主机不是 IP 地址形式的内网地址
340-
# 这样可以防止直接在 URL 中使用内网 IP
341-
try:
342-
# 尝试解析为 IP 地址
343-
ip_addr = ipaddress.ip_address(actual_host)
344-
if is_private_ip(actual_host):
345-
raise ValueError("Access to internal IP addresses is blocked")
346-
except ValueError as e:
347-
# 如果不是 IP 地址(是域名),则继续检查
348-
if "internal IP" in str(e):
349-
raise
350-
# 对于域名,检查其解析结果
351-
if is_private_ip(actual_host):
352-
raise ValueError("Access to internal IP addresses is blocked")
353-
354-
# 8. 重新构建干净的 URL,移除可能的认证信息
355-
clean_netloc = actual_host
356-
if parsed.port:
357-
clean_netloc = f"{actual_host}:{parsed.port}"
358-
359-
clean_url = urlunparse((
360-
parsed.scheme,
361-
clean_netloc,
362-
parsed.path,
363-
parsed.params,
364-
parsed.query,
365-
'' # 移除 fragment,防止片段注入
366-
))
367-
368-
return clean_url
369-
370-
371-
def validate_url(url: str):
372-
"""验证 URL 是否安全(保留向后兼容)"""
373-
return validate_and_normalize_url(url)
196+
except Exception as e:
197+
raise AppApiException(500, str(e))
198+
if int(response.get('Content-Length')) > file_limit:
199+
raise AppApiException(500, _('File size exceeds limit'))
200+
return {
201+
'status_code': response.get('status_code'),
202+
'Content-Type': response.get('Content-Type'),
203+
'Content-Length': response.get('Content-Length'),
204+
'content': response.get('content'),
205+
}

0 commit comments

Comments
 (0)