Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Validate translate-from-URL fetches against SSRF
download_with_limit (gui.py) and translate() (high_level.py) fetched
caller-supplied URLs server-side with no scheme or address validation, so
a user could make the server request loopback, private-range, or
link-local (cloud metadata) addresses and other internal services; the
core path also followed redirects. When the target returned a PDF its
content was returned to the caller.

Add assert_public_http_url (require http/https and a globally-routable
resolved address) and fetch_public_url (redirects re-validated per hop) in
high_level.py, and apply both to the two URL-fetch sinks.
  • Loading branch information
carfeii committed Sep 11, 2026
commit b6e31b53269e5607a79631485f068a446b7f94bc
20 changes: 18 additions & 2 deletions pdf2zh/gui.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
import uuid
from asyncio import CancelledError
from pathlib import Path
from urllib.parse import urljoin
import typing as T

import gradio as gr
Expand All @@ -16,7 +17,7 @@
import logging

from pdf2zh import __version__
from pdf2zh.high_level import translate
from pdf2zh.high_level import translate, assert_public_http_url
from pdf2zh.doclayout import ModelInstance
from pdf2zh.config import ConfigManager
from pdf2zh.translator import (
Expand Down Expand Up @@ -183,7 +184,22 @@ def download_with_limit(url: str, save_path: str, size_limit: int) -> str:
"""
chunk_size = 1024
total_size = 0
with requests.get(url, stream=True, timeout=10) as response:
# Re-validate every redirect hop so a public URL cannot redirect the
# server-side fetch to an internal address (SSRF).
for _ in range(6):
assert_public_http_url(url)
response = requests.get(url, stream=True, timeout=10, allow_redirects=False)
if response.is_redirect or response.is_permanent_redirect:
location = response.headers.get("Location")
response.close()
if not location:
raise gr.Error("Invalid redirect from URL")
url = urljoin(url, location)
continue
break
else:
raise gr.Error("Too many redirects")
with response:
response.raise_for_status()
content = response.headers.get("Content-Disposition")
try: # filename from header
Expand Down
53 changes: 52 additions & 1 deletion pdf2zh/high_level.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,15 +2,18 @@

import asyncio
import io
import ipaddress
import os
import re
import socket
import sys
import tempfile
import logging
from asyncio import CancelledError
from pathlib import Path
from string import Template
from typing import Any, BinaryIO, List, Optional, Dict
from urllib.parse import urljoin, urlparse

import numpy as np
import requests
Expand All @@ -35,6 +38,54 @@

logger = logging.getLogger(__name__)

_MAX_URL_REDIRECTS = 5


def assert_public_http_url(url: str) -> None:
"""Reject non-http(s) URLs and URLs resolving to a non-public address.

pdf2zh fetches user-supplied URLs server-side (translating a document from a
link). Without this check that is a server-side request forgery primitive:
a caller can target loopback, private-range, link-local (cloud metadata),
or other internal addresses. Every address the host resolves to must be
global.
"""
parsed = urlparse(url)
if parsed.scheme not in ("http", "https"):
raise PDFValueError("Only http/https URLs are allowed")
host = parsed.hostname
if not host:
raise PDFValueError("Invalid URL")
port = parsed.port or (443 if parsed.scheme == "https" else 80)
try:
addrinfo = socket.getaddrinfo(host, port, proto=socket.IPPROTO_TCP)
except socket.gaierror as e:
raise PDFValueError("Could not resolve URL host") from e
for _family, _type, _proto, _canon, sockaddr in addrinfo:
ip = ipaddress.ip_address(sockaddr[0])
mapped = getattr(ip, "ipv4_mapped", None)
if mapped is not None:
ip = mapped
if not ip.is_global or ip.is_reserved:
raise PDFValueError("URL host resolves to a non-public address")


def fetch_public_url(url: str, **kwargs) -> requests.Response:
"""GET *url* with SSRF protection, re-validating every redirect hop."""
for _ in range(_MAX_URL_REDIRECTS + 1):
assert_public_http_url(url)
response = requests.get(url, allow_redirects=False, **kwargs)
if response.is_redirect or response.is_permanent_redirect:
location = response.headers.get("Location")
response.close()
if not location:
raise PDFValueError("Invalid redirect from URL")
url = urljoin(url, location)
continue
return response
raise PDFValueError("Too many redirects")


noto_list = [
"am", # Amharic
"ar", # Arabic
Expand Down Expand Up @@ -445,7 +496,7 @@ def translate(
):
print("Online files detected, downloading...")
try:
r = requests.get(file, allow_redirects=True)
r = fetch_public_url(file)
if r.status_code == 200:
with tempfile.NamedTemporaryFile(
suffix=".pdf", delete=False
Expand Down