Files
jacuzzi/backend/main.py
Simon d508263946 Don't re-resolve a favorite whose URL is already the media file
Some channels hand back the media URL itself as an item's url. Since the
favorites fix, opening one of those sent it to /api/resolve first, so
yt-dlp fetched the media just to report the URL we already had. On a
signed link (`?secure=<ts>-<token>`) that is a second request against
something that may be single-use or IP-bound, and the request that
matters -- the playback fetch -- is then refused. Such URLs now play
directly, with no resolve round trip, as they did before.

Alongside that, three things that make expiry survivable:

/api/stream, after its existing referer-less retry, now retries a 403
completely bare (Range only). Signed CDN links are routinely served to a
plain browser request and refused when it carries extras -- a
`Sec-Fetch-Mode: navigate` on a media subresource, say, which is what
yt-dlp's generic extractor hands back and no real player would send.

When every source fails, the player re-resolves once and retries instead
of giving up, since the likeliest cause is that signed URLs went stale in
a long-open tab rather than the video being gone. A manual quality pick
is dropped for that retry, as it names one of the URLs that just failed.

Favorites stored by older versions still carry a `meta` blob of resolved
formats, long expired; it's now stripped on read so nothing can reach for
one.

Verified: a favorite whose url is a .mp4 plays with zero /api/resolve
calls, straight from that URL; playback, prefetch, feed paging, HUD,
rotation, momentum and the version check all still pass.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01QBDkEXP4htyXTCZUwMLphd
2026-09-05 20:41:31 +00:00

1090 lines
47 KiB
Python

from flask import Flask, request, Response, send_from_directory, jsonify
import os
import re
import requests
from flask_cors import CORS
import urllib.parse
from requests.adapters import HTTPAdapter
from urllib3.util import Retry
import yt_dlp
from yt_dlp.networking.impersonate import ImpersonateTarget
from curl_cffi import requests as impersonate_requests
import threading
import io
import time
import hashlib
from urllib.parse import urljoin
# Browser to impersonate at the TLS/HTTP layer. Some origins (e.g. the
# "animeidhentai" hottub channel) fingerprint clients and reset/403 anything
# that isn't a real browser, so impersonation must be on by default.
IMPERSONATE_TARGET = os.getenv('STREAM_IMPERSONATE', 'chrome').strip() or 'chrome'
# curl_cffi sessions wrap a single libcurl handle and are not safe to share
# across threads; keep one per worker thread so the Flask `threaded=True`
# server can proxy concurrent segments without corrupting state.
_thread_local = threading.local()
def get_impersonate_session():
sess = getattr(_thread_local, 'session', None)
if sess is None:
sess = impersonate_requests.Session(impersonate=IMPERSONATE_TARGET)
_thread_local.session = sess
return sess
def _is_tls_verify_error(err):
message = str(err).lower()
return 'certificate' in message or 'curl: (60)' in message or 'ssl: ' in message
# Hosts already proven to fail certificate verification. A video is fetched in
# many range requests, so remembering the host keeps us from paying for a
# doomed TLS handshake on every one of them.
_tls_unverified_hosts = set()
def impersonate_get(url, **kwargs):
"""Upstream GET that survives an origin with a broken certificate.
Some media hosts serve expired certs (heavyfetish's stNN CDN, for one), which
a browser refuses outright -- part of why this proxy exists. The viewer's
connection to *us* stays verified either way, so rather than failing the
stream we retry once with verification off, and say so in the log. Set
STREAM_TLS_VERIFY_ONLY=1 to keep the hard failure instead."""
host = urllib.parse.urlparse(url).netloc
if host in _tls_unverified_hosts:
return get_impersonate_session().get(url, verify=False, **kwargs)
try:
return get_impersonate_session().get(url, **kwargs)
except Exception as err:
strict = os.getenv('STREAM_TLS_VERIFY_ONLY', '').strip().lower() in ('1', 'true', 'yes', 'on')
if strict or not _is_tls_verify_error(err):
raise
app.logger.warning("[stream] TLS verification failed for %s (%s); retrying unverified", url, err)
_tls_unverified_hosts.add(host)
return get_impersonate_session().get(url, verify=False, **kwargs)
# Stream params that have dedicated meaning and must never be treated as headers.
# `referer` is mapped to a real Referer header by collect_passthrough_headers, but
# `live` is purely a playback hint and must not leak upstream as a header.
STREAM_RESERVED_PARAMS = {'url', 'live'}
# Headers that affect the transport layer rather than the resource itself; allowing
# these to be forwarded could enable request smuggling or vhost-routing abuse.
STREAM_DISALLOWED_HEADER_NAMES = {'host', 'content-length', 'transfer-encoding', 'connection', 'expect'}
# Headers curl_cffi sets coherently for the impersonated browser. Forwarding the
# client's (or extractor's) own values for these would contradict the spoofed TLS
# fingerprint and defeat impersonation, so they are never relayed upstream.
STREAM_IMPERSONATION_MANAGED_HEADERS = {
'user-agent', 'accept', 'accept-encoding', 'accept-language',
'sec-ch-ua', 'sec-ch-ua-mobile', 'sec-ch-ua-platform',
}
# RFC 7230 token charset for header field-names.
HEADER_NAME_RE = re.compile(r"^[!#$%&'*+\-.^_`|~0-9A-Za-z]+$")
# Reject control characters (CR/LF/NUL etc.) that could be used for header injection.
HEADER_VALUE_BAD_CHARS_RE = re.compile(r'[\x00-\x08\x0a-\x1f\x7f]')
MAX_HEADER_VALUE_LENGTH = 4096
def collect_passthrough_headers(source):
"""Treat any request param other than the reserved ones as an HTTP header to
forward to yt-dlp/upstream. Validates names and values to prevent header
injection (CRLF splitting) and disallows transport-level headers."""
headers = {}
if not source:
return headers
for key in source:
if key.lower() in STREAM_RESERVED_PARAMS:
continue
if key.lower() in STREAM_DISALLOWED_HEADER_NAMES:
continue
if not HEADER_NAME_RE.match(key):
continue
value = source.get(key)
if value is None:
continue
value = str(value)
if not value or len(value) > MAX_HEADER_VALUE_LENGTH:
continue
if HEADER_VALUE_BAD_CHARS_RE.search(value):
continue
header_name = 'Referer' if key.lower() == 'referer' else key
headers[header_name] = value
return headers
# Serve frontend static files under `/static` to avoid colliding with API routes
app = Flask(__name__, static_folder='../frontend', static_url_path='/static')
app.url_map.strict_slashes = False
# Use flask-cors for API routes
CORS(app, resources={r"/api/*": {"origins": "*"}})
# Configure a requests session with retries
session = requests.Session()
retries = Retry(total=2, backoff_factor=0.2, status_forcelist=(500, 502, 503, 504))
adapter = HTTPAdapter(max_retries=retries)
session.mount('http://', adapter)
session.mount('https://', adapter)
@app.route('/api/status', methods=['POST', 'GET'])
def proxy_status():
if request.method == 'POST':
# Safely get the json body
client_data = request.get_json() or {}
target_server = client_data.get('server')
else:
target_server = request.args.get('server')
if not target_server:
return jsonify({"error": "No server provided"}), 400
if target_server.endswith('/'):
target_server = target_server[:-1]
target_server = f"{target_server.strip()}/api/status"
# Validate target URL
parsed = urllib.parse.urlparse(target_server)
if parsed.scheme not in ('http', 'https') or not parsed.netloc:
return jsonify({"error": "Invalid target URL"}), 400
try:
# Forward a small set of safe request headers
safe_request_headers = {}
for k in ('User-Agent', 'Accept', 'Accept-Encoding', 'Accept-Language', 'Range'):
if k in request.headers:
safe_request_headers[k] = request.headers[k]
# Remove hop-by-hop request headers per RFC
for hop in ('Connection', 'Keep-Alive', 'Proxy-Authenticate', 'Proxy-Authorization', 'TE', 'Trailers', 'Transfer-Encoding', 'Upgrade'):
safe_request_headers.pop(hop, None)
# Stream the GET via a session with small retry policy
resp = session.get(target_server, headers=safe_request_headers, timeout=5, stream=True)
hop_by_hop = {
'connection', 'keep-alive', 'proxy-authenticate', 'proxy-authorization',
'te', 'trailers', 'transfer-encoding', 'upgrade'
}
forwarded_headers = []
for name, value in resp.headers.items():
if name.lower() in hop_by_hop:
continue
if name.lower() == 'content-length':
# Let Flask set Content-Length if needed for the assembled response
continue
forwarded_headers.append((name, value))
def generate():
try:
for chunk in resp.iter_content(1024 * 16):
if chunk:
yield chunk
finally:
resp.close()
return Response(generate(), status=resp.status_code, headers=forwarded_headers)
except Exception as e:
return jsonify({"error": str(e)}), 500
@app.route('/api/videos', methods=['POST'])
def videos_proxy():
client_data = request.get_json() or {}
target_server = client_data.get('server')
client_data.pop('server', None) # Remove server from payload
if not target_server:
return jsonify({"error": "No server provided"}), 400
if target_server.endswith('/'):
target_server = target_server[:-1]
target_server = f"{target_server.strip()}/api/videos"
# Validate target URL
parsed = urllib.parse.urlparse(target_server)
if parsed.scheme not in ('http', 'https') or not parsed.netloc:
return jsonify({"error": "Invalid target URL"}), 400
try:
resp = session.post(target_server, json=client_data,timeout=5)
return Response(resp.content, status=resp.status_code, content_type=resp.headers.get('Content-Type', 'application/json'))
except Exception as e:
return jsonify({"error": str(e)}), 500
# Cache of yt-dlp metadata resolutions so the on-demand probe endpoint doesn't
# re-extract the same video on every hover/scroll. Signed media URLs expire, so
# entries are intentionally short-lived.
RESOLVE_CACHE_TTL = 300
_resolve_cache = {}
_resolve_cache_lock = threading.Lock()
# Per-format fields the frontend needs to rank formats and build stream/probe
# URLs (see resolveStreamSources/rankFormats in videos.js). Everything else in a
# yt-dlp format dict is dropped to keep the payload small.
_RESOLVE_FORMAT_FIELDS = ('url', 'http_headers', 'height', 'width', 'tbr',
'fps', 'vcodec', 'acodec', 'ext', 'video_ext', 'quality')
# Some channels surface pages that yt-dlp can't extract because the video is
# embedded in a third-party JS player iframe (e.g. the xtremestream family used
# by tube.perverzija.com). The player page declares its HLS playlist URL as
# `m3u8_loader_url + video_id`, so we scrape the page for the embed iframe, then
# read those two variables out of the player to reconstruct the stream URL.
_EMBED_IFRAME_RE = re.compile(r'''<iframe[^>]+src=["']([^"']+)''', re.I)
_EMBED_LOADER_RE = re.compile(r'''m3u8_loader_url\s*=\s*[`'"]([^`'"]+)[`'"]''')
_EMBED_VIDEOID_RE = re.compile(r'''video_id\s*=\s*[`'"]([^`'"]+)[`'"]''')
def resolve_unsupported_embed(page_url):
"""Best-effort resolver for iframe-embedded JS players yt-dlp can't handle.
Returns an info-like dict ({'url','formats','http_headers','is_live'}) whose
single format is the embed's HLS playlist, or None if nothing was found."""
try:
sess = get_impersonate_session()
page = sess.get(page_url, headers={'Referer': page_url}, timeout=15)
embed_url = None
for src in _EMBED_IFRAME_RE.findall(page.text):
candidate = urljoin(page_url, src)
if '/player/' in candidate or 'index.php?data=' in candidate:
embed_url = candidate
break
if not embed_url:
return None
player = sess.get(embed_url, headers={'Referer': page_url}, timeout=15)
loader = _EMBED_LOADER_RE.search(player.text)
video_id = _EMBED_VIDEOID_RE.search(player.text)
if not (loader and video_id):
return None
stream_url = loader.group(1) + video_id.group(1)
parsed = urllib.parse.urlparse(embed_url)
referer = f"{parsed.scheme}://{parsed.netloc}/"
headers = {'Referer': referer}
return {
'url': stream_url,
'is_live': False,
'http_headers': headers,
'formats': [{'url': stream_url, 'ext': 'm3u8', 'protocol': 'm3u8', 'http_headers': headers}],
}
except Exception:
return None
@app.route('/api/resolve', methods=['POST', 'GET'])
def resolve_video():
"""Resolve a page URL to its playable formats via yt-dlp and return them as
JSON. The frontend calls this on demand (when a card is hovered or scrolled
into view) to learn the real media URLs so it can background-probe them for
direct, proxy-free playability."""
if request.method == 'POST':
source = request.json or {}
video_url = source.get('url')
else:
source = request.args
video_url = request.args.get('url')
if not video_url:
return jsonify({"error": "No URL provided"}), 400
now = time.time()
with _resolve_cache_lock:
cached = _resolve_cache.get(video_url)
if cached and cached[0] > now:
return jsonify(cached[1])
ydl_opts = {
'quiet': True,
'no_warnings': True,
'skip_download': True,
# Match /api/stream so the resolved formats reflect what playback will
# actually fetch from fingerprinting origins.
'impersonate': ImpersonateTarget.from_str(IMPERSONATE_TARGET),
}
passthrough_headers = collect_passthrough_headers(source)
if passthrough_headers:
ydl_opts['http_headers'] = passthrough_headers
try:
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
info = ydl.extract_info(video_url, download=False)
except Exception as e:
# Many channels point at sites yt-dlp can't extract ("Unsupported URL").
# That's not fatal here -- the embed fallback below may still find a
# stream, and otherwise we return empty formats so playback falls back to
# the proxy.
app.logger.info("[resolve] yt-dlp could not extract %s: %s", video_url, e)
info = None
# Fall back to scraping iframe-embedded JS players yt-dlp doesn't support.
if not (info and (info.get('formats') or info.get('url'))):
embed = resolve_unsupported_embed(video_url)
if embed:
info = embed
formats = []
for fmt in ((info.get('formats') if info else None) or []):
if not fmt.get('url'):
continue
formats.append({k: fmt.get(k) for k in _RESOLVE_FORMAT_FIELDS if fmt.get(k) is not None})
result = {
'url': info.get('url') if info else None,
'http_headers': (info.get('http_headers') if info else None) or {},
'isLive': bool(info.get('is_live')) if info else False,
'formats': formats,
}
with _resolve_cache_lock:
# Drop expired entries so the cache doesn't grow without bound.
for key in [k for k, v in _resolve_cache.items() if v[0] <= now]:
_resolve_cache.pop(key, None)
_resolve_cache[video_url] = (now + RESOLVE_CACHE_TTL, result)
return jsonify(result)
@app.route('/api/image', methods=['GET', 'HEAD'])
def image_proxy():
image_url = request.args.get('url')
if not image_url:
return jsonify({"error": "No URL provided"}), 400
parsed = urllib.parse.urlparse(image_url)
if parsed.scheme not in ('http', 'https') or not parsed.netloc:
return jsonify({"error": "Invalid target URL"}), 400
try:
safe_request_headers = {}
for k in ('User-Agent', 'Accept', 'Accept-Encoding', 'Accept-Language'):
if k in request.headers:
safe_request_headers[k] = request.headers[k]
resp = session.get(image_url, headers=safe_request_headers, stream=True, timeout=15, allow_redirects=True)
hop_by_hop = {
'connection', 'keep-alive', 'proxy-authenticate', 'proxy-authorization',
'te', 'trailers', 'transfer-encoding', 'upgrade'
}
forwarded_headers = []
for name, value in resp.headers.items():
if name.lower() in hop_by_hop:
continue
forwarded_headers.append((name, value))
if request.method == 'HEAD':
resp.close()
return Response("", status=resp.status_code, headers=forwarded_headers)
def generate():
try:
for chunk in resp.iter_content(1024 * 16):
if chunk:
yield chunk
finally:
resp.close()
return Response(generate(), status=resp.status_code, headers=forwarded_headers)
except Exception as e:
return jsonify({"error": str(e)}), 500
# Captures the path *relative to the frontend dir* (the manifest's key), since
# the served URL carries an extra `static/` prefix.
_ASSET_REF_RE = re.compile(r'(src|href)="static/((?:js|css)/[^"?#]+)"')
@app.route('/')
def index():
"""Serve index.html with each local asset URL stamped with its content hash.
index.html itself is always revalidated, but the assets it names are not
under our control once a CDN or a phone has them: Cloudflare rewrites our
`no-cache` to `max-age=14400`, and an iOS home-screen app will happily run
four-hour-old JavaScript. A content hash in the query gives every deploy new
URLs, which no cache can satisfy from an old copy -- so a reload always
lands on the build that's actually deployed."""
hashes = _version_payload().get('files', {})
def stamp(match):
attr, rel = match.group(1), match.group(2)
digest = hashes.get(rel)
return f'{attr}="static/{rel}?v={digest}"' if digest else match.group(0)
try:
with open(os.path.join(_FRONTEND_DIR, 'index.html'), encoding='utf-8') as fh:
html = _ASSET_REF_RE.sub(stamp, fh.read())
except OSError:
return send_from_directory(app.static_folder, 'index.html')
resp = Response(html, mimetype='text/html')
resp.headers['Cache-Control'] = 'no-cache, must-revalidate'
return resp
@app.route('/favicon.ico')
def favicon():
return send_from_directory(app.static_folder, 'favicon.ico')
# --- Frontend asset version tracking -------------------------------------
# The client polls /api/version and, when a tracked file's content hash
# changes, hot-swaps CSS in place or reloads the page. This lets a deploy
# reach already-open tabs without a manual refresh.
_FRONTEND_DIR = os.path.abspath(app.static_folder)
_VERSION_EXTS = ('.html', '.css', '.js')
_version_cache = {'mtime': None, 'payload': None}
_version_lock = threading.Lock()
def _scan_frontend_files():
"""Map served relative paths -> absolute paths for tracked frontend files."""
files = {}
for root, _dirs, names in os.walk(_FRONTEND_DIR):
for name in names:
if os.path.splitext(name)[1].lower() not in _VERSION_EXTS:
continue
path = os.path.join(root, name)
rel = os.path.relpath(path, _FRONTEND_DIR).replace(os.sep, '/')
files[rel] = path
return files
def _compute_version_payload(files):
"""Hash each tracked file's contents plus a combined version fingerprint."""
file_hashes = {}
combined = hashlib.md5()
for rel in sorted(files):
try:
with open(files[rel], 'rb') as fh:
digest = hashlib.md5(fh.read()).hexdigest()
except OSError:
continue
file_hashes[rel] = digest
combined.update(rel.encode('utf-8'))
combined.update(digest.encode('utf-8'))
return {'version': combined.hexdigest(), 'files': file_hashes}
def _version_payload():
files = _scan_frontend_files()
# Use the newest mtime across tracked files as a cheap cache key so frequent
# polls only re-hash contents when something on disk actually changed.
try:
latest_mtime = max((os.path.getmtime(p) for p in files.values()), default=0)
except OSError:
latest_mtime = 0
with _version_lock:
if _version_cache['mtime'] != latest_mtime or _version_cache['payload'] is None:
_version_cache['payload'] = _compute_version_payload(files)
_version_cache['mtime'] = latest_mtime
return _version_cache['payload']
@app.route('/api/version', methods=['GET'])
def frontend_version():
resp = jsonify(_version_payload())
resp.headers['Cache-Control'] = 'no-store'
return resp
@app.route('/api/stream', methods=['POST', 'GET', 'HEAD'])
def stream_video():
# Note: <video> tags perform GET. To support your POST requirement,
# we handle the URL via JSON post or URL params.
debug_param = os.getenv('STREAM_DEBUG', '').strip().lower()
debug_enabled = debug_param in ('1', 'true', 'yes', 'on')
cookie_param = os.getenv('STREAM_FORWARD_COOKIES', '').strip().lower()
forward_cookies = cookie_param in ('1', 'true', 'yes', 'on')
def dbg(message):
if debug_enabled:
app.logger.info("[stream_video] %s", message)
video_url = ""
if request.method == 'POST':
video_url = request.json.get('url')
live_hint = bool((request.json or {}).get('live'))
else:
video_url = request.args.get('url')
live_hint = str(request.args.get('live', '')).strip().lower() in ('1', 'true', 'yes', 'on')
if not video_url:
return jsonify({"error": "No URL provided"}), 400
dbg(f"method={request.method} url={video_url} live={live_hint}")
def media_path(url):
# Some sites serve media from a path with a trailing slash
# (heavyfetish: /get_file/.../11097_720p.mp4/). Without stripping it,
# every extension test below misses and the URL takes the yt-dlp branch
# instead -- a full extraction per request, including every seek.
return urllib.parse.urlparse(url).path.lower().rstrip('/')
def is_hls(url):
return '.m3u8' in media_path(url)
def is_dash(url):
return media_path(url).endswith('.mpd')
def guess_content_type(url):
path = media_path(url)
if path.endswith('.m3u8'):
return 'application/vnd.apple.mpegurl'
if path.endswith('.mpd'):
return 'application/dash+xml'
if path.endswith('.mp4') or path.endswith('.m4v') or path.endswith('.m4s'):
return 'video/mp4'
if path.endswith('.webm'):
return 'video/webm'
if path.endswith('.ts'):
return 'video/mp2t'
if path.endswith('.mov'):
return 'video/quicktime'
if path.endswith('.m4a'):
return 'audio/mp4'
if path.endswith('.mp3'):
return 'audio/mpeg'
if path.endswith('.ogg') or path.endswith('.oga'):
return 'audio/ogg'
return None
def is_direct_media(url):
path = media_path(url)
return any(path.endswith(ext) for ext in ('.mp4', '.m4v', '.m4s', '.ts', '.webm', '.mov'))
def looks_like_m3u8_bytes(chunk):
if not chunk:
return False
sample = chunk.lstrip(b'\xef\xbb\xbf')
return b'#EXTM3U' in sample[:1024]
def looks_like_mp4_bytes(chunk):
if not chunk or len(chunk) < 8:
return False
return chunk[4:8] == b'ftyp'
def build_upstream_headers(referer):
# We fetch upstream through curl_cffi with browser impersonation, which
# supplies a coherent User-Agent / Accept / Sec-CH-UA / Accept-Encoding
# set matching the impersonated browser. Forwarding the client's own
# values (the player may be Firefox while we impersonate Chrome) would
# contradict the TLS fingerprint and defeat impersonation, so we only
# pass headers the origin genuinely needs for authorization.
headers = {
'Referer': referer,
'Origin': referer,
}
if forward_cookies and 'Cookie' in request.headers:
headers['Cookie'] = request.headers['Cookie']
dbg("forwarding cookies")
# Relay the per-format headers (e.g. Cookie) the frontend forwarded as
# query params so cookie/token-authorized origins serve the media.
# Referer is already set above, and impersonation-managed headers are left
# to curl_cffi to keep the request coherent with the spoofed fingerprint.
for key, value in collect_passthrough_headers(request.args).items():
lower = key.lower()
if lower == 'referer' or lower in STREAM_IMPERSONATION_MANAGED_HEADERS:
continue
if value:
headers[key] = value
# Remove keys with None values
return {k: v for k, v in headers.items() if v}
def build_forwarded_headers(resp, target_url=None, content_type_override=None):
hop_by_hop = {
'connection', 'keep-alive', 'proxy-authenticate', 'proxy-authorization',
'te', 'trailers', 'transfer-encoding', 'upgrade'
}
forwarded_headers = []
response_content_type = None
for name, value in resp.headers.items():
if name.lower() in hop_by_hop:
continue
if name.lower() == 'content-length':
forwarded_headers.append((name, value))
continue
if name.lower() == 'content-type':
response_content_type = value
if name.lower() == 'content-type' and content_type_override:
continue
forwarded_headers.append((name, value))
if not content_type_override:
if not response_content_type or 'application/octet-stream' in response_content_type:
content_type_override = guess_content_type(target_url or resp.url)
if content_type_override:
forwarded_headers.append(('Content-Type', content_type_override))
dbg(f"content_type_override={content_type_override}")
return forwarded_headers
def proxy_response(target_url, content_type_override=None, referer_override=None, upstream_headers=None):
# Extract the base domain to spoof the referer
request_referer = request.args.get('referer')
if referer_override:
referer = referer_override
elif request_referer:
referer = request_referer
else:
parsed_uri = urllib.parse.urlparse(target_url)
referer = f"{parsed_uri.scheme}://{parsed_uri.netloc}/"
dbg(f"proxy_response target={target_url} referer={referer}")
safe_request_headers = build_upstream_headers(referer)
if isinstance(upstream_headers, dict):
for key, value in upstream_headers.items():
if value:
safe_request_headers[key] = value
# Pass through Range headers so the browser can 'sniff' the video
if 'Range' in request.headers:
safe_request_headers['Range'] = request.headers['Range']
resp = impersonate_get(target_url, headers=safe_request_headers, stream=True, timeout=30, allow_redirects=True)
# Some channel proxies (e.g. the "animeidhentai" hottub proxy) use
# inverted hotlink protection: they 403 any request that carries a
# Referer/Origin and only serve referer-less ones. Other CDNs require
# the spoofed referer. Satisfy both by retrying once without it.
if resp.status_code == 403 and ('Referer' in safe_request_headers or 'Origin' in safe_request_headers):
dbg("upstream 403 with referer; retrying without referer/origin")
resp.close()
referer_less = {k: v for k, v in safe_request_headers.items() if k not in ('Referer', 'Origin')}
resp = impersonate_get(target_url, headers=referer_less, stream=True, timeout=30, allow_redirects=True)
# Still refused: strip everything the extractor asked us to relay and go
# in bare (Range only, plus whatever impersonation supplies). Signed CDN
# links are often served fine to a plain browser request and refused when
# it carries extras -- a `Sec-Fetch-Mode: navigate` on a media
# subresource, say, which is exactly what yt-dlp's generic extractor
# hands back and what a real player would never send.
if resp.status_code == 403 and len(safe_request_headers) > (1 if 'Range' in safe_request_headers else 0):
dbg("upstream still 403; retrying bare (range only)")
resp.close()
bare = {}
if 'Range' in safe_request_headers:
bare['Range'] = safe_request_headers['Range']
resp = impersonate_get(target_url, headers=bare, stream=True, timeout=30, allow_redirects=True)
if debug_enabled:
dbg(f"upstream status={resp.status_code} content_type={resp.headers.get('Content-Type')} content_length={resp.headers.get('Content-Length')}")
content_iter = None
first_chunk = b""
if request.method != 'HEAD':
content_iter = resp.iter_content(chunk_size=1024 * 16)
try:
first_chunk = next(content_iter)
except StopIteration:
first_chunk = b""
if looks_like_m3u8_bytes(first_chunk):
remaining = b"".join(chunk for chunk in content_iter if chunk)
body_bytes = first_chunk + remaining
base_url = resp.url
encoding = resp.encoding
resp.close()
dbg("detected m3u8 by content sniff")
upstream_for_playlist = dict(safe_request_headers)
upstream_for_playlist.pop('Range', None)
return proxy_hls_playlist(
target_url,
referer_hint=referer,
upstream_headers=upstream_for_playlist,
prefetched_body=body_bytes,
prefetched_base_url=base_url,
prefetched_encoding=encoding,
)
forwarded_headers = build_forwarded_headers(
resp,
target_url=target_url,
content_type_override=content_type_override,
)
if request.method == 'HEAD':
resp.close()
return Response("", status=resp.status_code, headers=forwarded_headers)
def generate():
try:
if first_chunk:
yield first_chunk
for chunk in content_iter or resp.iter_content(chunk_size=1024 * 16):
if chunk:
yield chunk
finally:
resp.close()
return Response(generate(), status=resp.status_code, headers=forwarded_headers)
def decode_playlist_body(body_bytes, encoding=None):
if not body_bytes:
return ""
enc = encoding or "utf-8"
try:
return body_bytes.decode(enc, errors="replace")
except LookupError:
return body_bytes.decode("utf-8", errors="replace")
def passthrough_param_suffix():
# The relayed headers (e.g. Cookie) the upstream needs for authorization,
# encoded as &Name=value so they ride along on every proxied child URL
# (variant playlists, segments). Referer is appended separately by each
# rewriter; impersonation-managed headers stay with curl_cffi.
parts = []
for key, value in collect_passthrough_headers(request.args).items():
lower = key.lower()
if lower == 'referer' or lower in STREAM_IMPERSONATION_MANAGED_HEADERS:
continue
if not value:
continue
parts.append(f"&{urllib.parse.quote(key)}={urllib.parse.quote(str(value))}")
return ''.join(parts)
def rewrite_hls_playlist(body_text, base_url, referer):
extra = passthrough_param_suffix()
def proxied_url(target):
absolute = urljoin(base_url, target)
return f"/api/stream?url={urllib.parse.quote(absolute, safe='')}&referer={urllib.parse.quote(referer, safe='')}{extra}"
lines = body_text.splitlines()
rewritten = []
for line in lines:
stripped = line.strip()
if not stripped or stripped.startswith('#'):
# Rewrite URI attributes inside tags (keys/maps)
if 'URI="' in line:
def repl(match):
uri = match.group(1)
return f'URI="{proxied_url(uri)}"'
import re
line = re.sub(r'URI="([^"]+)"', repl, line)
rewritten.append(line)
continue
rewritten.append(proxied_url(stripped))
body = "\n".join(rewritten)
return Response(body, status=200, content_type='application/vnd.apple.mpegurl')
def proxy_hls_playlist(playlist_url, referer_hint=None, prefetched_body=None, prefetched_base_url=None, prefetched_encoding=None, upstream_headers=None):
dbg(f"proxy_hls_playlist url={playlist_url} referer_hint={referer_hint}")
base_url = prefetched_base_url or playlist_url
body_text = None
if prefetched_body is None:
headers = build_upstream_headers(referer_hint or "")
if isinstance(upstream_headers, dict):
for key, value in upstream_headers.items():
if value:
headers[key] = value
resp = impersonate_get(playlist_url, headers=headers, stream=True, timeout=30)
# See proxy_response: retry without referer for inverted hotlink
# protection that 403s any refered request.
if resp.status_code == 403 and ('Referer' in headers or 'Origin' in headers):
dbg("playlist upstream 403 with referer; retrying without referer/origin")
resp.close()
referer_less = {k: v for k, v in headers.items() if k not in ('Referer', 'Origin')}
resp = impersonate_get(playlist_url, headers=referer_less, stream=True, timeout=30)
base_url = resp.url
if resp.status_code >= 400:
forwarded_headers = build_forwarded_headers(resp, target_url=base_url)
if request.method == 'HEAD':
resp.close()
return Response("", status=resp.status_code, headers=forwarded_headers)
def generate():
try:
for chunk in resp.iter_content(chunk_size=1024 * 16):
if chunk:
yield chunk
finally:
resp.close()
return Response(generate(), status=resp.status_code, headers=forwarded_headers)
if request.method == 'HEAD':
forwarded_headers = build_forwarded_headers(resp, target_url=base_url)
resp.close()
return Response("", status=resp.status_code, headers=forwarded_headers)
content_iter = resp.iter_content(chunk_size=1024 * 16)
try:
first_chunk = next(content_iter)
except StopIteration:
first_chunk = b""
if looks_like_m3u8_bytes(first_chunk):
remaining = b"".join(chunk for chunk in content_iter if chunk)
body_bytes = first_chunk + remaining
body_text = decode_playlist_body(body_bytes, resp.encoding)
resp.close()
else:
content_type_override = None
if looks_like_mp4_bytes(first_chunk):
content_type_override = 'video/mp4'
forwarded_headers = build_forwarded_headers(
resp,
target_url=base_url,
content_type_override=content_type_override,
)
def generate():
try:
if first_chunk:
yield first_chunk
for chunk in content_iter:
if chunk:
yield chunk
finally:
resp.close()
return Response(generate(), status=resp.status_code, headers=forwarded_headers)
else:
body_text = decode_playlist_body(prefetched_body, prefetched_encoding)
if referer_hint:
referer = referer_hint
else:
referer = f"{urllib.parse.urlparse(base_url).scheme}://{urllib.parse.urlparse(base_url).netloc}/"
if request.method == 'HEAD':
return Response("", status=200, content_type='application/vnd.apple.mpegurl')
return rewrite_hls_playlist(body_text, base_url, referer)
if is_hls(video_url):
try:
dbg("detected input as hls")
referer_hint = request.args.get('referer')
if not referer_hint:
parsed = urllib.parse.urlparse(video_url)
referer_hint = f"{parsed.scheme}://{parsed.netloc}/"
return proxy_hls_playlist(video_url, referer_hint)
except Exception as e:
return jsonify({"error": str(e)}), 500
if is_direct_media(video_url):
try:
dbg("detected input as direct media")
return proxy_response(video_url)
except Exception as e:
return jsonify({"error": str(e)}), 500
def extract_referer(headers):
if not isinstance(headers, dict):
return None
return headers.get('Referer') or headers.get('referer')
def build_master_playlist(info, referer):
"""Synthesize an HLS master playlist from yt-dlp's parsed formats.
yt-dlp already downloads and parses the upstream master during
extraction, so we reconstruct an equivalent master that points each
variant playlist at our proxy. This avoids re-fetching the upstream
master (some sites issue a single-use session token on it, which the
extraction already consumed) and works generically for any HLS source
that exposes separate audio/video renditions. Returns the playlist
text, or None if there aren't enough HLS formats to build one.
"""
formats = info.get('formats') or []
def codec_present(value):
return value not in (None, '', 'none')
def is_hls_format(fmt):
url = fmt.get('url') or ''
return bool(url) and ('m3u8' in str(fmt.get('protocol') or '') or is_hls(url))
audio_fmts, video_fmts = [], []
for fmt in formats:
if not is_hls_format(fmt):
continue
if codec_present(fmt.get('vcodec')):
video_fmts.append(fmt)
elif codec_present(fmt.get('acodec')):
audio_fmts.append(fmt)
if not video_fmts:
return None
extra = passthrough_param_suffix()
def proxied(url):
return (f"/api/stream?url={urllib.parse.quote(url, safe='')}"
f"&referer={urllib.parse.quote(referer, safe='')}{extra}")
lines = ['#EXTM3U', '#EXT-X-VERSION:3']
audio_group = None
if audio_fmts:
audio_group = 'aud'
for index, fmt in enumerate(audio_fmts):
name = (fmt.get('format_note') or fmt.get('language')
or fmt.get('format_id') or f'audio{index}')
attrs = [
'TYPE=AUDIO',
f'GROUP-ID="{audio_group}"',
f'NAME="{name}"',
f'DEFAULT={"YES" if index == 0 else "NO"}',
'AUTOSELECT=YES',
]
if fmt.get('language'):
attrs.append(f'LANGUAGE="{fmt["language"]}"')
attrs.append(f'URI="{proxied(fmt["url"])}"')
lines.append('#EXT-X-MEDIA:' + ','.join(attrs))
for fmt in video_fmts:
bitrate = fmt.get('tbr') or fmt.get('vbr')
bandwidth = int(float(bitrate) * 1000) if bitrate else 1000000
codecs = []
if codec_present(fmt.get('vcodec')):
codecs.append(fmt['vcodec'])
if audio_group and codec_present(audio_fmts[0].get('acodec')):
codecs.append(audio_fmts[0]['acodec'])
elif codec_present(fmt.get('acodec')):
codecs.append(fmt['acodec'])
attrs = [f'BANDWIDTH={bandwidth}']
if fmt.get('width') and fmt.get('height'):
attrs.append(f'RESOLUTION={int(fmt["width"])}x{int(fmt["height"])}')
if fmt.get('fps'):
attrs.append(f'FRAME-RATE={float(fmt["fps"]):.3f}')
if codecs:
attrs.append(f'CODECS="{",".join(codecs)}"')
if audio_group:
attrs.append(f'AUDIO="{audio_group}"')
lines.append('#EXT-X-STREAM-INF:' + ','.join(attrs))
lines.append(proxied(fmt['url']))
return '\n'.join(lines) + '\n'
try:
# Configure yt-dlp options
ydl_opts = {
# Prefer HLS when available to enable chunked streaming in the browser.
# Live cam streams expose only separate video-only and audio-only HLS
# tracks (no muxed format), so `best` alone raises "Requested format
# is not available". Fall back to the best video-only rendition; we
# then hand the browser the master manifest below so its HLS player
# can pull in the matching audio track.
'format': 'best[protocol*=m3u8]/best[ext=mp4]/best/bestvideo[protocol*=m3u8]/bestvideo',
'format_sort': ['res', 'fps', 'vcodec:avc1', 'acodec:aac'],
'quiet': False,
'no_warnings': False,
# Impersonate a real browser by default so origins that fingerprint
# clients (e.g. the "animeidhentai" hottub channel) don't 403.
'impersonate': ImpersonateTarget.from_str(IMPERSONATE_TARGET),
}
passthrough_source = request.json if request.method == 'POST' else request.args
passthrough_headers = collect_passthrough_headers(passthrough_source)
dbg(f"passthrough_headers={list(passthrough_headers.keys())}")
if passthrough_headers:
ydl_opts.setdefault('http_headers', {}).update(passthrough_headers)
with yt_dlp.YoutubeDL(ydl_opts) as ydl:
# Extract the info
try:
info = ydl.extract_info(video_url, download=False)
except Exception as ydl_err:
# yt-dlp can't extract iframe-embedded JS players; scrape the
# embed for its HLS playlist and proxy that directly instead.
embed = resolve_unsupported_embed(video_url)
if not embed:
raise
dbg(f"embed fallback resolved {video_url} -> {embed['url']}")
if request.method == 'HEAD':
return Response("", status=200, content_type='application/vnd.apple.mpegurl')
return proxy_hls_playlist(embed['url'], embed['http_headers'].get('Referer'),
upstream_headers=embed['http_headers'])
dbg(f"yt_dlp extractor={info.get('extractor')} protocol={info.get('protocol')}")
# Try to get the URL from the info dict (works for progressive downloads)
stream_url = info.get('url')
protocol = info.get('protocol')
selected_format = None
# If no direct URL, try to get it from formats
if 'formats' in info:
if info.get('format_id'):
for fmt in info['formats']:
if fmt.get('format_id') == info.get('format_id'):
selected_format = fmt
break
if not selected_format and stream_url:
for fmt in info['formats']:
if fmt.get('url') == stream_url:
selected_format = fmt
break
if not selected_format:
for fmt in info['formats']:
if fmt.get('url'):
selected_format = fmt
break
if not stream_url and selected_format:
stream_url = selected_format.get('url')
if not stream_url:
return jsonify({"error": "Could not extract stream URL"}), 500
upstream_headers = None
if selected_format and isinstance(selected_format.get('http_headers'), dict):
upstream_headers = selected_format['http_headers']
elif isinstance(info.get('http_headers'), dict):
upstream_headers = info['http_headers']
referer_hint = None
if upstream_headers:
referer_hint = extract_referer(upstream_headers)
if not referer_hint:
parsed = urllib.parse.urlparse(video_url)
referer_hint = f"{parsed.scheme}://{parsed.netloc}/"
# When the chosen rendition carries no muxed audio (live cam
# streams) or the source is live, the browser needs a *master*
# playlist so its HLS player can combine the separate audio + video
# renditions. We synthesize that master from yt-dlp's already-parsed
# formats rather than re-fetching the upstream master.
def format_lacks_audio(fmt):
if not isinstance(fmt, dict):
return False
acodec = str(fmt.get('acodec') or '').lower()
vcodec = str(fmt.get('vcodec') or '').lower()
return vcodec not in ('', 'none') and acodec in ('', 'none')
needs_master = bool(info.get('is_live') or live_hint or format_lacks_audio(selected_format))
master_playlist = build_master_playlist(info, referer_hint) if needs_master else None
dbg(f"is_live={info.get('is_live')} needs_master={needs_master} synthesized={bool(master_playlist)}")
if master_playlist:
if request.method == 'HEAD':
return Response("", status=200, content_type='application/vnd.apple.mpegurl')
return Response(master_playlist, status=200, content_type='application/vnd.apple.mpegurl')
# Synthesis unavailable (e.g. a single muxed variant): fall back to
# the upstream master URL when one exists, else the variant itself.
master_fallback = None
if needs_master:
master_fallback = (selected_format or {}).get('manifest_url') or info.get('manifest_url')
if request.method == 'HEAD' and selected_format:
head_manifest = master_fallback or selected_format.get('manifest_url')
if head_manifest:
location = f"/api/stream?url={urllib.parse.quote(head_manifest, safe='')}"
return Response("", status=301, headers=[('Location', location)])
dbg(f"resolved stream_url={stream_url} referer_hint={referer_hint}")
if protocol and 'm3u8' in protocol:
dbg("protocol indicates hls")
return proxy_hls_playlist(master_fallback or stream_url, referer_hint, upstream_headers=upstream_headers)
if is_hls(stream_url):
dbg("stream_url is hls")
return proxy_hls_playlist(master_fallback or stream_url, referer_hint, upstream_headers=upstream_headers)
if is_dash(stream_url):
dbg("stream_url is dash")
return proxy_response(stream_url, content_type_override='application/dash+xml', referer_override=referer_hint, upstream_headers=upstream_headers)
dbg("stream_url is direct media")
return proxy_response(stream_url, referer_override=referer_hint, upstream_headers=upstream_headers)
except Exception as e:
return jsonify({"error": str(e)}), 500
if __name__ == '__main__':
# threaded=True allows multiple segments to be proxied at once
app.run(host='0.0.0.0', port=5000, threaded=True)