Initial commit: BiliDownloader Web 版:Python 零依赖后端 + shadcn/ui 前端

This commit is contained in:
WpyQwq
2026-09-19 11:49:36 +08:00
commit f42cf11831
41 changed files with 7576 additions and 0 deletions
+513
View File
@@ -0,0 +1,513 @@
# -*- coding: utf-8 -*-
"""
engine.py —— BiliDownloader 下载引擎(被 server.py 导入)
这里的核心逻辑全部来自已经实测跑通的 bili_download.py,行为保持一致:
* WBI 签名的站内搜索(老的无签名接口已被 B 站封禁)
* 视频详情 / 分P / 清晰度列表(playurl qn=127 拿全部可用清晰度)
* 多线程分段加速下载:Range 分片 + 并发 + 每片独立重试续传 + 旁路状态文件跨进程续传
* ffmpeg 无损封装为 mp4
只依赖 Python 标准库;ffmpeg 需在 PATH 或常见安装位置。
"""
from __future__ import annotations
import hashlib
import json
import os
import re
import shutil
import subprocess
import threading
import time
import urllib.parse
import urllib.request
UA = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36")
API_VIEW = "https://api.bilibili.com/x/web-interface/view?bvid={bvid}"
API_NAV = "https://api.bilibili.com/x/web-interface/nav"
API_SEARCH = "https://api.bilibili.com/x/web-interface/wbi/search/type"
API_PLAYURL = ("https://api.bilibili.com/x/player/playurl?bvid={bvid}&cid={cid}"
"&qn=127&fnver=0&fnval=4048&fourk=1")
QN_NAME = {
127: "8K 超高清", 126: "杜比视界", 125: "HDR 真彩", 120: "4K 超清",
116: "1080P60", 112: "1080P 高码率", 80: "1080P 高清", 74: "720P60",
64: "720P 高清", 32: "480P 清晰", 16: "360P 流畅", 6: "240P 极速",
}
# WBI 签名置换表(与 bilibili-API-collect 公开算法一致)
MIXIN_KEY_ENC_TAB = [
46, 47, 18, 2, 53, 8, 23, 32, 15, 50, 10, 31, 58, 3, 45, 35, 27, 43, 5, 49,
33, 9, 42, 19, 29, 28, 14, 39, 12, 38, 41, 13, 37, 48, 7, 16, 24, 55, 40, 61,
26, 17, 0, 1, 60, 51, 30, 4, 22, 25, 54, 21, 56, 59, 6, 63, 57, 62, 11, 36,
20, 34, 44, 52,
]
CHUNK = 1 << 18 # 256KB
SEG_MIN = 2 << 20 # 小于 2MB 不值得分片
class BiliError(RuntimeError):
"""带 B 站错误码的异常。"""
def __init__(self, code: int, message: str):
super().__init__(f"B 站接口错误 {code}: {message}")
self.code = code
self.message = message
def human(n: float) -> str:
for unit in ("B", "KB", "MB", "GB"):
if n < 1024 or unit == "GB":
return f"{n:.1f}{unit}" if unit != "B" else f"{int(n)}B"
n /= 1024
return f"{n:.1f}GB"
# ------------------------------------------------------------------ HTTP
def _headers(referer: str, cookie: str | None, extra: dict | None = None) -> dict:
h = {"User-Agent": UA, "Referer": referer, "Accept": "*/*",
"Accept-Encoding": "identity", "Connection": "keep-alive"}
if cookie:
h["Cookie"] = cookie
if extra:
h.update(extra)
return h
def _get(url: str, referer: str = "https://www.bilibili.com",
cookie: str | None = None, timeout: int = 30) -> bytes:
req = urllib.request.Request(url, headers=_headers(referer, cookie))
with urllib.request.urlopen(req, timeout=timeout) as r:
return r.read()
def _api(url: str, cookie: str | None = None, referer: str = "https://www.bilibili.com") -> dict:
raw = _get(url, referer, cookie)
try:
payload = json.loads(raw.decode("utf-8"))
except Exception as exc:
raise BiliError(-1, "B 站返回了非 JSON 内容(可能触发了风控)") from exc
if payload.get("code") != 0:
raise BiliError(payload.get("code", -1), payload.get("message", ""))
return payload["data"]
# ------------------------------------------------------------------ WBI 签名
def _mixin_key(img_url: str, sub_url: str) -> str:
img = os.path.basename(urllib.parse.urlparse(img_url).path).split(".")[0]
sub = os.path.basename(urllib.parse.urlparse(sub_url).path).split(".")[0]
raw = img + sub
return "".join(raw[i] for i in MIXIN_KEY_ENC_TAB)[:32]
def _sign(params: dict, mixin_key: str) -> str:
params = dict(params)
params["wts"] = int(time.time())
items = sorted(params.items())
query = urllib.parse.urlencode(
[(k, "".join(c for c in str(v) if c not in "!'()*")) for k, v in items])
w_rid = hashlib.md5((query + mixin_key).encode()).hexdigest()
return f"{query}&w_rid={w_rid}"
# ------------------------------------------------------------------ 业务接口
def login_status(cookie: str | None) -> dict:
"""未登录时 nav 返回 code=-101,但 data.wbi_img 仍然可用,所以这里不抛异常。"""
try:
raw = _get(API_NAV, cookie=cookie)
payload = json.loads(raw.decode("utf-8"))
except Exception as exc:
return {"isLogin": False, "uname": "", "error": str(exc)}
data = payload.get("data") or {}
return {"isLogin": bool(data.get("isLogin")), "uname": data.get("uname") or "",
"code": payload.get("code")}
def search(keyword: str, page: int = 1, page_size: int = 20,
cookie: str | None = None) -> dict:
"""站内搜索。必须走 WBI 签名接口 —— 老的无签名接口已被封(返回非 JSON)。"""
nav = json.loads(_get(API_NAV, cookie=cookie).decode("utf-8"))
wbi = (nav.get("data") or {}).get("wbi_img") or {}
img_url, sub_url = wbi.get("img_url"), wbi.get("sub_url")
if not img_url or not sub_url:
raise BiliError(-1, "无法获取 WBI 密钥")
query = _sign({"search_type": "video", "keyword": keyword,
"page": page, "page_size": page_size}, _mixin_key(img_url, sub_url))
payload = json.loads(_get(f"{API_SEARCH}?{query}", cookie=cookie).decode("utf-8"))
if payload.get("code") != 0:
raise BiliError(payload.get("code", -1), payload.get("message", ""))
data = payload.get("data") or {}
results = []
for item in (data.get("result") or []):
title = re.sub(r"<em[^>]*>|</em>", "", item.get("title") or "")
pic = item.get("pic") or ""
if pic.startswith("//"):
pic = "https:" + pic
results.append({
"bvid": item.get("bvid"),
"title": title,
"author": item.get("author"),
"duration": item.get("duration"), # 字符串,如 "2:28"
"play": item.get("play"),
"danmaku": item.get("danmaku"),
"pic": pic,
"description": item.get("description") or item.get("desc") or "",
})
return {"items": results, "page": page,
"numResults": data.get("numResults"),
"hasMore": len(results) >= page_size}
def video_detail(bvid: str, cookie: str | None = None) -> dict:
data = _api(API_VIEW.format(bvid=bvid), cookie)
pic = data.get("pic") or ""
if pic.startswith("//"):
pic = "https:" + pic
return {
"bvid": data.get("bvid"),
"title": data.get("title"),
"pic": pic,
"duration": data.get("duration"), # int 秒
"desc": data.get("desc") or "",
"owner": (data.get("owner") or {}).get("name"),
"stat": data.get("stat") or {},
"pages": [{"cid": p.get("cid"), "page": p.get("page"),
"part": p.get("part"), "duration": p.get("duration")}
for p in (data.get("pages") or [])],
}
def play_url(bvid: str, cid: int, cookie: str | None = None) -> dict:
"""qn=127 请求最高清晰度,让服务端把可用的清晰度都返回。
★ 关键:清晰度列表必须按 `dash.video` 里**真正拿到的流**来建,不能用
`accept_quality` / `support_formats` —— 那两个只反映视频"标称"支持什么,
与当前账号能否下载无关。实测:未登录时 accept_quality 仍列出 120(4K),
但 dash.video 里只有 32(480P)。若按 accept_quality 建列表,用户会选到根本拿不到的清晰度,
点下载才报错。
"""
data = _api(API_PLAYURL.format(bvid=bvid, cid=cid), cookie)
accept = data.get("accept_quality") or []
formats = {f.get("quality"): f for f in (data.get("support_formats") or [])}
dash = data.get("dash") or {}
videos = [{"quality": s.get("id"), "baseUrl": s.get("baseUrl"),
"backupUrl": s.get("backupUrl") or [], "codecs": s.get("codecs") or "",
"width": s.get("width"), "height": s.get("height"),
"bandwidth": s.get("bandwidth")}
for s in (dash.get("video") or []) if s.get("baseUrl")]
audios = [{"baseUrl": s.get("baseUrl"), "backupUrl": s.get("backupUrl") or [],
"codecs": s.get("codecs") or "", "bandwidth": s.get("bandwidth")}
for s in (dash.get("audio") or []) if s.get("baseUrl")]
# 同一清晰度可能有多种编码,取码率最高的一条来展示尺寸
best: dict[int, dict] = {}
for s in videos:
q = s["quality"]
if q is None:
continue
if q not in best or (s["bandwidth"] or 0) > (best[q]["bandwidth"] or 0):
best[q] = s
qualities = []
for q in sorted(best, reverse=True):
s = best[q]
f = formats.get(q) or {}
qualities.append({
"quality": q,
"label": f.get("new_description") or QN_NAME.get(q, str(q)),
"width": s.get("width"), "height": s.get("height"),
"codecs": s.get("codecs"),
"bandwidth": s.get("bandwidth"),
})
return {"qualities": qualities, "video": videos, "audio": audios,
# 实际可下载的最高清晰度(用它判断是否被"未登录"限制到 480P)
"maxQuality": max(best) if best else 0,
# 视频标称支持的清晰度,仅作参考展示
"advertisedQuality": max(accept) if accept else 0,
"duration": dash.get("duration") or 0}
# ------------------------------------------------------------------ 下载
def _state_path(dest: str) -> str:
return dest + ".parts.json"
def _resource_key(url: str) -> str:
"""取 URL 中标识资源的稳定部分。
baseUrl 带 e= 时效签名,重跑时签名字段与 CDN 域名都会变;只有 path
(含 avid/cid/清晰度)保持不变,所以续传状态必须以 path 为键。
"""
return urllib.parse.urlparse(url).path
def _load_state(dest: str, url: str, total: int, n: int) -> list[int]:
if not os.path.exists(dest) or os.path.getsize(dest) != total:
return [0] * n
try:
with open(_state_path(dest), "r", encoding="utf-8") as f:
st = json.load(f)
if st.get("key") != _resource_key(url) or st.get("total") != total:
return [0] * n
offs = [int(x) for x in st["offsets"]]
if len(offs) != n:
return [0] * n
return [max(0, min(o, total)) for o in offs]
except Exception:
return [0] * n
def _save_state(dest: str, url: str, total: int, offsets: list[int]) -> None:
tmp = _state_path(dest) + ".tmp"
try:
with open(tmp, "w", encoding="utf-8") as f:
json.dump({"key": _resource_key(url), "total": total, "offsets": offsets}, f)
os.replace(tmp, _state_path(dest))
except Exception:
pass
def probe(url: str, referer: str, cookie: str | None) -> tuple[int, bool]:
"""探测总长度与是否支持 Range 分片。"""
req = urllib.request.Request(
url, headers=_headers(referer, cookie, {"Range": "bytes=0-0"}))
with urllib.request.urlopen(req, timeout=30) as r:
if r.status == 206:
cr = r.headers.get("Content-Range", "")
return (int(cr.split("/")[-1]) if "/" in cr else 0), True
return int(r.headers.get("Content-Length") or 0), False
def _download_range(url: str, fd, referer: str, cookie: str | None,
start: int, end: int, on_bytes, cancel, retries: int = 24,
on_chunk=None) -> None:
cur = start
attempt = 0
while cur <= end:
if cancel is not None and cancel.is_set():
raise InterruptedError("已取消")
try:
req = urllib.request.Request(
url, headers=_headers(referer, cookie, {"Range": f"bytes={cur}-{end}"}))
with urllib.request.urlopen(req, timeout=45) as r:
if r.status != 206:
raise IOError(f"服务器未返回 206 (got {r.status})")
os.lseek(fd, cur, os.SEEK_SET)
while True:
chunk = r.read(CHUNK)
if not chunk:
break
os.write(fd, chunk)
cur += len(chunk)
on_bytes(len(chunk))
if on_chunk:
on_chunk(cur)
if cur <= end:
raise IOError("连接提前结束")
attempt = 0
except InterruptedError:
raise
except Exception as e:
attempt += 1
if attempt > retries:
raise RuntimeError(f"分片重试 {retries} 次仍失败: {e}")
time.sleep(min(0.5 * attempt, 5))
def download_fast(url: str, dest: str, referer: str, cookie: str | None,
threads: int = 16, on_progress=None, cancel=None,
label: str = "视频") -> dict:
"""多线程分段加速下载,支持跨进程断点续传。返回统计信息。"""
total, ranges_ok = probe(url, referer, cookie)
state = {"done": 0, "total": total, "t0": time.time(), "resumed": 0}
def report():
if not on_progress:
return
el = max(time.time() - state["t0"], 1e-6)
on_progress({
"label": label,
"downloaded": state["done"],
"total": total,
"percent": (state["done"] * 100 / total) if total else 0,
"speed": state["done"] / el,
"threads": 1 if not ranges_ok else max(1, min(threads, 64)),
"resumed": state["resumed"],
})
lock = threading.Lock()
def on_bytes(n):
with lock:
state["done"] += n
if not ranges_ok or total < SEG_MIN or threads <= 1:
have = os.path.getsize(dest) if os.path.exists(dest) else 0
if have and have < total:
state["done"] = have
state["resumed"] = have
report()
mode = "ab" if have else "wb"
req_headers = _headers(referer, cookie,
{"Range": f"bytes={have}-"} if have else None)
with urllib.request.urlopen(urllib.request.Request(url, headers=req_headers),
timeout=45) as r:
if have and r.status != 206:
have = 0
mode = "wb"
state["done"] = 0
with open(dest, mode) as f:
while True:
if cancel is not None and cancel.is_set():
raise InterruptedError("已取消")
chunk = r.read(CHUNK)
if not chunk:
break
f.write(chunk)
on_bytes(len(chunk))
report()
return {"total": total, "threads": 1, "resumed": state["resumed"]}
n = max(1, min(threads, 64))
seg = total // n
spans = [(i * seg, total - 1 if i == n - 1 else (i + 1) * seg - 1) for i in range(n)]
offsets = _load_state(dest, url, total, n)
if not os.path.exists(dest) or os.path.getsize(dest) != total:
with open(dest, "wb") as f:
f.truncate(total)
offsets = [0] * n
resume_done = sum(offsets)
if resume_done:
state["done"] = resume_done
state["resumed"] = resume_done
report()
last_flush = [time.time()]
errors: list[Exception] = []
state_lock = threading.Lock()
def flush(force=False):
with state_lock:
now = time.time()
if not force and now - last_flush[0] < 2.0:
return
last_flush[0] = now
_save_state(dest, url, total, offsets)
def worker(i, s, e):
try:
fd = os.open(dest, os.O_RDWR | getattr(os, "O_BINARY", 0))
try:
def on_chunk(cur_abs, i=i, s=s):
with state_lock:
offsets[i] = cur_abs - s + 1
flush()
_download_range(url, fd, referer, cookie, s + offsets[i], e,
on_bytes, cancel, on_chunk=on_chunk)
finally:
os.close(fd)
with state_lock:
offsets[i] = e - s + 1
except InterruptedError as ex:
errors.append(ex)
except Exception as ex:
errors.append(ex)
t0 = time.time()
reporter_stop = threading.Event()
def reporter():
while not reporter_stop.wait(0.4):
report()
threading.Thread(target=reporter, daemon=True).start()
ts = [threading.Thread(target=worker, args=(i, s, e), daemon=True)
for i, (s, e) in enumerate(spans)]
for t in ts:
t.start()
for t in ts:
t.join()
reporter_stop.set()
if any(isinstance(e, InterruptedError) for e in errors):
flush(force=True)
raise InterruptedError("已取消")
if errors:
flush(force=True)
raise RuntimeError(f"下载失败(进度已保存, 重跑可续传): {errors[0]}")
if os.path.getsize(dest) != total:
raise RuntimeError(f"大小不符: {os.path.getsize(dest)} != {total}")
if os.path.exists(_state_path(dest)):
os.remove(_state_path(dest))
report()
return {"total": total, "threads": n, "resumed": resume_done,
"elapsed": time.time() - t0}
# ------------------------------------------------------------------ ffmpeg
def find_ffmpeg() -> str | None:
exe = shutil.which("ffmpeg")
if exe:
return exe
for p in (r"C:\ffmpeg\bin\ffmpeg.exe",
r"C:\Users\Administrator\miniconda3\Library\bin\ffmpeg.exe"):
if os.path.exists(p):
return p
return None
def merge(video: str, audio: str | None, out: str) -> None:
ff = find_ffmpeg()
if not ff:
raise RuntimeError("找不到 ffmpeg")
cmd = [ff, "-y", "-hide_banner", "-loglevel", "error", "-i", video]
if audio:
cmd += ["-i", audio, "-map", "0:v:0", "-map", "1:a:0", "-c", "copy"]
else:
cmd += ["-c", "copy"]
cmd += ["-movflags", "+faststart", out]
subprocess.run(cmd, check=True, creationflags=getattr(subprocess, "CREATE_NO_WINDOW", 0))
# ------------------------------------------------------------------ 工具
def safe_name(name: str) -> str:
name = re.sub(r'[\\/:*?"<>|\r\n\t]', "_", name).strip(" .")
return (name or "video")[:120]
def pick_streams(play: dict, quality: int, prefer_avc: bool = True):
"""在指定清晰度下选码流:优先 H.264(兼容性最好),音频取码率最大。"""
pool = [s for s in play["video"] if s["quality"] == quality]
if not pool:
return None, None
if prefer_avc:
avc = [s for s in pool if s["codecs"].startswith(("avc", "av01"))]
pool = avc or pool
video = max(pool, key=lambda s: s["bandwidth"])
if not play["audio"]:
return video, None
audio = max(play["audio"], key=lambda s: s["bandwidth"])
return video, audio