Files
RaptDramaDump/rapt_scraper.py
T

462 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
RaptDrama Passion 分类全量抓取。
默认用游客 autologin。也可以用邮箱和密码调用 /app/open/emailLogin 换取 token,
或加上 --from-device,用项目内置 adb 读取设备里已登录账号的 accessToken。
凭证放在 apasstk 头里,正文是 JSON。已购买或已解锁的集会返回 m3u8。
重复运行会重新拉最新目录。邮箱和密码用参数或环境变量传入,不要写进源码。
"""
import argparse
import gzip
import html
import json
import os
import re
import socket
import subprocess
import time
import urllib.error
import urllib.request
from urllib.parse import urljoin
socket.setdefaulttimeout(20)
BASE = "https://apis.raptdrama.com"
PASSION_TYPE = 20
OUT_NAME = "raptdrama_passion_full.json"
PREFS = "/data/data/com.leivideo.raptdrama/shared_prefs/FlutterSharedPreferences.xml"
HEADERS = {
"Content-Type": "application/json",
"Accept": "application/json",
"User-Agent": "RaptDrama/1.1.81 (Android 12; 23113RKC6C; Redmi)",
"x-device-fingerprint": "0a4e4b212cbb0a2434cf804c8e30ab58",
"x-os-version": "12",
"x-device-physical": "true",
"x-device-model": "23113RKC6C",
"accept-language": "en-US",
"accept-language-custom": "en-US",
"x-os-name": "Android",
"x-device-brand": "Redmi",
"x-device-id": "rd_c9f0fa17a39f96aa",
"x-app-build": "139",
"x-device-type": "phone",
"x-app-version": "1.1.81",
}
def account_label(user):
name = user.get("displayName") or user.get("nickname") or "device"
uid = user.get("uid") or user.get("id") or ""
if user.get("isGuest"):
provider = "guest"
else:
provider = user.get("loginProvider") or "account"
vip = "VIP" if user.get("isVip") else "非VIP"
return f"{name} uid={uid} {provider} {vip}"
def bundled_adb():
name = "adb.exe" if os.name == "nt" else "adb"
path = os.path.join(os.path.dirname(os.path.abspath(__file__)), "tools", "platform-tools", name)
if not os.path.isfile(path):
raise RuntimeError(f"项目内没有 adb:{path}")
return path
def find_adb(explicit):
if explicit:
if not os.path.isfile(explicit):
raise RuntimeError(f"找不到 adb:{explicit}")
return explicit
return bundled_adb()
def adb_exec(adb_path, args, timeout=20):
return subprocess.run([adb_path, *args], capture_output=True, timeout=timeout)
def list_ready_devices(adb_path):
proc = adb_exec(adb_path, ["devices"])
serials = []
for line in proc.stdout.decode("utf-8", "replace").splitlines()[1:]:
parts = line.split()
if len(parts) >= 2 and parts[1] == "device":
serials.append(parts[0])
return serials
def resolve_serial(adb_path, serial):
serial = (serial or "").strip()
if serial:
if ":" in serial:
adb_exec(adb_path, ["connect", serial], timeout=15)
ready = list_ready_devices(adb_path)
if serial not in ready:
shown = "、".join(ready) or "无"
raise RuntimeError(f"设备 {serial} 未就绪。当前已连接:{shown}")
return serial
ready = list_ready_devices(adb_path)
if len(ready) == 1:
return ready[0]
if not ready:
raise RuntimeError(
"没有检测到设备。手机请打开 USB 调试并在手机上允许这台电脑;"
"模拟器请加上 --serial IP:端口。"
)
raise RuntimeError("检测到多台设备,请用 --serial 指定其一:\n" + "\n".join(ready))
def shell_text(adb_path, serial, shell_args, timeout=15):
try:
proc = adb_exec(adb_path, ["-s", serial, "shell", *shell_args], timeout=timeout)
except subprocess.TimeoutExpired:
return 1, "", "timeout"
return proc.returncode, proc.stdout.decode("utf-8", "replace"), proc.stderr.decode("utf-8", "replace")
def read_prefs_text(adb_path, serial):
def usable(text):
return "flutter.user_data" in text
_, text, _ = shell_text(adb_path, serial, ["cat", PREFS])
if usable(text):
return text
adb_exec(adb_path, ["-s", serial, "root"], timeout=15)
if ":" in serial:
adb_exec(adb_path, ["connect", serial], timeout=15)
for _ in range(10):
time.sleep(0.5)
_, text, _ = shell_text(adb_path, serial, ["cat", PREFS])
if usable(text):
return text
for shell_args in (["su", "-c", f"cat {PREFS}"], ["su", "0", "cat", PREFS]):
_, text, _ = shell_text(adb_path, serial, shell_args, timeout=8)
if usable(text):
return text
raise RuntimeError(
"读不到应用私有目录。此应用是正式包,未 root 的手机无法读取其中的登录凭证。"
"请使用已 root 的手机,或带 root 的模拟器。"
)
def load_token_from_device(adb_path, serial):
serial = resolve_serial(adb_path, serial)
text = read_prefs_text(adb_path, serial)
match = re.search(r'name="flutter\.user_data"[^>]*>(.*?)</string>', text, re.S)
if not match:
raise RuntimeError("登录信息里没有 user_data,请先在 App 里登录")
user = json.loads(html.unescape(match.group(1)))
token = user.get("accessToken") or ""
if not token:
raise RuntimeError("当前账号没有 accessToken,请先在 App 里登录")
return token, f"{account_label(user)} device={serial}"
def load_guest_token():
data = api("/app/open/autologin", None, "POST", {})
token = (data or {}).get("token") or ""
if not token:
raise RuntimeError("游客登录没有返回 token")
return token, (data or {}).get("nickname") or "guest"
def load_email_token(email, password):
data = api("/app/open/emailLogin", None, "POST", {
"email": email,
"password": password,
})
token = (data or {}).get("token") or ""
if not token:
raise RuntimeError("邮箱登录没有返回 token")
nickname = (data or {}).get("nickname") or email
uid = (data or {}).get("id") or ""
return token, f"{nickname} uid={uid} email"
def load_token(from_device, adb_path, serial, email, password):
if from_device:
return load_token_from_device(find_adb(adb_path), serial)
email = (email or "").strip()
password = password or ""
if email or password:
if not email or not password:
raise RuntimeError("邮箱登录需要同时提供邮箱和密码")
return load_email_token(email, password)
env = os.environ.get("RAPTD_TOKEN", "").strip()
if env:
return env, "RAPTD_TOKEN"
email = (os.environ.get("RAPTD_EMAIL") or "").strip()
password = os.environ.get("RAPTD_PASSWORD") or ""
if email or password:
if not email or not password:
raise RuntimeError("邮箱登录需要同时提供 RAPTD_EMAIL 和 RAPTD_PASSWORD")
return load_email_token(email, password)
return load_guest_token()
def _read_response(req):
last_error = None
for attempt in range(6):
try:
with urllib.request.urlopen(req, timeout=25) as resp:
return resp.read()
except (urllib.error.URLError, TimeoutError, OSError) as exc:
last_error = exc
time.sleep(min(10, 1.5 * (attempt + 1)))
raise RuntimeError(f"请求失败: {last_error}")
def api(path, token, method="GET", body=None):
headers = dict(HEADERS)
if token:
headers["apasstk"] = token
data = None if body is None else json.dumps(body).encode()
req = urllib.request.Request(BASE + path, data=data, headers=headers, method=method)
raw = _read_response(req)
if raw[:2] == b"\x1f\x8b":
raw = gzip.decompress(raw)
payload = json.loads(raw.decode("utf-8"))
if payload.get("code") != 200:
raise RuntimeError(f"{method} {path} -> {payload.get('code')} {payload.get('msg')}")
return payload.get("data")
def http_text(url):
req = urllib.request.Request(url, headers={"User-Agent": "ExoPlayer"})
raw = _read_response(req)
return raw.decode("utf-8", "replace")
def list_passion(token):
dramas = []
seen = set()
page = 1
while page <= 50:
data = api("/app/open/classify_video", token, "POST", {
"page": page, "limit": 10, "type": PASSION_TYPE,
})
items = data if isinstance(data, list) else (data or {}).get("list") or []
fresh = [item for item in items if item.get("id") not in seen]
if not fresh:
break
for item in fresh:
seen.add(item.get("id"))
dramas.append(item)
page += 1
time.sleep(0.05)
return dramas
def list_chapters(token, vid):
chapters = []
start = 0
while start < 1000:
data = api(
f"/app/video/getchapters?vid={vid}&start={start}&limit=200",
token,
)
items = (data or {}).get("list") or []
chapters.extend(items)
if len(items) < 200:
break
start += 200
return chapters
def _playlist_items(data):
if isinstance(data, list):
return data, len(data)
items = (data or {}).get("list") or []
total = (data or {}).get("count")
return items, total
def _playlist_complete(items, total):
if not items:
return False
if total is None:
return True
return len(items) >= int(total)
def list_playlist_v2(token, vid):
episodes = []
page = 1
total = None
while page <= 40:
data = api("/app/video/playlistv2", token, "POST", {
"vid": vid,
"page": page,
"before": 1 if page == 1 else 0,
})
items, page_total = _playlist_items(data)
if page_total is not None:
total = page_total
if not items:
break
episodes.extend(items)
if total is not None and len(episodes) >= int(total):
break
page += 1
time.sleep(0.05)
return episodes
def list_playlist(token, vid):
try:
data = api("/app/video/playlist", token, "POST", {
"vid": vid, "page": 1, "before": 1,
})
items, total = _playlist_items(data)
if _playlist_complete(items, total):
return items
except Exception as exc:
print(f" playlist 失败,改用 playlistv2:{exc}", flush=True)
return list_playlist_v2(token, vid)
print(f" playlist 结果不完整,改用 playlistv2", flush=True)
return list_playlist_v2(token, vid)
def expand_hls(master_url):
"""把主 m3u8 展开成子播放列表和 ts 绝对地址。失败时只保留主地址。"""
if not master_url:
return None, []
try:
master = http_text(master_url)
except Exception:
return None, []
variant = None
for line in master.splitlines():
line = line.strip()
if line and not line.startswith("#"):
variant = urljoin(master_url, line)
break
if not variant:
return None, []
try:
media = http_text(variant)
except Exception:
return variant, []
segments = []
for line in media.splitlines():
line = line.strip()
if line and not line.startswith("#"):
segments.append(urljoin(variant, line))
return variant, segments
def parse_args():
parser = argparse.ArgumentParser(description="抓取 RaptDrama Passion 目录和播放地址")
parser.add_argument(
"--from-device",
action="store_true",
help="从已连接设备读取当前登录账号的 accessToken",
)
parser.add_argument(
"--serial",
default=os.environ.get("RAPTD_SERIAL", ""),
help="设备序列号。只连着一台时可以省略;模拟器填写 IP:端口",
)
parser.add_argument(
"--adb",
default=os.environ.get("RAPTD_ADB", ""),
help="adb 路径。默认使用项目内 tools/platform-tools",
)
parser.add_argument(
"--email",
default="",
help="邮箱登录账号。也可设置环境变量 RAPTD_EMAIL",
)
parser.add_argument(
"--password",
default="",
help="邮箱登录密码。也可设置环境变量 RAPTD_PASSWORD",
)
return parser.parse_args()
def main():
args = parse_args()
out_dir = os.path.dirname(os.path.abspath(__file__))
out_file = os.path.join(out_dir, OUT_NAME)
if args.from_device:
mode = "从设备读取登录凭证"
elif args.email or args.password:
mode = "邮箱登录"
elif os.environ.get("RAPTD_TOKEN"):
mode = "使用 RAPTD_TOKEN"
elif os.environ.get("RAPTD_EMAIL") or os.environ.get("RAPTD_PASSWORD"):
mode = "邮箱登录"
else:
mode = "游客登录"
print(mode, flush=True)
token, user = load_token(args.from_device, args.adb, args.serial, args.email, args.password)
print(f" {user}", flush=True)
print("获取 Passion 分类", flush=True)
dramas = list_passion(token)
print(f" {len(dramas)} 部", flush=True)
results = []
for index, info in enumerate(dramas, 1):
vid = info["id"]
title = (info.get("title") or "").strip()
print(f" [{index}/{len(dramas)}] {title}", flush=True)
chapters = list_chapters(token, vid)
vip_by_id = {ch.get("id"): str(ch.get("isvip")) == "1" for ch in chapters}
title_by_id = {ch.get("id"): ch.get("title") or "" for ch in chapters}
playlist = list_playlist(token, vid)
episodes = []
for item in playlist:
cid = item.get("cid")
src = item.get("src") or ""
variant, segments = expand_hls(src)
episodes.append({
"id": cid,
"number": item.get("idx"),
"title": title_by_id.get(cid) or item.get("msg") or "",
"is_vip": vip_by_id.get(cid, not bool(src)),
"cdn_url": src or None,
"media_playlist": variant,
"segments": segments,
})
free = sum(1 for ep in episodes if not ep["is_vip"])
with_cdn = sum(1 for ep in episodes if ep["cdn_url"])
with_ts = sum(1 for ep in episodes if ep["segments"])
print(
f" [{index}/{len(dramas)}] {title}: {len(episodes)} 集, 免费 {free}, 有地址 {with_cdn}, 已展开 {with_ts}",
flush=True,
)
results.append({
"id": vid,
"title": title,
"image": info.get("image") or "",
"desc": (info.get("desc") or "").strip(),
"score": info.get("score") or "",
"view": info.get("view") or 0,
"status": info.get("forstausen") or "",
"classify": info.get("classify") or [],
"total_episodes": len(episodes),
"episodes": episodes,
})
with open(out_file, "w", encoding="utf-8") as handle:
json.dump(results, handle, ensure_ascii=False, indent=2)
total_eps = sum(item["total_episodes"] for item in results)
cdn_eps = sum(1 for item in results for ep in item["episodes"] if ep.get("cdn_url"))
ts_eps = sum(1 for item in results for ep in item["episodes"] if ep.get("segments"))
print(f"完成: {len(results)} 部, {total_eps} 集, 有 m3u8 {cdn_eps}, 已展开 ts {ts_eps}", flush=True)
print(f"保存至 {out_file}", flush=True)
if __name__ == "__main__":
main()