From 3ba54523596687980807223fbfa27f971547954a Mon Sep 17 00:00:00 2001 From: Nixevol Date: Sat, 26 Sep 2026 22:44:27 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E6=89=A9=E5=B1=95=E9=A6=96=E9=A1=B5?= =?UTF-8?q?=E5=88=86=E7=B1=BB=E3=80=81New=20Releases=20=E5=92=8C=20Recomme?= =?UTF-8?q?nd=20=E5=85=A8=E9=87=8F=E6=8A=93=E5=8F=96?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .gitignore | 2 + docs/接口清单.md | 42 ++++++++- rapt_scraper.py | 239 +++++++++++++++++++++++++++++++++++++++++++---- 3 files changed, 258 insertions(+), 25 deletions(-) diff --git a/.gitignore b/.gitignore index 1833b45..bf62a43 100644 --- a/.gitignore +++ b/.gitignore @@ -4,3 +4,5 @@ __pycache__/ .env *.apk raptdrama_passion_full.json +raptdrama_catalog.json +video/ diff --git a/docs/接口清单.md b/docs/接口清单.md index 4ba3c26..44b9195 100644 --- a/docs/接口清单.md +++ b/docs/接口清单.md @@ -165,7 +165,27 @@ | label_origin | string | | | image | string 或 null | | -已见到的分类:`18` Comedy,`20` Passion,`21` Newest,`22` Western,`23` Comic drama。 +已见到的分类:`18` Comedy,`20` Passion,`21` Newest,`22` Western,`23` Comic drama。首页顶栏就是这五个标签,脚本用 `sections` 记成 `comedy`、`passion`、`newest`、`western`、`comic`。 + +### 首页区块对应 + +发现页上能对上接口的区块: + +| 界面 | 接口 | 脚本里的 `sections` | +| --- | --- | --- | +| 顶栏 Comedy / Passion / Newest / Western / Comic drama | `POST /app/open/classify_video`,`type` 分别为 18 / 20 / 21 / 22 / 23 | `comedy` / `passion` / `newest` / `western` / `comic` | +| New Releases | `GET /app/open/orgin` | `new_releases` | +| Recommend | `POST /app/open/recommend`,按 `page` 翻页 | `recommend` | + +同一部剧可以同时出现在多个区块。脚本按剧 id 合并,播放地址只请求一次,`sections` 列出它出现过的区块,例如 `["passion","new_releases","recommend"]`。 + +Trending Now 对不上单独的短列表。安装包里发现页有 `trendingDramas` 和 `_buildTrendingSection`,但没有单独的拉取方法或 `/app/open/` 路径。核对过的结果: + +- `POST /app/open/extra_broadcast` 是历史页里的 extra broadcast,不是发现页的 Trending Now。当前 9 部里没有截图上的 Bai Jie: A Young Wife、Desire Manor、The Love Curse。改 `type` 或 `ishot` 仍是这 9 部。 +- `GET /app/open/orgin` 只有 Bai Jie: A Young Wife,没有另外两部。 +- Recommend 全量里三部都在,但分散在第 4、7、11 页,不是这一行的短列表。 +- 三部的 `videoinfo.ishot` 都是 `0`。猜的 `/app/open/banner`、`/hot`、`/ranking` 返回 404。 +- `/app/open/foryou` 不作为首页来源,脚本不调用。 ### POST /app/open/classify_video @@ -183,9 +203,19 @@ ### POST /app/open/recommend -请求体:`{}` +首页 Recommend。`page` 从 1 开始,`limit` 用 10。`start` 会被忽略,不能用来翻页。空对象 `{}` 和带 `page`/`limit` 的第一页一样,大约 10 条,不是全站列表。 -`data` 是剧目数组。空对象也只返回约 10 条,不是全站列表。 +请求体: + +```json +{"page": 1, "limit": 10} +``` + +每页 10 条,相邻页的剧 id 不同。某一页条数不足 10 时还要继续请求下一页。只有空页,或这一页的 id 全部在前面出现过,才停止。脚本最多翻 80 页,避免死循环。 + +2026-09-26 用游客 token 翻页:第 1–33 页各 10 条,第 34 页 2 条,第 35 页为空。去重后 332 部。 + +`data` 是剧目数组。 ### POST /app/open/foryou @@ -204,11 +234,13 @@ 请求体:`{}` -`data` 是剧目数组。 +`data` 是剧目数组。这是历史页的列表,不是首页 Trending Now。不传分页参数时返回固定的一小批剧;带 `page` 也不会换成 Trending Now 那一排。 ### GET /app/open/orgin -无请求体。`data` 是剧目数组。 +无请求体。`POST` 空对象同样返回这一批。`data` 是剧目数组。 + +这是首页 New Releases。当前一次返回 10 部,按较新的剧排在前面。截图里能看到的 Reaping What She Sowed、The Fall of a Mountain Flower、The Delivery Guy's Rise 是这 10 部的最后三部。加上 `type=trending`,或 `page`/`limit`,仍然只有 10 条,不是另一份 Trending Now 列表。 ### POST /app/open/search diff --git a/rapt_scraper.py b/rapt_scraper.py index 8ea90b8..da7c567 100644 --- a/rapt_scraper.py +++ b/rapt_scraper.py @@ -1,9 +1,17 @@ """ -RaptDrama Passion 分类全量抓取。 +RaptDrama 首页目录抓取。 + +抓取顶部分类(Comedy、Passion、Newest、Western、Comic drama)、 +New Releases,以及 Recommend 的全部分页。同一部剧只拉一次播放地址, +并用 sections 标记它出现在哪些区块。 + +首页 Trending Now 没有单独的 open 接口,安装包里的 extra_broadcast +对不上截图标题,因此不写入。/app/open/foryou 按要求不调用。 默认用游客 autologin。也可以用邮箱和密码调用 /app/open/emailLogin 换取 token, 或加上 --from-device,用项目内置 adb 读取设备里已登录账号的 accessToken。 凭证放在 apasstk 头里,正文是 JSON。已购买或已解锁的集会返回 m3u8。 +拉完目录后会询问是否把这些视频下载到项目的 video 目录,并按剧名分文件夹。 重复运行会重新拉最新目录。邮箱和密码用参数或环境变量传入,不要写进源码。 """ @@ -23,8 +31,17 @@ from urllib.parse import urljoin socket.setdefaulttimeout(20) BASE = "https://apis.raptdrama.com" -PASSION_TYPE = 20 -OUT_NAME = "raptdrama_passion_full.json" +OUT_NAME = "raptdrama_catalog.json" +# 顶部分类标签。id 来自 classifyv2,列表接口的参数名是 type。 +CLASSIFY_SECTIONS = ( + ("comedy", "Comedy", 18), + ("passion", "Passion", 20), + ("newest", "Newest", 21), + ("western", "Western", 22), + ("comic", "Comic drama", 23), +) +CLASSIFY_MAX_PAGES = 50 +RECOMMEND_MAX_PAGES = 80 PREFS = "/data/data/com.leivideo.raptdrama/shared_prefs/FlutterSharedPreferences.xml" HEADERS = { @@ -232,24 +249,102 @@ def http_text(url): return raw.decode("utf-8", "replace") -def list_passion(token): +def _as_drama_list(data): + if isinstance(data, list): + return [item for item in data if isinstance(item, dict)] + if isinstance(data, dict): + items = data.get("list") + if isinstance(items, list): + return [item for item in items if isinstance(item, dict)] + return [] + + +def list_paged(token, path, method, make_body, max_pages): + """按页拉取,直到空页或整页都是已经见过的 id。max_pages 防止死循环。""" dramas = [] seen = set() page = 1 - while page <= 50: - data = api("/app/open/classify_video", token, "POST", { - "page": page, "limit": 10, "type": PASSION_TYPE, - }) - items = data if isinstance(data, list) else (data or {}).get("list") or [] - fresh = [item for item in items if item.get("id") not in seen] - if not fresh: + stop_page = page + reason = "达到页数上限" + while page <= max_pages: + stop_page = page + data = api(path, token, method, make_body(page)) + items = _as_drama_list(data) + if not items: + reason = "空页" break - for item in fresh: - seen.add(item.get("id")) - dramas.append(item) + fresh = [] + for item in items: + vid = item.get("id") + if vid in seen: + continue + seen.add(vid) + fresh.append(item) + if not fresh: + reason = "整页都是重复 id" + break + dramas.extend(fresh) page += 1 time.sleep(0.05) - return dramas + return dramas, stop_page, reason + + +def list_classify(token, type_id): + return list_paged( + token, + "/app/open/classify_video", + "POST", + lambda page: {"page": page, "limit": 10, "type": type_id}, + CLASSIFY_MAX_PAGES, + ) + + +def list_new_releases(token): + items = _as_drama_list(api("/app/open/orgin", token, "GET")) + return items, 1, "单次返回" + + +def list_recommend(token): + return list_paged( + token, + "/app/open/recommend", + "POST", + lambda page: {"page": page, "limit": 10}, + RECOMMEND_MAX_PAGES, + ) + + +def add_section(catalog, order, section, items): + for item in items: + vid = item.get("id") + if vid is None: + continue + entry = catalog.get(vid) + if entry is None: + entry = dict(item) + entry["sections"] = [] + catalog[vid] = entry + order.append(vid) + if section not in entry["sections"]: + entry["sections"].append(section) + + +def collect_home(token): + """合并首页区块。同一 id 只保留一份剧目,sections 记录出现过的区块。""" + catalog = {} + order = [] + for key, label, type_id in CLASSIFY_SECTIONS: + items, stop_page, reason = list_classify(token, type_id) + add_section(catalog, order, key, items) + print(f" {label} {len(items)} 部,停在第 {stop_page} 页({reason})", flush=True) + items, _, reason = list_new_releases(token) + add_section(catalog, order, "new_releases", items) + print(f" New Releases {len(items)} 部({reason})", flush=True) + items, stop_page, reason = list_recommend(token) + add_section(catalog, order, "recommend", items) + print(f" Recommend {len(items)} 部,停在第 {stop_page} 页({reason})", flush=True) + print(f" 去重后 {len(order)} 部", flush=True) + return [catalog[vid] for vid in order] def list_chapters(token, vid): @@ -350,8 +445,102 @@ def expand_hls(master_url): return variant, segments +def safe_name(text, fallback): + cleaned = re.sub(r'[<>:"/\\|?*\x00-\x1f]', " ", text or "") + cleaned = re.sub(r"\s+", " ", cleaned).strip().rstrip(". ") + return (cleaned[:80] or fallback) + + +def download_bytes(url): + req = urllib.request.Request(url, headers={"User-Agent": "ExoPlayer"}) + return _read_response(req) + + +def localize_playlist(text, local_name_for_uri): + lines = [] + for line in text.splitlines(): + stripped = line.strip() + if stripped and not stripped.startswith("#"): + lines.append(local_name_for_uri(stripped)) + else: + lines.append(line) + return "\n".join(lines) + "\n" + + +def download_episode(episode, episode_dir): + master_url = episode.get("cdn_url") + if not master_url: + return False + os.makedirs(episode_dir, exist_ok=True) + master_text = download_bytes(master_url).decode("utf-8", "replace") + variant_rel = next( + (line.strip() for line in master_text.splitlines() if line.strip() and not line.startswith("#")), + None, + ) + if not variant_rel: + return False + variant_url = urljoin(master_url, variant_rel) + media_text = download_bytes(variant_url).decode("utf-8", "replace") + segment_rels = [ + line.strip() for line in media_text.splitlines() + if line.strip() and not line.startswith("#") + ] + if not segment_rels: + return False + names = [os.path.basename(urljoin(variant_url, rel).split("?", 1)[0]) for rel in segment_rels] + for rel, name in zip(segment_rels, names): + target = os.path.join(episode_dir, name) + if os.path.isfile(target) and os.path.getsize(target) > 0: + continue + payload = download_bytes(urljoin(variant_url, rel)) + with open(target, "wb") as handle: + handle.write(payload) + with open(os.path.join(episode_dir, "video.m3u8"), "w", encoding="utf-8", newline="\n") as handle: + handle.write(localize_playlist(media_text, lambda uri: os.path.basename(urljoin(variant_url, uri).split("?", 1)[0]))) + with open(os.path.join(episode_dir, "playlist.m3u8"), "w", encoding="utf-8", newline="\n") as handle: + handle.write(localize_playlist(master_text, lambda _uri: "video.m3u8")) + return all(os.path.isfile(os.path.join(episode_dir, name)) and os.path.getsize(os.path.join(episode_dir, name)) > 0 for name in names) + + +def download_videos(results, video_root): + used_names = {} + saved = 0 + failed = 0 + for drama in results: + episodes = [ep for ep in drama["episodes"] if ep.get("cdn_url")] + if not episodes: + continue + title = safe_name(drama.get("title"), f"drama-{drama.get('id')}") + if title in used_names: + title = safe_name(f"{title} {drama.get('id')}", title) + used_names[title] = True + drama_dir = os.path.join(video_root, title) + for episode in episodes: + number = episode.get("number") or 0 + ep_title = safe_name(episode.get("title"), f"EP.{number}") + episode_dir = os.path.join(drama_dir, f"{int(number):02d} {ep_title}") + label = f"{title}/{int(number):02d}" + try: + ok = download_episode(episode, episode_dir) + except Exception as exc: + ok = False + print(f" 下载失败 {label}: {exc}", flush=True) + if ok: + saved += 1 + print(f" 已保存 {label}", flush=True) + else: + failed += 1 + return saved, failed + + +def ask_download(count, video_root): + print(f"\n可下载 {count} 集,目录:{video_root}", flush=True) + answer = input("是否立即下载全部视频?输入 y 下载,其他键跳过: ").strip().lower() + return answer in ("y", "yes") + + def parse_args(): - parser = argparse.ArgumentParser(description="抓取 RaptDrama Passion 目录和播放地址") + parser = argparse.ArgumentParser(description="抓取 RaptDrama 首页目录和播放地址") parser.add_argument( "--from-device", action="store_true", @@ -398,15 +587,15 @@ def main(): token, user = load_token(args.from_device, args.adb, args.serial, args.email, args.password) print(f" {user}", flush=True) - print("获取 Passion 分类", flush=True) - dramas = list_passion(token) - print(f" {len(dramas)} 部", flush=True) + print("获取首页目录", flush=True) + dramas = collect_home(token) results = [] for index, info in enumerate(dramas, 1): vid = info["id"] title = (info.get("title") or "").strip() - print(f" [{index}/{len(dramas)}] {title}", flush=True) + sections = info.get("sections") or [] + print(f" [{index}/{len(dramas)}] {title} [{', '.join(sections)}]", flush=True) chapters = list_chapters(token, vid) vip_by_id = {ch.get("id"): str(ch.get("isvip")) == "1" for ch in chapters} title_by_id = {ch.get("id"): ch.get("title") or "" for ch in chapters} @@ -444,6 +633,7 @@ def main(): "view": info.get("view") or 0, "status": info.get("forstausen") or "", "classify": info.get("classify") or [], + "sections": info.get("sections") or [], "total_episodes": len(episodes), "episodes": episodes, }) @@ -456,6 +646,15 @@ def main(): print(f"完成: {len(results)} 部, {total_eps} 集, 有 m3u8 {cdn_eps}, 已展开 ts {ts_eps}", flush=True) print(f"保存至 {out_file}", flush=True) + video_root = os.path.join(out_dir, "video") + if cdn_eps and ask_download(cdn_eps, video_root): + print("开始下载", flush=True) + saved, failed = download_videos(results, video_root) + print(f"下载结束: 成功 {saved} 集, 失败 {failed} 集", flush=True) + print(f"文件在 {video_root}", flush=True) + elif cdn_eps: + print("已跳过下载", flush=True) + if __name__ == "__main__": main()