feat: 扩展首页分类、New Releases 和 Recommend 全量抓取
This commit is contained in:
@@ -4,3 +4,5 @@ __pycache__/
|
||||
.env
|
||||
*.apk
|
||||
raptdrama_passion_full.json
|
||||
raptdrama_catalog.json
|
||||
video/
|
||||
|
||||
+37
-5
@@ -165,7 +165,27 @@
|
||||
| label_origin | string | |
|
||||
| image | string 或 null | |
|
||||
|
||||
已见到的分类:`18` Comedy,`20` Passion,`21` Newest,`22` Western,`23` Comic drama。
|
||||
已见到的分类:`18` Comedy,`20` Passion,`21` Newest,`22` Western,`23` Comic drama。首页顶栏就是这五个标签,脚本用 `sections` 记成 `comedy`、`passion`、`newest`、`western`、`comic`。
|
||||
|
||||
### 首页区块对应
|
||||
|
||||
发现页上能对上接口的区块:
|
||||
|
||||
| 界面 | 接口 | 脚本里的 `sections` |
|
||||
| --- | --- | --- |
|
||||
| 顶栏 Comedy / Passion / Newest / Western / Comic drama | `POST /app/open/classify_video`,`type` 分别为 18 / 20 / 21 / 22 / 23 | `comedy` / `passion` / `newest` / `western` / `comic` |
|
||||
| New Releases | `GET /app/open/orgin` | `new_releases` |
|
||||
| Recommend | `POST /app/open/recommend`,按 `page` 翻页 | `recommend` |
|
||||
|
||||
同一部剧可以同时出现在多个区块。脚本按剧 id 合并,播放地址只请求一次,`sections` 列出它出现过的区块,例如 `["passion","new_releases","recommend"]`。
|
||||
|
||||
Trending Now 对不上单独的短列表。安装包里发现页有 `trendingDramas` 和 `_buildTrendingSection`,但没有单独的拉取方法或 `/app/open/` 路径。核对过的结果:
|
||||
|
||||
- `POST /app/open/extra_broadcast` 是历史页里的 extra broadcast,不是发现页的 Trending Now。当前 9 部里没有截图上的 Bai Jie: A Young Wife、Desire Manor、The Love Curse。改 `type` 或 `ishot` 仍是这 9 部。
|
||||
- `GET /app/open/orgin` 只有 Bai Jie: A Young Wife,没有另外两部。
|
||||
- Recommend 全量里三部都在,但分散在第 4、7、11 页,不是这一行的短列表。
|
||||
- 三部的 `videoinfo.ishot` 都是 `0`。猜的 `/app/open/banner`、`/hot`、`/ranking` 返回 404。
|
||||
- `/app/open/foryou` 不作为首页来源,脚本不调用。
|
||||
|
||||
### POST /app/open/classify_video
|
||||
|
||||
@@ -183,9 +203,19 @@
|
||||
|
||||
### POST /app/open/recommend
|
||||
|
||||
请求体:`{}`
|
||||
首页 Recommend。`page` 从 1 开始,`limit` 用 10。`start` 会被忽略,不能用来翻页。空对象 `{}` 和带 `page`/`limit` 的第一页一样,大约 10 条,不是全站列表。
|
||||
|
||||
`data` 是剧目数组。空对象也只返回约 10 条,不是全站列表。
|
||||
请求体:
|
||||
|
||||
```json
|
||||
{"page": 1, "limit": 10}
|
||||
```
|
||||
|
||||
每页 10 条,相邻页的剧 id 不同。某一页条数不足 10 时还要继续请求下一页。只有空页,或这一页的 id 全部在前面出现过,才停止。脚本最多翻 80 页,避免死循环。
|
||||
|
||||
2026-09-26 用游客 token 翻页:第 1–33 页各 10 条,第 34 页 2 条,第 35 页为空。去重后 332 部。
|
||||
|
||||
`data` 是剧目数组。
|
||||
|
||||
### POST /app/open/foryou
|
||||
|
||||
@@ -204,11 +234,13 @@
|
||||
|
||||
请求体:`{}`
|
||||
|
||||
`data` 是剧目数组。
|
||||
`data` 是剧目数组。这是历史页的列表,不是首页 Trending Now。不传分页参数时返回固定的一小批剧;带 `page` 也不会换成 Trending Now 那一排。
|
||||
|
||||
### GET /app/open/orgin
|
||||
|
||||
无请求体。`data` 是剧目数组。
|
||||
无请求体。`POST` 空对象同样返回这一批。`data` 是剧目数组。
|
||||
|
||||
这是首页 New Releases。当前一次返回 10 部,按较新的剧排在前面。截图里能看到的 Reaping What She Sowed、The Fall of a Mountain Flower、The Delivery Guy's Rise 是这 10 部的最后三部。加上 `type=trending`,或 `page`/`limit`,仍然只有 10 条,不是另一份 Trending Now 列表。
|
||||
|
||||
### POST /app/open/search
|
||||
|
||||
|
||||
+219
-20
@@ -1,9 +1,17 @@
|
||||
"""
|
||||
RaptDrama Passion 分类全量抓取。
|
||||
RaptDrama 首页目录抓取。
|
||||
|
||||
抓取顶部分类(Comedy、Passion、Newest、Western、Comic drama)、
|
||||
New Releases,以及 Recommend 的全部分页。同一部剧只拉一次播放地址,
|
||||
并用 sections 标记它出现在哪些区块。
|
||||
|
||||
首页 Trending Now 没有单独的 open 接口,安装包里的 extra_broadcast
|
||||
对不上截图标题,因此不写入。/app/open/foryou 按要求不调用。
|
||||
|
||||
默认用游客 autologin。也可以用邮箱和密码调用 /app/open/emailLogin 换取 token,
|
||||
或加上 --from-device,用项目内置 adb 读取设备里已登录账号的 accessToken。
|
||||
凭证放在 apasstk 头里,正文是 JSON。已购买或已解锁的集会返回 m3u8。
|
||||
拉完目录后会询问是否把这些视频下载到项目的 video 目录,并按剧名分文件夹。
|
||||
|
||||
重复运行会重新拉最新目录。邮箱和密码用参数或环境变量传入,不要写进源码。
|
||||
"""
|
||||
@@ -23,8 +31,17 @@ from urllib.parse import urljoin
|
||||
socket.setdefaulttimeout(20)
|
||||
|
||||
BASE = "https://apis.raptdrama.com"
|
||||
PASSION_TYPE = 20
|
||||
OUT_NAME = "raptdrama_passion_full.json"
|
||||
OUT_NAME = "raptdrama_catalog.json"
|
||||
# 顶部分类标签。id 来自 classifyv2,列表接口的参数名是 type。
|
||||
CLASSIFY_SECTIONS = (
|
||||
("comedy", "Comedy", 18),
|
||||
("passion", "Passion", 20),
|
||||
("newest", "Newest", 21),
|
||||
("western", "Western", 22),
|
||||
("comic", "Comic drama", 23),
|
||||
)
|
||||
CLASSIFY_MAX_PAGES = 50
|
||||
RECOMMEND_MAX_PAGES = 80
|
||||
PREFS = "/data/data/com.leivideo.raptdrama/shared_prefs/FlutterSharedPreferences.xml"
|
||||
|
||||
HEADERS = {
|
||||
@@ -232,24 +249,102 @@ def http_text(url):
|
||||
return raw.decode("utf-8", "replace")
|
||||
|
||||
|
||||
def list_passion(token):
|
||||
def _as_drama_list(data):
|
||||
if isinstance(data, list):
|
||||
return [item for item in data if isinstance(item, dict)]
|
||||
if isinstance(data, dict):
|
||||
items = data.get("list")
|
||||
if isinstance(items, list):
|
||||
return [item for item in items if isinstance(item, dict)]
|
||||
return []
|
||||
|
||||
|
||||
def list_paged(token, path, method, make_body, max_pages):
|
||||
"""按页拉取,直到空页或整页都是已经见过的 id。max_pages 防止死循环。"""
|
||||
dramas = []
|
||||
seen = set()
|
||||
page = 1
|
||||
while page <= 50:
|
||||
data = api("/app/open/classify_video", token, "POST", {
|
||||
"page": page, "limit": 10, "type": PASSION_TYPE,
|
||||
})
|
||||
items = data if isinstance(data, list) else (data or {}).get("list") or []
|
||||
fresh = [item for item in items if item.get("id") not in seen]
|
||||
if not fresh:
|
||||
stop_page = page
|
||||
reason = "达到页数上限"
|
||||
while page <= max_pages:
|
||||
stop_page = page
|
||||
data = api(path, token, method, make_body(page))
|
||||
items = _as_drama_list(data)
|
||||
if not items:
|
||||
reason = "空页"
|
||||
break
|
||||
for item in fresh:
|
||||
seen.add(item.get("id"))
|
||||
dramas.append(item)
|
||||
fresh = []
|
||||
for item in items:
|
||||
vid = item.get("id")
|
||||
if vid in seen:
|
||||
continue
|
||||
seen.add(vid)
|
||||
fresh.append(item)
|
||||
if not fresh:
|
||||
reason = "整页都是重复 id"
|
||||
break
|
||||
dramas.extend(fresh)
|
||||
page += 1
|
||||
time.sleep(0.05)
|
||||
return dramas
|
||||
return dramas, stop_page, reason
|
||||
|
||||
|
||||
def list_classify(token, type_id):
|
||||
return list_paged(
|
||||
token,
|
||||
"/app/open/classify_video",
|
||||
"POST",
|
||||
lambda page: {"page": page, "limit": 10, "type": type_id},
|
||||
CLASSIFY_MAX_PAGES,
|
||||
)
|
||||
|
||||
|
||||
def list_new_releases(token):
|
||||
items = _as_drama_list(api("/app/open/orgin", token, "GET"))
|
||||
return items, 1, "单次返回"
|
||||
|
||||
|
||||
def list_recommend(token):
|
||||
return list_paged(
|
||||
token,
|
||||
"/app/open/recommend",
|
||||
"POST",
|
||||
lambda page: {"page": page, "limit": 10},
|
||||
RECOMMEND_MAX_PAGES,
|
||||
)
|
||||
|
||||
|
||||
def add_section(catalog, order, section, items):
|
||||
for item in items:
|
||||
vid = item.get("id")
|
||||
if vid is None:
|
||||
continue
|
||||
entry = catalog.get(vid)
|
||||
if entry is None:
|
||||
entry = dict(item)
|
||||
entry["sections"] = []
|
||||
catalog[vid] = entry
|
||||
order.append(vid)
|
||||
if section not in entry["sections"]:
|
||||
entry["sections"].append(section)
|
||||
|
||||
|
||||
def collect_home(token):
|
||||
"""合并首页区块。同一 id 只保留一份剧目,sections 记录出现过的区块。"""
|
||||
catalog = {}
|
||||
order = []
|
||||
for key, label, type_id in CLASSIFY_SECTIONS:
|
||||
items, stop_page, reason = list_classify(token, type_id)
|
||||
add_section(catalog, order, key, items)
|
||||
print(f" {label} {len(items)} 部,停在第 {stop_page} 页({reason})", flush=True)
|
||||
items, _, reason = list_new_releases(token)
|
||||
add_section(catalog, order, "new_releases", items)
|
||||
print(f" New Releases {len(items)} 部({reason})", flush=True)
|
||||
items, stop_page, reason = list_recommend(token)
|
||||
add_section(catalog, order, "recommend", items)
|
||||
print(f" Recommend {len(items)} 部,停在第 {stop_page} 页({reason})", flush=True)
|
||||
print(f" 去重后 {len(order)} 部", flush=True)
|
||||
return [catalog[vid] for vid in order]
|
||||
|
||||
|
||||
def list_chapters(token, vid):
|
||||
@@ -350,8 +445,102 @@ def expand_hls(master_url):
|
||||
return variant, segments
|
||||
|
||||
|
||||
def safe_name(text, fallback):
|
||||
cleaned = re.sub(r'[<>:"/\\|?*\x00-\x1f]', " ", text or "")
|
||||
cleaned = re.sub(r"\s+", " ", cleaned).strip().rstrip(". ")
|
||||
return (cleaned[:80] or fallback)
|
||||
|
||||
|
||||
def download_bytes(url):
|
||||
req = urllib.request.Request(url, headers={"User-Agent": "ExoPlayer"})
|
||||
return _read_response(req)
|
||||
|
||||
|
||||
def localize_playlist(text, local_name_for_uri):
|
||||
lines = []
|
||||
for line in text.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped and not stripped.startswith("#"):
|
||||
lines.append(local_name_for_uri(stripped))
|
||||
else:
|
||||
lines.append(line)
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def download_episode(episode, episode_dir):
|
||||
master_url = episode.get("cdn_url")
|
||||
if not master_url:
|
||||
return False
|
||||
os.makedirs(episode_dir, exist_ok=True)
|
||||
master_text = download_bytes(master_url).decode("utf-8", "replace")
|
||||
variant_rel = next(
|
||||
(line.strip() for line in master_text.splitlines() if line.strip() and not line.startswith("#")),
|
||||
None,
|
||||
)
|
||||
if not variant_rel:
|
||||
return False
|
||||
variant_url = urljoin(master_url, variant_rel)
|
||||
media_text = download_bytes(variant_url).decode("utf-8", "replace")
|
||||
segment_rels = [
|
||||
line.strip() for line in media_text.splitlines()
|
||||
if line.strip() and not line.startswith("#")
|
||||
]
|
||||
if not segment_rels:
|
||||
return False
|
||||
names = [os.path.basename(urljoin(variant_url, rel).split("?", 1)[0]) for rel in segment_rels]
|
||||
for rel, name in zip(segment_rels, names):
|
||||
target = os.path.join(episode_dir, name)
|
||||
if os.path.isfile(target) and os.path.getsize(target) > 0:
|
||||
continue
|
||||
payload = download_bytes(urljoin(variant_url, rel))
|
||||
with open(target, "wb") as handle:
|
||||
handle.write(payload)
|
||||
with open(os.path.join(episode_dir, "video.m3u8"), "w", encoding="utf-8", newline="\n") as handle:
|
||||
handle.write(localize_playlist(media_text, lambda uri: os.path.basename(urljoin(variant_url, uri).split("?", 1)[0])))
|
||||
with open(os.path.join(episode_dir, "playlist.m3u8"), "w", encoding="utf-8", newline="\n") as handle:
|
||||
handle.write(localize_playlist(master_text, lambda _uri: "video.m3u8"))
|
||||
return all(os.path.isfile(os.path.join(episode_dir, name)) and os.path.getsize(os.path.join(episode_dir, name)) > 0 for name in names)
|
||||
|
||||
|
||||
def download_videos(results, video_root):
|
||||
used_names = {}
|
||||
saved = 0
|
||||
failed = 0
|
||||
for drama in results:
|
||||
episodes = [ep for ep in drama["episodes"] if ep.get("cdn_url")]
|
||||
if not episodes:
|
||||
continue
|
||||
title = safe_name(drama.get("title"), f"drama-{drama.get('id')}")
|
||||
if title in used_names:
|
||||
title = safe_name(f"{title} {drama.get('id')}", title)
|
||||
used_names[title] = True
|
||||
drama_dir = os.path.join(video_root, title)
|
||||
for episode in episodes:
|
||||
number = episode.get("number") or 0
|
||||
ep_title = safe_name(episode.get("title"), f"EP.{number}")
|
||||
episode_dir = os.path.join(drama_dir, f"{int(number):02d} {ep_title}")
|
||||
label = f"{title}/{int(number):02d}"
|
||||
try:
|
||||
ok = download_episode(episode, episode_dir)
|
||||
except Exception as exc:
|
||||
ok = False
|
||||
print(f" 下载失败 {label}: {exc}", flush=True)
|
||||
if ok:
|
||||
saved += 1
|
||||
print(f" 已保存 {label}", flush=True)
|
||||
else:
|
||||
failed += 1
|
||||
return saved, failed
|
||||
|
||||
|
||||
def ask_download(count, video_root):
|
||||
print(f"\n可下载 {count} 集,目录:{video_root}", flush=True)
|
||||
answer = input("是否立即下载全部视频?输入 y 下载,其他键跳过: ").strip().lower()
|
||||
return answer in ("y", "yes")
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="抓取 RaptDrama Passion 目录和播放地址")
|
||||
parser = argparse.ArgumentParser(description="抓取 RaptDrama 首页目录和播放地址")
|
||||
parser.add_argument(
|
||||
"--from-device",
|
||||
action="store_true",
|
||||
@@ -398,15 +587,15 @@ def main():
|
||||
token, user = load_token(args.from_device, args.adb, args.serial, args.email, args.password)
|
||||
print(f" {user}", flush=True)
|
||||
|
||||
print("获取 Passion 分类", flush=True)
|
||||
dramas = list_passion(token)
|
||||
print(f" {len(dramas)} 部", flush=True)
|
||||
print("获取首页目录", flush=True)
|
||||
dramas = collect_home(token)
|
||||
|
||||
results = []
|
||||
for index, info in enumerate(dramas, 1):
|
||||
vid = info["id"]
|
||||
title = (info.get("title") or "").strip()
|
||||
print(f" [{index}/{len(dramas)}] {title}", flush=True)
|
||||
sections = info.get("sections") or []
|
||||
print(f" [{index}/{len(dramas)}] {title} [{', '.join(sections)}]", flush=True)
|
||||
chapters = list_chapters(token, vid)
|
||||
vip_by_id = {ch.get("id"): str(ch.get("isvip")) == "1" for ch in chapters}
|
||||
title_by_id = {ch.get("id"): ch.get("title") or "" for ch in chapters}
|
||||
@@ -444,6 +633,7 @@ def main():
|
||||
"view": info.get("view") or 0,
|
||||
"status": info.get("forstausen") or "",
|
||||
"classify": info.get("classify") or [],
|
||||
"sections": info.get("sections") or [],
|
||||
"total_episodes": len(episodes),
|
||||
"episodes": episodes,
|
||||
})
|
||||
@@ -456,6 +646,15 @@ def main():
|
||||
print(f"完成: {len(results)} 部, {total_eps} 集, 有 m3u8 {cdn_eps}, 已展开 ts {ts_eps}", flush=True)
|
||||
print(f"保存至 {out_file}", flush=True)
|
||||
|
||||
video_root = os.path.join(out_dir, "video")
|
||||
if cdn_eps and ask_download(cdn_eps, video_root):
|
||||
print("开始下载", flush=True)
|
||||
saved, failed = download_videos(results, video_root)
|
||||
print(f"下载结束: 成功 {saved} 集, 失败 {failed} 集", flush=True)
|
||||
print(f"文件在 {video_root}", flush=True)
|
||||
elif cdn_eps:
|
||||
print("已跳过下载", flush=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
Reference in New Issue
Block a user