Files
StudioLift/skills/youtube-studio-csv-download/scripts/youtube_export_download.py
Sidney Zhang 9020752eed feat(scripts): 添加 PEP 723 脚本元数据并修复 HTTP/2 伪头过滤问题
- 为 build_studio_urls.py 和 lookup_groups.py 添加 PEP 723 依赖声明
- 修复 HTTP/2 伪头导致 requests 抛 InvalidHeader 的问题
- 添加 CDP 端口连通性预检查及启动指引
2026-08-27 13:33:34 +08:00

327 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# -*- coding: utf-8 -*-
# /// script
# requires-python = ">=3.10"
# dependencies = [
# "playwright>=1.40",
# ]
# ///
r"""
YouTube Studio 内容管理器「导出当前视图 → 逗号分隔值 (.csv)」下载脚本
(拦截响应 + 解码 zip + 完整性校验 + 自动保存到下载目录 + 重名去重)。
机制(已实证):前端点击导出后向后端
POST https://studio.youtube.com/youtubei/v1/yta_web/csv_export?alt=json
后端把打好的 zip 以 base64 内联在响应 `zippedData` 字段里(开头 `UEsDBBQ` 即 ZIP 文件头 `PK`)。
注意两个字段细节(均为实证结论):
- `zippedData` 是 **URL-safe base64**-/_ 代替 +//,且省略 = 填充);
- 当前界面导出入口是右上角**下载图标按钮**:只有 `aria-label="导出当前视图"`,没有可见文本。
本脚本拦截该响应base64 解码并校验 zip 完整性后按 `<维度标签> <起始日>_<结束日> <账号名>.zip`
写盘,重名自动加 ` (n)` 后缀n 从 1 起;未捕获到有效响应时自动重试一次触发导出。
登录态:脚本不负责登录,必须复用已登录 YouTube Studio 的浏览器会话,二选一:
- --user-data-dir + --channel chrome|msedge :用已登录的用户数据目录启动(需先关闭该浏览器)
- --connect http://localhost:9222 :附加到已在调试端口运行的浏览器
用法:
python youtube_export_download.py --selftest
python youtube_export_download.py --url "<explore URL>" --channel chrome --user-data-dir "C:/Users/<you>/AppData/Local/Google/Chrome/User Data"
python youtube_export_download.py --url "<explore URL>" --connect http://localhost:9222
"""
import argparse
import base64
import io
import json
import os
import re
import sys
import time
import zipfile
DOWNLOAD_DIR = r"D:\Downloads" # 默认下载目录,可用 --download-dir 覆盖
EXPORT_RESPONSE_TIMEOUT = 30 # 每次触发导出后等待 csv_export 响应的秒数
MAX_EXPORT_ATTEMPTS = 2 # 未捕获到有效响应时的最大触发次数1 次重试)
# 维度类型 -> 文件名前缀标签(生产环境建议从页面「维度」按钮文本读取,这里兜底映射)。
DIMENSION_LABEL = {
"VIDEO": "内容",
"USER": "频道",
"CONTENT_OWNER": "内容",
}
CSV_EXPORT_PATH = "/youtubei/v1/yta_web/csv_export"
def dedup_path(directory, filename):
"""返回不冲突的落盘路径;重名按 `名称 (n).后缀` 递增n 从 1 起。"""
directory = os.path.abspath(directory)
base, ext = os.path.splitext(filename)
candidate = os.path.join(directory, filename)
n = 1
while os.path.exists(candidate):
candidate = os.path.join(directory, f"{base} ({n}){ext}")
n += 1
return candidate
def build_export_filename(export_query, account_name, dimension_label=None):
"""从 csv_export 请求体 `exportQuery` 反推文件名。"""
def fmt_dateid(yyyymmdd):
s = str(yyyymmdd)
return f"{s[0:4]}-{s[4:6]}-{s[6:8]}"
date_range = None
dimension = None
nodes = export_query.get("joinRequest", {}).get("nodes") or []
for node in nodes:
q = node.get("value", {}).get("query") or {}
if not date_range:
tr = q.get("timeRange", {}).get("dateIdRange")
if tr and tr.get("inclusiveStart"):
date_range = (tr["inclusiveStart"], tr.get("exclusiveEnd"))
dims = q.get("dimensions") or []
if dimension is None and dims:
dimension = dims[0].get("type")
if not date_range:
raise ValueError("无法从 exportQuery 解析日期范围")
if dimension_label is None:
dimension_label = DIMENSION_LABEL.get(dimension or "", "")
start, end = date_range
return f"{dimension_label} {fmt_dateid(start)}_{fmt_dateid(end)} {account_name}.zip"
def decode_zipped_data(payload):
"""把 csv_export 响应 payload 里的 zippedData 解码为 zip 字节流。
该字段是 URL-safe base64-/_ 代替 +//,且省略 = 填充):标准 b64decode 遇
-/_ 或非 4 对齐长度会抛 Incorrect padding或静默解出损坏字节流。先补齐填充
再用 altchars 兼容 URL-safe 与标准两种字符表;解码后校验 PK 文件头。
"""
zipped = payload.get("zippedData")
if not zipped:
raise ValueError("响应中缺少 zippedData 字段")
z = str(zipped).strip()
data = base64.b64decode(z + "=" * (-len(z) % 4), altchars=b"-_")
if not data.startswith(b"PK"):
raise ValueError(
"zippedData 解码结果不是 zip 字节流(缺 PK 文件头),"
"响应可能被网关改写或导出接口已变动")
return data
def verify_zip(data):
"""校验内存 zip 完整性:结构可读、成员 CRC 全过;否则抛 ValueError。"""
try:
with zipfile.ZipFile(io.BytesIO(data)) as zf:
bad = zf.testzip()
except zipfile.BadZipFile as e:
raise ValueError(f"zip 结构损坏: {e}") from e
if bad is not None:
raise ValueError(f"zip 成员 CRC 校验失败: {bad}")
def trigger_export(page):
"""点「导出当前视图」→「逗号分隔值 (.csv)」。
导出入口是右上角的下载图标按钮:只有 aria-label、无可见文本
优先按 label 定位get_by_text 兜底旧版有可见文本的界面)。
"""
page.get_by_label("导出当前视图").or_(page.get_by_text("导出当前视图")).first.click()
page.get_by_text("逗号分隔值 (.csv)").click()
def intercept_and_save(page, account_name, download_dir):
"""给 page 绑定 response 拦截器:命中 csv_export 就把 zip 保存到下载目录。"""
import pathlib
saved = []
def on_response(response):
if CSV_EXPORT_PATH not in response.url:
return
try:
payload = response.json()
data = decode_zipped_data(payload)
verify_zip(data)
filename = None
try:
body = json.loads(response.request.post_data or "{}")
filename = build_export_filename(body.get("exportQuery", {}), account_name)
except Exception:
filename = "export.zip" # 反推失败兜底,避免丢内容
filename = re.sub(r"[\\/:*?\"<>|]", "_", filename) # Windows 非法字符
path = dedup_path(download_dir, filename)
pathlib.Path(path).write_bytes(data)
saved.append((filename, len(data), path))
except Exception as e: # noqa: BLE001
print(f"[interceptor] 处理 csv_export 响应失败: {e}", file=sys.stderr)
page.on("response", on_response)
return saved
def _default_user_data_dir(channel):
"""返回指定浏览器的默认用户数据目录Windows用于复用已登录会话。"""
_local = os.environ.get("LOCALAPPDATA") or os.path.expanduser(r"~\AppData\Local")
if channel == "msedge":
return os.path.join(_local, "Microsoft", "Edge", "User Data")
return os.path.join(_local, "Google", "Chrome", "User Data")
def run(url, user_data_dir=None, channel=None, cdp_url=None,
download_dir=None, account_name=None):
"""启动/连接浏览器并触发导出。必须复用已登录会话,否则跳 Google 登录页。"""
from playwright.sync_api import sync_playwright
download_dir = download_dir or DOWNLOAD_DIR
os.makedirs(download_dir, exist_ok=True)
with sync_playwright() as p:
browser = None
context = None
if cdp_url:
browser = p.chromium.connect_over_cdp(cdp_url)
context = browser.contexts[0] if browser.contexts else \
browser.new_context(accept_downloads=True)
page = context.new_page()
page.goto(url, wait_until="domcontentloaded")
elif user_data_dir:
context = p.chromium.launch_persistent_context(
user_data_dir=user_data_dir or _default_user_data_dir(channel),
channel=channel,
headless=False,
accept_downloads=True,
args=["--disable-blink-features=AutomationControlled"],
)
page = context.new_page()
page.goto(url, wait_until="domcontentloaded")
else:
browser = p.chromium.launch(headless=False, channel=channel)
context = browser.new_context(accept_downloads=True)
page = context.new_page()
page.goto(url, wait_until="domcontentloaded")
if "accounts.google" in page.url:
print("[!] 当前会话未登录,已跳转到 Google 登录页。", file=sys.stderr)
print(" 请用 --user-data-dir 或 --connect 复用已登录浏览器后重试。",
file=sys.stderr)
if not account_name:
account_name = "WL Media"
try:
account_name = page.locator(
"ytcp-account-item button, .account-switcher button"
).first.inner_text(timeout=5000).strip() or account_name
except Exception:
pass
saved = intercept_and_save(page, account_name, download_dir)
# 触发导出并等待响应;未捕获到有效 zip未触发/响应无效)自动重试一次
for attempt in range(1, MAX_EXPORT_ATTEMPTS + 1):
trigger_export(page)
deadline = time.time() + EXPORT_RESPONSE_TIMEOUT
while not saved and time.time() < deadline:
page.wait_for_timeout(500)
if saved:
break
if attempt < MAX_EXPORT_ATTEMPTS:
print(f"[重试] 第 {attempt} 次未捕获到有效导出响应,自动重试……",
file=sys.stderr)
if not saved:
print("未捕获到 csv_export 响应,请确认已点击导出且登录态有效。")
else:
for name, size, path in saved:
print(f"已保存: {path} ({size} bytes)")
if browser is not None:
browser.close()
elif context is not None:
context.close()
def selftest():
import tempfile
from pathlib import Path
with tempfile.TemporaryDirectory() as td:
p0 = Path(dedup_path(td, "需求文件.zip"))
assert p0.name == "需求文件.zip", p0.name
p0.write_bytes(b"a")
p1 = Path(dedup_path(td, "需求文件.zip"))
assert p1.name == "需求文件 (1).zip", p1.name
p1.write_bytes(b"b")
p2 = Path(dedup_path(td, "需求文件.zip"))
assert p2.name == "需求文件 (2).zip", p2.name
p2.write_bytes(b"c")
assert Path(dedup_path(td, "需求文件.zip")).name == "需求文件 (3).zip"
assert Path(dedup_path(td, "其他.zip")).name == "其他.zip"
export_query = {
"joinRequest": {"nodes": [{"value": {"query": {
"dimensions": [{"type": "VIDEO"}],
"timeRange": {"dateIdRange": {
"inclusiveStart": 20260723, "exclusiveEnd": 20260820}},
}}}]},
}
name = build_export_filename(export_query, "WL Media")
assert name == "内容 2026-07-23_2026-08-20 WL Media.zip", name
# zip 解码:标准 base64 与 URL-safe 无填充变体都要解出同一字节流
buf = io.BytesIO()
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
zf.writestr("a.csv", "x,y\n1,2")
raw = buf.getvalue()
std = base64.b64encode(raw).decode("ascii")
urlsafe_nopad = std.replace("+", "-").replace("/", "_").rstrip("=")
assert decode_zipped_data({"zippedData": std}) == raw
assert decode_zipped_data({"zippedData": urlsafe_nopad}) == raw
verify_zip(raw)
# 非 zip 字节流要报 PK 文件头错误,而不是静默落盘损坏文件
try:
decode_zipped_data({"zippedData": base64.b64encode(b"not a zip").decode()})
raise AssertionError("非 zip 数据应抛 ValueError")
except ValueError as e:
assert "PK" in str(e), e
print("selftest OK去重、文件名反推、zip 解码与完整性校验全部通过")
if __name__ == "__main__":
ap = argparse.ArgumentParser()
ap.add_argument("--selftest", action="store_true", help="仅跑去重/命名自测")
ap.add_argument("--url", help="explore URL")
ap.add_argument("--user-data-dir", help="浏览器用户数据目录(复用登录态,需先关闭该浏览器)")
ap.add_argument("--channel", choices=["chrome", "msedge"],
help="浏览器品牌,复用登录态时必填其一")
ap.add_argument("--connect", help="通过 CDP 附加到已打开浏览器,如 http://localhost:9222")
ap.add_argument("--download-dir", help=f"下载目录,默认 {DOWNLOAD_DIR}")
ap.add_argument("--account-name", help="账号名(用于文件名),默认从页面读取")
args = ap.parse_args()
if args.selftest:
selftest()
elif args.url:
run(args.url, user_data_dir=args.user_data_dir, channel=args.channel,
cdp_url=args.connect, download_dir=args.download_dir,
account_name=args.account_name)
else:
selftest()
print("\n实际运行需先准备依赖环境uv sync 或直接 uv run见 uv-env-setup 技能),"
"再复用登录态(二选一):\n"
" 方式 A: python youtube_export_download.py --url \"<explore URL>\" "
"--channel chrome --user-data-dir <你的 Chrome User Data 目录>\n"
" 方式 B: python youtube_export_download.py --url \"<explore URL>\" "
"--connect http://localhost:9222")