Compare commits

..
3 Commits
2 changed files with 534 additions and 35 deletions
+359 -34
View File
@@ -20,6 +20,7 @@ import time
from dataclasses import dataclass from dataclasses import dataclass
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
from urllib.parse import quote
DEFAULT_USER_URL = ( DEFAULT_USER_URL = (
"https://www.douyin.com/user/" "https://www.douyin.com/user/"
@@ -30,6 +31,8 @@ DEFAULT_BROWSER_PORT = 9223
LISTEN_TARGET = "web/aweme/post/" LISTEN_TARGET = "web/aweme/post/"
RECOMMENDATION_LISTEN_TARGET = "aweme/v2/web/module/feed/" RECOMMENDATION_LISTEN_TARGET = "aweme/v2/web/module/feed/"
SINGLE_VIDEO_LISTEN_TARGET = "web/aweme/detail/" SINGLE_VIDEO_LISTEN_TARGET = "web/aweme/detail/"
SEARCH_LISTEN_TARGET = "aweme/v1/web/general/search/single"
MAX_FILENAME_BYTES = 240
INVALID_FILENAME_CHARS = re.compile(r'[\\/:*?"<>|\r\n\t]') INVALID_FILENAME_CHARS = re.compile(r'[\\/:*?"<>|\r\n\t]')
RECOMMENDATION_URL_PATTERN = re.compile(r"^https?://www\.douyin\.com/?(?:jingxuan)?(?:\?.*)?$") RECOMMENDATION_URL_PATTERN = re.compile(r"^https?://www\.douyin\.com/?(?:jingxuan)?(?:\?.*)?$")
CREATOR_URL_PATTERN = re.compile(r"^https?://www\.douyin\.com/user/[^/?#]+(?:\?.*)?$") CREATOR_URL_PATTERN = re.compile(r"^https?://www\.douyin\.com/user/[^/?#]+(?:\?.*)?$")
@@ -45,11 +48,48 @@ class ResolvedTarget:
aweme_id: str | None = None aweme_id: str | None = None
@dataclass(frozen=True)
class ScrollSettings:
mode: str = "human"
min_wait: float = 2.0
max_wait: float = 8.0
reverse_scroll_probability: float = 0.2
max_runtime: float = 600.0
min_scroll: int = 300
max_scroll: int = 900
min_reverse_scroll: int = 80
max_reverse_scroll: int = 250
@dataclass(frozen=True)
class HumanScrollPlan:
down_distance: int
down_wait: float
reverse_distance: int = 0
reverse_wait: float = 0.0
settle_wait: float = 0.0
def sanitize_filename(value: str, fallback: str = "untitled") -> str: def sanitize_filename(value: str, fallback: str = "untitled") -> str:
cleaned = INVALID_FILENAME_CHARS.sub("_", value).strip(" ._") cleaned = INVALID_FILENAME_CHARS.sub("_", value).strip(" ._")
return cleaned or fallback return cleaned or fallback
def truncate_utf8_bytes(value: str, max_bytes: int) -> str:
if len(value.encode("utf-8")) <= max_bytes:
return value
result = ""
used = 0
for character in value:
character_bytes = len(character.encode("utf-8"))
if used + character_bytes > max_bytes:
break
result += character
used += character_bytes
return result.rstrip(" ._")
def is_recommendation_url(value: str) -> bool: def is_recommendation_url(value: str) -> bool:
return bool(RECOMMENDATION_URL_PATTERN.match(value.strip())) return bool(RECOMMENDATION_URL_PATTERN.match(value.strip()))
@@ -77,6 +117,10 @@ def build_video_page_url(aweme_id: str) -> str:
return f"https://www.douyin.com/video/{aweme_id}" return f"https://www.douyin.com/video/{aweme_id}"
def build_search_page_url(keyword: str) -> str:
return f"https://www.douyin.com/search/{quote(keyword)}?type=general"
def parse_target_input(value: str, source: str) -> ResolvedTarget: def parse_target_input(value: str, source: str) -> ResolvedTarget:
normalized = value.strip() normalized = value.strip()
if is_recommendation_url(normalized): if is_recommendation_url(normalized):
@@ -181,11 +225,20 @@ def build_output_path(
author_name: str | None = None, author_name: str | None = None,
) -> Path: ) -> Path:
safe_title = sanitize_filename(title, fallback="untitled") safe_title = sanitize_filename(title, fallback="untitled")
suffix = f"-{video_id}.mp4"
if author_name: if author_name:
safe_author = sanitize_filename(author_name, fallback="unknown") safe_author = sanitize_filename(author_name, fallback="unknown")
filename = f"[{safe_author}]{safe_title}-{video_id}.mp4" prefix = f"[{safe_author}]"
else: else:
filename = f"{safe_title}-{video_id}.mp4" prefix = ""
title_budget = MAX_FILENAME_BYTES - len(prefix.encode("utf-8")) - len(suffix.encode("utf-8"))
if title_budget < 1:
prefix_budget = MAX_FILENAME_BYTES - len(suffix.encode("utf-8")) - 1
prefix = truncate_utf8_bytes(prefix, max(1, prefix_budget))
title_budget = MAX_FILENAME_BYTES - len(prefix.encode("utf-8")) - len(suffix.encode("utf-8"))
filename = f"{prefix}{truncate_utf8_bytes(safe_title, max(1, title_budget))}{suffix}"
return output_dir / filename return output_dir / filename
@@ -278,6 +331,25 @@ def parse_single_aweme_item(body: Any) -> dict[str, str]:
raise ValueError("接口响应中缺少可下载的单视频数据。") raise ValueError("接口响应中缺少可下载的单视频数据。")
def parse_search_items(body: Any) -> list[dict[str, str]]:
if not isinstance(body, dict):
raise ValueError("接口响应不是字典,无法解析。")
data = body.get("data")
if not isinstance(data, list):
raise ValueError("搜索接口响应中缺少 data。")
aweme_list = []
for entry in data:
if not isinstance(entry, dict):
continue
aweme_info = entry.get("aweme_info")
if isinstance(aweme_info, dict):
aweme_list.append(aweme_info)
return parse_aweme_items({"aweme_list": aweme_list})
def build_headers(referer: str) -> dict[str, str]: def build_headers(referer: str) -> dict[str, str]:
return { return {
"referer": referer, "referer": referer,
@@ -319,7 +391,8 @@ def create_page(chromium_page_cls: Any, chromium_options_cls: Any, browser_port:
def wait_for_aweme_packet(page: Any, timeout: int) -> Any | None: def wait_for_aweme_packet(page: Any, timeout: int) -> Any | None:
try: try:
return page.listen.wait(timeout=timeout) packet = page.listen.wait(timeout=timeout)
return packet if packet else None
except Exception as exc: except Exception as exc:
print(f"[WARN] 等待接口数据超时或失败: {exc}") print(f"[WARN] 等待接口数据超时或失败: {exc}")
return None return None
@@ -330,12 +403,93 @@ def scroll_to_next_page(page: Any) -> None:
time.sleep(2) time.sleep(2)
def human_like_scroll(page: Any) -> None: def create_human_scroll_plan(
"""模拟人类滚动行为:随机滚动距离和随机停顿时间""" settings: ScrollSettings,
scroll_distance = random.randint(300, 800) random_module: Any = random,
page.run_js(f"window.scrollBy(0, {scroll_distance});") ) -> HumanScrollPlan:
sleep_time = random.uniform(1.5, 4.0) down_distance = random_module.randint(settings.min_scroll, settings.max_scroll)
time.sleep(sleep_time) down_wait = random_module.uniform(settings.min_wait, settings.max_wait)
settle_wait = random_module.uniform(settings.min_wait, settings.max_wait)
reverse_distance = 0
reverse_wait = 0.0
if random_module.random() < settings.reverse_scroll_probability:
reverse_distance = random_module.randint(
settings.min_reverse_scroll,
settings.max_reverse_scroll,
)
reverse_wait = random_module.uniform(1.0, min(3.0, settings.max_wait))
return HumanScrollPlan(
down_distance=down_distance,
down_wait=down_wait,
reverse_distance=reverse_distance,
reverse_wait=reverse_wait,
settle_wait=settle_wait,
)
def run_scroll_step(page: Any, distance: int) -> bool:
script = f"""
const distance = {distance};
function findMainScrollContainer() {{
const preferredSelectors = ['.tKqwmYAX', '.route-scroll-container', '.semi-tabs-content'];
for (const selector of preferredSelectors) {{
const el = document.querySelector(selector);
if (el && el.scrollHeight > el.clientHeight + 20) {{
return el;
}}
}}
const candidates = Array.from(document.querySelectorAll('*'))
.filter((el) => {{
const rect = el.getBoundingClientRect();
return rect.width > 300
&& rect.height > 200
&& el.scrollHeight > el.clientHeight + 20;
}})
.sort((a, b) => {{
const areaA = a.getBoundingClientRect().width * a.getBoundingClientRect().height;
const areaB = b.getBoundingClientRect().width * b.getBoundingClientRect().height;
return areaB - areaA;
}});
return candidates[0] || null;
}}
const scrollTarget = findMainScrollContainer();
if (scrollTarget) {{
scrollTarget.scrollBy(0, distance);
return true;
}}
return false;
"""
scrolled_container = bool(page.run_js(script))
if not scrolled_container:
page.run_js(f"window.scrollBy(0, {distance});")
return scrolled_container
def run_human_scroll_sequence(page: Any, plan: HumanScrollPlan) -> None:
run_scroll_step(page, plan.down_distance)
print(f"[INFO] 向下滚动 {plan.down_distance}px,停留 {plan.down_wait:.1f}s")
time.sleep(plan.down_wait)
if plan.reverse_distance > 0:
run_scroll_step(page, -plan.reverse_distance)
print(f"[INFO] 小幅回滚 {plan.reverse_distance}px,停留 {plan.reverse_wait:.1f}s")
time.sleep(plan.reverse_wait)
forward_distance = plan.reverse_distance * 2
run_scroll_step(page, forward_distance)
if plan.settle_wait > 0:
print(f"[INFO] 继续停留 {plan.settle_wait:.1f}s")
time.sleep(plan.settle_wait)
def human_like_scroll(page: Any, settings: ScrollSettings | None = None) -> None:
scroll_settings = settings or ScrollSettings()
run_human_scroll_sequence(page, create_human_scroll_plan(scroll_settings))
def download_video( def download_video(
@@ -435,6 +589,7 @@ def collect_recommendations(
timeout: int, timeout: int,
output_dir: Path, output_dir: Path,
browser_port: int | None, browser_port: int | None,
scroll_settings: ScrollSettings | None = None,
) -> int: ) -> int:
requests_module, chromium_page_cls, chromium_options_cls = import_runtime_dependencies() requests_module, chromium_page_cls, chromium_options_cls = import_runtime_dependencies()
headers = build_headers("https://www.douyin.com/") headers = build_headers("https://www.douyin.com/")
@@ -450,16 +605,22 @@ def collect_recommendations(
downloaded = 0 downloaded = 0
seen_ids: set[str] = set() seen_ids: set[str] = set()
consecutive_empty = 0 consecutive_empty = 0
max_consecutive_empty = 3 max_consecutive_empty = 6
settings = scroll_settings or ScrollSettings()
started_at = time.monotonic()
while downloaded < max_videos: while downloaded < max_videos:
if settings.max_runtime > 0 and time.monotonic() - started_at >= settings.max_runtime:
print("[INFO] 已达到最大运行时间,结束抓取。")
break
packet = wait_for_aweme_packet(page, timeout=timeout) packet = wait_for_aweme_packet(page, timeout=timeout)
if packet is None: if packet is None:
consecutive_empty += 1 consecutive_empty += 1
if consecutive_empty >= max_consecutive_empty: if consecutive_empty >= max_consecutive_empty:
print("[INFO] 连续多次未获取到新数据,结束抓取。") print("[INFO] 连续多次未获取到新数据,结束抓取。")
break break
human_like_scroll(page) human_like_scroll(page, settings=settings)
continue continue
try: try:
@@ -470,14 +631,14 @@ def collect_recommendations(
consecutive_empty += 1 consecutive_empty += 1
if consecutive_empty >= max_consecutive_empty: if consecutive_empty >= max_consecutive_empty:
break break
human_like_scroll(page) human_like_scroll(page, settings=settings)
continue continue
if not items: if not items:
consecutive_empty += 1 consecutive_empty += 1
if consecutive_empty >= max_consecutive_empty: if consecutive_empty >= max_consecutive_empty:
break break
human_like_scroll(page) human_like_scroll(page, settings=settings)
continue continue
consecutive_empty = 0 consecutive_empty = 0
@@ -518,7 +679,109 @@ def collect_recommendations(
if consecutive_empty >= max_consecutive_empty: if consecutive_empty >= max_consecutive_empty:
break break
human_like_scroll(page) human_like_scroll(page, settings=settings)
return downloaded
def collect_search_results(
keyword: str,
max_videos: int,
timeout: int,
output_dir: Path,
browser_port: int | None,
scroll_settings: ScrollSettings | None = None,
) -> int:
requests_module, chromium_page_cls, chromium_options_cls = import_runtime_dependencies()
search_url = build_search_page_url(keyword)
headers = build_headers(search_url)
if browser_port is not None:
ensure_browser_debug_port_ready(browser_port)
page = create_page(chromium_page_cls, chromium_options_cls, browser_port)
page.listen.start(SEARCH_LISTEN_TARGET)
print(f"[INFO] 正在打开抖音搜索页:{keyword}。若出现登录或验证码,请先在浏览器窗口里完成。")
page.get(search_url)
time.sleep(3)
downloaded = 0
seen_ids: set[str] = set()
consecutive_empty = 0
max_consecutive_empty = 6
settings = scroll_settings or ScrollSettings()
started_at = time.monotonic()
while downloaded < max_videos:
if settings.max_runtime > 0 and time.monotonic() - started_at >= settings.max_runtime:
print("[INFO] 已达到最大运行时间,结束抓取。")
break
packet = wait_for_aweme_packet(page, timeout=timeout)
if packet is None:
consecutive_empty += 1
if consecutive_empty >= max_consecutive_empty:
print("[INFO] 连续多次未获取到新搜索数据,结束抓取。")
break
human_like_scroll(page, settings=settings)
continue
try:
payload = extract_aweme_payload(packet.response)
items = parse_search_items(payload)
except Exception as exc:
print(f"[WARN] 解析搜索接口数据失败: {exc}")
consecutive_empty += 1
if consecutive_empty >= max_consecutive_empty:
break
human_like_scroll(page, settings=settings)
continue
if not items:
consecutive_empty += 1
if consecutive_empty >= max_consecutive_empty:
break
human_like_scroll(page, settings=settings)
continue
consecutive_empty = 0
new_items_in_batch = 0
for item in items:
if item["video_id"] in seen_ids:
continue
if downloaded >= max_videos:
break
seen_ids.add(item["video_id"])
output_path = build_output_path(
title=item["title"],
video_id=item["video_id"],
output_dir=output_dir,
author_name=item.get("author_name"),
)
try:
download_video(
requests_module=requests_module,
headers=headers,
video_url=item["video_url"],
output_path=output_path,
)
except Exception as exc:
print(f"[WARN] 下载失败 {item['video_id']}: {exc}")
continue
downloaded += 1
new_items_in_batch += 1
print(f"[OK] 已保存: {output_path}")
if new_items_in_batch == 0:
consecutive_empty += 1
if consecutive_empty >= max_consecutive_empty:
break
human_like_scroll(page, settings=settings)
return downloaded return downloaded
@@ -596,6 +859,41 @@ def build_parser() -> argparse.ArgumentParser:
default=50, default=50,
help="推荐流最大抓取数量,默认 50", help="推荐流最大抓取数量,默认 50",
) )
parser.add_argument(
"--search-keyword",
default=None,
help="搜索关键词;提供后抓取搜索结果页视频",
)
parser.add_argument(
"--scroll-mode",
choices=["human"],
default="human",
help="推荐流滚动模式,默认 human",
)
parser.add_argument(
"--min-wait",
type=float,
default=2.0,
help="推荐流每次滚动后的最短等待秒数,默认 2",
)
parser.add_argument(
"--max-wait",
type=float,
default=8.0,
help="推荐流每次滚动后的最长等待秒数,默认 8",
)
parser.add_argument(
"--reverse-scroll-probability",
type=float,
default=0.2,
help="推荐流小幅回滚概率,取值 0 到 1,默认 0.2",
)
parser.add_argument(
"--max-runtime",
type=float,
default=600.0,
help="推荐流最大运行秒数,默认 600;设置为 0 表示不限制",
)
return parser return parser
@@ -611,34 +909,61 @@ def main(argv: list[str] | None = None) -> int:
parser.error("--browser-port 必须大于 0") parser.error("--browser-port 必须大于 0")
if args.max_videos <= 0: if args.max_videos <= 0:
parser.error("--max-videos 必须大于 0") parser.error("--max-videos 必须大于 0")
if args.min_wait < 0:
parser.error("--min-wait 不能小于 0")
if args.max_wait < args.min_wait:
parser.error("--max-wait 必须大于或等于 --min-wait")
if not 0 <= args.reverse_scroll_probability <= 1:
parser.error("--reverse-scroll-probability 必须在 0 到 1 之间")
if args.max_runtime < 0:
parser.error("--max-runtime 不能小于 0")
scroll_settings = ScrollSettings(
mode=args.scroll_mode,
min_wait=args.min_wait,
max_wait=args.max_wait,
reverse_scroll_probability=args.reverse_scroll_probability,
max_runtime=args.max_runtime,
)
try: try:
target = resolve_cli_target(args.target, browser_port=args.browser_port) if args.search_keyword:
if target.kind == "creator": total = collect_search_results(
total = collect_videos( keyword=args.search_keyword,
user_url=target.value,
max_pages=args.pages,
timeout=args.timeout,
output_dir=Path(args.output_dir),
browser_port=args.browser_port,
auto_scroll=args.pages > 1,
)
elif target.kind == "recommendation":
total = collect_recommendations(
max_videos=args.max_videos, max_videos=args.max_videos,
timeout=args.timeout, timeout=args.timeout,
output_dir=Path(args.output_dir), output_dir=Path(args.output_dir),
browser_port=args.browser_port, browser_port=args.browser_port,
) scroll_settings=scroll_settings,
elif target.kind == "single-video":
total = collect_single_video(
target=target,
timeout=args.timeout,
output_dir=Path(args.output_dir),
browser_port=args.browser_port,
) )
else: else:
raise RuntimeError(f"不支持的目标类型: {target.kind}") target = resolve_cli_target(args.target, browser_port=args.browser_port)
if target.kind == "creator":
total = collect_videos(
user_url=target.value,
max_pages=args.pages,
timeout=args.timeout,
output_dir=Path(args.output_dir),
browser_port=args.browser_port,
auto_scroll=args.pages > 1,
)
elif target.kind == "recommendation":
total = collect_recommendations(
max_videos=args.max_videos,
timeout=args.timeout,
output_dir=Path(args.output_dir),
browser_port=args.browser_port,
scroll_settings=scroll_settings,
)
elif target.kind == "single-video":
total = collect_single_video(
target=target,
timeout=args.timeout,
output_dir=Path(args.output_dir),
browser_port=args.browser_port,
)
else:
raise RuntimeError(f"不支持的目标类型: {target.kind}")
except RuntimeError as exc: except RuntimeError as exc:
print(f"[ERROR] {exc}") print(f"[ERROR] {exc}")
return 1 return 1
+175 -1
View File
@@ -60,6 +60,26 @@ class FakeRuntimePage:
raise AssertionError(f"unexpected scroll script: {script}") raise AssertionError(f"unexpected scroll script: {script}")
class FakeScrollPage:
def __init__(self):
self.scripts = []
def run_js(self, script):
self.scripts.append(script)
class FakeContainerScrollPage:
def __init__(self, container_found=True):
self.container_found = container_found
self.scripts = []
def run_js(self, script):
self.scripts.append(script)
if "findMainScrollContainer" in script:
return self.container_found
return None
class DouyinModuleTests(unittest.TestCase): class DouyinModuleTests(unittest.TestCase):
def test_module_can_import_without_optional_runtime_dependencies(self) -> None: def test_module_can_import_without_optional_runtime_dependencies(self) -> None:
module = importlib.import_module("Douyin") module = importlib.import_module("Douyin")
@@ -98,6 +118,16 @@ class DouyinModuleTests(unittest.TestCase):
) )
self.assertEqual(output_path.as_posix(), "video/[测试博主]测试标题-123456.mp4") self.assertEqual(output_path.as_posix(), "video/[测试博主]测试标题-123456.mp4")
def test_build_output_path_limits_long_filename(self) -> None:
module = importlib.import_module("Douyin")
output_path = module.build_output_path(
title="超长标题" * 100,
video_id="7619989983668240802",
author_name="超长博主名" * 20,
)
self.assertLessEqual(len(output_path.name.encode("utf-8")), 240)
self.assertTrue(output_path.name.endswith("-7619989983668240802.mp4"))
def test_extract_aweme_payload_uses_dict_body(self) -> None: def test_extract_aweme_payload_uses_dict_body(self) -> None:
module = importlib.import_module("Douyin") module = importlib.import_module("Douyin")
response = FakeResponse({"aweme_list": []}, "") response = FakeResponse({"aweme_list": []}, "")
@@ -111,11 +141,79 @@ class DouyinModuleTests(unittest.TestCase):
{"aweme_list": [{"aweme_id": "1"}]}, {"aweme_list": [{"aweme_id": "1"}]},
) )
def test_wait_for_aweme_packet_treats_false_listener_result_as_missing(self) -> None:
module = importlib.import_module("Douyin")
page = mock.MagicMock()
page.listen.wait.return_value = False
self.assertIsNone(module.wait_for_aweme_packet(page, timeout=10))
def test_build_browser_address_from_port(self) -> None: def test_build_browser_address_from_port(self) -> None:
module = importlib.import_module("Douyin") module = importlib.import_module("Douyin")
self.assertEqual(module.build_browser_address(9223), "127.0.0.1:9223") self.assertEqual(module.build_browser_address(9223), "127.0.0.1:9223")
self.assertIsNone(module.build_browser_address(None)) self.assertIsNone(module.build_browser_address(None))
def test_default_scroll_settings_uses_human_mode(self) -> None:
module = importlib.import_module("Douyin")
settings = module.ScrollSettings()
self.assertEqual(settings.mode, "human")
self.assertEqual(settings.min_wait, 2.0)
self.assertEqual(settings.max_wait, 8.0)
self.assertEqual(settings.reverse_scroll_probability, 0.2)
def test_create_human_scroll_plan_uses_configured_ranges(self) -> None:
module = importlib.import_module("Douyin")
settings = module.ScrollSettings(
min_wait=2.0,
max_wait=4.0,
min_scroll=300,
max_scroll=900,
reverse_scroll_probability=0.0,
)
plan = module.create_human_scroll_plan(settings, random_module=module.random.Random(7))
self.assertGreaterEqual(plan.down_distance, 300)
self.assertLessEqual(plan.down_distance, 900)
self.assertGreaterEqual(plan.down_wait, 2.0)
self.assertLessEqual(plan.down_wait, 4.0)
self.assertEqual(plan.reverse_distance, 0)
def test_create_human_scroll_plan_can_include_reverse_scroll(self) -> None:
module = importlib.import_module("Douyin")
settings = module.ScrollSettings(reverse_scroll_probability=1.0)
plan = module.create_human_scroll_plan(settings, random_module=module.random.Random(3))
self.assertGreaterEqual(plan.reverse_distance, 80)
self.assertLessEqual(plan.reverse_distance, 250)
self.assertGreater(plan.reverse_wait, 0)
def test_run_human_scroll_sequence_scrolls_down_and_optionally_back_up(self) -> None:
module = importlib.import_module("Douyin")
page = FakeScrollPage()
plan = module.HumanScrollPlan(
down_distance=500,
down_wait=2.5,
reverse_distance=120,
reverse_wait=1.0,
settle_wait=3.0,
)
with mock.patch.object(module.time, "sleep") as mocked_sleep:
module.run_human_scroll_sequence(page, plan)
self.assertIn("window.scrollBy(0, 500);", page.scripts)
self.assertIn("window.scrollBy(0, -120);", page.scripts)
self.assertIn("window.scrollBy(0, 240);", page.scripts)
mocked_sleep.assert_has_calls([mock.call(2.5), mock.call(1.0), mock.call(3.0)])
def test_run_scroll_step_prefers_main_scroll_container(self) -> None:
module = importlib.import_module("Douyin")
page = FakeContainerScrollPage(container_found=True)
self.assertTrue(module.run_scroll_step(page, 500))
self.assertIn("const distance = 500;", page.scripts[-1])
self.assertIn("scrollTarget.scrollBy(0, distance);", page.scripts[-1])
def test_run_scroll_step_falls_back_to_window_when_container_is_missing(self) -> None:
module = importlib.import_module("Douyin")
page = FakeContainerScrollPage(container_found=False)
self.assertFalse(module.run_scroll_step(page, 500))
self.assertEqual(page.scripts[-1], "window.scrollBy(0, 500);")
def test_ensure_browser_debug_port_ready_accepts_open_port(self) -> None: def test_ensure_browser_debug_port_ready_accepts_open_port(self) -> None:
module = importlib.import_module("Douyin") module = importlib.import_module("Douyin")
connection = mock.MagicMock() connection = mock.MagicMock()
@@ -344,6 +442,38 @@ class DouyinModuleTests(unittest.TestCase):
"https://www.douyin.com/video/7619989983668240802", "https://www.douyin.com/video/7619989983668240802",
) )
def test_build_search_page_url_encodes_keyword(self) -> None:
module = importlib.import_module("Douyin")
self.assertEqual(
module.build_search_page_url("猫咪"),
"https://www.douyin.com/search/%E7%8C%AB%E5%92%AA?type=general",
)
def test_parse_search_items_extracts_aweme_info(self) -> None:
module = importlib.import_module("Douyin")
payload = {
"data": [
{
"type": 1,
"aweme_info": {
"aweme_id": "7319795133048769829",
"desc": "猫咪视频",
"author": {"nickname": "奶芙芙", "uid": "75478174642"},
"video": {
"play_addr_lowbr": {
"url_list": ["https://v26-web.douyinvod.com/example/search.mp4"]
}
},
},
}
]
}
items = module.parse_search_items(payload)
self.assertEqual(len(items), 1)
self.assertEqual(items[0]["video_id"], "7319795133048769829")
self.assertEqual(items[0]["author_name"], "奶芙芙")
self.assertEqual(items[0]["video_url"], "https://v26-web.douyinvod.com/example/search.mp4")
def test_collect_recommendations_downloads_videos_with_author_prefix(self) -> None: def test_collect_recommendations_downloads_videos_with_author_prefix(self) -> None:
module = importlib.import_module("Douyin") module = importlib.import_module("Douyin")
packet = FakePacket( packet = FakePacket(
@@ -367,7 +497,7 @@ class DouyinModuleTests(unittest.TestCase):
with mock.patch.object(module, "import_runtime_dependencies", return_value=(object(), object(), object())): with mock.patch.object(module, "import_runtime_dependencies", return_value=(object(), object(), object())):
with mock.patch.object(module, "create_page", return_value=page): with mock.patch.object(module, "create_page", return_value=page):
with mock.patch.object(module, "download_video") as mocked_download: with mock.patch.object(module, "download_video") as mocked_download:
with mock.patch.object(module, "scroll_to_next_page"): with mock.patch.object(module, "human_like_scroll"):
downloaded = module.collect_recommendations( downloaded = module.collect_recommendations(
max_videos=50, max_videos=50,
timeout=10, timeout=10,
@@ -455,6 +585,49 @@ class DouyinModuleTests(unittest.TestCase):
args = module.build_parser().parse_args(["--max-videos", "30"]) args = module.build_parser().parse_args(["--max-videos", "30"])
self.assertEqual(args.max_videos, 30) self.assertEqual(args.max_videos, 30)
def test_build_parser_has_human_scroll_arguments(self) -> None:
module = importlib.import_module("Douyin")
args = module.build_parser().parse_args(
[
"--scroll-mode",
"human",
"--min-wait",
"3",
"--max-wait",
"9",
"--reverse-scroll-probability",
"0.4",
"--max-runtime",
"600",
]
)
self.assertEqual(args.scroll_mode, "human")
self.assertEqual(args.min_wait, 3)
self.assertEqual(args.max_wait, 9)
self.assertEqual(args.reverse_scroll_probability, 0.4)
self.assertEqual(args.max_runtime, 600)
def test_build_parser_has_search_keyword_argument(self) -> None:
module = importlib.import_module("Douyin")
args = module.build_parser().parse_args(["--search-keyword", "猫咪"])
self.assertEqual(args.search_keyword, "猫咪")
def test_main_dispatches_search_flow_for_search_keyword(self) -> None:
module = importlib.import_module("Douyin")
stdout = io.StringIO()
with redirect_stdout(stdout):
with mock.patch.object(module, "collect_search_results", return_value=7) as mocked_collect:
exit_code = module.main(["--search-keyword", "猫咪"])
self.assertEqual(exit_code, 0)
mocked_collect.assert_called_once_with(
keyword="猫咪",
max_videos=50,
timeout=10,
output_dir=module.Path("video"),
browser_port=9223,
scroll_settings=module.ScrollSettings(),
)
def test_build_parser_defaults_to_zero_argument_current_page_flow(self) -> None: def test_build_parser_defaults_to_zero_argument_current_page_flow(self) -> None:
module = importlib.import_module("Douyin") module = importlib.import_module("Douyin")
args = module.build_parser().parse_args([]) args = module.build_parser().parse_args([])
@@ -488,6 +661,7 @@ class DouyinModuleTests(unittest.TestCase):
timeout=10, timeout=10,
output_dir=module.Path("video"), output_dir=module.Path("video"),
browser_port=9223, browser_port=9223,
scroll_settings=module.ScrollSettings(),
) )
def test_main_without_target_dispatches_current_page_creator_flow(self) -> None: def test_main_without_target_dispatches_current_page_creator_flow(self) -> None: