Skip to content

Commit d2bc105

Browse files
chuanxu742-glitch2233admin
authored andcommitted
feat(browser-act): add Kuaishou search pack
1 parent 5e04641 commit d2bc105

5 files changed

Lines changed: 184 additions & 0 deletions

File tree

Lines changed: 45 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,45 @@
1+
---
2+
name: kuaishou-search
3+
description: "Extract structured public Kuaishou video search results from the current browser page."
4+
---
5+
6+
# Kuaishou — Video Search
7+
8+
> Search keyword → bounded structured video results
9+
10+
## Prerequisites
11+
12+
- Browser Act is available.
13+
- The target browser can reach `kuaishou.com`.
14+
- The current session may need human login or verification.
15+
16+
## Execution
17+
18+
Navigate to:
19+
20+
```text
21+
https://www.kuaishou.com/search/video?searchKey={query}
22+
```
23+
24+
Wait for the page to settle, then run:
25+
26+
```bash
27+
python scripts/extract-search.py --max-results 10
28+
```
29+
30+
The result contains the video URL, caption, author, cover, playable media
31+
URL when exposed by the page, publication timestamp, tags, and bounded
32+
engagement statistics.
33+
34+
## Operational boundary
35+
36+
This pack reads the public search state already present in the browser. It
37+
never automates login, captcha solving, or anti-bot bypass. Login, verification,
38+
regional restrictions, and blocked responses are human-handled conditions.
39+
40+
## Limitations
41+
42+
The manifest extracts the initial search result state only. Cursor-based
43+
follow-up requests and comment collection are intentionally out of scope for
44+
this pack; they require a separate pagination contract and should not be
45+
silently represented as complete results.
Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,15 @@
1+
{
2+
"domain": "video-platforms",
3+
"capability": "kuaishou-search",
4+
"param_schema": [
5+
{"name": "query", "required": true},
6+
{"name": "max_results", "required": false, "default": "10"}
7+
],
8+
"steps": [
9+
{"op": "navigate", "url_template": "https://www.kuaishou.com/search/video?searchKey={query}"},
10+
{"op": "wait", "wait_mode": "stable"},
11+
{"op": "eval_script", "script": "scripts/extract-search.py", "args": ["--max-results", "{max_results}"]}
12+
],
13+
"pagination": {"mode": "none"},
14+
"success": {"min_count": 1, "required_field": "url"}
15+
}
Lines changed: 88 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,88 @@
1+
import argparse
2+
3+
4+
def main() -> None:
5+
parser = argparse.ArgumentParser()
6+
parser.add_argument("--max-results", type=int, default=10)
7+
args = parser.parse_args()
8+
max_results = max(1, min(args.max_results, 50))
9+
10+
js = r"""(() => {
11+
const clean = (value) => String(value || '').replace(/\s+/g, ' ').trim();
12+
const firstUrl = (value) => {
13+
if (typeof value === 'string' && value) return value;
14+
if (Array.isArray(value)) {
15+
for (const item of value) {
16+
const result = firstUrl(item);
17+
if (result) return result;
18+
}
19+
}
20+
if (value && typeof value === 'object') {
21+
for (const key of ['url', 'src', 'srcNoWatermark', 'playUrl']) {
22+
const result = firstUrl(value[key]);
23+
if (result) return result;
24+
}
25+
}
26+
return null;
27+
};
28+
const isoTime = (value) => {
29+
const timestamp = Number(value);
30+
if (!Number.isFinite(timestamp) || timestamp <= 0) return null;
31+
const date = new Date(timestamp < 100000000000 ? timestamp * 1000 : timestamp);
32+
return Number.isNaN(date.getTime()) ? null : date.toISOString();
33+
};
34+
const stateValues = Object.values(window.INIT_STATE || {});
35+
const state = stateValues.find((value) => value && Array.isArray(value.feeds)) || {feeds: []};
36+
const items = state.feeds.map((feed) => {
37+
if (!feed || typeof feed !== 'object') return null;
38+
const photo = feed.photo && typeof feed.photo === 'object' ? feed.photo : null;
39+
if (!photo || !clean(photo.id)) return null;
40+
const author = feed.author && typeof feed.author === 'object' ? feed.author : {};
41+
const comment = feed.comment && typeof feed.comment === 'object' ? feed.comment : {};
42+
const photoId = clean(photo.id);
43+
const caption = clean(photo.caption);
44+
const coverUrl = firstUrl(photo.coverUrl);
45+
const playUrl = firstUrl(photo.manifestH265) || firstUrl(photo.manifest);
46+
const statistics = {
47+
like_count: photo.likeCount,
48+
comment_count: comment.us_c,
49+
collect_count: photo.collectCount,
50+
view_count: photo.viewCount,
51+
share_count: photo.shareCount,
52+
};
53+
Object.keys(statistics).forEach((key) => {
54+
if (statistics[key] === null || statistics[key] === undefined) delete statistics[key];
55+
});
56+
return {
57+
title: caption || `Kuaishou video ${photoId}`,
58+
content: caption,
59+
author: clean(author.name),
60+
author_id: clean(author.id) || null,
61+
author_avatar: firstUrl(author.headerUrl),
62+
url: `https://www.kuaishou.com/short-video/${encodeURIComponent(photoId)}`,
63+
photo_id: photoId,
64+
create_time: photo.timestamp || null,
65+
published_at: isoTime(photo.timestamp),
66+
cover_url: coverUrl,
67+
play_url: playUrl,
68+
statistics,
69+
media: {
70+
type: 'video',
71+
play_url: playUrl,
72+
cover_url: coverUrl,
73+
duration_ms: photo.duration || null,
74+
width: photo.width || null,
75+
height: photo.height || null,
76+
},
77+
tags: Array.isArray(feed.tags)
78+
? feed.tags.map((tag) => clean(tag && tag.name)).filter(Boolean)
79+
: [],
80+
};
81+
}).filter(Boolean).slice(0, MAX_RESULTS);
82+
return {count: items.length, items};
83+
})()"""
84+
print(js.replace("MAX_RESULTS", str(max_results)))
85+
86+
87+
if __name__ == "__main__":
88+
main()

‎tests/integration/test_browser_act_packs_api.py‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -9,6 +9,7 @@
99
SEEDED_WITH_MANIFEST = {
1010
"ecommerce/taobao-keyword-search",
1111
"search-research/google-search-serp",
12+
"video-platforms/kuaishou-search",
1213
}
1314

1415

Lines changed: 35 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,35 @@
1+
"""Contract checks for the Kuaishou Browser Act pack."""
2+
3+
import subprocess
4+
import sys
5+
from pathlib import Path
6+
7+
from backend.browser_act_packs.catalog import PackCatalog
8+
from backend.browser_act_packs.manifest import load_manifest
9+
10+
_PACK = Path(PackCatalog().root) / "video-platforms" / "kuaishou-search"
11+
12+
13+
def test_kuaishou_manifest_points_to_bounded_search_script() -> None:
14+
manifest = load_manifest(_PACK / "channel.manifest.json")
15+
16+
assert manifest.domain == "video-platforms"
17+
assert manifest.capability == "kuaishou-search"
18+
assert manifest.success.required_field == "url"
19+
assert manifest.pagination.mode == "none"
20+
assert manifest.steps[-1].script == "scripts/extract-search.py"
21+
assert (_PACK / manifest.steps[-1].script).is_file()
22+
23+
24+
def test_kuaishou_script_emits_requested_result_bound() -> None:
25+
script = _PACK / "scripts" / "extract-search.py"
26+
result = subprocess.run(
27+
[sys.executable, str(script), "--max-results", "7"],
28+
check=True,
29+
capture_output=True,
30+
text=True,
31+
)
32+
33+
assert "MAX_RESULTS" not in result.stdout
34+
assert ".slice(0, 7)" in result.stdout
35+
assert "kuaishou.com/short-video" in result.stdout

0 commit comments

Comments
 (0)