curl -sS -X POST "https://api.adscrawl.net/spa-extract" \
-H "content-type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"url": "https://example.com/dashboard",
"mode": "extract",
"waitUntil": "domcontentloaded",
"waitFor": { "selector": "h1", "timeoutMs": 15000 },
"fields": {
"title": { "source": "dom", "selector": "h1", "parse": "string" },
"subtitle": { "source": "dom", "selector": "p.subtitle", "parse": "string" },
"count": { "source": "dom", "selector": ".metric-value", "parse": "number" }
},
"countryCode": "GLOBAL",
"userAgentMode": "random"
}'
const response = await fetch("https://api.adscrawl.net/spa-extract", {
method: "POST",
headers: {
"content-type": "application/json",
"x-api-key": "YOUR_API_KEY",
},
body: JSON.stringify({
url: "https://example.com/dashboard",
mode: "extract",
waitUntil: "domcontentloaded",
waitFor: { selector: "h1", timeoutMs: 15000 },
fields: {
title: { source: "dom", selector: "h1", parse: "string" },
subtitle: { source: "dom", selector: "p.subtitle", parse: "string" },
count: { source: "dom", selector: ".metric-value", parse: "number" },
},
countryCode: "GLOBAL",
userAgentMode: "random",
}),
});
const result = await response.json();
console.log(result.data);
import httpx
response = httpx.post(
"https://api.adscrawl.net/spa-extract",
headers={
"content-type": "application/json",
"x-api-key": "YOUR_API_KEY",
},
json={
"url": "https://example.com/dashboard",
"mode": "extract",
"waitUntil": "domcontentloaded",
"waitFor": {"selector": "h1", "timeoutMs": 15000},
"fields": {
"title": {"source": "dom", "selector": "h1", "parse": "string"},
"subtitle": {"source": "dom", "selector": "p.subtitle", "parse": "string"},
"count": {"source": "dom", "selector": ".metric-value", "parse": "number"},
},
"countryCode": "GLOBAL",
"userAgentMode": "random",
},
)
response.raise_for_status()
print(response.json()["data"])
{
"mode": "extract",
"page": {
"url": "https://example.com/dashboard",
"title": "Example Dashboard"
},
"data": {
"title": "Example Dashboard"
},
"missingFields": []
}
{
"mode": "inspect",
"page": {
"url": "https://example.com/dashboard",
"title": "Example Dashboard"
},
"candidates": {
"dom": { "metrics": [], "tables": [] },
"network": []
},
"suggestedPlan": {
"fields": {},
"schema": {
"type": "object",
"properties": {},
"additionalProperties": false
}
}
}
{
"mode": "extract",
"page": {
"url": "https://trends.google.com/trends/explore?q=adspower%2Cplaywright",
"title": "Google Trends"
},
"data": {
"averages": [
{ "query": "adspower", "extractedValue": 42 },
{ "query": "playwright", "extractedValue": 71 }
],
"interestOverTime": []
},
"missingFields": [],
"source": "sunbrowser",
"cached": false,
"stale": false,
"collectedAt": "2026-08-03T02:04:05.123456789Z",
"attempts": 2
}
{
"code": "INSUFFICIENT_CREDITS"
}
{
"code": "SPA_REQUIRED_FIELDS_MISSING"
}
浏览器任务
提取 SPA 数据
通过自定义 CSS 或网络字段、内置模板或 inspect 模式发现提取候选方案,从 JavaScript 密集型 SPA 中提取结构化数据。
POST
/
spa-extract
curl -sS -X POST "https://api.adscrawl.net/spa-extract" \
-H "content-type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"url": "https://example.com/dashboard",
"mode": "extract",
"waitUntil": "domcontentloaded",
"waitFor": { "selector": "h1", "timeoutMs": 15000 },
"fields": {
"title": { "source": "dom", "selector": "h1", "parse": "string" },
"subtitle": { "source": "dom", "selector": "p.subtitle", "parse": "string" },
"count": { "source": "dom", "selector": ".metric-value", "parse": "number" }
},
"countryCode": "GLOBAL",
"userAgentMode": "random"
}'
const response = await fetch("https://api.adscrawl.net/spa-extract", {
method: "POST",
headers: {
"content-type": "application/json",
"x-api-key": "YOUR_API_KEY",
},
body: JSON.stringify({
url: "https://example.com/dashboard",
mode: "extract",
waitUntil: "domcontentloaded",
waitFor: { selector: "h1", timeoutMs: 15000 },
fields: {
title: { source: "dom", selector: "h1", parse: "string" },
subtitle: { source: "dom", selector: "p.subtitle", parse: "string" },
count: { source: "dom", selector: ".metric-value", parse: "number" },
},
countryCode: "GLOBAL",
userAgentMode: "random",
}),
});
const result = await response.json();
console.log(result.data);
import httpx
response = httpx.post(
"https://api.adscrawl.net/spa-extract",
headers={
"content-type": "application/json",
"x-api-key": "YOUR_API_KEY",
},
json={
"url": "https://example.com/dashboard",
"mode": "extract",
"waitUntil": "domcontentloaded",
"waitFor": {"selector": "h1", "timeoutMs": 15000},
"fields": {
"title": {"source": "dom", "selector": "h1", "parse": "string"},
"subtitle": {"source": "dom", "selector": "p.subtitle", "parse": "string"},
"count": {"source": "dom", "selector": ".metric-value", "parse": "number"},
},
"countryCode": "GLOBAL",
"userAgentMode": "random",
},
)
response.raise_for_status()
print(response.json()["data"])
{
"mode": "extract",
"page": {
"url": "https://example.com/dashboard",
"title": "Example Dashboard"
},
"data": {
"title": "Example Dashboard"
},
"missingFields": []
}
{
"mode": "inspect",
"page": {
"url": "https://example.com/dashboard",
"title": "Example Dashboard"
},
"candidates": {
"dom": { "metrics": [], "tables": [] },
"network": []
},
"suggestedPlan": {
"fields": {},
"schema": {
"type": "object",
"properties": {},
"additionalProperties": false
}
}
}
{
"mode": "extract",
"page": {
"url": "https://trends.google.com/trends/explore?q=adspower%2Cplaywright",
"title": "Google Trends"
},
"data": {
"averages": [
{ "query": "adspower", "extractedValue": 42 },
{ "query": "playwright", "extractedValue": 71 }
],
"interestOverTime": []
},
"missingFields": [],
"source": "sunbrowser",
"cached": false,
"stale": false,
"collectedAt": "2026-08-03T02:04:05.123456789Z",
"attempts": 2
}
{
"code": "INSUFFICIENT_CREDITS"
}
{
"code": "SPA_REQUIRED_FIELDS_MISSING"
}
提交页面 URL 和提取配置,从 JavaScript 渲染的 SPA 中检索结构化数据。你可以使用自定义字段定义、内置模板或
inspect 模式来发现页面上可用的数据。
string
目标 SPA 的完整 HTTP(S) URL,使用 80 或 443 端口。大多数模板都需要此字段。对于
google-trends-explore,建议改用 keyword 字段。SimilarWeb 需要完整的页面 URL,而不是裸域名。string
推荐用于
google-trends-explore。提供 1 到 5 个唯一的、非空的逗号分隔关键词,每个最多 100 个 Unicode 字符(例如 "adspower,playwright")。服务器会自动构建规范的 Trends URL。显式的 null、数字或空字符串值将返回 INVALID_KEYWORD,且不会回退到 url。string
提取模式。默认值为
"extract"。| Value | Behaviour |
|---|---|
"extract" | 运行字段或模板提取并返回结构化 data。 |
"inspect" | 返回页面上发现的 DOM 和网络候选方案,以及一个 suggestedPlan,你可以将其复制到未来的提取请求中。 |
string
内置站点模板 ID。提供后,
fields 由模板而非你的请求提供。可用模板 ID:"similarweb-overview"— 网站流量、参与度和排名指标。"google-trends-explore"— 最多 5 个关键词的随时间变化兴趣数据和相对平均值。"chrome-web-store-app-info"— Chrome Web Store 中的扩展元数据(附带 CRX manifest 回退)。
object
所选模板声明的模板特定参数。请参阅
GET /spa-extract/templates 中的模板 input 规范。string
提取前等待的导航事件。自定义提取默认使用
"domcontentloaded"。站点模板优先使用请求值,然后是模板默认值,最后是 "domcontentloaded"。| Value | Behaviour |
|---|---|
"domcontentloaded" | 等待 DOMContentLoaded — 推荐用于大多数 SPA。 |
"load" | 等待 window.load,包括图片和样式表。 |
"networkidle" | 等待至少 500 毫秒内没有任何网络连接。在长轮询站点上可能超时。 |
object
array
在提取前按数组顺序执行的可选页面交互。每个步骤的上限为 30 秒。
显示 action 类型
显示 action 类型
object
{ type: "wait", milliseconds: number } — 暂停 0 到 30,000 毫秒。object
{ type: "waitForSelector", selector: string, timeoutMs?: number } — 等待元素变为可见。object
{ type: "click", selector: string } — 点击第一个匹配的元素。object
{ type: "fill", selector: string, value: string } — 清除并填充第一个匹配的输入框。object
{ type: "press", selector: string, key: string } — 在第一个匹配的元素上按下按键。object
{ type: "scroll", selector?: string, x?: number, y?: number } — 滚动元素到视图中或滚动页面。未提供 selector 时 y 默认值为 800。object
extract 模式下命名字段定义的映射。每个键都会成为响应 data 对象中的一个属性。在不使用模板时使用此字段。显示 field definition
显示 field definition
string
必填
"dom" 从页面 HTML 中读取,或 "network" 从拦截的 JSON 响应中读取。string
DOM 字段的 CSS 选择器。
string
DOM 读取模式:
"text"(默认)、"html" 或 "attribute"。string
属性名称,在
value 为 "attribute" 时使用。string
用于匹配网络字段响应 URL 的子字符串。
string
匹配的网络响应内的 JSON 路径,例如
$.data.metrics[0].value。boolean
DOM 字段是否应返回所有匹配元素的数组。
string
将值强制转换为
"string"、"number"、"integer"、"boolean" 或 "json"。string
可选的正则表达式。当包含捕获组时,使用组 1。
boolean
缺少必填字段将返回
422 SPA_REQUIRED_FIELDS_MISSING。object
fields 的遗留别名,仅在 fields 不存在时使用。字符串值被视为 DOM CSS 选择器。不验证响应 JSON,也不返回 schemaValid 或 schemaErrors 字段。建议所有新集成都使用 fields。number
任务完成的最长等待时间,单位为毫秒。必须是正整数且不超过
3,600,000。string
浏览器语言环境,例如
"en-US"。省略时可能遵循受信任代理的元数据。string
IANA 时区标识符,例如
"Asia/Shanghai"。省略时可能遵循受信任代理的元数据。object
string
管理的住宅代理区域。不能与
proxy 同时使用。"GLOBAL"— 从 15 个热门区域中动态选择出口。- 两位字母国家代码 — 优先使用受信任代理并支持动态回退。
- 省略 — 自动随机选择受信任代理。
string
"random" 让服务器从其库中选择 User-Agent。未显式提供 userAgent 的请求默认使用 "random"。string
当
userAgentMode 为 "random" 时使用的操作系统。可选值:"windows"(默认)或 "macos"。string
显式指定 User-Agent 字符串。覆盖随机选择。
object
浏览器指纹设置。省略时,所有信号默认随机,同时保持操作系统、GPU、CPU、内存、字体和设备信号的一致性。
显示 fingerprint 字段
显示 fingerprint 字段
string
"forward"、"real" 或 "disabled"。string
"random" 或 "real"。控制 WebGL 厂商和渲染器元数据。string
"random"、"real" 或 "disabled"。string
"random" 或 "real"。控制 WebGL 图像噪声。string
"random" 或 "real"。控制 canvas 噪声。string
"random" 或 "real"。控制音频指纹噪声。string
"random" 或 "real"。控制布局测量噪声。string
"random" 或 "real"。返回与操作系统匹配的语音列表。string
"random" 或 "real"。返回与操作系统匹配的字体列表。string
"random" 或 "real"。生成一致的 CPU 线程数和内存配对。string
"random"、"enabled" 或 "disabled"。array
在导航前注入浏览器上下文的 Cookie 列表。
google-trends-explore 会忽略省略、null 或空数组值。在 Trends 请求中提供任何非空或格式错误的 Cookie 将返回 400 INVALID_TRENDS_COOKIES。响应
string
"extract" 或 "inspect",与请求模式一致。object
提取的结构化数据。键与你的
fields 定义或模板的输出字段匹配。array
未找到的字段键列表。
string
"sunbrowser" — 结果来自实时浏览器采集。"cache" — 结果来自服务器缓存。出现在 Trends 响应中。boolean
此响应是否来自缓存。
boolean
缓存结果是否已过期。
string
结果实际采集时的 RFC3339Nano 时间戳。
number
采集尝试次数。缓存命中响应使用
0;过期响应使用 Worker 报告时的尝试次数。object
在
inspect 模式下出现。页面上发现的 DOM 和网络提取候选方案。object
在
inspect 模式下出现。包含 fields 和 schema 的计划,你可以将其复制到未来的 extract 请求中。| Code | Meaning |
|---|---|
| 200 | 检查结果或提取的结构化数据。 |
| 400 | URL、关键词、代理、区域、User-Agent 或 Trends Cookie 无效。 |
| 401 | x-api-key 缺失或无效。 |
| 402 | 余额不足(INSUFFICIENT_CREDITS)。 |
| 404 | Chrome Web Store 列表不可用,且 CRX 回退无法恢复。 |
| 422 | 必填字段缺失、模板或字段配置无效,或 Worker 拒绝了任务。 |
| 429 | 任务请求频率受限,或 Google Trends 上游 429。 |
| 502 | 代理或目标失败、Trends 数据无效,或所有候选身份已耗尽。 |
| 503 | 队列、Worker 或管理资源不可用;或 Trends 身份计划已过期。 |
| 504 | 任务、导航或代理超时;或 Trends 响应未观察到。 |
| 500 | 未分类的任务执行失败。 |
一次请求验证后会消耗 1 积分,在任务入队之前扣除。请求体限制为 1 MiB。
更多示例
curl -sS -X POST "https://api.adscrawl.net/spa-extract" \
-H "content-type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"url": "https://example.com/dashboard",
"mode": "extract",
"waitUntil": "domcontentloaded",
"waitFor": { "selector": "h1", "timeoutMs": 15000 },
"fields": {
"title": { "source": "dom", "selector": "h1", "parse": "string" },
"subtitle": { "source": "dom", "selector": "p.subtitle", "parse": "string" },
"count": { "source": "dom", "selector": ".metric-value", "parse": "number" }
},
"countryCode": "GLOBAL",
"userAgentMode": "random"
}'
curl -sS -X POST "https://api.adscrawl.net/spa-extract" \
-H "content-type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"template": "similarweb-overview",
"url": "https://www.similarweb.com/website/dolphin-anty.com/#overview",
"countryCode": "GLOBAL"
}'
curl -sS -X POST "https://api.adscrawl.net/spa-extract" \
-H "content-type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"template": "google-trends-explore",
"keyword": "adspower,playwright"
}'
curl -sS -X POST "https://api.adscrawl.net/spa-extract" \
-H "content-type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"url": "https://example.com/dashboard",
"mode": "inspect"
}'
curl -sS -X POST "https://api.adscrawl.net/spa-extract" \
-H "content-type: application/json" \
-H "x-api-key: YOUR_API_KEY" \
-d '{
"url": "https://example.com/dashboard",
"mode": "extract",
"waitUntil": "domcontentloaded",
"waitFor": { "selector": "h1", "timeoutMs": 15000 },
"fields": {
"title": { "source": "dom", "selector": "h1", "parse": "string" },
"subtitle": { "source": "dom", "selector": "p.subtitle", "parse": "string" },
"count": { "source": "dom", "selector": ".metric-value", "parse": "number" }
},
"countryCode": "GLOBAL",
"userAgentMode": "random"
}'
const response = await fetch("https://api.adscrawl.net/spa-extract", {
method: "POST",
headers: {
"content-type": "application/json",
"x-api-key": "YOUR_API_KEY",
},
body: JSON.stringify({
url: "https://example.com/dashboard",
mode: "extract",
waitUntil: "domcontentloaded",
waitFor: { selector: "h1", timeoutMs: 15000 },
fields: {
title: { source: "dom", selector: "h1", parse: "string" },
subtitle: { source: "dom", selector: "p.subtitle", parse: "string" },
count: { source: "dom", selector: ".metric-value", parse: "number" },
},
countryCode: "GLOBAL",
userAgentMode: "random",
}),
});
const result = await response.json();
console.log(result.data);
import httpx
response = httpx.post(
"https://api.adscrawl.net/spa-extract",
headers={
"content-type": "application/json",
"x-api-key": "YOUR_API_KEY",
},
json={
"url": "https://example.com/dashboard",
"mode": "extract",
"waitUntil": "domcontentloaded",
"waitFor": {"selector": "h1", "timeoutMs": 15000},
"fields": {
"title": {"source": "dom", "selector": "h1", "parse": "string"},
"subtitle": {"source": "dom", "selector": "p.subtitle", "parse": "string"},
"count": {"source": "dom", "selector": ".metric-value", "parse": "number"},
},
"countryCode": "GLOBAL",
"userAgentMode": "random",
},
)
response.raise_for_status()
print(response.json()["data"])
{
"mode": "extract",
"page": {
"url": "https://example.com/dashboard",
"title": "Example Dashboard"
},
"data": {
"title": "Example Dashboard"
},
"missingFields": []
}
{
"mode": "inspect",
"page": {
"url": "https://example.com/dashboard",
"title": "Example Dashboard"
},
"candidates": {
"dom": { "metrics": [], "tables": [] },
"network": []
},
"suggestedPlan": {
"fields": {},
"schema": {
"type": "object",
"properties": {},
"additionalProperties": false
}
}
}
{
"mode": "extract",
"page": {
"url": "https://trends.google.com/trends/explore?q=adspower%2Cplaywright",
"title": "Google Trends"
},
"data": {
"averages": [
{ "query": "adspower", "extractedValue": 42 },
{ "query": "playwright", "extractedValue": 71 }
],
"interestOverTime": []
},
"missingFields": [],
"source": "sunbrowser",
"cached": false,
"stale": false,
"collectedAt": "2026-08-03T02:04:05.123456789Z",
"attempts": 2
}
{
"code": "INSUFFICIENT_CREDITS"
}
{
"code": "SPA_REQUIRED_FIELDS_MISSING"
}