diff --git a/cli/src/commands.rs b/cli/src/commands.rs
index cd4992e2a6..021aaa11d5 100644
--- a/cli/src/commands.rs
+++ b/cli/src/commands.rs
@@ -1919,6 +1919,27 @@ fn parse_command_inner(args: &[String], flags: &Flags) -> Result/ [args] [--write-fixture]`: a normal run,
+ // then main.rs compares the result's shape with the stored fixture.
+ let (verify, rest): (Option, Vec<&str>) = if rest.first() == Some(&"verify") {
+ let write = rest.contains(&"--write-fixture");
+ let kept = rest[1..]
+ .iter()
+ .filter(|a| **a != "--write-fixture")
+ .copied()
+ .collect();
+ (Some(write), kept)
+ } else {
+ (None, rest.to_vec())
+ };
let spec = rest.first().ok_or(ParseError::InvalidValue {
message: "site requires / (run `chrome-use site list`)".to_string(),
usage: "site / [args]",
@@ -2000,6 +2021,7 @@ fn parse_command_inner(args: &[String], flags: &Flags) -> Result = clean[at + 1..]
+ .iter()
+ .filter(|a| !(verify && a.as_str() == "--write-fixture"))
+ .cloned()
+ .collect();
+ let mut env = opencli::run(spec, &entry, &rest, &flags.session);
+ let mut ok = env.get("success").and_then(|v| v.as_bool()) == Some(true);
+ let result = env.get("data").cloned().unwrap_or(Value::Null);
+ if ok && verify {
+ let (vok, report) = site::verify_result(spec, &result, write_fixture);
+ if !vok {
+ ok = false;
+ let issues: Vec = report
+ .get("issues")
+ .and_then(|x| x.as_array())
+ .map(|a| {
+ a.iter()
+ .filter_map(|i| i.as_str().map(String::from))
+ .collect()
+ })
+ .unwrap_or_default();
+ env["error"] = json!(format!("site verify {spec}: {}", issues.join("; ")));
+ }
+ env["verify"] = report;
+ }
+ if flags.json {
+ let mut data = json!({ "result": result, "source": opencli::SOURCE_LABEL });
+ if let Some(v) = env.get("verify") {
+ data["verify"] = v.clone();
+ }
+ println!(
+ "{}",
+ json!({ "success": ok, "data": data, "error": if ok { Value::Null } else { env.get("error").cloned().unwrap_or(Value::Null) } })
+ );
+ } else if ok {
+ eprintln!("{}", color::dim(&format!("site {spec} (via OpenCLI)")));
+ println!(
+ "{}",
+ serde_json::to_string_pretty(&result).unwrap_or_default()
+ );
+ if let Some(v) = env.get("verify") {
+ if v.get("recorded").and_then(|x| x.as_bool()) == Some(true) {
+ eprintln!(
+ "{} site verify: fixture recorded",
+ color::success_indicator()
+ );
+ } else {
+ eprintln!("{} site verify: ok", color::success_indicator());
+ }
+ }
+ } else {
+ let err = env
+ .get("error")
+ .and_then(|v| v.as_str())
+ .unwrap_or("failed");
+ match env
+ .get("hint")
+ .and_then(|v| v.as_str())
+ .filter(|h| !h.is_empty())
+ {
+ Some(h) => eprintln!("{} {err} — {h}", color::error_indicator()),
+ None => eprintln!("{} {err}", color::error_indicator()),
+ }
+ }
+ exit(if ok { 0 } else { 1 });
+ }
+ }
match clean.get(1).map(|s| s.as_str()) {
Some("update") => {
let rt = tokio::runtime::Runtime::new().expect("Failed to create tokio runtime");
@@ -1631,9 +1707,23 @@ fn main() {
return;
}
Some("list") => {
+ let theirs: Vec = {
+ let ours = site::list_adapters().unwrap_or_default();
+ let mut v: Vec = opencli::manifest()
+ .iter()
+ .filter_map(opencli::spec_of)
+ .filter(|s| !ours.contains(s))
+ .collect();
+ v.sort();
+ v.dedup();
+ v
+ };
match site::list_adapters() {
Ok(list) if flags.json => {
- println!("{}", json!({ "success": true, "adapters": list }))
+ println!(
+ "{}",
+ json!({ "success": true, "adapters": list, "opencli": theirs })
+ )
}
Ok(list) if list.is_empty() => {
println!("no site adapters installed — run `chrome-use site update`")
@@ -1642,6 +1732,9 @@ fn main() {
for a in &list {
println!("{a}");
}
+ for a in &theirs {
+ println!("{a} {}", color::dim("(opencli)"));
+ }
eprintln!(
"{}",
color::dim(&format!(
@@ -1659,6 +1752,16 @@ fn main() {
}
Some("info") => {
let spec = clean.get(2).cloned().unwrap_or_default();
+ if opencli::handles(&spec) {
+ if let Some(entry) = opencli::lookup(&spec) {
+ println!(
+ "{}",
+ serde_json::to_string_pretty(&opencli::info(&entry))
+ .unwrap_or_default()
+ );
+ return;
+ }
+ }
match site::load_adapter(&spec) {
Ok(a) => println!(
"{}",
@@ -1781,6 +1884,9 @@ fn main() {
}
return;
}
+ // `site analyze [url]` / `site verify / …` → daemon dispatch
+ // (commands.rs builds them).
+ Some("analyze") | Some("verify") => {}
// `site / [args]` → fall through to the daemon dispatch.
Some(spec) if spec.contains('/') => {
// #125: an adapter arg whose name collides with a reserved global
@@ -1815,7 +1921,7 @@ fn main() {
}
_ => {
eprintln!(
- "{} usage: chrome-use site / [args] | site update | site list | \
+ "{} usage: chrome-use site / [args] | site analyze [url] | site verify / [args] [--write-fixture] | site update | site list | \
site info / | site sources | site add|remove ",
color::error_indicator()
);
@@ -3106,6 +3212,44 @@ fn main() {
}
}
}
+ // `site verify`: compare the result's shape with the stored fixture
+ // (or record it). A mismatch fails the command like an adapter error.
+ if let Some(v) = cmd.get("verify").filter(|v| !v.is_null()) {
+ if resp.success {
+ let spec = v.get("spec").and_then(|x| x.as_str()).unwrap_or("");
+ let write = v.get("writeFixture").and_then(|x| x.as_bool()) == Some(true);
+ let result = resp
+ .data
+ .as_ref()
+ .and_then(|d| d.get("result"))
+ .cloned()
+ .unwrap_or(Value::Null);
+ let (ok, report) = site::verify_result(spec, &result, write);
+ if !ok {
+ resp.success = false;
+ let issues: Vec = report
+ .get("issues")
+ .and_then(|x| x.as_array())
+ .map(|a| {
+ a.iter()
+ .filter_map(|i| i.as_str().map(String::from))
+ .collect()
+ })
+ .unwrap_or_default();
+ resp.error = Some(format!(
+ "site verify {spec}: {}",
+ if issues.is_empty() {
+ "could not record the fixture".to_string()
+ } else {
+ issues.join("; ")
+ }
+ ));
+ }
+ if let Some(d) = resp.data.as_mut().and_then(|d| d.as_object_mut()) {
+ d.insert("verify".into(), report);
+ }
+ }
+ }
if let Some(err) = resp.error.as_mut() {
if err.contains("has NO snapshot refs") {
// Keep what to do last: agents read errors through `tail -1`.
diff --git a/cli/src/native/actions.rs b/cli/src/native/actions.rs
index 4945f15703..99286116bf 100644
--- a/cli/src/native/actions.rs
+++ b/cli/src/native/actions.rs
@@ -2012,6 +2012,7 @@ pub async fn execute_command(cmd: &Value, state: &mut DaemonState) -> Value {
"content" => handle_content(state).await,
"evaluate" => handle_evaluate(cmd, state).await,
"site" => handle_site(cmd, state).await,
+ "site_analyze" => handle_site_analyze(cmd, state).await,
"script" => super::script::handle_script(cmd, state).await,
"close" => handle_close(state).await,
"keep" => handle_keep(cmd, state).await,
@@ -4794,6 +4795,37 @@ async fn resolve_iframe_selector(
/// already loaded the adapter and built the `script`; here we just place the page
/// and evaluate. Never disrupts the user's foreground tab — navigation happens on
/// the daemon's own tab (same as every other command on the relay).
+/// `site analyze [url]`: scan the current page (after opening `url`, if given)
+/// for what an adapter should read — API calls the page made, state it embeds,
+/// anti-bot vendors — and recommend a data source.
+async fn handle_site_analyze(cmd: &Value, state: &mut DaemonState) -> Result {
+ if cmd.get("url").and_then(|v| v.as_str()).is_some() {
+ handle_navigate(cmd, state).await?;
+ }
+ let mgr = state.browser.as_ref().ok_or("site analyze: no browser")?;
+ let raw = mgr.evaluate(crate::site::ANALYZE_JS, None).await?;
+ let strings = |key: &str| -> Vec {
+ raw.get("signals")
+ .and_then(|s| s.get(key))
+ .and_then(|v| v.as_array())
+ .map(|a| {
+ a.iter()
+ .filter_map(|x| x.as_str().map(String::from))
+ .collect()
+ })
+ .unwrap_or_default()
+ };
+ let signals = humanize::DetectSignals {
+ cookie_names: strings("cookies"),
+ script_urls: strings("scripts"),
+ window_globals: strings("globals"),
+ };
+ let vendors = humanize::detected_vendors(&signals);
+ let host = raw.get("host").and_then(|v| v.as_str()).unwrap_or("");
+ let adapters = crate::site::adapters_for_domain(host);
+ Ok(crate::site::analyze_report(&raw, &vendors, &adapters))
+}
+
async fn handle_site(cmd: &Value, state: &mut DaemonState) -> Result {
let domain = cmd
.get("domain")
diff --git a/cli/src/native/humanize.rs b/cli/src/native/humanize.rs
index 7977253df1..ef89d2b0a2 100644
--- a/cli/src/native/humanize.rs
+++ b/cli/src/native/humanize.rs
@@ -375,6 +375,25 @@ const VENDOR_MARKERS: &[(&str, &str)] = &[
("__cf_bm", "cloudflare-bot-mgmt"),
];
+/// The anti-bot vendors whose markers appear in `signals`, deduped, in marker
+/// order (for `site analyze`).
+pub fn detected_vendors(signals: &DetectSignals) -> Vec<&'static str> {
+ let hay: Vec = signals
+ .cookie_names
+ .iter()
+ .chain(signals.script_urls.iter())
+ .chain(signals.window_globals.iter())
+ .map(|s| s.to_ascii_lowercase())
+ .collect();
+ let mut out: Vec<&'static str> = Vec::new();
+ for (marker, vendor) in VENDOR_MARKERS {
+ if hay.iter().any(|h| h.contains(marker)) && !out.contains(vendor) {
+ out.push(vendor);
+ }
+ }
+ out
+}
+
/// Decide the level for a page. Returns `Human` if any known anti-bot vendor is
/// present, otherwise `baseline`. Misses just stay at baseline and false hits
/// only cost a little latency, so matching is deliberately liberal.
diff --git a/cli/src/opencli.rs b/cli/src/opencli.rs
new file mode 100644
index 0000000000..dbbdba4a31
--- /dev/null
+++ b/cli/src/opencli.rs
@@ -0,0 +1,323 @@
+//! OpenCLI compatibility: run jackwener/OpenCLI adapters as `site` commands.
+//!
+//! OpenCLI's adapters run in Node and call a `page` object, so they cannot be
+//! evaluated in a tab like our adapters. Instead `site update` installs a pinned
+//! `@jackwener/opencli` package (public npm tarball, `--ignore-scripts`) into
+//! `~/.chrome-use/opencli`, and `opencli_runner.mjs` runs an adapter with
+//! OpenCLI's own registry, argument coercion, pipeline executor and `BasePage`
+//! helpers, over a page whose transport is chrome-use. Nothing is converted or
+//! copied into our packs.
+//!
+//! Precedence: a chrome-use adapter (official, configured, then community) with
+//! the same `name/cmd` always wins; OpenCLI only answers names we don't have,
+//! and it is listed last in the domain hint.
+
+use std::path::{Path, PathBuf};
+use std::process::{Command, Stdio};
+
+use serde_json::{json, Value};
+
+/// The OpenCLI release we install. Pinned rather than `latest` because its
+/// adapters run as local Node code with the user's privileges; a version bump
+/// is a deliberate change here. `AGENT_BROWSER_OPENCLI_VERSION` overrides.
+pub const PINNED_VERSION: &str = "1.8.8";
+pub const PACKAGE: &str = "@jackwener/opencli";
+pub const SOURCE_LABEL: &str = "opencli";
+
+const RUNNER_JS: &str = include_str!("opencli_runner.mjs");
+
+pub fn disabled() -> bool {
+ std::env::var_os("AGENT_BROWSER_SITES_NO_OPENCLI").is_some()
+}
+
+fn version() -> String {
+ std::env::var("AGENT_BROWSER_OPENCLI_VERSION")
+ .ok()
+ .filter(|v| !v.trim().is_empty())
+ .unwrap_or_else(|| PINNED_VERSION.to_string())
+}
+
+/// `~/.chrome-use/opencli` — npm prefix holding the package and the runner.
+pub fn root() -> Option {
+ std::env::var_os("HOME").map(|h| PathBuf::from(h).join(".chrome-use").join("opencli"))
+}
+
+fn pkg_dir(root: &Path) -> PathBuf {
+ root.join("node_modules").join("@jackwener").join("opencli")
+}
+
+fn installed_version(root: &Path) -> Option {
+ let text = std::fs::read_to_string(pkg_dir(root).join("package.json")).ok()?;
+ let v: Value = serde_json::from_str(&text).ok()?;
+ v.get("version").and_then(|x| x.as_str()).map(String::from)
+}
+
+fn on_path(bin: &str) -> bool {
+ Command::new(bin)
+ .arg("--version")
+ .stdout(Stdio::null())
+ .stderr(Stdio::null())
+ .status()
+ .is_ok_and(|s| s.success())
+}
+
+/// Install or update the pinned package. Best-effort: Ok(None) when skipped
+/// (disabled, or no node/npm on PATH), Ok(Some(n)) with the command count.
+pub fn sync() -> Result
+
OpenCLI commands work too
+
+ With Node.js 20+ on PATH, site update also installs OpenCLI (about 180 sites and 1,300 commands; a pinned version, installed with --ignore-scripts).
+ A name/cmd that neither adapter pack has runs through OpenCLI's own runtime, driving your current chrome-use session and its logins, e.g.
+ chrome-use site hackernews/best --limit 5 --json. They show as (opencli) in site list, site info shows their args, and they come last in the site hint.
+ On a shared name ours win. AGENT_BROWSER_SITES_NO_OPENCLI=1 turns them off.
+
+
+
Writing your own: analyze and verify
+
+ Do the action that loads the data on the page first (search, scroll, open the list), then run chrome-use site analyze. It lists the API calls the page made, the state it embeds (__NEXT_DATA__, __INITIAL_STATE__, …) and any anti-bot vendor, and recommends a data source.
+ In order of preference: a public API, the site's own JSON API called from the page with its cookies, embedded page state, then the DOM. Each step down breaks more often.
+ Once it works, chrome-use site verify <name>/<cmd> [args] --write-fixture records the shape of a good result; later, site verify fails (exit 1) when a field disappears, changes type, or a list comes back empty. The fixture keeps the shape only, never values.
+
+
ℹ️ Exception on a fresh environment
Only on a brand-new environment where the adapter sources haven't been pulled yet might naming a
diff --git a/docs/site-adapters.html b/docs/site-adapters.html
index 09eab5d8ec..e7fd1cc7c3 100644
--- a/docs/site-adapters.html
+++ b/docs/site-adapters.html
@@ -258,6 +258,21 @@
+ 先在页面上把加载数据的操作做一遍(搜索、滚动、打开列表),再跑 chrome-use site analyze:它列出页面发过的接口请求、页面里嵌着的初始数据(__NEXT_DATA__、__INITIAL_STATE__ 等)和检测到的反爬厂商,并推荐取数方式。
+ 优先顺序:公开 API,网站自己的 JSON 接口(在页面里带 cookie 调用),页面嵌入的数据,最后才是 DOM;越往后越容易坏。
+ 写好后用 chrome-use site verify <name>/<cmd> [参数] --write-fixture 记下一次正确结果的结构;之后 site verify 发现字段消失、类型变了或列表变空就失败(退出码 1)。fixture 只存结构,不存数据。
+
+
ℹ️ 全新环境的例外
只有在刚装好、适配器源还没拉取的全新环境里,点名一个 site <name>/<cmd>
diff --git a/skill-data/core/references/site-adapters.md b/skill-data/core/references/site-adapters.md
index c3f7e0ebc7..a8d7cbd37b 100644
--- a/skill-data/core/references/site-adapters.md
+++ b/skill-data/core/references/site-adapters.md
@@ -20,8 +20,15 @@ chrome-use site github/issues owner/repo --json # run it → JSON (navigates t
- It navigates to the adapter's domain (reusing the current tab if you're already on it), so
login-gated feeds (`bilibili/feed`, `twitter/...`) work because they run as *you*.
- If no adapter fits, fall back to the normal `snapshot`/`eval` loop. chrome-use fetches and
- runs two default sources: the [bb-sites](https://github.com/epiral/bb-sites) community pack
- and the official [chrome-use-sites](https://github.com/leeguooooo/chrome-use-sites) pack.
+ runs two default sources: the official [chrome-use-sites](https://github.com/leeguooooo/chrome-use-sites)
+ pack and the [bb-sites](https://github.com/epiral/bb-sites) community pack. On a shared
+ `name/cmd` the official one wins.
+- **OpenCLI commands work too.** When Node.js 20+ is on PATH, `site update` also installs
+ [OpenCLI](https://github.com/jackwener/OpenCLI) (~1,300 commands over ~180 sites). A
+ `name/cmd` that neither pack has runs through OpenCLI's own runtime, driving this same
+ session. They show as `(opencli)` in `site list`, `site info` shows their args, and they come
+ last in the `siteAdapters` hint. Same command, same JSON: `chrome-use site hackernews/best
+ --limit 5 --json`. `AGENT_BROWSER_SITES_NO_OPENCLI=1` turns them off.
> **Auto-trigger — act on it.** chrome-use keeps both packs synced automatically (first use +
> weekly), and whenever you reach a page whose domain has adapters it tells you: on every
@@ -43,13 +50,21 @@ If you work on the same site a lot and no adapter covers it, chrome-use adds
**Ask the user** whether to turn the steps you keep repeating there into an adapter. Don't write
one without a yes. If they agree:
-1. Find the data source with `network requests` / `eval` (the site's own JSON API beats DOM
- scraping), then write `~/.chrome-use/my-sites//.js` in the format above
- (`@meta` with `name`, `description`, `domain`, `args`, `readOnly`; then the `async function`).
+1. First check `chrome-use site list | grep `: OpenCLI may already cover it. Otherwise
+ do the action that loads the data (search, scroll, open the list), then run
+ `chrome-use site analyze`. It lists the API calls the page made, the state it embeds
+ (`__NEXT_DATA__`, `__INITIAL_STATE__`, …) and any anti-bot vendor, and picks a strategy.
+ Prefer, in this order: a public API; the site's own JSON API called from the page
+ (`fetch(url, {credentials: 'include'})`); embedded page state; the DOM. Each step down breaks
+ more often. Write `~/.chrome-use/my-sites//.js` in the format above (`@meta` with
+ `name`, `description`, `domain`, `args`, `readOnly`; then the `async function`).
2. Register the folder once and sync: `chrome-use site add ~/.chrome-use/my-sites`, then
`chrome-use site update`. Keep your own adapters in that folder, not in `~/.chrome-use/sites`,
because a sync rewrites `~/.chrome-use/sites`.
-3. Run it: `chrome-use site / --json`. If it's generally useful, offer to send it to
+3. Run it: `chrome-use site / --json`. When the result looks right, record it:
+ `chrome-use site verify / [args] --write-fixture`. Later, `site verify /`
+ fails (exit 1) when a field disappears, changes type, or a list comes back empty, so a broken
+ adapter shows up before you trust its output. The fixture keeps the shape only, never values. If it's generally useful, offer to send it to
[chrome-use-sites](https://github.com/leeguooooo/chrome-use-sites). That opens a PR from the
user's account, so ask before you do it.