From d0ca2adaccac3e55ef32c0e1862441506ffc2cc5 Mon Sep 17 00:00:00 2001 From: leeguooooo Date: Tue, 6 Oct 2026 13:56:36 +0900 Subject: [PATCH 1/5] feat(site): site analyze and site verify --- cli/src/commands.rs | 22 +++ cli/src/main.rs | 40 +++- cli/src/native/actions.rs | 32 +++ cli/src/native/humanize.rs | 19 ++ cli/src/output.rs | 84 ++++++++ cli/src/site.rs | 394 +++++++++++++++++++++++++++++++++++++ 6 files changed, 590 insertions(+), 1 deletion(-) diff --git a/cli/src/commands.rs b/cli/src/commands.rs index cd4992e2a6..021aaa11d5 100644 --- a/cli/src/commands.rs +++ b/cli/src/commands.rs @@ -1919,6 +1919,27 @@ fn parse_command_inner(args: &[String], flags: &Flags) -> Result/ [args] [--write-fixture]`: a normal run, + // then main.rs compares the result's shape with the stored fixture. + let (verify, rest): (Option, Vec<&str>) = if rest.first() == Some(&"verify") { + let write = rest.contains(&"--write-fixture"); + let kept = rest[1..] + .iter() + .filter(|a| **a != "--write-fixture") + .copied() + .collect(); + (Some(write), kept) + } else { + (None, rest.to_vec()) + }; let spec = rest.first().ok_or(ParseError::InvalidValue { message: "site requires / (run `chrome-use site list`)".to_string(), usage: "site / [args]", @@ -2000,6 +2021,7 @@ fn parse_command_inner(args: &[String], flags: &Flags) -> Result = report + .get("issues") + .and_then(|x| x.as_array()) + .map(|a| { + a.iter() + .filter_map(|i| i.as_str().map(String::from)) + .collect() + }) + .unwrap_or_default(); + resp.error = Some(format!( + "site verify {spec}: {}", + if issues.is_empty() { + "could not record the fixture".to_string() + } else { + issues.join("; ") + } + )); + } + if let Some(d) = resp.data.as_mut().and_then(|d| d.as_object_mut()) { + d.insert("verify".into(), report); + } + } + } if let Some(err) = resp.error.as_mut() { if err.contains("has NO snapshot refs") { // Keep what to do last: agents read errors through `tail -1`. diff --git a/cli/src/native/actions.rs b/cli/src/native/actions.rs index 4945f15703..99286116bf 100644 --- a/cli/src/native/actions.rs +++ b/cli/src/native/actions.rs @@ -2012,6 +2012,7 @@ pub async fn execute_command(cmd: &Value, state: &mut DaemonState) -> Value { "content" => handle_content(state).await, "evaluate" => handle_evaluate(cmd, state).await, "site" => handle_site(cmd, state).await, + "site_analyze" => handle_site_analyze(cmd, state).await, "script" => super::script::handle_script(cmd, state).await, "close" => handle_close(state).await, "keep" => handle_keep(cmd, state).await, @@ -4794,6 +4795,37 @@ async fn resolve_iframe_selector( /// already loaded the adapter and built the `script`; here we just place the page /// and evaluate. Never disrupts the user's foreground tab — navigation happens on /// the daemon's own tab (same as every other command on the relay). +/// `site analyze [url]`: scan the current page (after opening `url`, if given) +/// for what an adapter should read — API calls the page made, state it embeds, +/// anti-bot vendors — and recommend a data source. +async fn handle_site_analyze(cmd: &Value, state: &mut DaemonState) -> Result { + if cmd.get("url").and_then(|v| v.as_str()).is_some() { + handle_navigate(cmd, state).await?; + } + let mgr = state.browser.as_ref().ok_or("site analyze: no browser")?; + let raw = mgr.evaluate(crate::site::ANALYZE_JS, None).await?; + let strings = |key: &str| -> Vec { + raw.get("signals") + .and_then(|s| s.get(key)) + .and_then(|v| v.as_array()) + .map(|a| { + a.iter() + .filter_map(|x| x.as_str().map(String::from)) + .collect() + }) + .unwrap_or_default() + }; + let signals = humanize::DetectSignals { + cookie_names: strings("cookies"), + script_urls: strings("scripts"), + window_globals: strings("globals"), + }; + let vendors = humanize::detected_vendors(&signals); + let host = raw.get("host").and_then(|v| v.as_str()).unwrap_or(""); + let adapters = crate::site::adapters_for_domain(host); + Ok(crate::site::analyze_report(&raw, &vendors, &adapters)) +} + async fn handle_site(cmd: &Value, state: &mut DaemonState) -> Result { let domain = cmd .get("domain") diff --git a/cli/src/native/humanize.rs b/cli/src/native/humanize.rs index 7977253df1..ef89d2b0a2 100644 --- a/cli/src/native/humanize.rs +++ b/cli/src/native/humanize.rs @@ -375,6 +375,25 @@ const VENDOR_MARKERS: &[(&str, &str)] = &[ ("__cf_bm", "cloudflare-bot-mgmt"), ]; +/// The anti-bot vendors whose markers appear in `signals`, deduped, in marker +/// order (for `site analyze`). +pub fn detected_vendors(signals: &DetectSignals) -> Vec<&'static str> { + let hay: Vec = signals + .cookie_names + .iter() + .chain(signals.script_urls.iter()) + .chain(signals.window_globals.iter()) + .map(|s| s.to_ascii_lowercase()) + .collect(); + let mut out: Vec<&'static str> = Vec::new(); + for (marker, vendor) in VENDOR_MARKERS { + if hay.iter().any(|h| h.contains(marker)) && !out.contains(vendor) { + out.push(vendor); + } + } + out +} + /// Decide the level for a page. Returns `Human` if any known anti-bot vendor is /// present, otherwise `baseline`. Misses just stay at baseline and false hits /// only cost a little latency, so matching is deliberately liberal. diff --git a/cli/src/output.rs b/cli/src/output.rs index 37d628e24a..7febe4dbc4 100644 --- a/cli/src/output.rs +++ b/cli/src/output.rs @@ -300,6 +300,61 @@ fn stderr_is_discarded() -> bool { } } +fn print_site_analyze(data: &serde_json::Value, strategy: &str, next: &[serde_json::Value]) { + let host = data.get("host").and_then(|v| v.as_str()).unwrap_or(""); + println!("site analyze: {host}"); + let api = data + .get("api") + .and_then(|v| v.as_array()) + .cloned() + .unwrap_or_default(); + let seen = data + .get("requestsSeen") + .and_then(|v| v.as_u64()) + .unwrap_or(0); + println!(" API candidates ({} of {seen} requests):", api.len()); + for a in api.iter().take(8) { + let url = a.get("url").and_then(|v| v.as_str()).unwrap_or(""); + let why: Vec<&str> = a + .get("reasons") + .and_then(|v| v.as_array()) + .map(|r| r.iter().filter_map(|x| x.as_str()).collect()) + .unwrap_or_default(); + println!(" {url}"); + println!(" {}", color::dim(&why.join(", "))); + } + let state = data + .get("state") + .and_then(|v| v.as_array()) + .cloned() + .unwrap_or_default(); + if !state.is_empty() { + println!(" Embedded state:"); + for st in state.iter().take(8) { + let name = st.get("name").and_then(|v| v.as_str()).unwrap_or(""); + let size = st.get("size").and_then(|v| v.as_i64()).unwrap_or(0); + let keys: Vec<&str> = st + .get("keys") + .and_then(|v| v.as_array()) + .map(|r| r.iter().filter_map(|x| x.as_str()).collect()) + .unwrap_or_default(); + println!(" {name} ({size} chars) {}", color::dim(&keys.join(", "))); + } + } + if let Some(v) = data + .get("antiBot") + .and_then(|v| v.as_array()) + .filter(|a| !a.is_empty()) + { + let names: Vec<&str> = v.iter().filter_map(|x| x.as_str()).collect(); + println!(" Anti-bot: {}", names.join(", ")); + } + println!(" Strategy: {strategy}"); + for (i, step) in next.iter().filter_map(|s| s.as_str()).enumerate() { + println!(" {}. {step}", i + 1); + } +} + pub fn print_response_with_opts(resp: &Response, action: Option<&str>, opts: &OutputOptions) { print_response_body(resp, action, opts); // Every successful text response gets its observation, including branches @@ -389,6 +444,35 @@ fn print_response_body(resp: &Response, action: Option<&str>, opts: &OutputOptio eprintln!("site adapter suggestion: {msg}"); } } + // `site verify`: the verdict, to stderr so the result stays parseable. + if let Some(v) = data.get("verify") { + if v.get("recorded").and_then(|x| x.as_bool()) == Some(true) { + let at = v.get("fixture").and_then(|x| x.as_str()).unwrap_or(""); + eprintln!( + "{} site verify: fixture recorded → {at}", + color::success_indicator() + ); + } else if v.get("ok").and_then(|x| x.as_bool()) == Some(true) { + match v.get("next").and_then(|x| x.as_str()) { + Some(next) => eprintln!( + "{} site verify: result non-empty; {next}", + color::success_indicator() + ), + None => eprintln!( + "{} site verify: matches the fixture", + color::success_indicator() + ), + } + } + } + // `site analyze`: a report, not a page result. + if let (Some(strategy), Some(next)) = ( + data.get("strategy").and_then(|v| v.as_str()), + data.get("next").and_then(|v| v.as_array()), + ) { + print_site_analyze(data, strategy, next); + return; + } // `open` that landed on a page refusing this browser's sign-in (#387). if let Some(h) = data.get("humanCheck") { print_human_check(h); diff --git a/cli/src/site.rs b/cli/src/site.rs index 3c33f9d18b..a30135bda1 100644 --- a/cli/src/site.rs +++ b/cli/src/site.rs @@ -1344,6 +1344,355 @@ pub fn unix_now() -> u64 { now_secs() } +/// Page-side half of `site analyze`: what an adapter author needs to pick a +/// data source. Requests come from the Resource Timing buffer, so they cover +/// what the page has loaded so far — interact first (search, scroll, open a +/// list), then analyze, to catch the request that action made. +pub const ANALYZE_JS: &str = r#"(() => { + const out = { url: location.href, host: location.hostname, title: document.title }; + const host = location.hostname.replace(/^www\./, ''); + const base = host.split('.').slice(-2).join('.'); + const noise = /google-analytics|googletagmanager|doubleclick|facebook\.net|hotjar|sentry|segment\.(io|com)|mixpanel|clarity\.ms|bat\.bing|newrelic|datadoghq|amplitude|\/collect\b|\/log(ging)?\b|\/track(ing)?\b|\/beacon\b|\/metrics?\b|\/report\b|\/telemetry\b|\/pixel\b/i; + const seen = new Set(); + const api = []; + for (const e of performance.getEntriesByType('resource')) { + if (!['fetch', 'xmlhttprequest'].includes(e.initiatorType)) continue; + let u; try { u = new URL(e.name); } catch (_) { continue; } + const key = u.origin + u.pathname; + if (seen.has(key)) continue; + seen.add(key); + const sameSite = u.hostname === location.hostname || u.hostname.endsWith('.' + base) || u.hostname === base; + const reasons = []; + let score = 0; + if (noise.test(e.name)) { score -= 5; reasons.push('analytics/telemetry'); } + if (sameSite) { score += 2; reasons.push('same site'); } + if (/\/(api|ajax|graphql|gql|rest|v\d+|x\/|web-interface|rpc|data)\b/i.test(u.pathname)) { score += 3; reasons.push('api-like path'); } + if (/\.json\b/i.test(u.pathname)) { score += 2; reasons.push('json'); } + if (/graphql|gql/i.test(u.pathname)) reasons.push('graphql'); + if (/[?&](page|cursor|offset|limit|size|count|keyword|q|query|id|uid)=/i.test(u.search)) { score += 1; reasons.push('paging/query params'); } + if (e.transferSize > 2000) { score += 1; reasons.push('sizeable body'); } + api.push({ url: e.name.length > 300 ? e.name.slice(0, 300) + '…' : e.name, type: e.initiatorType, sameSite, bytes: e.transferSize || 0, score, reasons }); + } + api.sort((a, b) => b.score - a.score); + out.api = api.filter(a => a.score > 0).slice(0, 12); + out.requestsSeen = api.length; + + const known = ['__NEXT_DATA__', '__NUXT__', '__NUXT_DATA__', '__INITIAL_STATE__', '__INITIAL_DATA__', '__INITIAL_PROPS__', '__PRELOADED_STATE__', '__APOLLO_STATE__', '__REDUX_STATE__', '__SSR_DATA__', '__remixContext', '__UNIVERSAL_DATA_FOR_REHYDRATION__', 'ytInitialData', 'ytInitialPlayerResponse', '__pinia', '__INITIAL_SSR_STATE__', 'g_initialProps', '__STATE__']; + const names = new Set(known.filter(k => { try { return window[k] != null; } catch (_) { return false; } })); + for (const k of Object.getOwnPropertyNames(window)) { + if (names.size > 20) break; + if (/^__.*(STATE|DATA|PROPS|CONTEXT|STORE)__?$/i.test(k) || /^(initial|preloaded|ssr)(State|Data|Props)$/i.test(k)) { + try { if (window[k] && typeof window[k] === 'object') names.add(k); } catch (_) {} + } + } + const describe = v => { + let size = 0; try { size = JSON.stringify(v).length; } catch (_) { size = -1; } + const keys = v && typeof v === 'object' ? Object.keys(v).slice(0, 10) : []; + return { size, keys }; + }; + out.state = []; + for (const k of names) { try { out.state.push({ name: 'window.' + k, ...describe(window[k]) }); } catch (_) {} } + for (const el of document.querySelectorAll('script[type="application/json"], script[type="application/ld+json"]')) { + if (out.state.length > 25) break; + let v; try { v = JSON.parse(el.textContent); } catch (_) { continue; } + const sel = el.id ? 'script#' + el.id : 'script[type="' + el.type + '"]'; + out.state.push({ name: sel, ...describe(v) }); + } + out.state = out.state.filter(s => s.size === -1 || s.size > 200); + + out.webpack = Object.getOwnPropertyNames(window).filter(k => /^webpackChunk|^webpackJsonp/.test(k)).slice(0, 3); + out.signals = { + cookies: document.cookie.split(';').map(c => c.trim().split('=')[0]).filter(Boolean), + scripts: Array.from(document.scripts, s => s.src || '').filter(Boolean), + globals: Object.getOwnPropertyNames(window).filter(k => /_px|bmak|_abck|datadome|reese84|kpsdk|incap_ses|visid_incap|akam/i.test(k)), + }; + out.loggedInHint = document.cookie.length > 0; + return out; +})()"#; + +/// Turn the page scan into the `site analyze` report: drop the raw signals, +/// attach anti-bot vendors and installed adapters, and recommend a data source +/// in the order that breaks least often — a site's JSON API called from the +/// page, then state the page already embeds, then the DOM. +pub fn analyze_report(raw: &Value, vendors: &[&str], adapters: &[String]) -> Value { + let mut report = raw.clone(); + if let Some(o) = report.as_object_mut() { + o.remove("signals"); + o.remove("loggedInHint"); + } + let api = raw + .get("api") + .and_then(|v| v.as_array()) + .cloned() + .unwrap_or_default(); + let state = raw + .get("state") + .and_then(|v| v.as_array()) + .cloned() + .unwrap_or_default(); + let webpack = raw + .get("webpack") + .and_then(|v| v.as_array()) + .is_some_and(|a| !a.is_empty()); + let host = raw.get("host").and_then(|v| v.as_str()).unwrap_or(""); + + let mut steps: Vec = Vec::new(); + let strategy; + if !adapters.is_empty() { + steps.push(format!( + "Adapters already exist for {host}: {}. Run `chrome-use site info /` before writing a new one.", + adapters.join(", ") + )); + } + if let Some(top) = api + .iter() + .find(|a| a.get("sameSite").and_then(|v| v.as_bool()) == Some(true)) + { + strategy = "page-fetch"; + let url = top.get("url").and_then(|v| v.as_str()).unwrap_or(""); + steps.push(format!( + "Call the site's own API from the page: `fetch(url, {{credentials: 'include'}})`, starting from {url}. Check its JSON with `chrome-use eval` first, then map it to the fields the user needs." + )); + } else if let Some(s) = state.first() { + strategy = "page-state"; + let name = s.get("name").and_then(|v| v.as_str()).unwrap_or(""); + steps.push(format!( + "No API call seen yet, but the page embeds its data in {name}. Read it in the adapter (no extra request). If the data you need comes from a later action, do that action and analyze again." + )); + } else { + strategy = "dom"; + steps.push( + "No API call or embedded state found. Do the action that loads the data (search, scroll, open a list) and run `site analyze` again; fall back to reading the DOM only if nothing shows up — it breaks the most often.".to_string(), + ); + } + if webpack { + steps.push("The page is a webpack bundle: its own modules (e.g. signed request helpers) can be reached via the webpackChunk global if the API needs a signature.".to_string()); + } + if !vendors.is_empty() { + steps.push(format!( + "Anti-bot protection detected ({}). Keep requests inside the page (same-origin fetch with the page's cookies), keep the call rate low, and never replay them from outside the browser.", + vendors.join(", ") + )); + } + steps.push("Write the adapter to ~/.chrome-use/my-sites//.js, register it once with `chrome-use site add ~/.chrome-use/my-sites`, run `site update`, then `chrome-use site verify / --write-fixture` to record what a good result looks like. Guide: `chrome-use skills get core/site-adapters`.".to_string()); + + if let Some(o) = report.as_object_mut() { + o.insert("antiBot".into(), json!(vendors)); + o.insert("adapters".into(), json!(adapters)); + o.insert("strategy".into(), json!(strategy)); + o.insert("next".into(), json!(steps)); + } + report +} + +/// `~/.chrome-use/site-fixtures//.json` — the recorded shape of a +/// good result, for `site verify`. +pub fn fixture_path(spec: &str) -> Option { + let (name, cmd) = spec.split_once('/')?; + dirs_home().map(|h| { + h.join(".chrome-use") + .join("site-fixtures") + .join(name) + .join(format!("{cmd}.json")) + }) +} + +/// A structural summary of an adapter result: types per field, depth-limited, +/// with array items merged over the first few elements. Values are dropped, so +/// a fixture holds no user data. +pub fn result_shape(v: &Value) -> Value { + shape_at(v, 0) +} + +fn shape_at(v: &Value, depth: usize) -> Value { + match v { + Value::Null => json!("null"), + Value::Bool(_) => json!("boolean"), + Value::Number(_) => json!("number"), + Value::String(_) => json!("string"), + Value::Array(a) => { + let mut items = Value::Null; + if depth < 4 { + for el in a.iter().take(5) { + items = merge_shape(items, shape_at(el, depth + 1)); + } + } + json!({ "array": items, "nonEmpty": !a.is_empty() }) + } + Value::Object(o) => { + if depth >= 4 { + return json!("object"); + } + let fields: serde_json::Map = o + .iter() + .map(|(k, v)| (k.clone(), shape_at(v, depth + 1))) + .collect(); + json!({ "object": fields }) + } + } +} + +/// Merge two shapes of sibling array items: keep fields present in any item, +/// and let a concrete type win over "null". +fn merge_shape(a: Value, b: Value) -> Value { + match (a, b) { + (Value::Null, b) => b, + (a, Value::String(t)) if t == "null" => a, + (Value::String(t), b) if t == "null" => b, + (Value::Object(mut x), Value::Object(y)) => { + if let (Some(Value::Object(fx)), Some(Value::Object(fy))) = + (x.get("object").cloned(), y.get("object")) + { + let mut merged = fx; + for (k, v) in fy { + let cur = merged.remove(k).unwrap_or(Value::Null); + merged.insert(k.clone(), merge_shape(cur, v.clone())); + } + x.insert("object".into(), Value::Object(merged)); + } + Value::Object(x) + } + (a, _) => a, + } +} + +/// Differences that mean the adapter broke: a field the fixture had is gone, +/// a field changed type, or a list that had rows came back empty. New fields +/// and null values are fine. +pub fn shape_diff(expected: &Value, actual: &Value) -> Vec { + let mut out = Vec::new(); + diff_at(expected, actual, "result", &mut out); + out +} + +fn kind(v: &Value) -> &str { + match v { + Value::String(t) => t.as_str(), + Value::Object(o) if o.contains_key("array") => "array", + Value::Object(o) if o.contains_key("object") => "object", + _ => "unknown", + } +} + +fn diff_at(exp: &Value, act: &Value, path: &str, out: &mut Vec) { + let (ke, ka) = (kind(exp), kind(act)); + if ke == "null" || ka == "null" || ke == "unknown" { + return; + } + if ke != ka { + out.push(format!("{path}: was {ke}, now {ka}")); + return; + } + match ke { + "array" => { + let had = exp + .get("nonEmpty") + .and_then(|v| v.as_bool()) + .unwrap_or(false); + let has = act + .get("nonEmpty") + .and_then(|v| v.as_bool()) + .unwrap_or(false); + if had && !has { + out.push(format!("{path}: was a non-empty list, now empty")); + return; + } + if let (Some(e), Some(a)) = (exp.get("array"), act.get("array")) { + if !a.is_null() { + diff_at(e, a, &format!("{path}[]"), out); + } + } + } + "object" => { + let (Some(fe), Some(fa)) = ( + exp.get("object").and_then(|v| v.as_object()), + act.get("object").and_then(|v| v.as_object()), + ) else { + return; + }; + for (k, ve) in fe { + match fa.get(k) { + Some(va) => diff_at(ve, va, &format!("{path}.{k}"), out), + None => out.push(format!("{path}.{k}: missing")), + } + } + } + _ => {} + } +} + +/// `site verify`: check a run's result against the stored fixture, or record +/// one. Returns (ok, report). +pub fn verify_result(spec: &str, result: &Value, write_fixture: bool) -> (bool, Value) { + let shape = result_shape(result); + let path = fixture_path(spec); + let empty = match result { + Value::Null => true, + Value::Array(a) => a.is_empty(), + Value::Object(o) => o.is_empty(), + Value::String(s) => s.is_empty(), + _ => false, + }; + if write_fixture { + if empty { + return ( + false, + json!({ "spec": spec, "ok": false, "issues": ["result is empty; not recording it as a fixture"] }), + ); + } + let fixture = json!({ "spec": spec, "recordedAt": now_secs(), "shape": shape }); + let written = path.as_ref().is_some_and(|p| { + p.parent() + .is_some_and(|d| std::fs::create_dir_all(d).is_ok()) + && std::fs::write( + p, + serde_json::to_string_pretty(&fixture).unwrap_or_default(), + ) + .is_ok() + }); + return ( + written, + json!({ + "spec": spec, + "ok": written, + "fixture": path.map(|p| p.display().to_string()), + "recorded": written, + }), + ); + } + let stored = path + .as_ref() + .and_then(|p| std::fs::read_to_string(p).ok()) + .and_then(|t| serde_json::from_str::(&t).ok()); + let Some(stored) = stored else { + let mut issues = Vec::new(); + if empty { + issues.push("result is empty".to_string()); + } + return ( + !empty, + json!({ + "spec": spec, + "ok": !empty, + "fixture": null, + "issues": issues, + "next": format!("no fixture yet: run `chrome-use site verify {spec} … --write-fixture` once the result looks right"), + }), + ); + }; + let issues = shape_diff(stored.get("shape").unwrap_or(&Value::Null), &shape); + let ok = issues.is_empty(); + ( + ok, + json!({ + "spec": spec, + "ok": ok, + "fixture": path.map(|p| p.display().to_string()), + "issues": issues, + }), + ) +} + /// Map CLI args to the adapter's `args` object. Positional args fill the adapter's /// declared `args` keys in order; `--key value` overrides by name. The adapter /// validates required args itself. @@ -1394,6 +1743,51 @@ mod tests { assert_eq!(source_rank(None), source_rank(Some(&community))); } + #[test] + fn analyze_prefers_same_site_api_then_state_then_dom() { + let api = json!({"host":"x.com","api":[{"url":"https://x.com/i/api/graphql/Q","sameSite":true}],"state":[],"webpack":[],"signals":{}}); + let r = analyze_report(&api, &[], &[]); + assert_eq!(r["strategy"], "page-fetch"); + assert!(r.get("signals").is_none()); + let state = + json!({"host":"a.com","api":[],"state":[{"name":"window.__NEXT_DATA__","size":900}]}); + assert_eq!(analyze_report(&state, &[], &[])["strategy"], "page-state"); + let dom = json!({"host":"a.com","api":[{"url":"https://cdn.other.net/x","sameSite":false}],"state":[]}); + let r = analyze_report(&dom, &["akamai"], &["a/b".to_string()]); + assert_eq!(r["strategy"], "dom"); + assert_eq!(r["antiBot"][0], "akamai"); + let next = r["next"].to_string(); + assert!(next.contains("a/b") && next.contains("Anti-bot")); + } + + #[test] + fn verify_flags_missing_fields_type_changes_and_emptied_lists() { + let good = json!({"items":[{"id":1,"title":"a","tag":null},{"id":2,"title":"b","tag":"x"}],"total":2}); + let exp = result_shape(&good); + assert!(shape_diff(&exp, &result_shape(&good)).is_empty()); + // extra field and null value are fine + let extra = json!({"items":[{"id":3,"title":null,"tag":"y","new":true}],"total":1}); + assert!(shape_diff(&exp, &result_shape(&extra)).is_empty()); + let broken = json!({"items":[{"id":"3"}],"total":1}); + let d = shape_diff(&exp, &result_shape(&broken)); + assert!( + d.iter() + .any(|x| x.contains("result.items[].id: was number, now string")), + "{d:?}" + ); + assert!( + d.iter() + .any(|x| x.contains("result.items[].title: missing")), + "{d:?}" + ); + let emptied = json!({"items":[],"total":0}); + let d = shape_diff(&exp, &result_shape(&emptied)); + assert!( + d.iter().any(|x| x.contains("non-empty list, now empty")), + "{d:?}" + ); + } + #[test] fn adapter_suggestion_thresholds() { let now = 10_000_000; From 8c1984252d69cf7f57a39bcf9422ed3065403130 Mon Sep 17 00:00:00 2001 From: leeguooooo Date: Tue, 6 Oct 2026 14:02:40 +0900 Subject: [PATCH 2/5] feat(site): run OpenCLI adapters through its own runtime over chrome-use --- cli/src/main.rs | 105 +++++++++++- cli/src/opencli.rs | 323 +++++++++++++++++++++++++++++++++++++ cli/src/opencli_runner.mjs | 136 ++++++++++++++++ cli/src/output.rs | 12 +- cli/src/site.rs | 39 ++++- 5 files changed, 612 insertions(+), 3 deletions(-) create mode 100644 cli/src/opencli.rs create mode 100644 cli/src/opencli_runner.mjs diff --git a/cli/src/main.rs b/cli/src/main.rs index 8ed9bcb5c5..b185206cc3 100644 --- a/cli/src/main.rs +++ b/cli/src/main.rs @@ -16,6 +16,7 @@ mod install; mod jev; mod mcp; mod native; +mod opencli; mod output; mod ownership; mod read; @@ -1611,6 +1612,81 @@ fn main() { ), } } + // A `name/cmd` we have no adapter for but OpenCLI does: run it with + // OpenCLI's runtime over this session (opencli.rs). `site verify` too. + { + let verify = clean.get(1).map(|s| s.as_str()) == Some("verify"); + let at = if verify { 2 } else { 1 }; + if let Some(spec) = clean.get(at).filter(|s| opencli::handles(s)) { + let entry = opencli::lookup(spec).unwrap_or_default(); + let write_fixture = verify && clean.iter().any(|a| a == "--write-fixture"); + let rest: Vec = clean[at + 1..] + .iter() + .filter(|a| !(verify && a.as_str() == "--write-fixture")) + .cloned() + .collect(); + let mut env = opencli::run(spec, &entry, &rest, &flags.session); + let mut ok = env.get("success").and_then(|v| v.as_bool()) == Some(true); + let result = env.get("data").cloned().unwrap_or(Value::Null); + if ok && verify { + let (vok, report) = site::verify_result(spec, &result, write_fixture); + if !vok { + ok = false; + let issues: Vec = report + .get("issues") + .and_then(|x| x.as_array()) + .map(|a| { + a.iter() + .filter_map(|i| i.as_str().map(String::from)) + .collect() + }) + .unwrap_or_default(); + env["error"] = json!(format!("site verify {spec}: {}", issues.join("; "))); + } + env["verify"] = report; + } + if flags.json { + let mut data = json!({ "result": result, "source": opencli::SOURCE_LABEL }); + if let Some(v) = env.get("verify") { + data["verify"] = v.clone(); + } + println!( + "{}", + json!({ "success": ok, "data": data, "error": if ok { Value::Null } else { env.get("error").cloned().unwrap_or(Value::Null) } }) + ); + } else if ok { + eprintln!("{}", color::dim(&format!("site {spec} (via OpenCLI)"))); + println!( + "{}", + serde_json::to_string_pretty(&result).unwrap_or_default() + ); + if let Some(v) = env.get("verify") { + if v.get("recorded").and_then(|x| x.as_bool()) == Some(true) { + eprintln!( + "{} site verify: fixture recorded", + color::success_indicator() + ); + } else { + eprintln!("{} site verify: ok", color::success_indicator()); + } + } + } else { + let err = env + .get("error") + .and_then(|v| v.as_str()) + .unwrap_or("failed"); + match env + .get("hint") + .and_then(|v| v.as_str()) + .filter(|h| !h.is_empty()) + { + Some(h) => eprintln!("{} {err} — {h}", color::error_indicator()), + None => eprintln!("{} {err}", color::error_indicator()), + } + } + exit(if ok { 0 } else { 1 }); + } + } match clean.get(1).map(|s| s.as_str()) { Some("update") => { let rt = tokio::runtime::Runtime::new().expect("Failed to create tokio runtime"); @@ -1631,9 +1707,23 @@ fn main() { return; } Some("list") => { + let theirs: Vec = { + let ours = site::list_adapters().unwrap_or_default(); + let mut v: Vec = opencli::manifest() + .iter() + .filter_map(opencli::spec_of) + .filter(|s| !ours.contains(s)) + .collect(); + v.sort(); + v.dedup(); + v + }; match site::list_adapters() { Ok(list) if flags.json => { - println!("{}", json!({ "success": true, "adapters": list })) + println!( + "{}", + json!({ "success": true, "adapters": list, "opencli": theirs }) + ) } Ok(list) if list.is_empty() => { println!("no site adapters installed — run `chrome-use site update`") @@ -1642,6 +1732,9 @@ fn main() { for a in &list { println!("{a}"); } + for a in &theirs { + println!("{a} {}", color::dim("(opencli)")); + } eprintln!( "{}", color::dim(&format!( @@ -1659,6 +1752,16 @@ fn main() { } Some("info") => { let spec = clean.get(2).cloned().unwrap_or_default(); + if opencli::handles(&spec) { + if let Some(entry) = opencli::lookup(&spec) { + println!( + "{}", + serde_json::to_string_pretty(&opencli::info(&entry)) + .unwrap_or_default() + ); + return; + } + } match site::load_adapter(&spec) { Ok(a) => println!( "{}", diff --git a/cli/src/opencli.rs b/cli/src/opencli.rs new file mode 100644 index 0000000000..dbbdba4a31 --- /dev/null +++ b/cli/src/opencli.rs @@ -0,0 +1,323 @@ +//! OpenCLI compatibility: run jackwener/OpenCLI adapters as `site` commands. +//! +//! OpenCLI's adapters run in Node and call a `page` object, so they cannot be +//! evaluated in a tab like our adapters. Instead `site update` installs a pinned +//! `@jackwener/opencli` package (public npm tarball, `--ignore-scripts`) into +//! `~/.chrome-use/opencli`, and `opencli_runner.mjs` runs an adapter with +//! OpenCLI's own registry, argument coercion, pipeline executor and `BasePage` +//! helpers, over a page whose transport is chrome-use. Nothing is converted or +//! copied into our packs. +//! +//! Precedence: a chrome-use adapter (official, configured, then community) with +//! the same `name/cmd` always wins; OpenCLI only answers names we don't have, +//! and it is listed last in the domain hint. + +use std::path::{Path, PathBuf}; +use std::process::{Command, Stdio}; + +use serde_json::{json, Value}; + +/// The OpenCLI release we install. Pinned rather than `latest` because its +/// adapters run as local Node code with the user's privileges; a version bump +/// is a deliberate change here. `AGENT_BROWSER_OPENCLI_VERSION` overrides. +pub const PINNED_VERSION: &str = "1.8.8"; +pub const PACKAGE: &str = "@jackwener/opencli"; +pub const SOURCE_LABEL: &str = "opencli"; + +const RUNNER_JS: &str = include_str!("opencli_runner.mjs"); + +pub fn disabled() -> bool { + std::env::var_os("AGENT_BROWSER_SITES_NO_OPENCLI").is_some() +} + +fn version() -> String { + std::env::var("AGENT_BROWSER_OPENCLI_VERSION") + .ok() + .filter(|v| !v.trim().is_empty()) + .unwrap_or_else(|| PINNED_VERSION.to_string()) +} + +/// `~/.chrome-use/opencli` — npm prefix holding the package and the runner. +pub fn root() -> Option { + std::env::var_os("HOME").map(|h| PathBuf::from(h).join(".chrome-use").join("opencli")) +} + +fn pkg_dir(root: &Path) -> PathBuf { + root.join("node_modules").join("@jackwener").join("opencli") +} + +fn installed_version(root: &Path) -> Option { + let text = std::fs::read_to_string(pkg_dir(root).join("package.json")).ok()?; + let v: Value = serde_json::from_str(&text).ok()?; + v.get("version").and_then(|x| x.as_str()).map(String::from) +} + +fn on_path(bin: &str) -> bool { + Command::new(bin) + .arg("--version") + .stdout(Stdio::null()) + .stderr(Stdio::null()) + .status() + .is_ok_and(|s| s.success()) +} + +/// Install or update the pinned package. Best-effort: Ok(None) when skipped +/// (disabled, or no node/npm on PATH), Ok(Some(n)) with the command count. +pub fn sync() -> Result, String> { + if disabled() { + return Ok(None); + } + if !on_path("node") || !on_path("npm") { + return Ok(None); + } + let root = root().ok_or("opencli: cannot resolve home dir")?; + std::fs::create_dir_all(&root).map_err(|e| e.to_string())?; + let want = version(); + if installed_version(&root).as_deref() != Some(want.as_str()) { + let out = Command::new("npm") + .arg("install") + .arg("--prefix") + .arg(&root) + .arg(format!("{PACKAGE}@{want}")) + .args([ + "--omit=dev", + "--ignore-scripts", + "--no-audit", + "--no-fund", + "--no-save", + "--loglevel=error", + ]) + .stdin(Stdio::null()) + .output() + .map_err(|e| format!("opencli: npm: {e}"))?; + if !out.status.success() { + return Err(format!( + "opencli: npm install {PACKAGE}@{want} failed: {}", + String::from_utf8_lossy(&out.stderr).trim() + )); + } + } + write_runner(&root)?; + Ok(Some(manifest().len())) +} + +fn write_runner(root: &Path) -> Result { + let path = root.join("runner.mjs"); + if std::fs::read_to_string(&path).ok().as_deref() != Some(RUNNER_JS) { + std::fs::write(&path, RUNNER_JS).map_err(|e| format!("opencli: write runner: {e}"))?; + } + Ok(path) +} + +/// The package's `cli-manifest.json` entries (one per command), or empty. +pub fn manifest() -> Vec { + if disabled() { + return Vec::new(); + } + let Some(root) = root() else { + return Vec::new(); + }; + std::fs::read_to_string(pkg_dir(&root).join("cli-manifest.json")) + .ok() + .and_then(|t| serde_json::from_str::>(&t).ok()) + .unwrap_or_default() +} + +pub fn spec_of(entry: &Value) -> Option { + let site = entry.get("site")?.as_str()?; + let name = entry.get("name")?.as_str()?; + Some(format!("{site}/{name}")) +} + +/// The manifest entry for `site/name`, if OpenCLI has it. +pub fn lookup(spec: &str) -> Option { + manifest() + .into_iter() + .find(|e| spec_of(e).as_deref() == Some(spec)) +} + +/// Whether `spec` should run through OpenCLI: we have no adapter by that name +/// and OpenCLI does. +pub fn handles(spec: &str) -> bool { + spec.contains('/') && crate::site::load_adapter(spec).is_err() && lookup(spec).is_some() +} + +/// `site info` for an OpenCLI command, shaped like an adapter's @meta. +pub fn info(entry: &Value) -> Value { + let mut args = serde_json::Map::new(); + for a in entry + .get("args") + .and_then(|v| v.as_array()) + .into_iter() + .flatten() + { + if let Some(name) = a.get("name").and_then(|v| v.as_str()) { + args.insert(name.to_string(), a.clone()); + } + } + json!({ + "name": spec_of(entry), + "description": entry.get("description"), + "domain": entry.get("domain"), + "readOnly": entry.get("access").and_then(|v| v.as_str()) == Some("read"), + "strategy": entry.get("strategy"), + "args": args, + "columns": entry.get("columns"), + "source": format!("{SOURCE_LABEL} ({PACKAGE}@{})", version()), + }) +} + +/// Map CLI args onto the command's declared args: positional args fill the +/// ones marked `positional` in order, `--key value` sets by name, and a bare +/// `--flag` sets a boolean arg. Values stay strings; OpenCLI coerces them. +pub fn map_args(entry: &Value, rest: &[String]) -> Result, String> { + let declared: Vec<&Value> = entry + .get("args") + .and_then(|v| v.as_array()) + .map(|a| a.iter().collect()) + .unwrap_or_default(); + let positional: Vec<&str> = declared + .iter() + .filter(|a| a.get("positional").and_then(|v| v.as_bool()) == Some(true)) + .filter_map(|a| a.get("name").and_then(|v| v.as_str())) + .collect(); + let is_bool = |name: &str| { + declared.iter().any(|a| { + a.get("name").and_then(|v| v.as_str()) == Some(name) + && matches!( + a.get("type").and_then(|v| v.as_str()), + Some("bool") | Some("boolean") + ) + }) + }; + let mut out = serde_json::Map::new(); + let mut pos = positional.iter(); + let mut it = rest.iter().peekable(); + while let Some(a) = it.next() { + if let Some(key) = a.strip_prefix("--") { + let (key, inline) = match key.split_once('=') { + Some((k, v)) => (k, Some(v.to_string())), + None => (key, None), + }; + let value = match inline { + Some(v) => v, + None if is_bool(key) && it.peek().is_none_or(|n| n.starts_with("--")) => { + "true".to_string() + } + None => it + .next() + .cloned() + .ok_or_else(|| format!("site: --{key} needs a value"))?, + }; + out.insert(key.to_string(), Value::String(value)); + } else { + let name = pos.next().ok_or_else(|| { + format!( + "site: unexpected argument `{a}` (positional args: {})", + if positional.is_empty() { + "none".to_string() + } else { + positional.join(", ") + } + ) + })?; + out.insert(name.to_string(), Value::String(a.clone())); + } + } + Ok(out) +} + +/// Run `site/name` through OpenCLI's runtime. Returns the runner's envelope: +/// `{success, data}` or `{success: false, error, hint?}`. +pub fn run(spec: &str, entry: &Value, rest: &[String], session: &str) -> Value { + let fail = |e: String| json!({ "success": false, "error": e }); + if !on_path("node") { + return fail(format!( + "site {spec} comes from OpenCLI, which needs Node.js 20+ on PATH (https://nodejs.org)" + )); + } + let Some(root) = root() else { + return fail("opencli: cannot resolve home dir".into()); + }; + let runner = match write_runner(&root) { + Ok(p) => p, + Err(e) => return fail(e), + }; + let kwargs = match map_args(entry, rest) { + Ok(k) => k, + Err(e) => return fail(e), + }; + let (site, name) = spec.split_once('/').unwrap_or((spec, "")); + let chrome_use = std::env::current_exe() + .map(|p| p.display().to_string()) + .unwrap_or_else(|_| "chrome-use".to_string()); + let req = json!({ + "pkgDir": pkg_dir(&root), + "chromeUse": chrome_use, + "session": session, + "site": site, + "name": name, + "modulePath": entry.get("modulePath"), + "kwargs": kwargs, + }); + let req_path = + std::env::temp_dir().join(format!("cu-opencli-{}.json", uuid::Uuid::new_v4().simple())); + if let Err(e) = std::fs::write(&req_path, req.to_string()) { + return fail(format!("opencli: write request: {e}")); + } + let out = Command::new("node") + .arg(&runner) + .arg(&req_path) + .stdin(Stdio::null()) + .stderr(Stdio::inherit()) + .output(); + let _ = std::fs::remove_file(&req_path); + let out = match out { + Ok(o) => o, + Err(e) => return fail(format!("opencli: node: {e}")), + }; + let stdout = String::from_utf8_lossy(&out.stdout); + stdout + .lines() + .rev() + .find_map(|l| serde_json::from_str::(l).ok()) + .unwrap_or_else(|| fail(format!("opencli: {spec} produced no result"))) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn entry() -> Value { + json!({ + "site": "bilibili", "name": "feed", "domain": "www.bilibili.com", "access": "read", + "args": [ + {"name": "uid", "positional": true}, + {"name": "limit", "type": "int"}, + {"name": "verbose", "type": "bool"} + ] + }) + } + + #[test] + fn maps_positional_named_and_bare_bool_args() { + let s = |v: &[&str]| v.iter().map(|x| x.to_string()).collect::>(); + let m = map_args(&entry(), &s(&["123", "--limit", "5", "--verbose"])).unwrap(); + assert_eq!(m["uid"], "123"); + assert_eq!(m["limit"], "5"); + assert_eq!(m["verbose"], "true"); + let m = map_args(&entry(), &s(&["--limit=7"])).unwrap(); + assert_eq!(m["limit"], "7"); + assert!(map_args(&entry(), &s(&["a", "b"])).is_err()); + assert!(map_args(&entry(), &s(&["--limit"])).is_err()); + } + + #[test] + fn info_reads_like_adapter_meta() { + let i = info(&entry()); + assert_eq!(i["name"], "bilibili/feed"); + assert_eq!(i["readOnly"], true); + assert!(i["args"].get("limit").is_some()); + assert!(i["source"].as_str().unwrap().starts_with("opencli")); + } +} diff --git a/cli/src/opencli_runner.mjs b/cli/src/opencli_runner.mjs new file mode 100644 index 0000000000..55af2a4e5f --- /dev/null +++ b/cli/src/opencli_runner.mjs @@ -0,0 +1,136 @@ +// chrome-use ⇄ OpenCLI bridge. Runs one OpenCLI adapter with OpenCLI's own +// runtime (registry, argument coercion, pipeline executor, BasePage helpers) +// and a page whose transport is chrome-use, so the adapter drives the user's +// real, logged-in Chrome through the chrome-use daemon. +// +// Invoked by `chrome-use site /` when no chrome-use adapter has +// that name: node opencli_runner.mjs +// request: { pkgDir, chromeUse, session, site, name, modulePath, kwargs } +// stdout: one JSON line { success, data | error, hint? } +import { spawn } from 'node:child_process'; +import { readFileSync, mkdtempSync, rmSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import { join } from 'node:path'; +import { pathToFileURL } from 'node:url'; + +const req = JSON.parse(readFileSync(process.argv[2], 'utf8')); +const dist = join(req.pkgDir, 'dist', 'src'); +const mod = (p) => import(pathToFileURL(join(dist, p)).href); + +const { BasePage } = await mod('browser/base-page.js'); +const { buildEvaluateExpression } = await mod('browser/utils.js'); +const { getRegistry } = await mod('registry-api.js'); +const { executePipeline } = await mod('pipeline/index.js'); +const { coerceAndValidateArgs } = await mod('execution.js'); + +function cu(args, stdin) { + return new Promise((resolve, reject) => { + const env = { ...process.env }; + if (req.session) env.AGENT_BROWSER_SESSION = req.session; + const child = spawn(req.chromeUse, ['--json', ...args], { env, stdio: ['pipe', 'pipe', 'pipe'] }); + let out = ''; + let err = ''; + child.stdout.on('data', (d) => { out += d; }); + child.stderr.on('data', (d) => { err += d; }); + child.on('error', reject); + child.on('close', () => { + let resp; + try { resp = JSON.parse(out.trim().split('\n').pop()); } catch (_) { + return reject(new Error(`chrome-use ${args[0]}: ${(err || out).trim().slice(0, 500)}`)); + } + if (!resp.success) return reject(new Error(`chrome-use ${args[0]}: ${resp.error}`)); + resolve(resp.data); + }); + child.stdin.end(stdin ?? ''); + }); +} + +class ChromeUsePage extends BasePage { + async goto(url, options) { + await cu(['open', url]); + this._lastUrl = url; + if (options?.waitUntil !== 'none') { + const { waitForDomStableJs } = await mod('browser/dom-helpers.js'); + const maxMs = options?.settleMs ?? 1000; + await this.evaluate(waitForDomStableJs(maxMs, Math.min(500, maxMs))).catch(() => {}); + } + } + async evaluate(input, ...args) { + const data = await cu(['eval', '--stdin'], buildEvaluateExpression(input, args)); + return data?.result; + } + async getCookies(opts = {}) { + const a = ['cookies', 'get']; + if (opts.url) a.push('--url', opts.url); + const data = await cu(a); + const cookies = Array.isArray(data?.cookies) ? data.cookies : []; + const d = opts.domain?.replace(/^\./, ''); + return d ? cookies.filter((c) => { const cd = String(c.domain || '').replace(/^\./, ''); return cd === d || cd.endsWith('.' + d) || d.endsWith('.' + cd); }) : cookies; + } + async screenshot(options = {}) { + const dir = mkdtempSync(join(tmpdir(), 'cu-oc-')); + const file = options.path || join(dir, 'shot.png'); + const a = ['screenshot', file]; + if (options.fullPage) a.push('--full'); + await cu(a); + const b64 = readFileSync(file).toString('base64'); + rmSync(dir, { recursive: true, force: true }); + return b64; + } + async tabs() { + const data = await cu(['tab', 'list']); + return data?.tabs ?? []; + } + async selectTab(target) { + await cu(['tab', String(target)]); + } + async newTab(url) { + const data = await cu(url ? ['tab', 'new', url] : ['tab', 'new']); + return data?.tabId; + } + async closeTab(target) { + await cu(target == null ? ['tab', 'close'] : ['tab', 'close', String(target)]); + } + async getCurrentUrl() { + const data = await cu(['get', 'url']); + return data?.url ?? null; + } + async setFileInput(files, selector) { + await cu(['upload', selector || 'input[type=file]', ...files]); + } + async insertText(text) { + await this.evaluate((t) => document.execCommand('insertText', false, t), text); + } +} + +function hostMatches(url, domain) { + try { const h = new URL(url).hostname; return h === domain || h.endsWith('.' + domain); } catch (_) { return false; } +} + +function emit(obj) { + process.stdout.write(JSON.stringify(obj) + '\n'); +} + +try { + await import(pathToFileURL(join(req.pkgDir, 'clis', req.modulePath)).href); + const cmd = getRegistry().get(`${req.site}/${req.name}`); + if (!cmd) throw new Error(`opencli: ${req.site}/${req.name} is not defined in ${req.modulePath}`); + const kwargs = coerceAndValidateArgs(cmd.args ?? [], req.kwargs ?? {}); + cmd.validateArgs?.(kwargs); + const page = cmd.browser === false ? null : new ChromeUsePage(); + if (page && cmd.navigateBefore !== false) { + const target = typeof cmd.navigateBefore === 'string' + ? cmd.navigateBefore + : cmd.domain ? `https://${cmd.domain}` : null; + const current = target ? await page.getCurrentUrl().catch(() => null) : null; + if (target && !(cmd.domain && hostMatches(current, cmd.domain))) await page.goto(target); + } + let result; + if (typeof cmd.func === 'function') result = await cmd.func(page, kwargs); + else if (Array.isArray(cmd.pipeline)) result = await executePipeline(page, cmd.pipeline, { args: kwargs }); + else throw new Error(`opencli: ${req.site}/${req.name} has neither func nor pipeline`); + emit({ success: true, data: result ?? null }); +} catch (e) { + emit({ success: false, error: e?.message || String(e), hint: e?.hint, code: e?.code }); + process.exitCode = 1; +} diff --git a/cli/src/output.rs b/cli/src/output.rs index 7febe4dbc4..5121a8e7d5 100644 --- a/cli/src/output.rs +++ b/cli/src/output.rs @@ -424,7 +424,17 @@ fn print_response_body(resp: &Response, action: Option<&str>, opts: &OutputOptio .unwrap_or_default(); if !cmds.is_empty() { eprintln!("site adapters for {domain} — prefer these for structured data:"); - eprintln!(" {}", color::dim(&cmds.join(", "))); + // OpenCLI can add dozens per site; the full list is in --json. + const SHOWN: usize = 12; + let mut line = cmds[..cmds.len().min(SHOWN)].join(", "); + if cmds.len() > SHOWN { + line.push_str(&format!( + " … +{} more (`chrome-use site list | grep {}`)", + cmds.len() - SHOWN, + cmds[0].split('/').next().unwrap_or("") + )); + } + eprintln!(" {}", color::dim(&line)); eprintln!( " {}", color::dim(&format!("e.g. chrome-use site {} --json", cmds[0])) diff --git a/cli/src/site.rs b/cli/src/site.rs index a30135bda1..a970f5f1ae 100644 --- a/cli/src/site.rs +++ b/cli/src/site.rs @@ -899,6 +899,13 @@ pub async fn update() -> Result { if let Ok(json) = serde_json::to_string(&provenance) { let _ = std::fs::write(dir.join(".provenance.json"), json); } + // OpenCLI's adapters run from its own package (see opencli.rs); a failure + // here never fails the update. + match tokio::task::spawn_blocking(crate::opencli::sync).await { + Ok(Err(e)) => eprintln!("site update: {e}"), + Err(e) => eprintln!("site update: opencli: {e}"), + _ => {} + } write_domain_index(&dir); if let Some(p) = last_update_path() { let _ = std::fs::write(p, now_secs().to_string()); @@ -1124,7 +1131,30 @@ fn write_domain_index(dir: &std::path::Path) { } } } - let ordered: std::collections::BTreeMap> = by_domain + // OpenCLI commands ride along after ours, for names we don't already have. + let ours: std::collections::HashSet = by_domain + .values() + .flat_map(|v| v.iter().map(|(_, s)| s.clone())) + .collect(); + let mut opencli_by_domain: std::collections::BTreeMap> = + Default::default(); + for entry in crate::opencli::manifest() { + let (Some(spec), Some(domain)) = ( + crate::opencli::spec_of(&entry), + entry.get("domain").and_then(|v| v.as_str()), + ) else { + continue; + }; + if domain.is_empty() || ours.contains(&spec) { + continue; + } + let read_only = entry.get("access").and_then(|v| v.as_str()) == Some("read"); + opencli_by_domain + .entry(domain.to_string()) + .or_default() + .push((read_only, spec)); + } + let mut ordered: std::collections::BTreeMap> = by_domain .into_iter() .map(|(domain, mut v)| { // ours (official / configured) before community, then read-only @@ -1139,6 +1169,13 @@ fn write_domain_index(dir: &std::path::Path) { (domain, v.into_iter().map(|(_, s)| s).collect()) }) .collect(); + for (domain, mut v) in opencli_by_domain { + v.sort_by(|a, b| b.0.cmp(&a.0).then_with(|| a.1.cmp(&b.1))); + ordered + .entry(domain) + .or_default() + .extend(v.into_iter().map(|(_, s)| s)); + } if let Ok(json) = serde_json::to_string(&ordered) { let _ = std::fs::write(dir.join(".index.json"), json); } From f16add4fff0ebfe058e62949b09d781b3c4ab074 Mon Sep 17 00:00:00 2001 From: leeguooooo Date: Tue, 6 Oct 2026 14:10:43 +0900 Subject: [PATCH 3/5] fix(site): let analyze/verify through the site subcommand guard; help text --- cli/src/main.rs | 5 ++++- cli/src/output.rs | 13 +++++++++++-- 2 files changed, 15 insertions(+), 3 deletions(-) diff --git a/cli/src/main.rs b/cli/src/main.rs index b185206cc3..bb84ae894b 100644 --- a/cli/src/main.rs +++ b/cli/src/main.rs @@ -1884,6 +1884,9 @@ fn main() { } return; } + // `site analyze [url]` / `site verify / …` → daemon dispatch + // (commands.rs builds them). + Some("analyze") | Some("verify") => {} // `site / [args]` → fall through to the daemon dispatch. Some(spec) if spec.contains('/') => { // #125: an adapter arg whose name collides with a reserved global @@ -1918,7 +1921,7 @@ fn main() { } _ => { eprintln!( - "{} usage: chrome-use site / [args] | site update | site list | \ + "{} usage: chrome-use site / [args] | site analyze [url] | site verify / [args] [--write-fixture] | site update | site list | \ site info / | site sources | site add|remove ", color::error_indicator() ); diff --git a/cli/src/output.rs b/cli/src/output.rs index 5121a8e7d5..781252b942 100644 --- a/cli/src/output.rs +++ b/cli/src/output.rs @@ -4947,11 +4947,20 @@ Usage: chrome-use site list List installed adapters (name/command) chrome-use site update Fetch/refresh the adapter packs chrome-use site info Show one pack's adapters and their args + chrome-use site analyze [url] Find a page's API calls, embedded state and + anti-bot vendors; recommend a data source + chrome-use site verify / [args] [--write-fixture] + Run it and compare the result's shape with a + recorded fixture (--write-fixture records one) An adapter extracts structured JSON from a site through its own API/DOM, in the site's real logged-in page — so it replaces a snapshot+click scrape -with one call. Adapters ship in community packs (epiral/bb-sites) and the -official leeguooooo/chrome-use-sites pack; `update` syncs both. +with one call. Adapters ship in the official leeguooooo/chrome-use-sites pack +and the community epiral/bb-sites pack; `update` syncs both. When Node.js 20+ +is on PATH, `update` also installs OpenCLI (jackwener/OpenCLI): a `name/command` +neither pack has runs through OpenCLI's own runtime over this session, marked +"(opencli)" in `site list`. Ours win on a shared name. +AGENT_BROWSER_SITES_NO_OPENCLI=1 turns OpenCLI off. The spec is always `name/command`. `site` alone, or a wrong spec, prints a one-line usage and points you at `site list`. From 3021b23dba6bebce748ba381cf494057f0af2a53 Mon Sep 17 00:00:00 2001 From: leeguooooo Date: Tue, 6 Oct 2026 14:11:15 +0900 Subject: [PATCH 4/5] docs: site analyze, verify and OpenCLI commands --- docs/en/site-adapters.html | 15 ++++++++++++ docs/site-adapters.html | 15 ++++++++++++ skill-data/core/references/site-adapters.md | 27 ++++++++++++++++----- 3 files changed, 51 insertions(+), 6 deletions(-) diff --git a/docs/en/site-adapters.html b/docs/en/site-adapters.html index 9439ebe801..48e8475ccf 100644 --- a/docs/en/site-adapters.html +++ b/docs/en/site-adapters.html @@ -266,6 +266,21 @@

A site you use a lot has no adapter

AGENT_BROWSER_SITES_NO_SUGGEST=1 turns it off.

+

OpenCLI commands work too

+

+ With Node.js 20+ on PATH, site update also installs OpenCLI (about 180 sites and 1,300 commands; a pinned version, installed with --ignore-scripts). + A name/cmd that neither adapter pack has runs through OpenCLI's own runtime, driving your current chrome-use session and its logins, e.g. + chrome-use site hackernews/best --limit 5 --json. They show as (opencli) in site list, site info shows their args, and they come last in the site hint. + On a shared name ours win. AGENT_BROWSER_SITES_NO_OPENCLI=1 turns them off. +

+ +

Writing your own: analyze and verify

+

+ Do the action that loads the data on the page first (search, scroll, open the list), then run chrome-use site analyze. It lists the API calls the page made, the state it embeds (__NEXT_DATA__, __INITIAL_STATE__, …) and any anti-bot vendor, and recommends a data source. + In order of preference: a public API, the site's own JSON API called from the page with its cookies, embedded page state, then the DOM. Each step down breaks more often. + Once it works, chrome-use site verify <name>/<cmd> [args] --write-fixture records the shape of a good result; later, site verify fails (exit 1) when a field disappears, changes type, or a list comes back empty. The fixture keeps the shape only, never values. +

+
ℹ️ Exception on a fresh environment
Only on a brand-new environment where the adapter sources haven't been pulled yet might naming a diff --git a/docs/site-adapters.html b/docs/site-adapters.html index 09eab5d8ec..e7fd1cc7c3 100644 --- a/docs/site-adapters.html +++ b/docs/site-adapters.html @@ -258,6 +258,21 @@

常用的站点还没有适配器

设 AGENT_BROWSER_SITES_NO_SUGGEST=1 关闭这个提示。

+

OpenCLI 的命令也能直接用

+

+ 装了 Node.js 20+ 时,site update 会顺带安装 OpenCLI(约 180 个站点、1300 多条命令,固定版本,--ignore-scripts)。 + 两个适配器库都没有的 name/cmd,会用 OpenCLI 自己的运行时执行,操作的还是你当前的 chrome-use 会话和登录态,比如 + chrome-use site hackernews/best --limit 5 --json。site list 里标着 (opencli),site info 能看参数,站点提示里排在最后。 + 同名时用我们自己的。AGENT_BROWSER_SITES_NO_OPENCLI=1 关闭。 +

+ +

自己写适配器:analyze 和 verify

+

+ 先在页面上把加载数据的操作做一遍(搜索、滚动、打开列表),再跑 chrome-use site analyze:它列出页面发过的接口请求、页面里嵌着的初始数据(__NEXT_DATA__、__INITIAL_STATE__ 等)和检测到的反爬厂商,并推荐取数方式。 + 优先顺序:公开 API,网站自己的 JSON 接口(在页面里带 cookie 调用),页面嵌入的数据,最后才是 DOM;越往后越容易坏。 + 写好后用 chrome-use site verify <name>/<cmd> [参数] --write-fixture 记下一次正确结果的结构;之后 site verify 发现字段消失、类型变了或列表变空就失败(退出码 1)。fixture 只存结构,不存数据。 +

+
ℹ️ 全新环境的例外
只有在刚装好、适配器源还没拉取的全新环境里,点名一个 site <name>/<cmd> diff --git a/skill-data/core/references/site-adapters.md b/skill-data/core/references/site-adapters.md index c3f7e0ebc7..a8d7cbd37b 100644 --- a/skill-data/core/references/site-adapters.md +++ b/skill-data/core/references/site-adapters.md @@ -20,8 +20,15 @@ chrome-use site github/issues owner/repo --json # run it → JSON (navigates t - It navigates to the adapter's domain (reusing the current tab if you're already on it), so login-gated feeds (`bilibili/feed`, `twitter/...`) work because they run as *you*. - If no adapter fits, fall back to the normal `snapshot`/`eval` loop. chrome-use fetches and - runs two default sources: the [bb-sites](https://github.com/epiral/bb-sites) community pack - and the official [chrome-use-sites](https://github.com/leeguooooo/chrome-use-sites) pack. + runs two default sources: the official [chrome-use-sites](https://github.com/leeguooooo/chrome-use-sites) + pack and the [bb-sites](https://github.com/epiral/bb-sites) community pack. On a shared + `name/cmd` the official one wins. +- **OpenCLI commands work too.** When Node.js 20+ is on PATH, `site update` also installs + [OpenCLI](https://github.com/jackwener/OpenCLI) (~1,300 commands over ~180 sites). A + `name/cmd` that neither pack has runs through OpenCLI's own runtime, driving this same + session. They show as `(opencli)` in `site list`, `site info` shows their args, and they come + last in the `siteAdapters` hint. Same command, same JSON: `chrome-use site hackernews/best + --limit 5 --json`. `AGENT_BROWSER_SITES_NO_OPENCLI=1` turns them off. > **Auto-trigger — act on it.** chrome-use keeps both packs synced automatically (first use + > weekly), and whenever you reach a page whose domain has adapters it tells you: on every @@ -43,13 +50,21 @@ If you work on the same site a lot and no adapter covers it, chrome-use adds **Ask the user** whether to turn the steps you keep repeating there into an adapter. Don't write one without a yes. If they agree: -1. Find the data source with `network requests` / `eval` (the site's own JSON API beats DOM - scraping), then write `~/.chrome-use/my-sites//.js` in the format above - (`@meta` with `name`, `description`, `domain`, `args`, `readOnly`; then the `async function`). +1. First check `chrome-use site list | grep `: OpenCLI may already cover it. Otherwise + do the action that loads the data (search, scroll, open the list), then run + `chrome-use site analyze`. It lists the API calls the page made, the state it embeds + (`__NEXT_DATA__`, `__INITIAL_STATE__`, …) and any anti-bot vendor, and picks a strategy. + Prefer, in this order: a public API; the site's own JSON API called from the page + (`fetch(url, {credentials: 'include'})`); embedded page state; the DOM. Each step down breaks + more often. Write `~/.chrome-use/my-sites//.js` in the format above (`@meta` with + `name`, `description`, `domain`, `args`, `readOnly`; then the `async function`). 2. Register the folder once and sync: `chrome-use site add ~/.chrome-use/my-sites`, then `chrome-use site update`. Keep your own adapters in that folder, not in `~/.chrome-use/sites`, because a sync rewrites `~/.chrome-use/sites`. -3. Run it: `chrome-use site / --json`. If it's generally useful, offer to send it to +3. Run it: `chrome-use site / --json`. When the result looks right, record it: + `chrome-use site verify / [args] --write-fixture`. Later, `site verify /` + fails (exit 1) when a field disappears, changes type, or a list comes back empty, so a broken + adapter shows up before you trust its output. The fixture keeps the shape only, never values. If it's generally useful, offer to send it to [chrome-use-sites](https://github.com/leeguooooo/chrome-use-sites). That opens a PR from the user's account, so ask before you do it. From ce18c2abfcfcdb262236b4328414d8e8339682c9 Mon Sep 17 00:00:00 2001 From: leeguooooo Date: Tue, 6 Oct 2026 14:17:46 +0900 Subject: [PATCH 5/5] fix(site): analyze skips extension globals and ad requests; OpenCLI ranks after ours on every domain spelling --- cli/src/site.rs | 72 ++++++++++++++++++++++++++++++++----------------- 1 file changed, 47 insertions(+), 25 deletions(-) diff --git a/cli/src/site.rs b/cli/src/site.rs index a970f5f1ae..12f057f3aa 100644 --- a/cli/src/site.rs +++ b/cli/src/site.rs @@ -1154,7 +1154,7 @@ fn write_domain_index(dir: &std::path::Path) { .or_default() .push((read_only, spec)); } - let mut ordered: std::collections::BTreeMap> = by_domain + let ordered: std::collections::BTreeMap> = by_domain .into_iter() .map(|(domain, mut v)| { // ours (official / configured) before community, then read-only @@ -1169,12 +1169,18 @@ fn write_domain_index(dir: &std::path::Path) { (domain, v.into_iter().map(|(_, s)| s).collect()) }) .collect(); - for (domain, mut v) in opencli_by_domain { - v.sort_by(|a, b| b.0.cmp(&a.0).then_with(|| a.1.cmp(&b.1))); - ordered - .entry(domain) - .or_default() - .extend(v.into_iter().map(|(_, s)| s)); + // OpenCLI goes in its own index so a lookup can always rank it after + // ours, even when the two packs spell the domain differently + // (`v2ex.com` vs `www.v2ex.com`). + let theirs: std::collections::BTreeMap> = opencli_by_domain + .into_iter() + .map(|(domain, mut v)| { + v.sort_by(|a, b| b.0.cmp(&a.0).then_with(|| a.1.cmp(&b.1))); + (domain, v.into_iter().map(|(_, s)| s).collect()) + }) + .collect(); + if let Ok(json) = serde_json::to_string(&theirs) { + let _ = std::fs::write(dir.join(".index-opencli.json"), json); } if let Ok(json) = serde_json::to_string(&ordered) { let _ = std::fs::write(dir.join(".index.json"), json); @@ -1236,22 +1242,30 @@ pub fn needs_refresh() -> bool { /// Reads the prebuilt `.index.json`; empty if the packs aren't synced yet. pub fn adapters_for_domain(host: &str) -> Vec { let host = host.trim_start_matches("www."); - let Some(raw) = index_path().and_then(|p| std::fs::read_to_string(p).ok()) else { - return Vec::new(); - }; - let Ok(idx) = serde_json::from_str::>>(&raw) - else { - return Vec::new(); - }; - // Preserve the index's per-domain ordering (read-only adapters first); just - // dedup if a host somehow matches multiple domain keys. + // Ours first, then OpenCLI's (its own index, see write_domain_index). let mut out: Vec = Vec::new(); - for (domain, specs) in idx { - let d = domain.trim_start_matches("www."); - if host == d || host.ends_with(&format!(".{d}")) { - for s in specs { - if !out.contains(&s) { - out.push(s); + let files = [ + index_path(), + sites_dir().map(|d| d.join(".index-opencli.json")), + ]; + for path in files.into_iter().flatten() { + if path.ends_with(".index-opencli.json") && crate::opencli::disabled() { + continue; + } + let Some(idx) = std::fs::read_to_string(&path).ok().and_then(|raw| { + serde_json::from_str::>>(&raw).ok() + }) else { + continue; + }; + // Preserve each index's per-domain ordering (read-only first); dedup + // when a host matches several domain keys. + for (domain, specs) in idx { + let d = domain.trim_start_matches("www."); + if host == d || host.ends_with(&format!(".{d}")) { + for s in specs { + if !out.contains(&s) { + out.push(s); + } } } } @@ -1389,7 +1403,7 @@ pub const ANALYZE_JS: &str = r#"(() => { const out = { url: location.href, host: location.hostname, title: document.title }; const host = location.hostname.replace(/^www\./, ''); const base = host.split('.').slice(-2).join('.'); - const noise = /google-analytics|googletagmanager|doubleclick|facebook\.net|hotjar|sentry|segment\.(io|com)|mixpanel|clarity\.ms|bat\.bing|newrelic|datadoghq|amplitude|\/collect\b|\/log(ging)?\b|\/track(ing)?\b|\/beacon\b|\/metrics?\b|\/report\b|\/telemetry\b|\/pixel\b/i; + const noise = /google-analytics|googletagmanager|doubleclick|googlesyndication|adtrafficquality|adservice|pagead|facebook\.net|hotjar|sentry|segment\.(io|com)|mixpanel|clarity\.ms|bat\.bing|newrelic|datadoghq|amplitude|\/collect\b|\/log(ging)?\b|\/track(ing)?\b|\/beacon\b|\/metrics?\b|\/report\b|\/telemetry\b|\/pixel\b/i; const seen = new Set(); const api = []; for (const e of performance.getEntriesByType('resource')) { @@ -1411,7 +1425,7 @@ pub const ANALYZE_JS: &str = r#"(() => { api.push({ url: e.name.length > 300 ? e.name.slice(0, 300) + '…' : e.name, type: e.initiatorType, sameSite, bytes: e.transferSize || 0, score, reasons }); } api.sort((a, b) => b.score - a.score); - out.api = api.filter(a => a.score > 0).slice(0, 12); + out.api = api.filter(a => a.score >= 2).slice(0, 12); out.requestsSeen = api.length; const known = ['__NEXT_DATA__', '__NUXT__', '__NUXT_DATA__', '__INITIAL_STATE__', '__INITIAL_DATA__', '__INITIAL_PROPS__', '__PRELOADED_STATE__', '__APOLLO_STATE__', '__REDUX_STATE__', '__SSR_DATA__', '__remixContext', '__UNIVERSAL_DATA_FOR_REHYDRATION__', 'ytInitialData', 'ytInitialPlayerResponse', '__pinia', '__INITIAL_SSR_STATE__', 'g_initialProps', '__STATE__']; @@ -1435,7 +1449,15 @@ pub const ANALYZE_JS: &str = r#"(() => { const sel = el.id ? 'script#' + el.id : 'script[type="' + el.type + '"]'; out.state.push({ name: sel, ...describe(v) }); } - out.state = out.state.filter(s => s.size === -1 || s.size > 200); + // Extension-injected globals (Vue/React devtools) and telemetry config are + // not the page's data. + const junk = /devtools|rum|analytics|tracking|gtm|sentry/i; + const seenState = new Set(); + out.state = out.state.filter(s => { + if (junk.test(s.name) || seenState.has(s.name)) return false; + seenState.add(s.name); + return s.size === -1 || s.size > 200; + }); out.webpack = Object.getOwnPropertyNames(window).filter(k => /^webpackChunk|^webpackJsonp/.test(k)).slice(0, 3); out.signals = {