From d0ca2adaccac3e55ef32c0e1862441506ffc2cc5 Mon Sep 17 00:00:00 2001
From: leeguooooo
Date: Tue, 6 Oct 2026 13:56:36 +0900
Subject: [PATCH 1/5] feat(site): site analyze and site verify
---
cli/src/commands.rs | 22 +++
cli/src/main.rs | 40 +++-
cli/src/native/actions.rs | 32 +++
cli/src/native/humanize.rs | 19 ++
cli/src/output.rs | 84 ++++++++
cli/src/site.rs | 394 +++++++++++++++++++++++++++++++++++++
6 files changed, 590 insertions(+), 1 deletion(-)
diff --git a/cli/src/commands.rs b/cli/src/commands.rs
index cd4992e2a6..021aaa11d5 100644
--- a/cli/src/commands.rs
+++ b/cli/src/commands.rs
@@ -1919,6 +1919,27 @@ fn parse_command_inner(args: &[String], flags: &Flags) -> Result/ [args] [--write-fixture]`: a normal run,
+ // then main.rs compares the result's shape with the stored fixture.
+ let (verify, rest): (Option, Vec<&str>) = if rest.first() == Some(&"verify") {
+ let write = rest.contains(&"--write-fixture");
+ let kept = rest[1..]
+ .iter()
+ .filter(|a| **a != "--write-fixture")
+ .copied()
+ .collect();
+ (Some(write), kept)
+ } else {
+ (None, rest.to_vec())
+ };
let spec = rest.first().ok_or(ParseError::InvalidValue {
message: "site requires / (run `chrome-use site list`)".to_string(),
usage: "site / [args]",
@@ -2000,6 +2021,7 @@ fn parse_command_inner(args: &[String], flags: &Flags) -> Result = report
+ .get("issues")
+ .and_then(|x| x.as_array())
+ .map(|a| {
+ a.iter()
+ .filter_map(|i| i.as_str().map(String::from))
+ .collect()
+ })
+ .unwrap_or_default();
+ resp.error = Some(format!(
+ "site verify {spec}: {}",
+ if issues.is_empty() {
+ "could not record the fixture".to_string()
+ } else {
+ issues.join("; ")
+ }
+ ));
+ }
+ if let Some(d) = resp.data.as_mut().and_then(|d| d.as_object_mut()) {
+ d.insert("verify".into(), report);
+ }
+ }
+ }
if let Some(err) = resp.error.as_mut() {
if err.contains("has NO snapshot refs") {
// Keep what to do last: agents read errors through `tail -1`.
diff --git a/cli/src/native/actions.rs b/cli/src/native/actions.rs
index 4945f15703..99286116bf 100644
--- a/cli/src/native/actions.rs
+++ b/cli/src/native/actions.rs
@@ -2012,6 +2012,7 @@ pub async fn execute_command(cmd: &Value, state: &mut DaemonState) -> Value {
"content" => handle_content(state).await,
"evaluate" => handle_evaluate(cmd, state).await,
"site" => handle_site(cmd, state).await,
+ "site_analyze" => handle_site_analyze(cmd, state).await,
"script" => super::script::handle_script(cmd, state).await,
"close" => handle_close(state).await,
"keep" => handle_keep(cmd, state).await,
@@ -4794,6 +4795,37 @@ async fn resolve_iframe_selector(
/// already loaded the adapter and built the `script`; here we just place the page
/// and evaluate. Never disrupts the user's foreground tab — navigation happens on
/// the daemon's own tab (same as every other command on the relay).
+/// `site analyze [url]`: scan the current page (after opening `url`, if given)
+/// for what an adapter should read — API calls the page made, state it embeds,
+/// anti-bot vendors — and recommend a data source.
+async fn handle_site_analyze(cmd: &Value, state: &mut DaemonState) -> Result {
+ if cmd.get("url").and_then(|v| v.as_str()).is_some() {
+ handle_navigate(cmd, state).await?;
+ }
+ let mgr = state.browser.as_ref().ok_or("site analyze: no browser")?;
+ let raw = mgr.evaluate(crate::site::ANALYZE_JS, None).await?;
+ let strings = |key: &str| -> Vec {
+ raw.get("signals")
+ .and_then(|s| s.get(key))
+ .and_then(|v| v.as_array())
+ .map(|a| {
+ a.iter()
+ .filter_map(|x| x.as_str().map(String::from))
+ .collect()
+ })
+ .unwrap_or_default()
+ };
+ let signals = humanize::DetectSignals {
+ cookie_names: strings("cookies"),
+ script_urls: strings("scripts"),
+ window_globals: strings("globals"),
+ };
+ let vendors = humanize::detected_vendors(&signals);
+ let host = raw.get("host").and_then(|v| v.as_str()).unwrap_or("");
+ let adapters = crate::site::adapters_for_domain(host);
+ Ok(crate::site::analyze_report(&raw, &vendors, &adapters))
+}
+
async fn handle_site(cmd: &Value, state: &mut DaemonState) -> Result {
let domain = cmd
.get("domain")
diff --git a/cli/src/native/humanize.rs b/cli/src/native/humanize.rs
index 7977253df1..ef89d2b0a2 100644
--- a/cli/src/native/humanize.rs
+++ b/cli/src/native/humanize.rs
@@ -375,6 +375,25 @@ const VENDOR_MARKERS: &[(&str, &str)] = &[
("__cf_bm", "cloudflare-bot-mgmt"),
];
+/// The anti-bot vendors whose markers appear in `signals`, deduped, in marker
+/// order (for `site analyze`).
+pub fn detected_vendors(signals: &DetectSignals) -> Vec<&'static str> {
+ let hay: Vec = signals
+ .cookie_names
+ .iter()
+ .chain(signals.script_urls.iter())
+ .chain(signals.window_globals.iter())
+ .map(|s| s.to_ascii_lowercase())
+ .collect();
+ let mut out: Vec<&'static str> = Vec::new();
+ for (marker, vendor) in VENDOR_MARKERS {
+ if hay.iter().any(|h| h.contains(marker)) && !out.contains(vendor) {
+ out.push(vendor);
+ }
+ }
+ out
+}
+
/// Decide the level for a page. Returns `Human` if any known anti-bot vendor is
/// present, otherwise `baseline`. Misses just stay at baseline and false hits
/// only cost a little latency, so matching is deliberately liberal.
diff --git a/cli/src/output.rs b/cli/src/output.rs
index 37d628e24a..7febe4dbc4 100644
--- a/cli/src/output.rs
+++ b/cli/src/output.rs
@@ -300,6 +300,61 @@ fn stderr_is_discarded() -> bool {
}
}
+fn print_site_analyze(data: &serde_json::Value, strategy: &str, next: &[serde_json::Value]) {
+ let host = data.get("host").and_then(|v| v.as_str()).unwrap_or("");
+ println!("site analyze: {host}");
+ let api = data
+ .get("api")
+ .and_then(|v| v.as_array())
+ .cloned()
+ .unwrap_or_default();
+ let seen = data
+ .get("requestsSeen")
+ .and_then(|v| v.as_u64())
+ .unwrap_or(0);
+ println!(" API candidates ({} of {seen} requests):", api.len());
+ for a in api.iter().take(8) {
+ let url = a.get("url").and_then(|v| v.as_str()).unwrap_or("");
+ let why: Vec<&str> = a
+ .get("reasons")
+ .and_then(|v| v.as_array())
+ .map(|r| r.iter().filter_map(|x| x.as_str()).collect())
+ .unwrap_or_default();
+ println!(" {url}");
+ println!(" {}", color::dim(&why.join(", ")));
+ }
+ let state = data
+ .get("state")
+ .and_then(|v| v.as_array())
+ .cloned()
+ .unwrap_or_default();
+ if !state.is_empty() {
+ println!(" Embedded state:");
+ for st in state.iter().take(8) {
+ let name = st.get("name").and_then(|v| v.as_str()).unwrap_or("");
+ let size = st.get("size").and_then(|v| v.as_i64()).unwrap_or(0);
+ let keys: Vec<&str> = st
+ .get("keys")
+ .and_then(|v| v.as_array())
+ .map(|r| r.iter().filter_map(|x| x.as_str()).collect())
+ .unwrap_or_default();
+ println!(" {name} ({size} chars) {}", color::dim(&keys.join(", ")));
+ }
+ }
+ if let Some(v) = data
+ .get("antiBot")
+ .and_then(|v| v.as_array())
+ .filter(|a| !a.is_empty())
+ {
+ let names: Vec<&str> = v.iter().filter_map(|x| x.as_str()).collect();
+ println!(" Anti-bot: {}", names.join(", "));
+ }
+ println!(" Strategy: {strategy}");
+ for (i, step) in next.iter().filter_map(|s| s.as_str()).enumerate() {
+ println!(" {}. {step}", i + 1);
+ }
+}
+
pub fn print_response_with_opts(resp: &Response, action: Option<&str>, opts: &OutputOptions) {
print_response_body(resp, action, opts);
// Every successful text response gets its observation, including branches
@@ -389,6 +444,35 @@ fn print_response_body(resp: &Response, action: Option<&str>, opts: &OutputOptio
eprintln!("site adapter suggestion: {msg}");
}
}
+ // `site verify`: the verdict, to stderr so the result stays parseable.
+ if let Some(v) = data.get("verify") {
+ if v.get("recorded").and_then(|x| x.as_bool()) == Some(true) {
+ let at = v.get("fixture").and_then(|x| x.as_str()).unwrap_or("");
+ eprintln!(
+ "{} site verify: fixture recorded → {at}",
+ color::success_indicator()
+ );
+ } else if v.get("ok").and_then(|x| x.as_bool()) == Some(true) {
+ match v.get("next").and_then(|x| x.as_str()) {
+ Some(next) => eprintln!(
+ "{} site verify: result non-empty; {next}",
+ color::success_indicator()
+ ),
+ None => eprintln!(
+ "{} site verify: matches the fixture",
+ color::success_indicator()
+ ),
+ }
+ }
+ }
+ // `site analyze`: a report, not a page result.
+ if let (Some(strategy), Some(next)) = (
+ data.get("strategy").and_then(|v| v.as_str()),
+ data.get("next").and_then(|v| v.as_array()),
+ ) {
+ print_site_analyze(data, strategy, next);
+ return;
+ }
// `open` that landed on a page refusing this browser's sign-in (#387).
if let Some(h) = data.get("humanCheck") {
print_human_check(h);
diff --git a/cli/src/site.rs b/cli/src/site.rs
index 3c33f9d18b..a30135bda1 100644
--- a/cli/src/site.rs
+++ b/cli/src/site.rs
@@ -1344,6 +1344,355 @@ pub fn unix_now() -> u64 {
now_secs()
}
+/// Page-side half of `site analyze`: what an adapter author needs to pick a
+/// data source. Requests come from the Resource Timing buffer, so they cover
+/// what the page has loaded so far — interact first (search, scroll, open a
+/// list), then analyze, to catch the request that action made.
+pub const ANALYZE_JS: &str = r#"(() => {
+ const out = { url: location.href, host: location.hostname, title: document.title };
+ const host = location.hostname.replace(/^www\./, '');
+ const base = host.split('.').slice(-2).join('.');
+ const noise = /google-analytics|googletagmanager|doubleclick|facebook\.net|hotjar|sentry|segment\.(io|com)|mixpanel|clarity\.ms|bat\.bing|newrelic|datadoghq|amplitude|\/collect\b|\/log(ging)?\b|\/track(ing)?\b|\/beacon\b|\/metrics?\b|\/report\b|\/telemetry\b|\/pixel\b/i;
+ const seen = new Set();
+ const api = [];
+ for (const e of performance.getEntriesByType('resource')) {
+ if (!['fetch', 'xmlhttprequest'].includes(e.initiatorType)) continue;
+ let u; try { u = new URL(e.name); } catch (_) { continue; }
+ const key = u.origin + u.pathname;
+ if (seen.has(key)) continue;
+ seen.add(key);
+ const sameSite = u.hostname === location.hostname || u.hostname.endsWith('.' + base) || u.hostname === base;
+ const reasons = [];
+ let score = 0;
+ if (noise.test(e.name)) { score -= 5; reasons.push('analytics/telemetry'); }
+ if (sameSite) { score += 2; reasons.push('same site'); }
+ if (/\/(api|ajax|graphql|gql|rest|v\d+|x\/|web-interface|rpc|data)\b/i.test(u.pathname)) { score += 3; reasons.push('api-like path'); }
+ if (/\.json\b/i.test(u.pathname)) { score += 2; reasons.push('json'); }
+ if (/graphql|gql/i.test(u.pathname)) reasons.push('graphql');
+ if (/[?&](page|cursor|offset|limit|size|count|keyword|q|query|id|uid)=/i.test(u.search)) { score += 1; reasons.push('paging/query params'); }
+ if (e.transferSize > 2000) { score += 1; reasons.push('sizeable body'); }
+ api.push({ url: e.name.length > 300 ? e.name.slice(0, 300) + '…' : e.name, type: e.initiatorType, sameSite, bytes: e.transferSize || 0, score, reasons });
+ }
+ api.sort((a, b) => b.score - a.score);
+ out.api = api.filter(a => a.score > 0).slice(0, 12);
+ out.requestsSeen = api.length;
+
+ const known = ['__NEXT_DATA__', '__NUXT__', '__NUXT_DATA__', '__INITIAL_STATE__', '__INITIAL_DATA__', '__INITIAL_PROPS__', '__PRELOADED_STATE__', '__APOLLO_STATE__', '__REDUX_STATE__', '__SSR_DATA__', '__remixContext', '__UNIVERSAL_DATA_FOR_REHYDRATION__', 'ytInitialData', 'ytInitialPlayerResponse', '__pinia', '__INITIAL_SSR_STATE__', 'g_initialProps', '__STATE__'];
+ const names = new Set(known.filter(k => { try { return window[k] != null; } catch (_) { return false; } }));
+ for (const k of Object.getOwnPropertyNames(window)) {
+ if (names.size > 20) break;
+ if (/^__.*(STATE|DATA|PROPS|CONTEXT|STORE)__?$/i.test(k) || /^(initial|preloaded|ssr)(State|Data|Props)$/i.test(k)) {
+ try { if (window[k] && typeof window[k] === 'object') names.add(k); } catch (_) {}
+ }
+ }
+ const describe = v => {
+ let size = 0; try { size = JSON.stringify(v).length; } catch (_) { size = -1; }
+ const keys = v && typeof v === 'object' ? Object.keys(v).slice(0, 10) : [];
+ return { size, keys };
+ };
+ out.state = [];
+ for (const k of names) { try { out.state.push({ name: 'window.' + k, ...describe(window[k]) }); } catch (_) {} }
+ for (const el of document.querySelectorAll('script[type="application/json"], script[type="application/ld+json"]')) {
+ if (out.state.length > 25) break;
+ let v; try { v = JSON.parse(el.textContent); } catch (_) { continue; }
+ const sel = el.id ? 'script#' + el.id : 'script[type="' + el.type + '"]';
+ out.state.push({ name: sel, ...describe(v) });
+ }
+ out.state = out.state.filter(s => s.size === -1 || s.size > 200);
+
+ out.webpack = Object.getOwnPropertyNames(window).filter(k => /^webpackChunk|^webpackJsonp/.test(k)).slice(0, 3);
+ out.signals = {
+ cookies: document.cookie.split(';').map(c => c.trim().split('=')[0]).filter(Boolean),
+ scripts: Array.from(document.scripts, s => s.src || '').filter(Boolean),
+ globals: Object.getOwnPropertyNames(window).filter(k => /_px|bmak|_abck|datadome|reese84|kpsdk|incap_ses|visid_incap|akam/i.test(k)),
+ };
+ out.loggedInHint = document.cookie.length > 0;
+ return out;
+})()"#;
+
+/// Turn the page scan into the `site analyze` report: drop the raw signals,
+/// attach anti-bot vendors and installed adapters, and recommend a data source
+/// in the order that breaks least often — a site's JSON API called from the
+/// page, then state the page already embeds, then the DOM.
+pub fn analyze_report(raw: &Value, vendors: &[&str], adapters: &[String]) -> Value {
+ let mut report = raw.clone();
+ if let Some(o) = report.as_object_mut() {
+ o.remove("signals");
+ o.remove("loggedInHint");
+ }
+ let api = raw
+ .get("api")
+ .and_then(|v| v.as_array())
+ .cloned()
+ .unwrap_or_default();
+ let state = raw
+ .get("state")
+ .and_then(|v| v.as_array())
+ .cloned()
+ .unwrap_or_default();
+ let webpack = raw
+ .get("webpack")
+ .and_then(|v| v.as_array())
+ .is_some_and(|a| !a.is_empty());
+ let host = raw.get("host").and_then(|v| v.as_str()).unwrap_or("");
+
+ let mut steps: Vec = Vec::new();
+ let strategy;
+ if !adapters.is_empty() {
+ steps.push(format!(
+ "Adapters already exist for {host}: {}. Run `chrome-use site info /` before writing a new one.",
+ adapters.join(", ")
+ ));
+ }
+ if let Some(top) = api
+ .iter()
+ .find(|a| a.get("sameSite").and_then(|v| v.as_bool()) == Some(true))
+ {
+ strategy = "page-fetch";
+ let url = top.get("url").and_then(|v| v.as_str()).unwrap_or("");
+ steps.push(format!(
+ "Call the site's own API from the page: `fetch(url, {{credentials: 'include'}})`, starting from {url}. Check its JSON with `chrome-use eval` first, then map it to the fields the user needs."
+ ));
+ } else if let Some(s) = state.first() {
+ strategy = "page-state";
+ let name = s.get("name").and_then(|v| v.as_str()).unwrap_or("");
+ steps.push(format!(
+ "No API call seen yet, but the page embeds its data in {name}. Read it in the adapter (no extra request). If the data you need comes from a later action, do that action and analyze again."
+ ));
+ } else {
+ strategy = "dom";
+ steps.push(
+ "No API call or embedded state found. Do the action that loads the data (search, scroll, open a list) and run `site analyze` again; fall back to reading the DOM only if nothing shows up — it breaks the most often.".to_string(),
+ );
+ }
+ if webpack {
+ steps.push("The page is a webpack bundle: its own modules (e.g. signed request helpers) can be reached via the webpackChunk global if the API needs a signature.".to_string());
+ }
+ if !vendors.is_empty() {
+ steps.push(format!(
+ "Anti-bot protection detected ({}). Keep requests inside the page (same-origin fetch with the page's cookies), keep the call rate low, and never replay them from outside the browser.",
+ vendors.join(", ")
+ ));
+ }
+ steps.push("Write the adapter to ~/.chrome-use/my-sites//.js, register it once with `chrome-use site add ~/.chrome-use/my-sites`, run `site update`, then `chrome-use site verify / --write-fixture` to record what a good result looks like. Guide: `chrome-use skills get core/site-adapters`.".to_string());
+
+ if let Some(o) = report.as_object_mut() {
+ o.insert("antiBot".into(), json!(vendors));
+ o.insert("adapters".into(), json!(adapters));
+ o.insert("strategy".into(), json!(strategy));
+ o.insert("next".into(), json!(steps));
+ }
+ report
+}
+
+/// `~/.chrome-use/site-fixtures//.json` — the recorded shape of a
+/// good result, for `site verify`.
+pub fn fixture_path(spec: &str) -> Option {
+ let (name, cmd) = spec.split_once('/')?;
+ dirs_home().map(|h| {
+ h.join(".chrome-use")
+ .join("site-fixtures")
+ .join(name)
+ .join(format!("{cmd}.json"))
+ })
+}
+
+/// A structural summary of an adapter result: types per field, depth-limited,
+/// with array items merged over the first few elements. Values are dropped, so
+/// a fixture holds no user data.
+pub fn result_shape(v: &Value) -> Value {
+ shape_at(v, 0)
+}
+
+fn shape_at(v: &Value, depth: usize) -> Value {
+ match v {
+ Value::Null => json!("null"),
+ Value::Bool(_) => json!("boolean"),
+ Value::Number(_) => json!("number"),
+ Value::String(_) => json!("string"),
+ Value::Array(a) => {
+ let mut items = Value::Null;
+ if depth < 4 {
+ for el in a.iter().take(5) {
+ items = merge_shape(items, shape_at(el, depth + 1));
+ }
+ }
+ json!({ "array": items, "nonEmpty": !a.is_empty() })
+ }
+ Value::Object(o) => {
+ if depth >= 4 {
+ return json!("object");
+ }
+ let fields: serde_json::Map = o
+ .iter()
+ .map(|(k, v)| (k.clone(), shape_at(v, depth + 1)))
+ .collect();
+ json!({ "object": fields })
+ }
+ }
+}
+
+/// Merge two shapes of sibling array items: keep fields present in any item,
+/// and let a concrete type win over "null".
+fn merge_shape(a: Value, b: Value) -> Value {
+ match (a, b) {
+ (Value::Null, b) => b,
+ (a, Value::String(t)) if t == "null" => a,
+ (Value::String(t), b) if t == "null" => b,
+ (Value::Object(mut x), Value::Object(y)) => {
+ if let (Some(Value::Object(fx)), Some(Value::Object(fy))) =
+ (x.get("object").cloned(), y.get("object"))
+ {
+ let mut merged = fx;
+ for (k, v) in fy {
+ let cur = merged.remove(k).unwrap_or(Value::Null);
+ merged.insert(k.clone(), merge_shape(cur, v.clone()));
+ }
+ x.insert("object".into(), Value::Object(merged));
+ }
+ Value::Object(x)
+ }
+ (a, _) => a,
+ }
+}
+
+/// Differences that mean the adapter broke: a field the fixture had is gone,
+/// a field changed type, or a list that had rows came back empty. New fields
+/// and null values are fine.
+pub fn shape_diff(expected: &Value, actual: &Value) -> Vec {
+ let mut out = Vec::new();
+ diff_at(expected, actual, "result", &mut out);
+ out
+}
+
+fn kind(v: &Value) -> &str {
+ match v {
+ Value::String(t) => t.as_str(),
+ Value::Object(o) if o.contains_key("array") => "array",
+ Value::Object(o) if o.contains_key("object") => "object",
+ _ => "unknown",
+ }
+}
+
+fn diff_at(exp: &Value, act: &Value, path: &str, out: &mut Vec) {
+ let (ke, ka) = (kind(exp), kind(act));
+ if ke == "null" || ka == "null" || ke == "unknown" {
+ return;
+ }
+ if ke != ka {
+ out.push(format!("{path}: was {ke}, now {ka}"));
+ return;
+ }
+ match ke {
+ "array" => {
+ let had = exp
+ .get("nonEmpty")
+ .and_then(|v| v.as_bool())
+ .unwrap_or(false);
+ let has = act
+ .get("nonEmpty")
+ .and_then(|v| v.as_bool())
+ .unwrap_or(false);
+ if had && !has {
+ out.push(format!("{path}: was a non-empty list, now empty"));
+ return;
+ }
+ if let (Some(e), Some(a)) = (exp.get("array"), act.get("array")) {
+ if !a.is_null() {
+ diff_at(e, a, &format!("{path}[]"), out);
+ }
+ }
+ }
+ "object" => {
+ let (Some(fe), Some(fa)) = (
+ exp.get("object").and_then(|v| v.as_object()),
+ act.get("object").and_then(|v| v.as_object()),
+ ) else {
+ return;
+ };
+ for (k, ve) in fe {
+ match fa.get(k) {
+ Some(va) => diff_at(ve, va, &format!("{path}.{k}"), out),
+ None => out.push(format!("{path}.{k}: missing")),
+ }
+ }
+ }
+ _ => {}
+ }
+}
+
+/// `site verify`: check a run's result against the stored fixture, or record
+/// one. Returns (ok, report).
+pub fn verify_result(spec: &str, result: &Value, write_fixture: bool) -> (bool, Value) {
+ let shape = result_shape(result);
+ let path = fixture_path(spec);
+ let empty = match result {
+ Value::Null => true,
+ Value::Array(a) => a.is_empty(),
+ Value::Object(o) => o.is_empty(),
+ Value::String(s) => s.is_empty(),
+ _ => false,
+ };
+ if write_fixture {
+ if empty {
+ return (
+ false,
+ json!({ "spec": spec, "ok": false, "issues": ["result is empty; not recording it as a fixture"] }),
+ );
+ }
+ let fixture = json!({ "spec": spec, "recordedAt": now_secs(), "shape": shape });
+ let written = path.as_ref().is_some_and(|p| {
+ p.parent()
+ .is_some_and(|d| std::fs::create_dir_all(d).is_ok())
+ && std::fs::write(
+ p,
+ serde_json::to_string_pretty(&fixture).unwrap_or_default(),
+ )
+ .is_ok()
+ });
+ return (
+ written,
+ json!({
+ "spec": spec,
+ "ok": written,
+ "fixture": path.map(|p| p.display().to_string()),
+ "recorded": written,
+ }),
+ );
+ }
+ let stored = path
+ .as_ref()
+ .and_then(|p| std::fs::read_to_string(p).ok())
+ .and_then(|t| serde_json::from_str::(&t).ok());
+ let Some(stored) = stored else {
+ let mut issues = Vec::new();
+ if empty {
+ issues.push("result is empty".to_string());
+ }
+ return (
+ !empty,
+ json!({
+ "spec": spec,
+ "ok": !empty,
+ "fixture": null,
+ "issues": issues,
+ "next": format!("no fixture yet: run `chrome-use site verify {spec} … --write-fixture` once the result looks right"),
+ }),
+ );
+ };
+ let issues = shape_diff(stored.get("shape").unwrap_or(&Value::Null), &shape);
+ let ok = issues.is_empty();
+ (
+ ok,
+ json!({
+ "spec": spec,
+ "ok": ok,
+ "fixture": path.map(|p| p.display().to_string()),
+ "issues": issues,
+ }),
+ )
+}
+
/// Map CLI args to the adapter's `args` object. Positional args fill the adapter's
/// declared `args` keys in order; `--key value` overrides by name. The adapter
/// validates required args itself.
@@ -1394,6 +1743,51 @@ mod tests {
assert_eq!(source_rank(None), source_rank(Some(&community)));
}
+ #[test]
+ fn analyze_prefers_same_site_api_then_state_then_dom() {
+ let api = json!({"host":"x.com","api":[{"url":"https://x.com/i/api/graphql/Q","sameSite":true}],"state":[],"webpack":[],"signals":{}});
+ let r = analyze_report(&api, &[], &[]);
+ assert_eq!(r["strategy"], "page-fetch");
+ assert!(r.get("signals").is_none());
+ let state =
+ json!({"host":"a.com","api":[],"state":[{"name":"window.__NEXT_DATA__","size":900}]});
+ assert_eq!(analyze_report(&state, &[], &[])["strategy"], "page-state");
+ let dom = json!({"host":"a.com","api":[{"url":"https://cdn.other.net/x","sameSite":false}],"state":[]});
+ let r = analyze_report(&dom, &["akamai"], &["a/b".to_string()]);
+ assert_eq!(r["strategy"], "dom");
+ assert_eq!(r["antiBot"][0], "akamai");
+ let next = r["next"].to_string();
+ assert!(next.contains("a/b") && next.contains("Anti-bot"));
+ }
+
+ #[test]
+ fn verify_flags_missing_fields_type_changes_and_emptied_lists() {
+ let good = json!({"items":[{"id":1,"title":"a","tag":null},{"id":2,"title":"b","tag":"x"}],"total":2});
+ let exp = result_shape(&good);
+ assert!(shape_diff(&exp, &result_shape(&good)).is_empty());
+ // extra field and null value are fine
+ let extra = json!({"items":[{"id":3,"title":null,"tag":"y","new":true}],"total":1});
+ assert!(shape_diff(&exp, &result_shape(&extra)).is_empty());
+ let broken = json!({"items":[{"id":"3"}],"total":1});
+ let d = shape_diff(&exp, &result_shape(&broken));
+ assert!(
+ d.iter()
+ .any(|x| x.contains("result.items[].id: was number, now string")),
+ "{d:?}"
+ );
+ assert!(
+ d.iter()
+ .any(|x| x.contains("result.items[].title: missing")),
+ "{d:?}"
+ );
+ let emptied = json!({"items":[],"total":0});
+ let d = shape_diff(&exp, &result_shape(&emptied));
+ assert!(
+ d.iter().any(|x| x.contains("non-empty list, now empty")),
+ "{d:?}"
+ );
+ }
+
#[test]
fn adapter_suggestion_thresholds() {
let now = 10_000_000;
From 8c1984252d69cf7f57a39bcf9422ed3065403130 Mon Sep 17 00:00:00 2001
From: leeguooooo
Date: Tue, 6 Oct 2026 14:02:40 +0900
Subject: [PATCH 2/5] feat(site): run OpenCLI adapters through its own runtime
over chrome-use
---
cli/src/main.rs | 105 +++++++++++-
cli/src/opencli.rs | 323 +++++++++++++++++++++++++++++++++++++
cli/src/opencli_runner.mjs | 136 ++++++++++++++++
cli/src/output.rs | 12 +-
cli/src/site.rs | 39 ++++-
5 files changed, 612 insertions(+), 3 deletions(-)
create mode 100644 cli/src/opencli.rs
create mode 100644 cli/src/opencli_runner.mjs
diff --git a/cli/src/main.rs b/cli/src/main.rs
index 8ed9bcb5c5..b185206cc3 100644
--- a/cli/src/main.rs
+++ b/cli/src/main.rs
@@ -16,6 +16,7 @@ mod install;
mod jev;
mod mcp;
mod native;
+mod opencli;
mod output;
mod ownership;
mod read;
@@ -1611,6 +1612,81 @@ fn main() {
),
}
}
+ // A `name/cmd` we have no adapter for but OpenCLI does: run it with
+ // OpenCLI's runtime over this session (opencli.rs). `site verify` too.
+ {
+ let verify = clean.get(1).map(|s| s.as_str()) == Some("verify");
+ let at = if verify { 2 } else { 1 };
+ if let Some(spec) = clean.get(at).filter(|s| opencli::handles(s)) {
+ let entry = opencli::lookup(spec).unwrap_or_default();
+ let write_fixture = verify && clean.iter().any(|a| a == "--write-fixture");
+ let rest: Vec = clean[at + 1..]
+ .iter()
+ .filter(|a| !(verify && a.as_str() == "--write-fixture"))
+ .cloned()
+ .collect();
+ let mut env = opencli::run(spec, &entry, &rest, &flags.session);
+ let mut ok = env.get("success").and_then(|v| v.as_bool()) == Some(true);
+ let result = env.get("data").cloned().unwrap_or(Value::Null);
+ if ok && verify {
+ let (vok, report) = site::verify_result(spec, &result, write_fixture);
+ if !vok {
+ ok = false;
+ let issues: Vec = report
+ .get("issues")
+ .and_then(|x| x.as_array())
+ .map(|a| {
+ a.iter()
+ .filter_map(|i| i.as_str().map(String::from))
+ .collect()
+ })
+ .unwrap_or_default();
+ env["error"] = json!(format!("site verify {spec}: {}", issues.join("; ")));
+ }
+ env["verify"] = report;
+ }
+ if flags.json {
+ let mut data = json!({ "result": result, "source": opencli::SOURCE_LABEL });
+ if let Some(v) = env.get("verify") {
+ data["verify"] = v.clone();
+ }
+ println!(
+ "{}",
+ json!({ "success": ok, "data": data, "error": if ok { Value::Null } else { env.get("error").cloned().unwrap_or(Value::Null) } })
+ );
+ } else if ok {
+ eprintln!("{}", color::dim(&format!("site {spec} (via OpenCLI)")));
+ println!(
+ "{}",
+ serde_json::to_string_pretty(&result).unwrap_or_default()
+ );
+ if let Some(v) = env.get("verify") {
+ if v.get("recorded").and_then(|x| x.as_bool()) == Some(true) {
+ eprintln!(
+ "{} site verify: fixture recorded",
+ color::success_indicator()
+ );
+ } else {
+ eprintln!("{} site verify: ok", color::success_indicator());
+ }
+ }
+ } else {
+ let err = env
+ .get("error")
+ .and_then(|v| v.as_str())
+ .unwrap_or("failed");
+ match env
+ .get("hint")
+ .and_then(|v| v.as_str())
+ .filter(|h| !h.is_empty())
+ {
+ Some(h) => eprintln!("{} {err} — {h}", color::error_indicator()),
+ None => eprintln!("{} {err}", color::error_indicator()),
+ }
+ }
+ exit(if ok { 0 } else { 1 });
+ }
+ }
match clean.get(1).map(|s| s.as_str()) {
Some("update") => {
let rt = tokio::runtime::Runtime::new().expect("Failed to create tokio runtime");
@@ -1631,9 +1707,23 @@ fn main() {
return;
}
Some("list") => {
+ let theirs: Vec = {
+ let ours = site::list_adapters().unwrap_or_default();
+ let mut v: Vec = opencli::manifest()
+ .iter()
+ .filter_map(opencli::spec_of)
+ .filter(|s| !ours.contains(s))
+ .collect();
+ v.sort();
+ v.dedup();
+ v
+ };
match site::list_adapters() {
Ok(list) if flags.json => {
- println!("{}", json!({ "success": true, "adapters": list }))
+ println!(
+ "{}",
+ json!({ "success": true, "adapters": list, "opencli": theirs })
+ )
}
Ok(list) if list.is_empty() => {
println!("no site adapters installed — run `chrome-use site update`")
@@ -1642,6 +1732,9 @@ fn main() {
for a in &list {
println!("{a}");
}
+ for a in &theirs {
+ println!("{a} {}", color::dim("(opencli)"));
+ }
eprintln!(
"{}",
color::dim(&format!(
@@ -1659,6 +1752,16 @@ fn main() {
}
Some("info") => {
let spec = clean.get(2).cloned().unwrap_or_default();
+ if opencli::handles(&spec) {
+ if let Some(entry) = opencli::lookup(&spec) {
+ println!(
+ "{}",
+ serde_json::to_string_pretty(&opencli::info(&entry))
+ .unwrap_or_default()
+ );
+ return;
+ }
+ }
match site::load_adapter(&spec) {
Ok(a) => println!(
"{}",
diff --git a/cli/src/opencli.rs b/cli/src/opencli.rs
new file mode 100644
index 0000000000..dbbdba4a31
--- /dev/null
+++ b/cli/src/opencli.rs
@@ -0,0 +1,323 @@
+//! OpenCLI compatibility: run jackwener/OpenCLI adapters as `site` commands.
+//!
+//! OpenCLI's adapters run in Node and call a `page` object, so they cannot be
+//! evaluated in a tab like our adapters. Instead `site update` installs a pinned
+//! `@jackwener/opencli` package (public npm tarball, `--ignore-scripts`) into
+//! `~/.chrome-use/opencli`, and `opencli_runner.mjs` runs an adapter with
+//! OpenCLI's own registry, argument coercion, pipeline executor and `BasePage`
+//! helpers, over a page whose transport is chrome-use. Nothing is converted or
+//! copied into our packs.
+//!
+//! Precedence: a chrome-use adapter (official, configured, then community) with
+//! the same `name/cmd` always wins; OpenCLI only answers names we don't have,
+//! and it is listed last in the domain hint.
+
+use std::path::{Path, PathBuf};
+use std::process::{Command, Stdio};
+
+use serde_json::{json, Value};
+
+/// The OpenCLI release we install. Pinned rather than `latest` because its
+/// adapters run as local Node code with the user's privileges; a version bump
+/// is a deliberate change here. `AGENT_BROWSER_OPENCLI_VERSION` overrides.
+pub const PINNED_VERSION: &str = "1.8.8";
+pub const PACKAGE: &str = "@jackwener/opencli";
+pub const SOURCE_LABEL: &str = "opencli";
+
+const RUNNER_JS: &str = include_str!("opencli_runner.mjs");
+
+pub fn disabled() -> bool {
+ std::env::var_os("AGENT_BROWSER_SITES_NO_OPENCLI").is_some()
+}
+
+fn version() -> String {
+ std::env::var("AGENT_BROWSER_OPENCLI_VERSION")
+ .ok()
+ .filter(|v| !v.trim().is_empty())
+ .unwrap_or_else(|| PINNED_VERSION.to_string())
+}
+
+/// `~/.chrome-use/opencli` — npm prefix holding the package and the runner.
+pub fn root() -> Option {
+ std::env::var_os("HOME").map(|h| PathBuf::from(h).join(".chrome-use").join("opencli"))
+}
+
+fn pkg_dir(root: &Path) -> PathBuf {
+ root.join("node_modules").join("@jackwener").join("opencli")
+}
+
+fn installed_version(root: &Path) -> Option {
+ let text = std::fs::read_to_string(pkg_dir(root).join("package.json")).ok()?;
+ let v: Value = serde_json::from_str(&text).ok()?;
+ v.get("version").and_then(|x| x.as_str()).map(String::from)
+}
+
+fn on_path(bin: &str) -> bool {
+ Command::new(bin)
+ .arg("--version")
+ .stdout(Stdio::null())
+ .stderr(Stdio::null())
+ .status()
+ .is_ok_and(|s| s.success())
+}
+
+/// Install or update the pinned package. Best-effort: Ok(None) when skipped
+/// (disabled, or no node/npm on PATH), Ok(Some(n)) with the command count.
+pub fn sync() -> Result
+
OpenCLI commands work too
+
+ With Node.js 20+ on PATH, site update also installs OpenCLI (about 180 sites and 1,300 commands; a pinned version, installed with --ignore-scripts).
+ A name/cmd that neither adapter pack has runs through OpenCLI's own runtime, driving your current chrome-use session and its logins, e.g.
+ chrome-use site hackernews/best --limit 5 --json. They show as (opencli) in site list, site info shows their args, and they come last in the site hint.
+ On a shared name ours win. AGENT_BROWSER_SITES_NO_OPENCLI=1 turns them off.
+
+
+
Writing your own: analyze and verify
+
+ Do the action that loads the data on the page first (search, scroll, open the list), then run chrome-use site analyze. It lists the API calls the page made, the state it embeds (__NEXT_DATA__, __INITIAL_STATE__, …) and any anti-bot vendor, and recommends a data source.
+ In order of preference: a public API, the site's own JSON API called from the page with its cookies, embedded page state, then the DOM. Each step down breaks more often.
+ Once it works, chrome-use site verify <name>/<cmd> [args] --write-fixture records the shape of a good result; later, site verify fails (exit 1) when a field disappears, changes type, or a list comes back empty. The fixture keeps the shape only, never values.
+
+
ℹ️ Exception on a fresh environment
Only on a brand-new environment where the adapter sources haven't been pulled yet might naming a
diff --git a/docs/site-adapters.html b/docs/site-adapters.html
index 09eab5d8ec..e7fd1cc7c3 100644
--- a/docs/site-adapters.html
+++ b/docs/site-adapters.html
@@ -258,6 +258,21 @@
+ 先在页面上把加载数据的操作做一遍(搜索、滚动、打开列表),再跑 chrome-use site analyze:它列出页面发过的接口请求、页面里嵌着的初始数据(__NEXT_DATA__、__INITIAL_STATE__ 等)和检测到的反爬厂商,并推荐取数方式。
+ 优先顺序:公开 API,网站自己的 JSON 接口(在页面里带 cookie 调用),页面嵌入的数据,最后才是 DOM;越往后越容易坏。
+ 写好后用 chrome-use site verify <name>/<cmd> [参数] --write-fixture 记下一次正确结果的结构;之后 site verify 发现字段消失、类型变了或列表变空就失败(退出码 1)。fixture 只存结构,不存数据。
+
+
ℹ️ 全新环境的例外
只有在刚装好、适配器源还没拉取的全新环境里,点名一个 site <name>/<cmd>
diff --git a/skill-data/core/references/site-adapters.md b/skill-data/core/references/site-adapters.md
index c3f7e0ebc7..a8d7cbd37b 100644
--- a/skill-data/core/references/site-adapters.md
+++ b/skill-data/core/references/site-adapters.md
@@ -20,8 +20,15 @@ chrome-use site github/issues owner/repo --json # run it → JSON (navigates t
- It navigates to the adapter's domain (reusing the current tab if you're already on it), so
login-gated feeds (`bilibili/feed`, `twitter/...`) work because they run as *you*.
- If no adapter fits, fall back to the normal `snapshot`/`eval` loop. chrome-use fetches and
- runs two default sources: the [bb-sites](https://github.com/epiral/bb-sites) community pack
- and the official [chrome-use-sites](https://github.com/leeguooooo/chrome-use-sites) pack.
+ runs two default sources: the official [chrome-use-sites](https://github.com/leeguooooo/chrome-use-sites)
+ pack and the [bb-sites](https://github.com/epiral/bb-sites) community pack. On a shared
+ `name/cmd` the official one wins.
+- **OpenCLI commands work too.** When Node.js 20+ is on PATH, `site update` also installs
+ [OpenCLI](https://github.com/jackwener/OpenCLI) (~1,300 commands over ~180 sites). A
+ `name/cmd` that neither pack has runs through OpenCLI's own runtime, driving this same
+ session. They show as `(opencli)` in `site list`, `site info` shows their args, and they come
+ last in the `siteAdapters` hint. Same command, same JSON: `chrome-use site hackernews/best
+ --limit 5 --json`. `AGENT_BROWSER_SITES_NO_OPENCLI=1` turns them off.
> **Auto-trigger — act on it.** chrome-use keeps both packs synced automatically (first use +
> weekly), and whenever you reach a page whose domain has adapters it tells you: on every
@@ -43,13 +50,21 @@ If you work on the same site a lot and no adapter covers it, chrome-use adds
**Ask the user** whether to turn the steps you keep repeating there into an adapter. Don't write
one without a yes. If they agree:
-1. Find the data source with `network requests` / `eval` (the site's own JSON API beats DOM
- scraping), then write `~/.chrome-use/my-sites//.js` in the format above
- (`@meta` with `name`, `description`, `domain`, `args`, `readOnly`; then the `async function`).
+1. First check `chrome-use site list | grep `: OpenCLI may already cover it. Otherwise
+ do the action that loads the data (search, scroll, open the list), then run
+ `chrome-use site analyze`. It lists the API calls the page made, the state it embeds
+ (`__NEXT_DATA__`, `__INITIAL_STATE__`, …) and any anti-bot vendor, and picks a strategy.
+ Prefer, in this order: a public API; the site's own JSON API called from the page
+ (`fetch(url, {credentials: 'include'})`); embedded page state; the DOM. Each step down breaks
+ more often. Write `~/.chrome-use/my-sites//.js` in the format above (`@meta` with
+ `name`, `description`, `domain`, `args`, `readOnly`; then the `async function`).
2. Register the folder once and sync: `chrome-use site add ~/.chrome-use/my-sites`, then
`chrome-use site update`. Keep your own adapters in that folder, not in `~/.chrome-use/sites`,
because a sync rewrites `~/.chrome-use/sites`.
-3. Run it: `chrome-use site / --json`. If it's generally useful, offer to send it to
+3. Run it: `chrome-use site / --json`. When the result looks right, record it:
+ `chrome-use site verify / [args] --write-fixture`. Later, `site verify /`
+ fails (exit 1) when a field disappears, changes type, or a list comes back empty, so a broken
+ adapter shows up before you trust its output. The fixture keeps the shape only, never values. If it's generally useful, offer to send it to
[chrome-use-sites](https://github.com/leeguooooo/chrome-use-sites). That opens a PR from the
user's account, so ask before you do it.
From ce18c2abfcfcdb262236b4328414d8e8339682c9 Mon Sep 17 00:00:00 2001
From: leeguooooo
Date: Tue, 6 Oct 2026 14:17:46 +0900
Subject: [PATCH 5/5] fix(site): analyze skips extension globals and ad
requests; OpenCLI ranks after ours on every domain spelling
---
cli/src/site.rs | 72 ++++++++++++++++++++++++++++++++-----------------
1 file changed, 47 insertions(+), 25 deletions(-)
diff --git a/cli/src/site.rs b/cli/src/site.rs
index a970f5f1ae..12f057f3aa 100644
--- a/cli/src/site.rs
+++ b/cli/src/site.rs
@@ -1154,7 +1154,7 @@ fn write_domain_index(dir: &std::path::Path) {
.or_default()
.push((read_only, spec));
}
- let mut ordered: std::collections::BTreeMap> = by_domain
+ let ordered: std::collections::BTreeMap> = by_domain
.into_iter()
.map(|(domain, mut v)| {
// ours (official / configured) before community, then read-only
@@ -1169,12 +1169,18 @@ fn write_domain_index(dir: &std::path::Path) {
(domain, v.into_iter().map(|(_, s)| s).collect())
})
.collect();
- for (domain, mut v) in opencli_by_domain {
- v.sort_by(|a, b| b.0.cmp(&a.0).then_with(|| a.1.cmp(&b.1)));
- ordered
- .entry(domain)
- .or_default()
- .extend(v.into_iter().map(|(_, s)| s));
+ // OpenCLI goes in its own index so a lookup can always rank it after
+ // ours, even when the two packs spell the domain differently
+ // (`v2ex.com` vs `www.v2ex.com`).
+ let theirs: std::collections::BTreeMap> = opencli_by_domain
+ .into_iter()
+ .map(|(domain, mut v)| {
+ v.sort_by(|a, b| b.0.cmp(&a.0).then_with(|| a.1.cmp(&b.1)));
+ (domain, v.into_iter().map(|(_, s)| s).collect())
+ })
+ .collect();
+ if let Ok(json) = serde_json::to_string(&theirs) {
+ let _ = std::fs::write(dir.join(".index-opencli.json"), json);
}
if let Ok(json) = serde_json::to_string(&ordered) {
let _ = std::fs::write(dir.join(".index.json"), json);
@@ -1236,22 +1242,30 @@ pub fn needs_refresh() -> bool {
/// Reads the prebuilt `.index.json`; empty if the packs aren't synced yet.
pub fn adapters_for_domain(host: &str) -> Vec {
let host = host.trim_start_matches("www.");
- let Some(raw) = index_path().and_then(|p| std::fs::read_to_string(p).ok()) else {
- return Vec::new();
- };
- let Ok(idx) = serde_json::from_str::>>(&raw)
- else {
- return Vec::new();
- };
- // Preserve the index's per-domain ordering (read-only adapters first); just
- // dedup if a host somehow matches multiple domain keys.
+ // Ours first, then OpenCLI's (its own index, see write_domain_index).
let mut out: Vec = Vec::new();
- for (domain, specs) in idx {
- let d = domain.trim_start_matches("www.");
- if host == d || host.ends_with(&format!(".{d}")) {
- for s in specs {
- if !out.contains(&s) {
- out.push(s);
+ let files = [
+ index_path(),
+ sites_dir().map(|d| d.join(".index-opencli.json")),
+ ];
+ for path in files.into_iter().flatten() {
+ if path.ends_with(".index-opencli.json") && crate::opencli::disabled() {
+ continue;
+ }
+ let Some(idx) = std::fs::read_to_string(&path).ok().and_then(|raw| {
+ serde_json::from_str::>>(&raw).ok()
+ }) else {
+ continue;
+ };
+ // Preserve each index's per-domain ordering (read-only first); dedup
+ // when a host matches several domain keys.
+ for (domain, specs) in idx {
+ let d = domain.trim_start_matches("www.");
+ if host == d || host.ends_with(&format!(".{d}")) {
+ for s in specs {
+ if !out.contains(&s) {
+ out.push(s);
+ }
}
}
}
@@ -1389,7 +1403,7 @@ pub const ANALYZE_JS: &str = r#"(() => {
const out = { url: location.href, host: location.hostname, title: document.title };
const host = location.hostname.replace(/^www\./, '');
const base = host.split('.').slice(-2).join('.');
- const noise = /google-analytics|googletagmanager|doubleclick|facebook\.net|hotjar|sentry|segment\.(io|com)|mixpanel|clarity\.ms|bat\.bing|newrelic|datadoghq|amplitude|\/collect\b|\/log(ging)?\b|\/track(ing)?\b|\/beacon\b|\/metrics?\b|\/report\b|\/telemetry\b|\/pixel\b/i;
+ const noise = /google-analytics|googletagmanager|doubleclick|googlesyndication|adtrafficquality|adservice|pagead|facebook\.net|hotjar|sentry|segment\.(io|com)|mixpanel|clarity\.ms|bat\.bing|newrelic|datadoghq|amplitude|\/collect\b|\/log(ging)?\b|\/track(ing)?\b|\/beacon\b|\/metrics?\b|\/report\b|\/telemetry\b|\/pixel\b/i;
const seen = new Set();
const api = [];
for (const e of performance.getEntriesByType('resource')) {
@@ -1411,7 +1425,7 @@ pub const ANALYZE_JS: &str = r#"(() => {
api.push({ url: e.name.length > 300 ? e.name.slice(0, 300) + '…' : e.name, type: e.initiatorType, sameSite, bytes: e.transferSize || 0, score, reasons });
}
api.sort((a, b) => b.score - a.score);
- out.api = api.filter(a => a.score > 0).slice(0, 12);
+ out.api = api.filter(a => a.score >= 2).slice(0, 12);
out.requestsSeen = api.length;
const known = ['__NEXT_DATA__', '__NUXT__', '__NUXT_DATA__', '__INITIAL_STATE__', '__INITIAL_DATA__', '__INITIAL_PROPS__', '__PRELOADED_STATE__', '__APOLLO_STATE__', '__REDUX_STATE__', '__SSR_DATA__', '__remixContext', '__UNIVERSAL_DATA_FOR_REHYDRATION__', 'ytInitialData', 'ytInitialPlayerResponse', '__pinia', '__INITIAL_SSR_STATE__', 'g_initialProps', '__STATE__'];
@@ -1435,7 +1449,15 @@ pub const ANALYZE_JS: &str = r#"(() => {
const sel = el.id ? 'script#' + el.id : 'script[type="' + el.type + '"]';
out.state.push({ name: sel, ...describe(v) });
}
- out.state = out.state.filter(s => s.size === -1 || s.size > 200);
+ // Extension-injected globals (Vue/React devtools) and telemetry config are
+ // not the page's data.
+ const junk = /devtools|rum|analytics|tracking|gtm|sentry/i;
+ const seenState = new Set();
+ out.state = out.state.filter(s => {
+ if (junk.test(s.name) || seenState.has(s.name)) return false;
+ seenState.add(s.name);
+ return s.size === -1 || s.size > 200;
+ });
out.webpack = Object.getOwnPropertyNames(window).filter(k => /^webpackChunk|^webpackJsonp/.test(k)).slice(0, 3);
out.signals = {