From b25958946c2e4a618d42362758d6e861e28cb735 Mon Sep 17 00:00:00 2001 From: leeguooooo Date: Mon, 15 Jun 2026 13:42:48 +0900 Subject: [PATCH] feat(text): 'get text' (no selector) defaults to cross-frame whole-page read MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit So an agent never silently misses iframed content (listing descriptions etc.) without having to know the --all-frames flag. Single-frame pages are unchanged (identical to the old body read); multi-frame pages now include child frames — a strict superset. Skill + help updated to make the default and 'frames'/--main discoverable. --- cli/src/commands.rs | 36 +++++++++++++++++++++--------------- cli/src/output.rs | 9 ++++----- skill-data/core/SKILL.md | 25 ++++++++++++++++--------- 3 files changed, 41 insertions(+), 29 deletions(-) diff --git a/cli/src/commands.rs b/cli/src/commands.rs index 80a67a2..bfa68e7 100644 --- a/cli/src/commands.rs +++ b/cli/src/commands.rs @@ -2463,16 +2463,16 @@ fn parse_get(rest: &[&str], id: &str) -> Result { if main { return Ok(json!({ "id": id, "action": "gettext", "main": true })); } - // `get text` with no selector returns the whole page's text (body) — - // a common convenience; previously it errored without a selector - // (issue #24-D). - let sel = rest - .iter() - .skip(1) - .find(|a| !a.starts_with("--")) - .copied() - .unwrap_or("body"); - Ok(json!({ "id": id, "action": "gettext", "selector": sel })) + // `get text` with no selector reads the WHOLE PAGE and now defaults + // to cross-frame aggregation, so an agent gets a page's iframed + // content (listing descriptions etc.) without having to know about + // `--all-frames` (#27). On a single-frame page this is identical to + // the old body read; multi-frame pages get the child frames too — + // a strict superset. An explicit selector stays element-scoped. + match rest.iter().skip(1).find(|a| !a.starts_with("--")).copied() { + Some(sel) => Ok(json!({ "id": id, "action": "gettext", "selector": sel })), + None => Ok(json!({ "id": id, "action": "gettext", "allFrames": true })), + } } Some("html") => { let sel = rest.get(1).ok_or_else(|| ParseError::MissingArguments { @@ -4830,15 +4830,21 @@ mod tests { } #[test] - fn test_get_text_defaults_to_body() { - // `get text` with no selector now returns the whole page (body) instead - // of erroring (issue #24-D). + fn test_get_text_defaults_to_all_frames() { + // `get text` with no selector now reads the whole page across ALL frames + // by default (#27), so iframed content isn't silently missed. (Was: a + // top-frame `body` read, #24-D.) let cmd = parse_command(&args("get text"), &default_flags()).unwrap(); assert_eq!(cmd["action"], "gettext"); - assert_eq!(cmd["selector"], "body"); - // An explicit selector still wins. + assert_eq!(cmd["allFrames"], true); + assert!(cmd.get("selector").is_none()); + // An explicit selector still wins and stays element-scoped. let cmd2 = parse_command(&args("get text h1"), &default_flags()).unwrap(); assert_eq!(cmd2["selector"], "h1"); + assert!(cmd2.get("allFrames").is_none()); + // `text` top-level shortcut behaves the same. + let cmd3 = parse_command(&args("text"), &default_flags()).unwrap(); + assert_eq!(cmd3["allFrames"], true); } #[test] diff --git a/cli/src/output.rs b/cli/src/output.rs index 54d2f79..7d1be9b 100644 --- a/cli/src/output.rs +++ b/cli/src/output.rs @@ -1959,8 +1959,7 @@ Usage: chrome-use get [args] Retrieves various types of information from elements or the page. Subcommands: - text Get text content of element - text --all-frames Aggregate text across ALL frames (incl. iframes) + text [selector] Element text; no selector = WHOLE PAGE, all frames text --main Main-content text only (skip nav/header/sidebar) html Get inner HTML of element value Get value of input element @@ -1977,8 +1976,8 @@ Global Options: --session Use specific session Examples: - chrome-use get text @e1 - chrome-use get text --all-frames # read iframed content (listing pages) + chrome-use get text # whole page across ALL frames (default) + chrome-use get text @e1 # one element chrome-use get text --main # main content, no nav/sidebar boilerplate chrome-use frames # list frames + where the text lives chrome-use get html "#content" @@ -3190,7 +3189,7 @@ Navigation: Get Info: chrome-use get [selector] text, html, value, attr , title, url, count, box, styles, cdp-url - text --all-frames (cross-frame), text --main (no boilerplate), frames (list) + text (no selector = whole page, all frames), text --main, frames (list) Check State: chrome-use is visible, enabled, checked diff --git a/skill-data/core/SKILL.md b/skill-data/core/SKILL.md index a3ad8e8..bb8fdcf 100644 --- a/skill-data/core/SKILL.md +++ b/skill-data/core/SKILL.md @@ -207,8 +207,8 @@ assigned fresh on every snapshot. For unstructured reading (no refs needed): ```bash -chrome-use get text @e1 # visible text of an element -chrome-use get text --all-frames # whole page, aggregated across ALL frames +chrome-use get text # WHOLE PAGE — all frames by default (see below) +chrome-use get text @e1 # visible text of one element (or a CSS selector) chrome-use get text --main # main content only — skip nav/header/sidebar chrome-use frames # list every frame + where the text lives chrome-use get html @e1 # innerHTML @@ -219,13 +219,20 @@ chrome-use get url # current URL chrome-use get count ".item" # count matching elements ``` -On listing/marketplace pages (Yahoo Auctions, Rakuten, Mercari shops) the seller's -description often lives in a **child frame** or is buried under a "related items" -sidebar, so a plain `get text body` returns only header/nav boilerplate. When the -text you expect is missing: run `chrome-use frames` to see where it is, then -`get text --all-frames` (reads every reachable frame incl. cross-origin iframes) -or `get text --main` (drops the global chrome). If the content is lazy-loaded, -`scroll` it into view first. +**Whole-page text is cross-frame by default.** `chrome-use get text` with no +selector aggregates visible text across **every** frame — top document plus +same-process child frames plus cross-origin iframes — so you never silently miss +content that lives in an iframe (Yahoo Auctions / Rakuten / Mercari shop +descriptions, embedded checkout/spec frames). Each child frame is delimited with +a `----- frame [kind] url -----` marker. You do **not** need to remember a flag — +the default already reads all frames. (`--all-frames` is still accepted as an +explicit alias.) + +So: when text looks missing or wrong, you don't have to guess — just +`chrome-use get text` reads everything. To **see** the structure (which frame +holds what), run `chrome-use frames`. To **cut boilerplate** (global nav/header/ +footer, "related items" sidebars), use `chrome-use get text --main`. If content +is lazy-loaded, `scroll` it into view first, then read. ## Interacting