diff --git a/crates/tinytools-std/src/network/fixtures/web_fetch.json b/crates/tinytools-std/src/network/fixtures/web_fetch.json index 18a24c8..a715af6 100644 --- a/crates/tinytools-std/src/network/fixtures/web_fetch.json +++ b/crates/tinytools-std/src/network/fixtures/web_fetch.json @@ -6,7 +6,7 @@ "schema": { "properties": { "max_bytes": { - "description": "Truncate body at this many bytes (default 1_000_000).", + "description": "Cap the returned text at this many bytes (default 1_000_000). Bounds the OUTPUT — the extracted markdown, or the raw body with raw:true — never the markup the extractor reads, so lowering it cannot cost you content the page actually had.", "minimum": 1, "type": "integer" }, diff --git a/crates/tinytools-std/src/network/web_fetch.rs b/crates/tinytools-std/src/network/web_fetch.rs index 1775e0b..9fba8a5 100644 --- a/crates/tinytools-std/src/network/web_fetch.rs +++ b/crates/tinytools-std/src/network/web_fetch.rs @@ -127,7 +127,11 @@ impl Tool for WebFetchTool { "url": { "type": "string", "description": "Absolute http(s) URL." }, "max_bytes": { "type": "integer", - "description": "Truncate body at this many bytes (default 1_000_000).", + "description": "Cap the returned text at this many bytes \ + (default 1_000_000). Bounds the OUTPUT — the extracted \ + markdown, or the raw body with raw:true — never the markup \ + the extractor reads, so lowering it cannot cost you content \ + the page actually had.", "minimum": 1 }, "raw": { @@ -305,12 +309,6 @@ impl WebFetchTool { } let downloaded = body.len(); - let (body, byte_capped) = if downloaded > max_bytes { - let cut = floor_char_boundary(&body, max_bytes); - (body[..cut].to_string(), true) - } else { - (body, false) - }; // Markdown by default. A page's prose is a small fraction of its // bytes; handing the raw document to the model (and to the payload @@ -319,24 +317,26 @@ impl WebFetchTool { // content transforms. let converted = !raw_requested && is_html(self.html.as_ref(), &body, content_type.as_deref()); - let content = if converted { - self.html.to_markdown(&body) - } else { - body - }; + let rendered = render_body(self.html.as_ref(), body, converted, max_bytes); - let extracted = content.len(); + let extracted = rendered.extracted; let mut header = format!("status={} url={final_url}", status.as_u16()); if converted { header.push_str(" content=markdown"); } - if byte_capped { - header.push_str(&format!(" download_capped_at={max_bytes}B")); + // `output_capped_at`, not the old `download_capped_at`: nothing here + // ever capped a download — `resp.text()` above materialises the whole + // body regardless — and naming it that sent a reader looking in the + // wrong place for the content that went missing. + if rendered.output_capped { + header.push_str(&format!(" output_capped_at={max_bytes}B")); } + append_markup_truncation_header(&mut header, &rendered); if converted && extracted < downloaded { header.push_str(&format!(" extracted={extracted}B_of_{downloaded}B")); } header.push('\n'); + let content = rendered.content; // Full extracted content. Bounding it — the head/tail window, the // spill to an artifact and the paging handle — belongs to @@ -412,6 +412,87 @@ fn error_body_excerpt( } } +/// Raw markup handed to the HTML extractor, at most. +/// +/// Not a byte budget — the caller's `max_bytes` owns that, applied to the +/// output. This exists only so a pathological document cannot cost unbounded +/// CPU in `to_markdown`, a cost the old pre-truncation ordering hid by +/// accident. 8 MiB is ~19x the largest real page measured here (a 434 KB +/// client-rendered spreadsheet) and ~8x the default `max_response_size`, so no +/// realistic page reaches it. +const EXTRACTOR_INPUT_CEILING: usize = 8 * 1024 * 1024; + +/// A body turned into what the caller reads. +struct RenderedBody { + /// The text to return, bounded by the caller's `max_bytes`. + content: String, + /// Length *before* that bound, so the header's ratio describes the + /// extraction rather than the truncation. + extracted: usize, + /// Whether `content` was cut to fit `max_bytes`. + output_capped: bool, + /// Whether the markup was cut before the extractor saw it, which only + /// happens past [`EXTRACTOR_INPUT_CEILING`]. + markup_truncated: bool, +} + +/// Add the extractor input ceiling to a fetch header when markup was cut. +fn append_markup_truncation_header(header: &mut String, rendered: &RenderedBody) { + if rendered.markup_truncated { + header.push_str(&format!(" markup_truncated_at={EXTRACTOR_INPUT_CEILING}B")); + } +} + +/// Convert, **then** bound. +/// +/// The order is the whole of this function. It used to be the other way round, +/// and the cap was therefore destroying the thing it was meant to measure: a +/// 432,864-byte client-rendered page fetched with `max_bytes: 50000` was cut at +/// byte 50,000 — mid-tag, mid-DOM — and the readability pass, handed that +/// wreckage, recovered only the `
the prose that matters
", + "f".repeat(filler) + ) +} + +#[test] +fn the_cap_applies_to_the_extracted_text_not_the_markup() { + // The regression this function exists for. A 432,864-byte page fetched at + // `max_bytes: 50000` used to come back as `extracted=48B_of_432864B` — the + // whole document reduced to its `visible
", + "x".repeat(EXTRACTOR_INPUT_CEILING) + ); + assert!(body.len() > EXTRACTOR_INPUT_CEILING); + let rendered = render_body(&TestHtml, body, true, 1_000); + assert!(rendered.markup_truncated); + let mut output = "status=200 url=https://example.com content=markdown".to_string(); + append_markup_truncation_header(&mut output, &rendered); + assert!( + output.contains(&format!("markup_truncated_at={EXTRACTOR_INPUT_CEILING}B")), + "header should disclose the extractor input ceiling: {output}" + ); +} + +#[test] +fn raw_output_is_bounded_by_the_same_cap() { + // With `raw: true` there is no extraction, so the cap applies to the body + // itself — the one case where cutting the input and cutting the output are + // the same act. + let rendered = render_body(&TestHtml, "abcdefghij".to_string(), false, 4); + assert_eq!(rendered.content, "abcd"); + assert!(rendered.output_capped); + assert_eq!(rendered.extracted, 10); +} + +#[test] +fn the_markup_ceiling_is_far_above_any_real_page() { + // Sized against the measured worst case, not picked round. A ceiling near + // the old default would reintroduce the bug for ordinary documents. + const { + assert!(EXTRACTOR_INPUT_CEILING >= 8 * 1024 * 1024); + assert!(EXTRACTOR_INPUT_CEILING > 433_638 * 10); + } +} + +#[test] +fn the_schema_says_which_side_of_the_extractor_it_bounds() { + // The description is a caller's only account of what the knob does, and + // the old wording ("Truncate body at this many bytes") is what made + // lowering it look free. + let tool = fetch(test_security(), vec![], None, None); + let schema = tool.parameters_schema(); + let desc = schema["properties"]["max_bytes"]["description"] + .as_str() + .expect("max_bytes documents itself"); + assert!(desc.contains("OUTPUT"), "{desc}"); + assert!(desc.contains("never the markup"), "{desc}"); +}