{
  "name": "forge-web",
  "language": "rust",
  "version": "0.2.0",
  "description": "Native web substrate for the Forge SDK \u2014 fetch, parse, extract, crawl, and compact web content",
  "manifest": "forge-rs/crates/forge-web/Cargo.toml",
  "manifestSha256": "4c2951edbca90cb8d1bfbb13f7bb6f2b2035bfdbb17de2cb49d154887e9f2272",
  "status": "source-reference",
  "registryPublicationVerified": false,
  "route": "/libraries/rust/forge-web",
  "features": {},
  "files": [
    {
      "path": "forge-rs/crates/forge-web/src/compact.rs",
      "sha256": "732cc4703c0bdaff99937e0286f177d874f42dc70adb07294405855697fe6928",
      "artifactSha256": "d45a63304309f553634b3d683ae4b39d7aaf8e5425b728ff8a456e501b434246",
      "url": "/reference/source/forge-rs/crates/forge-web/src/compact.rs.txt",
      "declarations": [
        {
          "name": "::web_compact_site",
          "line": 29,
          "signature": "pub async fn web_compact_site(\n    config: &WebSubstrateConfig,\n    root_url: &str,\n    max_depth: u32,\n    max_pages: u32,\n) -> WebResult<CompactedSite>;",
          "documentation": "Compacts a site's content into a [`CompactedSite`] for LLM context.\n\nCrawls the site, converts each page to Markdown, strips boilerplate,\nand estimates the total token count.\n\n# Arguments\n\n* `config` - The web substrate configuration.\n* `root_url` - The URL to start from.\n* `max_depth` - Maximum link-following depth.\n* `max_pages` - Maximum number of pages to include.\n\n# Returns\n\nA [`CompactedSite`] with page content and token estimates.\n\n# Errors\n"
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/config.rs",
      "sha256": "3fd9f95e5a6d2752b8fbd8bd12c2b52bcd34b9c44f8214c7347d30924e3c6b64",
      "artifactSha256": "6f9f8fa61738d01df601fd0fe5ba52079f222b13304ad5e6e8142e9cc73b2212",
      "url": "/reference/source/forge-rs/crates/forge-web/src/config.rs.txt",
      "declarations": [
        {
          "name": "::WebSubstrateConfig",
          "line": 35,
          "signature": "#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct WebSubstrateConfig {\n/// Maximum response body size in bytes.\n\n///\n\n/// Responses exceeding this limit will return\n\n/// [`WebError::ContentTooLarge`](crate::error::WebError::ContentTooLarge).\n\n///\n\n/// Default: 10,485,760 (10 MB).\n\npub max_fetch_size_bytes: u64,\n/// Maximum number of HTTP redirects to follow per request.\n\n///\n\n/// Exceeding this limit returns\n\n/// [`WebError::RedirectLimitExceeded`](crate::error::WebError::RedirectLimitExceeded).\n\n///\n\n/// Default: 5.\n\npub max_redirects: u32,\n/// Request timeout in milliseconds.\n\n///\n\n/// If the server does not respond within this duration, the request\n\n/// returns [`WebError::Timeout`](crate::error::WebError::Timeout).\n\n///\n\n/// Default: 30,000 (30 seconds).\n\npub request_timeout_ms: u64,\n/// IP address ranges to block for SSRF protection.\n\n///\n\n/// Each entry is a CIDR notation string (e.g., `10.0.0.0/8`). Requests\n\n/// whose resolved IP falls within any of these ranges will return\n\n/// [`WebError::SsrfBlocked`](crate::error::WebError::SsrfBlocked).\n\n///\n\n/// Default: private (RFC 1918), loopback, link-local, and IPv6 private ranges.\n\npub blocked_ip_ranges: Vec<String>,\n/// The `User-Agent` header sent with all outgoing requests.\n\n///\n\n/// Default: `\"forge-web/0.1 (+https://github.com/l1fe-labs/forge)\"`.\n\npub user_agent: String,\n/// Whether to respect `robots.txt` directives when crawling.\n\n///\n\n/// When `true`, the substrate will fetch and honor `robots.txt` rules\n\n/// before crawling a site. Disabling this is only appropriate for\n\n/// authorized internal crawling.\n\n///\n\n/// Default: `true`.\n\npub respect_robots_txt: bool\n}",
          "documentation": "Configuration for the web substrate.\n\nControls SSRF protection, redirect limits, content size caps, timeouts,\nand other behavioral defaults. All web operations (fetch, crawl, inspect,\ncompact) use this configuration.\n\n# Defaults\n\nThe default configuration provides a secure baseline:\n- 10 MB max fetch size\n- 5 max redirects\n- 30-second request timeout\n- SSRF protection enabled (blocks private, loopback, link-local ranges)\n- `robots.txt` respected\n\n# Examples\n\n```\nuse forge_web::config::WebSubstrateConfig;\n\nlet config = WebSubstrateConfig::default();\nassert_eq!(config.max_fetch_size_bytes, 10_485_760);\nassert_eq!(config.max_redirects, 5);\nassert_eq!(config.request_timeout_ms, 30_000);\nassert!(config.respect_robots_txt);\n```"
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/crawl.rs",
      "sha256": "833caa5c63776aff4342376cdde0a0a35c0d04c88ff9a3cd5c8ead35161a46df",
      "artifactSha256": "1fbe9f1c47e8e3c51ed6abd9e93012379a9d7c2ff4df56aebe31634194ab1d5e",
      "url": "/reference/source/forge-rs/crates/forge-web/src/crawl.rs.txt",
      "declarations": [
        {
          "name": "::web_crawl",
          "line": 28,
          "signature": "pub async fn web_crawl(\n    config: &WebSubstrateConfig,\n    root_url: &str,\n    max_depth: u32,\n    max_pages: u32,\n) -> WebResult<Vec<WebFetchResponse>>;",
          "documentation": "Crawls a site starting from the given root URL.\n\nPerforms a breadth-first crawl, fetching pages and extracting links up\nto the specified depth and page count limits.\n\n# Arguments\n\n* `config` - The web substrate configuration.\n* `root_url` - The URL to start crawling from.\n* `max_depth` - Maximum link-following depth (0 = root only).\n* `max_pages` - Maximum number of pages to fetch.\n\n# Returns\n\nA vector of [`WebFetchResponse`] values for all successfully fetched pages.\n\n# Errors\n"
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/error.rs",
      "sha256": "f1cf0da76c9cc6f04ff91e055e5f1dd883b84e94c77a565c6e244857e0cc911d",
      "artifactSha256": "9d40ce42a4941c706e9927cba75061a7432225ffd1be74d12645abadb67bce1e",
      "url": "/reference/source/forge-rs/crates/forge-web/src/error.rs.txt",
      "declarations": [
        {
          "name": "::WebError",
          "line": 26,
          "signature": "#[derive(Debug, Error)]\npub enum WebError {\n    /// An HTTP fetch operation failed.\n    ///\n    /// This covers network errors, DNS resolution failures, TLS handshake\n    /// failures, and other transport-level problems.\n    #[error(\"fetch failed for '{url}': {reason}\")]\n    FetchFailed {\n        /// The URL that was being fetched.\n        url: String,\n        /// What went wrong during the fetch.\n        reason: String,\n    },\n\n    /// HTML or document parsing failed.\n    ///\n    /// Returned when the web content cannot be parsed into the expected\n    /// structured format or does not satisfy the operation's parser contract.\n    #[error(\"parse failed for '{url}': {reason}\")]\n    ParseFailed {\n        /// The URL whose content could not be parsed.\n        url: String,\n        /// What went wrong during parsing.\n        reason: String,\n    },\n\n    /// Content extraction failed.\n    ///\n    /// Returned when a CSS selector, XPath, or JSONPath query cannot be\n    /// executed against the parsed content.\n    #[error(\"extraction failed for query '{query}' on '{url}': {reason}\")]\n    ExtractionFailed {\n        /// The URL whose content was being queried.\n        url: String,\n        /// The extraction query that failed.\n        query: String,\n        /// What went wrong during extraction.\n        reason: String,\n    },\n\n    /// A web search operation failed.\n    ///\n    /// Returned when a search query cannot be validated, a search endpoint\n    /// cannot be queried, or the response cannot be parsed into results.\n    #[error(\"search failed for query '{query}': {reason}\")]\n    SearchFailed {\n        /// The search query.\n        query: String,\n        /// What went wrong during the search.\n        reason: String,\n    },\n\n    /// A fetch was blocked because the resolved IP address falls within a\n    /// private, loopback, or link-local range (SSRF protection).\n    ///\n    /// This is a security control. The blocked IP ranges are configured via\n    /// [`WebSubstrateConfig::blocked_ip_ranges`](crate::config::WebSubstrateConfig).\n    #[error(\n        \"SSRF blocked: '{url}' resolved to blocked IP {ip} (private/loopback/link-local range)\"\n    )]\n    SsrfBlocked {\n        /// The URL that was being fetched.\n        url: String,\n        /// The IP address that triggered the block.\n        ip: String,\n    },\n\n    /// The maximum number of HTTP redirects was exceeded.\n    ///\n    /// Configure the limit via\n    /// [`WebSubstrateConfig::max_redirects`](crate::config::WebSubstrateConfig).\n    #[error(\"redirect limit exceeded for '{url}': followed {count} redirects (max {max})\")]\n    RedirectLimitExceeded {\n        /// The original URL that was being fetched.\n        url: String,\n        /// The number of redirects followed before the limit was hit.\n        count: u32,\n        /// The configured maximum number of redirects.\n        max: u32,\n    },\n\n    /// The response body exceeds the configured maximum size.\n    ///\n    /// Configure the limit via\n    /// [`WebSubstrateConfig::max_fetch_size_bytes`](crate::config::WebSubstrateConfig).\n    #[error(\n        \"content too large for '{url}': response size {size} bytes exceeds limit of {max} bytes\"\n    )]\n    ContentTooLarge {\n        /// The URL whose response was too large.\n        url: String,\n        /// The actual (or estimated) response size in bytes.\n        size: u64,\n        /// The configured maximum size in bytes.\n        max: u64,\n    },\n\n    /// The provided URL is invalid or cannot be parsed.\n    ///\n    /// Check that the URL includes a scheme (`http://` or `https://`),\n    /// a valid host, and well-formed path components.\n    #[error(\"invalid URL '{url}': {reason}\")]\n    InvalidUrl {\n        /// The URL string that failed validation.\n        url: String,\n        /// What is wrong with the URL.\n        reason: String,\n    },\n\n    /// The HTTP request timed out.\n    ///\n    /// Configure the timeout via\n    /// [`WebSubstrateConfig::request_timeout_ms`](crate::config::WebSubstrateConfig).\n    #[error(\"request timed out for '{url}' after {timeout_ms}ms\")]\n    Timeout {\n        /// The URL that timed out.\n        url: String,\n        /// The configured timeout in milliseconds.\n        timeout_ms: u64,\n    },\n\n    /// A boundary contract denied the operation.\n    ///\n    /// This occurs when the web substrate detects a security policy violation\n    /// such as a cross-scheme redirect downgrade (HTTPS to HTTP).\n    #[error(\"boundary contract denied for '{url}': {reason}\")]\n    BoundaryContractDenied {\n        /// The URL involved in the denied operation.\n        url: String,\n        /// Why the boundary contract was violated.\n        reason: String,\n    },\n}",
          "documentation": "Errors that can occur during web substrate operations.\n\nEvery variant includes actionable context: what failed, why, and what the\ndeveloper should check. No generic \"something went wrong\" messages.\n\n# Examples\n\n```\nuse forge_web::error::WebError;\n\nlet err = WebError::InvalidUrl {\n    url: \"not-a-url\".to_string(),\n    reason: \"missing scheme (expected http:// or https://)\".to_string(),\n};\nassert!(err.to_string().contains(\"not-a-url\"));\n```"
        },
        {
          "name": "::WebResult",
          "line": 160,
          "signature": "pub type WebResult<T> = Result<T, WebError>;",
          "documentation": "A specialized `Result` type for `forge-web` operations."
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/extract.rs",
      "sha256": "17a74f9fec1cdc51c8029733c6665014c6398fe8b6cccf1b94967bb20a34821e",
      "artifactSha256": "f65ea30161ea005eed1cc82d501c023b10796f48c145689fbc88031c4fb03315",
      "url": "/reference/source/forge-rs/crates/forge-web/src/extract.rs.txt",
      "declarations": [
        {
          "name": "::web_extract",
          "line": 29,
          "signature": "pub fn web_extract(\n    content: &str,\n    query: &WebExtractQuery,\n    url: &str,\n) -> WebResult<WebExtractResult>;",
          "documentation": "Extracts content from a parsed web page using a query.\n\n# Arguments\n\n* `content` - The HTML or JSON content to query against.\n* `query` - The extraction query (CSS selector, XPath, or JSONPath).\n* `url` - The source URL (used in error messages).\n\n# Returns\n\nA [`WebExtractResult`] containing the matched elements.\n\n# Errors\n\nReturns [`WebError::ExtractionFailed`] for invalid selectors or unsupported\nquery engines."
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/fetch.rs",
      "sha256": "86c52a3550e900cc3c60583c100ffa97ba6f5268ae02228e6e51e8312f344e0d",
      "artifactSha256": "e56003068a787b5d000b626163dae7767053b05a959b6e6f2ae0165200bb399b",
      "url": "/reference/source/forge-rs/crates/forge-web/src/fetch.rs.txt",
      "declarations": [
        {
          "name": "::web_fetch",
          "line": 79,
          "signature": "pub async fn web_fetch(\n    config: &WebSubstrateConfig,\n    request: &WebFetchRequest,\n) -> WebResult<WebFetchResponse>;",
          "documentation": "Fetches a URL with SSRF protection, redirect following, and size limits.\n\nThis is the core HTTP primitive for the web substrate. It performs:\n\n1. **URL validation** -- ensures the URL is well-formed and uses HTTP(S).\n2. **DNS resolution** -- resolves the hostname to IP addresses.\n3. **SSRF check** -- verifies resolved IPs are not in blocked ranges.\n4. **HTTP request** -- sends the request with configured timeout.\n5. **Redirect following** -- follows redirects up to the configured limit,\n   blocking HTTPS-to-HTTP downgrades.\n6. **Size enforcement** -- caps the response body at the configured limit.\n\n# Arguments\n\n* `config` - The web substrate configuration controlling limits and security.\n* `request` - The fetch request specifying URL, method, headers, and options.\n\n# Returns\n\nA [`WebFetchResponse`] containing the status, headers, body, content type,\nand final URL.\n\n# Errors\n\n- [`WebError::InvalidUrl`] if the URL is malformed.\n- [`WebError::SsrfBlocked`] if the resolved IP is in a blocked range.\n- [`WebError::BoundaryContractDenied`] if a redirect downgrades HTTPS to HTTP.\n- [`WebError::RedirectLimitExceeded`] if too many redirects occur.\n- [`WebError::ContentTooLarge`] if the response body exceeds the limit.\n- [`WebError::Timeout`] if the request exceeds the configured timeout.\n- [`WebError::FetchFailed`] for network errors, DNS failures, TLS errors, etc.\n\n# Examples\n\n```no_run\nuse forge_web::config::WebSubstrateConfig;\nuse forge_web::types::WebFetchRequest;\nuse forge_web::fetch::web_fetch;\n\n# async fn example() -> Result<(), forge_web::error::WebError> {\nlet config = WebSubstrateConfig::default();\nlet request = WebFetchRequest::get(\"https://example.com\")?;\nlet response = web_fetch(&config, &request).await?;\nassert_eq!(response.status, 200);\n# Ok(())\n# }\n```"
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/inspect.rs",
      "sha256": "e9506ed625c1493cf5f282a260daca10f2a0b273160eab8a5ce5630be75a2231",
      "artifactSha256": "f49a484c6bf17eb496a4eb3a952693669cc414697d03d7cdb40664576d1adb0f",
      "url": "/reference/source/forge-rs/crates/forge-web/src/inspect.rs.txt",
      "declarations": [
        {
          "name": "::web_inspect_site",
          "line": 26,
          "signature": "pub async fn web_inspect_site(\n    config: &WebSubstrateConfig,\n    root_url: &str,\n    max_depth: u32,\n    max_pages: u32,\n) -> WebResult<SiteMap>;",
          "documentation": "Inspects a site and returns its structure as a [`SiteMap`].\n\nCrawls the site from the root URL, building a map of pages, their titles,\noutgoing links, and discovery depths.\n\n# Arguments\n\n* `config` - The web substrate configuration.\n* `root_url` - The URL to start inspection from.\n* `max_depth` - Maximum link-following depth.\n* `max_pages` - Maximum number of pages to inspect.\n\n# Returns\n\nA [`SiteMap`] describing the site structure.\n\n# Errors\n"
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/lib.rs",
      "sha256": "26630838bdf402c32e999baa41f3bde35d394100ee9c67ab0f77f83a8a17a983",
      "artifactSha256": "59105636854fa4271fc4554dedb73f309cf833c985a599c06fd7fb05e542d9b5",
      "url": "/reference/source/forge-rs/crates/forge-web/src/lib.rs.txt",
      "declarations": [
        {
          "name": "compact",
          "line": 82,
          "signature": "pub mod compact;",
          "documentation": "# forge-web\n\nNative web substrate for the Forge SDK.\n\nThis crate provides first-class web capabilities as core runtime primitives:\nfetch, parse, extract, crawl, inspect, compact, and convert web content. These\noperations are built in to every Forge installation -- no external services,\nMCP servers, or third-party dependencies required for basic web access.\n\n# Capabilities and Readiness\n\n| Operation | Status | Description |\n|-----------|--------|-------------|\n| **Fetch** | **READY** | HTTP GET/POST with SSRF protection, redirect limits, and size caps |\n| **Tools** | **READY** | `web_fetch`, `web_parse`, `web_extract`, `web_to_markdown`, `web_crawl`, `web_inspect_site`, `web_compact_site`, and `web_search` execute through the Forge tool registry |\n| **Parse** | **READY** | HTML parser validation and raw HTML content representation |\n| **Extract** | **READY** | CSS selector extraction over HTML and native JSONPath extraction over JSON; XPath returns a typed unsupported error |\n| **Markdown** | **READY** | HTML to Markdown conversion for headings, paragraphs, lists, preformatted blocks, and blockquotes |\n| **Crawl** | **READY** | Same-origin breadth-first site crawl with depth and page limits |\n| **Inspect** | **READY** | Same-origin site map discovery with page titles, links, and crawl depth |\n| **Compact** | **READY** | Multi-page site compaction into Markdown pages with an estimated token count |\n| **Search** | **READY** | HTML search endpoint integration parsed into ranked title, URL, and snippet results |\n\n**Important for consumers**: `Fetch`, `Parse`, CSS `Extract`, JSONPath\n`Extract`, `Markdown`, `Crawl`, `Inspect`, `Compact`, `Search`, and their\nmatching tool executors are production-ready. JSONPath supports `$`, dot\nfields, array indexes, and `[*]` array wildcards. XPath returns an explicit\nunsupported-query error until its implementation slice is wired.\n\n# Security\n\nThe fetch layer enforces SSRF protection by default. All private, loopback,\nand link-local IP ranges are blocked. Cross-scheme downgrades (HTTPS to HTTP)\non redirects are denied. Content size and redirect count are capped.\n\n# Architecture\n\n```text\nWebSubstrateConfig\n       |\n       v\n   web_fetch() \u2500\u2500> WebFetchResponse\n       |\n       v\n   web_parse() \u2500\u2500> WebContent::Html\n       |\n       v\n   web_extract() \u2500\u2500> WebExtractResult (CSS + JSONPath ready)\n   web_to_markdown() \u2500\u2500> WebContent::Markdown\n   web_crawl() \u2500\u2500> Vec<WebFetchResponse>\n   web_inspect_site() \u2500\u2500> SiteMap\n   web_compact_site() \u2500\u2500> CompactedSite\n   web_search() \u2500\u2500> WebSearchResponse\n```\n\n# Tool Registration\n\nAll web operations are available as registered tools via\n[`tools::register_web_tools`]. Each tool has a `ToolDefinition` with a\nproper JSON Schema, description, and `ToolTier::External` classification.\nThe `web_fetch`, `web_parse`, `web_extract`, `web_to_markdown`,\n`web_crawl`, `web_inspect_site`, `web_compact_site`, and `web_search`\nexecutors call the native substrate.\n\n# Examples\n\n```no_run\nuse forge_web::config::WebSubstrateConfig;\nuse forge_web::types::WebFetchRequest;\nuse forge_web::fetch::web_fetch;\n\n# async fn example() -> Result<(), forge_web::error::WebError> {\nlet config = WebSubstrateConfig::default();\nlet request = WebFetchRequest::get(\"https://example.com\")?;\nlet response = web_fetch(&config, &request).await?;\nprintln!(\"Status: {}\", response.status);\n# Ok(())\n# }\n```"
        },
        {
          "name": "config",
          "line": 83,
          "signature": "pub mod config;",
          "documentation": ""
        },
        {
          "name": "crawl",
          "line": 84,
          "signature": "pub mod crawl;",
          "documentation": ""
        },
        {
          "name": "error",
          "line": 85,
          "signature": "pub mod error;",
          "documentation": ""
        },
        {
          "name": "extract",
          "line": 86,
          "signature": "pub mod extract;",
          "documentation": ""
        },
        {
          "name": "fetch",
          "line": 88,
          "signature": "#[cfg(not(target_arch = \"wasm32\"))]\npub mod fetch;",
          "documentation": ""
        },
        {
          "name": "inspect",
          "line": 89,
          "signature": "pub mod inspect;",
          "documentation": ""
        },
        {
          "name": "markdown",
          "line": 90,
          "signature": "pub mod markdown;",
          "documentation": ""
        },
        {
          "name": "parse",
          "line": 91,
          "signature": "pub mod parse;",
          "documentation": ""
        },
        {
          "name": "search",
          "line": 92,
          "signature": "pub mod search;",
          "documentation": ""
        },
        {
          "name": "tools",
          "line": 93,
          "signature": "pub mod tools;",
          "documentation": ""
        },
        {
          "name": "types",
          "line": 94,
          "signature": "pub mod types;",
          "documentation": ""
        },
        {
          "name": "prelude",
          "line": 97,
          "signature": "pub mod prelude;",
          "documentation": "Re-exports of the most commonly used types."
        },
        {
          "name": "pub use crate::config::WebSubstrateConfig;",
          "line": 98,
          "signature": "pub use crate::config::WebSubstrateConfig;",
          "documentation": ""
        },
        {
          "name": "pub use crate::error::{WebError, WebResult};",
          "line": 99,
          "signature": "pub use crate::error::{WebError, WebResult};",
          "documentation": ""
        },
        {
          "name": "pub use crate::fetch::web_fetch;",
          "line": 101,
          "signature": "#[cfg(not(target_arch = \"wasm32\"))]\npub use crate::fetch::web_fetch;",
          "documentation": ""
        },
        {
          "name": "pub use crate::tools::register_web_tools;",
          "line": 102,
          "signature": "pub use crate::tools::register_web_tools;",
          "documentation": ""
        },
        {
          "name": "pub use crate::types::{\n        CompactedSite, SiteMap, SiteNode, WebContent, WebExtractQuery, WebExtractResult,\n        WebFetchRequest, WebFetchResponse, WebSearchResponse, WebSearchResult,\n    };",
          "line": 103,
          "signature": "pub use crate::types::{\n        CompactedSite, SiteMap, SiteNode, WebContent, WebExtractQuery, WebExtractResult,\n        WebFetchRequest, WebFetchResponse, WebSearchResponse, WebSearchResult,\n    };",
          "documentation": ""
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/markdown.rs",
      "sha256": "97794948acd3030912822500d9733b89d7ba1f581cc82e3e89bbbae004c76b6f",
      "artifactSha256": "40fcddc273f737365eab4c0d34662b90834dccce578aa5991588a8e64ae6287b",
      "url": "/reference/source/forge-rs/crates/forge-web/src/markdown.rs.txt",
      "declarations": [
        {
          "name": "::web_to_markdown",
          "line": 20,
          "signature": "pub fn web_to_markdown(html: &str, url: &str) -> WebResult<WebContent>;",
          "documentation": "Converts HTML content to Markdown.\n\n# Arguments\n\n* `html` - The HTML string to convert.\n* `url` - The source URL (used in error messages).\n\n# Returns\n\nA [`WebContent::Markdown`] containing the converted content.\n\n# Errors\n\nReturns [`WebError::ParseFailed`] if the document cannot be converted."
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/parse.rs",
      "sha256": "69e4160eea14af0b4abdce36244340de00dddea2a6c4dcdff6edf46689872af5",
      "artifactSha256": "bf970e26dc7b70a6723b63b27ab1ecb11673988d4499dcb9062db8aaef05cd0d",
      "url": "/reference/source/forge-rs/crates/forge-web/src/parse.rs.txt",
      "declarations": [
        {
          "name": "::web_parse",
          "line": 21,
          "signature": "pub fn web_parse(html: &str, url: &str) -> WebResult<WebContent>;",
          "documentation": "Parses raw HTML content into a structured [`WebContent`] representation.\n\n# Arguments\n\n* `html` - The raw HTML string to parse.\n* `url` - The source URL (used in error messages).\n\n# Returns\n\nA [`WebContent::Html`] containing the parsed content.\n\n# Errors\n\nReturns [`WebError::ParseFailed`] when the document contains no readable\nHTML nodes."
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/search.rs",
      "sha256": "faf0d6bda12c857546c3abe9ad14fafb32ad0a28c939890f9a9c41f92773f610",
      "artifactSha256": "a7918b2eaa03eb14877aa7c0fef82fdc71cfcf354fc5ecd4f15c6bb63be07989",
      "url": "/reference/source/forge-rs/crates/forge-web/src/search.rs.txt",
      "declarations": [
        {
          "name": "::DEFAULT_SEARCH_ENDPOINT",
          "line": 11,
          "signature": "pub const DEFAULT_SEARCH_ENDPOINT: &str;",
          "documentation": "Default HTML search endpoint used by [`web_search`]."
        },
        {
          "name": "::web_search",
          "line": 17,
          "signature": "pub async fn web_search(\n    config: &WebSubstrateConfig,\n    query: &str,\n    num_results: u32,\n) -> WebResult<WebSearchResponse>;",
          "documentation": "Search the web using the configured first-party HTML search adapter.\n\nThe implementation intentionally goes through [`web_fetch`] so the same\nSSRF, redirect, timeout, and size controls apply to search traffic."
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/tools.rs",
      "sha256": "8cd7ed9a392ac34ef20f813e7d3a2e0a4a9a1907d30a392d0b8947ccb478b048",
      "artifactSha256": "6e0bf96aa145af8a58acddb5afe47ecb38cca033021f3a2f5e487fac385bfa34",
      "url": "/reference/source/forge-rs/crates/forge-web/src/tools.rs.txt",
      "declarations": [
        {
          "name": "::WEB_TOOL_NAMES",
          "line": 54,
          "signature": "pub const WEB_TOOL_NAMES: &[&str];",
          "documentation": "The names of all web tools registered by [`register_web_tools`]."
        },
        {
          "name": "::web_tool_definitions",
          "line": 90,
          "signature": "pub fn web_tool_definitions() -> Vec<ToolDefinition>;",
          "documentation": "Returns the tool definitions for all web substrate tools.\n\nEach tool definition includes:\n- A unique name matching the web substrate API function\n- A human-readable description suitable for LLM tool selection\n- A JSON Schema defining the expected parameters\n- `ToolTier::External` classification (web operations access external resources)\n\n# Returns\n\nA `Vec` of 8 [`ToolDefinition`] values, one for each web tool.\n\n# Examples\n\n```\nuse forge_web::tools::web_tool_definitions;\n\nlet defs = web_tool_definitions();\nassert_eq!(defs.len(), 8);\n\n// All tools are External tier\nfor def in &defs {\n    assert_eq!(def.tier(), forge_core::tool::ToolTier::External);\n}\n```"
        },
        {
          "name": "::register_web_tools",
          "line": 136,
          "signature": "pub fn register_web_tools(registry: &mut ToolRegistry) -> Result<(), ForgeToolError>;",
          "documentation": "Registers all 8 web substrate tools into the given [`ToolRegistry`].\n\nEach tool is registered with its definition (name, description, parameter\nschema, tier) and an executor. Fetch, parse, extract, Markdown conversion,\ncrawl, inspect, and compact tools run the real Forge web substrate. Search\nreturns a descriptive tool error until a search-provider integration is\nwired.\n\n# Arguments\n\n* `registry` - The tool registry to register web tools into.\n\n# Returns\n\n`Ok(())` if all 8 tools are registered successfully.\n\n# Errors\n\nReturns [`ForgeToolError::RegistryError`] if any tool name is already\nregistered in the registry.\n\n# Examples\n\n```\nuse forge_tool::registry::ToolRegistry;\nuse forge_web::tools::register_web_tools;\n\nlet mut registry = ToolRegistry::new();\nregister_web_tools(&mut registry).unwrap();\nassert_eq!(registry.len(), 8);\nassert!(registry.contains(\"web_fetch\"));\nassert!(registry.contains(\"web_search\"));\n```"
        }
      ]
    },
    {
      "path": "forge-rs/crates/forge-web/src/types.rs",
      "sha256": "f2545a2e9b88a253e9786d9a68a9448a9abbad1b7dccc547d88dc553bf9f0215",
      "artifactSha256": "834ba89425d5af56446029d591111b2f0894024baa07f98d29fbdf78c5d037d3",
      "url": "/reference/source/forge-rs/crates/forge-web/src/types.rs.txt",
      "declarations": [
        {
          "name": "::HttpMethod",
          "line": 14,
          "signature": "#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\npub enum HttpMethod {\n    /// HTTP GET request.\n    #[serde(rename = \"GET\")]\n    Get,\n    /// HTTP POST request.\n    #[serde(rename = \"POST\")]\n    Post,\n    /// HTTP HEAD request.\n    #[serde(rename = \"HEAD\")]\n    Head,\n}",
          "documentation": "An HTTP method for web fetch requests."
        },
        {
          "name": "::WebFetchRequest",
          "line": 47,
          "signature": "#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct WebFetchRequest {\n/// The target URL to fetch.\n\npub url: String,\n/// The HTTP method to use.\n\npub method: HttpMethod,\n/// HTTP headers to include in the request.\n\n///\n\n/// Keys are header names, values are header values. Uses `BTreeMap` for\n\n/// deterministic serialization.\n\npub headers: BTreeMap<String, String>,\n/// Request timeout in milliseconds. Overrides the config default if set.\n\npub timeout_ms: Option<u64>,\n/// Whether to follow HTTP redirects.\n\npub follow_redirects: bool,\n/// Maximum number of redirects to follow. Overrides the config default if set.\n\npub max_redirects: Option<u32>,\n/// Optional request body (for POST requests).\n\n#[serde(skip_serializing_if = \"Option::is_none\")]\npub body: Option<String>\n}",
          "documentation": "A web fetch request specifying URL, method, headers, and fetch behavior.\n\n# Examples\n\n```\nuse forge_web::types::WebFetchRequest;\n\nlet request = WebFetchRequest::get(\"https://example.com\").unwrap();\nassert_eq!(request.url, \"https://example.com/\");\n```"
        },
        {
          "name": "::WebFetchRequest::get",
          "line": 98,
          "signature": "pub fn get(url: &str) -> Result<Self, WebError>;",
          "documentation": "Creates a GET request for the given URL.\n\n# Arguments\n\n* `url` - The URL to fetch. Must be a valid HTTP or HTTPS URL.\n\n# Returns\n\nA `WebFetchRequest` configured for GET with default settings.\n\n# Errors\n\nReturns [`WebError::InvalidUrl`] if the URL cannot be parsed or uses\na non-HTTP scheme.\n\n# Examples\n\n```\nuse forge_web::types::WebFetchRequest;\n\nlet req = WebFetchRequest::get(\"https://example.com\").unwrap();\nassert_eq!(req.method, forge_web::types::HttpMethod::Get);\n```"
        },
        {
          "name": "::WebFetchRequest::post",
          "line": 134,
          "signature": "pub fn post(url: &str) -> Result<Self, WebError>;",
          "documentation": "Creates a POST request for the given URL.\n\n# Arguments\n\n* `url` - The URL to fetch. Must be a valid HTTP or HTTPS URL.\n\n# Returns\n\nA `WebFetchRequest` configured for POST with default settings.\n\n# Errors\n\nReturns [`WebError::InvalidUrl`] if the URL cannot be parsed or uses\na non-HTTP scheme.\n\n# Examples\n\n```\nuse forge_web::types::WebFetchRequest;\n\nlet req = WebFetchRequest::post(\"https://api.example.com/data\").unwrap();\nassert_eq!(req.method, forge_web::types::HttpMethod::Post);\n```"
        },
        {
          "name": "::WebFetchRequest::head",
          "line": 161,
          "signature": "pub fn head(url: &str) -> Result<Self, WebError>;",
          "documentation": "Creates a HEAD request for the given URL.\n\n# Arguments\n\n* `url` - The URL to fetch. Must be a valid HTTP or HTTPS URL.\n\n# Returns\n\nA `WebFetchRequest` configured for HEAD with default settings.\n\n# Errors\n\nReturns [`WebError::InvalidUrl`] if the URL cannot be parsed or uses\na non-HTTP scheme."
        },
        {
          "name": "::WebFetchRequest::with_header",
          "line": 184,
          "signature": "pub fn with_header(mut self, name: impl Into<String>, value: impl Into<String>) -> Self;",
          "documentation": "Adds a header to the request.\n\n# Arguments\n\n* `name` - The header name.\n* `value` - The header value.\n\n# Returns\n\nThe modified request (builder pattern)."
        },
        {
          "name": "::WebFetchRequest::with_body",
          "line": 198,
          "signature": "pub fn with_body(mut self, body: impl Into<String>) -> Self;",
          "documentation": "Sets the request body (for POST requests).\n\n# Arguments\n\n* `body` - The request body as a string.\n\n# Returns\n\nThe modified request (builder pattern)."
        },
        {
          "name": "::WebFetchRequest::with_timeout",
          "line": 212,
          "signature": "pub fn with_timeout(mut self, timeout_ms: u64) -> Self;",
          "documentation": "Sets the request timeout in milliseconds.\n\n# Arguments\n\n* `timeout_ms` - The timeout duration in milliseconds.\n\n# Returns\n\nThe modified request (builder pattern)."
        },
        {
          "name": "::WebFetchResponse",
          "line": 223,
          "signature": "#[derive(Debug, Clone, Serialize, Deserialize)]\npub struct WebFetchResponse {\n/// The HTTP status code (e.g., 200, 404, 500).\n\npub status: u16,\n/// Response headers. Uses `BTreeMap` for deterministic serialization.\n\npub headers: BTreeMap<String, String>,\n/// The response body as a string.\n\n///\n\n/// Binary responses are base64-encoded. Non-UTF-8 text responses use\n\n/// lossy conversion.\n\npub body: String,\n/// The detected content type from the `Content-Type` header.\n\n///\n\n/// `None` if no `Content-Type` header is present.\n\npub content_type: Option<String>,\n/// The final URL after following any redirects.\n\n///\n\n/// Matches the request URL if no redirects occurred.\n\npub final_url: String\n}",
          "documentation": "An HTTP response from a web fetch operation.\n\nContains the status code, response headers, body content, detected content\ntype, and the final URL after any redirects."
        },
        {
          "name": "::WebContent",
          "line": 253,
          "signature": "#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]\n#[serde(tag = \"type\", content = \"data\")]\npub enum WebContent {\n    /// Raw HTML content.\n    Html(String),\n    /// Markdown-converted content.\n    Markdown(String),\n    /// Plain text content (HTML tags stripped).\n    PlainText(String),\n    /// Structured JSON content.\n    Json(serde_json::Value),\n    /// Binary content (e.g., images, PDFs).\n    Binary(Vec<u8>),\n}",
          "documentation": "Represents web content in various formats.\n\nThe `WebContent` enum captures the different representations of web\ncontent that the substrate can produce."
        },
        {
          "name": "::WebExtractQuery",
          "line": 273,
          "signature": "#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]\n#[serde(tag = \"type\", content = \"expression\")]\npub enum WebExtractQuery {\n    /// A CSS selector query (e.g., `div.content > p`).\n    CssSelector(String),\n    /// An XPath query (e.g., `//div[@class='content']/p`).\n    XPath(String),\n    /// A JSONPath query (e.g., `$.data.items[*].name`).\n    JsonPath(String),\n}",
          "documentation": "A query for extracting content from a parsed web page.\n\nSupports CSS selectors, XPath expressions, and JSONPath queries. The native\nJSONPath evaluator supports `$`, dot fields, array indexes, and `[*]` array\nwildcards."
        },
        {
          "name": "::WebExtractResult",
          "line": 298,
          "signature": "#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]\npub struct WebExtractResult {\n/// The matched elements or values as strings.\n\npub matches: Vec<String>\n}",
          "documentation": "The result of a content extraction query.\n\nContains the matched elements as strings. For CSS selectors and XPath,\neach match is the outer HTML or text content of the matched element.\nFor JSONPath, each match is the JSON-serialized matched value."
        },
        {
          "name": "::SiteNode",
          "line": 305,
          "signature": "#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]\npub struct SiteNode {\n/// The URL of this page.\n\npub url: String,\n/// The page title extracted from the `<title>` tag, if available.\n\npub title: Option<String>,\n/// URLs linked from this page.\n\npub links: Vec<String>,\n/// The crawl depth at which this node was discovered (0 = root).\n\npub depth: u32\n}",
          "documentation": "A node in a site map representing a single page and its outgoing links."
        },
        {
          "name": "::SiteMap",
          "line": 320,
          "signature": "#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]\npub struct SiteMap {\n/// The root URL that the crawl started from.\n\npub root: String,\n/// All discovered site nodes, in breadth-first order.\n\npub nodes: Vec<SiteNode>\n}",
          "documentation": "A site map representing the structure of a website.\n\nBuilt by crawling from a root URL and following links to a configured depth."
        },
        {
          "name": "::CompactedSite",
          "line": 332,
          "signature": "#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]\npub struct CompactedSite {\n/// Map of URL to compacted Markdown content. Uses `BTreeMap` for\n\n/// deterministic serialization order.\n\npub pages: BTreeMap<String, String>,\n/// Estimated total token count across all pages.\n\n///\n\n/// Uses a rough heuristic of ~4 characters per token.\n\npub total_tokens_estimate: u64\n}",
          "documentation": "A compacted representation of a multi-page site, optimized for LLM context.\n\nEach page's content is converted to Markdown and stripped of boilerplate.\nThe total token estimate helps with context window management."
        },
        {
          "name": "::WebSearchResult",
          "line": 344,
          "signature": "#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\npub struct WebSearchResult {\n/// Human-readable search result title.\n\npub title: String,\n/// Canonical result URL.\n\npub url: String,\n/// Optional search-provider snippet or summary.\n\npub snippet: Option<String>\n}",
          "documentation": "A single web search result."
        },
        {
          "name": "::WebSearchResponse",
          "line": 355,
          "signature": "#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]\npub struct WebSearchResponse {\n/// Original user query after trimming leading/trailing whitespace.\n\npub query: String,\n/// Search results in provider rank order.\n\npub results: Vec<WebSearchResult>\n}",
          "documentation": "Search response returned by [`crate::search::web_search`]."
        },
        {
          "name": "::validate_url",
          "line": 378,
          "signature": "pub fn validate_url(url: &str) -> Result<String, WebError>;",
          "documentation": "Validates and normalizes a URL string.\n\nEnsures the URL uses HTTP or HTTPS scheme and is well-formed.\n\n# Arguments\n\n* `url` - The URL string to validate.\n\n# Returns\n\nThe normalized URL string on success.\n\n# Errors\n\nReturns [`WebError::InvalidUrl`] if the URL is malformed or uses a\nnon-HTTP scheme."
        }
      ]
    }
  ]
}
