Files
probo/pkg/agent/tools/browser/fetch_robots.go
Sacha Al Himdani 98f08b7439 Guard vetting agent HTTP tools against SSRF
The third-party vetting agent runs a suite of HTTP "security" tools on
the internal worker network against a caller-supplied URL that is only
validated for length and charset, not host. Several tools reached
internal, loopback, and link-local addresses:

  - analyze_csp used a bare http.Client with no host validation, no
    redirect control, and no rebinding-safe transport, reflecting the
    target's CSP header back to the caller.
  - check_security_headers, fetch_robots_txt, and fetch_sitemap
    validated only the initial host, then followed 3xx redirects with an
    ordinary client, yielding full-read SSRF via a redirect to an
    internal address.
  - check_cors validated the URL but still dialed through an ordinary
    transport, leaving it exposed to DNS-rebinding TOCTOU.

Route every one of these clients through the house-standard
httpclient.DefaultPooledClient(WithSSRFProtection()), which rejects
dials to loopback, private, CGNAT, link-local, ULA, IPv4-mapped, and
reserved ranges on the resolved peer IP at connect time (defeating DNS
rebinding on every redirect hop) and refuses cross-origin redirects.
download_pdf moves onto the same client, and the now-unused local
netcheck.NewPinnedTransport is removed. analyze_csp also gains an
up-front ValidatePublicURL check for a clean early error and scheme
enforcement.

Signed-off-by: Sacha Al Himdani <sacha@probo.com>
2026-07-08 11:23:00 +02:00

125 lines
3.6 KiB
Go

// Copyright (c) 2026 Probo Inc <hello@probo.com>.
//
// Permission to use, copy, modify, and/or distribute this software for any
// purpose with or without fee is hereby granted, provided that the above
// copyright notice and this permission notice appear in all copies.
//
// THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH
// REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
// AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,
// INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM
// LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR
// OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
// PERFORMANCE OF THIS SOFTWARE.
package browser
import (
"bufio"
"context"
"fmt"
"net/http"
"net/url"
"strings"
"time"
"go.gearno.de/kit/httpclient"
"go.probo.inc/probo/pkg/agent"
)
type (
robotsParams struct {
Domain string `json:"domain" jsonschema:"The domain to fetch robots.txt from (e.g. example.com)"`
}
robotsResult struct {
Found bool `json:"found"`
Sitemaps []string `json:"sitemaps,omitempty"`
Disallowed []string `json:"disallowed_paths,omitempty"`
ErrorDetail string `json:"error_detail,omitempty"`
}
)
func FetchRobotsTxtTool() agent.Tool {
client := httpclient.DefaultPooledClient(httpclient.WithSSRFProtection())
client.Timeout = 10 * time.Second
return agent.FunctionTool(
"fetch_robots_txt",
"Fetch and parse the robots.txt file for a domain. Returns sitemap URLs and disallowed paths, which can reveal hidden pages the crawler might miss.",
func(ctx context.Context, p robotsParams) (agent.ToolResult, error) {
if err := validatePublicDomain(p.Domain); err != nil {
return agent.ResultJSON(
robotsResult{
Found: false,
ErrorDetail: fmt.Sprintf("domain not allowed: %s", err),
},
), nil
}
u := &url.URL{
Scheme: "https",
Host: p.Domain,
Path: "/robots.txt",
}
req, err := http.NewRequestWithContext(ctx, http.MethodGet, u.String(), nil)
if err != nil {
return agent.ResultJSON(
robotsResult{
Found: false,
ErrorDetail: fmt.Sprintf("cannot create request: %s", err),
},
), nil
}
resp, err := client.Do(req)
if err != nil {
return agent.ResultJSON(
robotsResult{
Found: false,
ErrorDetail: fmt.Sprintf("cannot fetch robots.txt: %s", err),
},
), nil
}
defer func() { _ = resp.Body.Close() }()
if resp.StatusCode != http.StatusOK {
return agent.ResultJSON(
robotsResult{
Found: false,
ErrorDetail: fmt.Sprintf("robots.txt returned status %d", resp.StatusCode),
},
), nil
}
var result robotsResult
result.Found = true
scanner := bufio.NewScanner(resp.Body)
for scanner.Scan() {
line := strings.TrimSpace(scanner.Text())
// Directive names are case-insensitive but values
// (URLs, paths) are case-sensitive, so extract the
// original-case suffix from the raw line rather than
// reading it off the lowercased copy.
if after, ok := strings.CutPrefix(strings.ToLower(line), "sitemap:"); ok {
result.Sitemaps = append(result.Sitemaps, strings.TrimSpace(line[len(line)-len(after):]))
}
if after, ok := strings.CutPrefix(strings.ToLower(line), "disallow:"); ok {
path := strings.TrimSpace(line[len(line)-len(after):])
if path != "" && len(result.Disallowed) < 50 {
result.Disallowed = append(result.Disallowed, path)
}
}
}
return agent.ResultJSON(result), nil
},
)
}