The third-party vetting agent runs a suite of HTTP "security" tools on
the internal worker network against a caller-supplied URL that is only
validated for length and charset, not host. Several tools reached
internal, loopback, and link-local addresses:
- analyze_csp used a bare http.Client with no host validation, no
redirect control, and no rebinding-safe transport, reflecting the
target's CSP header back to the caller.
- check_security_headers, fetch_robots_txt, and fetch_sitemap
validated only the initial host, then followed 3xx redirects with an
ordinary client, yielding full-read SSRF via a redirect to an
internal address.
- check_cors validated the URL but still dialed through an ordinary
transport, leaving it exposed to DNS-rebinding TOCTOU.
Route every one of these clients through the house-standard
httpclient.DefaultPooledClient(WithSSRFProtection()), which rejects
dials to loopback, private, CGNAT, link-local, ULA, IPv4-mapped, and
reserved ranges on the resolved peer IP at connect time (defeating DNS
rebinding on every redirect hop) and refuses cross-origin redirects.
download_pdf moves onto the same client, and the now-unused local
netcheck.NewPinnedTransport is removed. analyze_csp also gains an
up-front ValidatePublicURL check for a clean early error and scheme
enforcement.
Signed-off-by: Sacha Al Himdani <sacha@probo.com>
171 lines
4.4 KiB
Go
171 lines
4.4 KiB
Go
// Copyright (c) 2026 Probo Inc <hello@probo.com>.
|
|
//
|
|
// Permission to use, copy, modify, and/or distribute this software for any
|
|
// purpose with or without fee is hereby granted, provided that the above
|
|
// copyright notice and this permission notice appear in all copies.
|
|
//
|
|
// THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH
|
|
// REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
|
|
// AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,
|
|
// INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM
|
|
// LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR
|
|
// OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
|
|
// PERFORMANCE OF THIS SOFTWARE.
|
|
|
|
package browser
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"os"
|
|
"path/filepath"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/pdfcpu/pdfcpu/pkg/api"
|
|
"github.com/pdfcpu/pdfcpu/pkg/pdfcpu/model"
|
|
"go.gearno.de/kit/httpclient"
|
|
"go.probo.inc/probo/pkg/agent"
|
|
)
|
|
|
|
type (
|
|
downloadPDFParams struct {
|
|
URL string `json:"url" jsonschema:"The URL of the PDF document to download and extract text from"`
|
|
}
|
|
|
|
downloadPDFResult struct {
|
|
Text string `json:"text"`
|
|
PageCount int `json:"page_count"`
|
|
ErrorDetail string `json:"error_detail,omitempty"`
|
|
}
|
|
)
|
|
|
|
func DownloadPDFTool() agent.Tool {
|
|
client := httpclient.DefaultPooledClient(httpclient.WithSSRFProtection())
|
|
client.Timeout = 30 * time.Second
|
|
|
|
return agent.FunctionTool(
|
|
"download_pdf",
|
|
"Download a PDF document from a URL and extract its text content. Use this for DPAs, SOC 2 reports, privacy policies, and other documents hosted as PDFs.",
|
|
func(ctx context.Context, p downloadPDFParams) (agent.ToolResult, error) {
|
|
if err := validatePublicURL(p.URL); err != nil {
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
ErrorDetail: fmt.Sprintf("URL not allowed: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, p.URL, nil)
|
|
if err != nil {
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
ErrorDetail: fmt.Sprintf("cannot create request: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
resp, err := client.Do(req)
|
|
if err != nil {
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
ErrorDetail: fmt.Sprintf("cannot download PDF: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
defer func() { _ = resp.Body.Close() }()
|
|
|
|
if resp.StatusCode != http.StatusOK {
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
ErrorDetail: fmt.Sprintf("PDF download returned status %d", resp.StatusCode),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
// Read PDF into memory (max 20MB).
|
|
body, err := io.ReadAll(io.LimitReader(resp.Body, 20*1024*1024))
|
|
if err != nil {
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
ErrorDetail: fmt.Sprintf("cannot read PDF body: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
// Write to temp file for pdfcpu.
|
|
tmpDir, err := os.MkdirTemp("", "pdf-extract-*")
|
|
if err != nil {
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
ErrorDetail: fmt.Sprintf("cannot create temp dir: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
defer func() { _ = os.RemoveAll(tmpDir) }()
|
|
|
|
tmpFile := filepath.Join(tmpDir, "input.pdf")
|
|
if err := os.WriteFile(tmpFile, body, 0o600); err != nil {
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
ErrorDetail: fmt.Sprintf("cannot write temp file: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
// Get page count.
|
|
conf := model.NewDefaultConfiguration()
|
|
|
|
pageCount, err := api.PageCountFile(tmpFile)
|
|
if err != nil {
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
ErrorDetail: fmt.Sprintf("cannot read PDF: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
// Extract content, digesting each page's content stream.
|
|
var sb strings.Builder
|
|
|
|
reader := bytes.NewReader(body)
|
|
digest := func(r io.Reader, _ int) error {
|
|
content, err := io.ReadAll(r)
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
sb.Write(content)
|
|
sb.WriteString("\n")
|
|
|
|
return nil
|
|
}
|
|
|
|
if err := api.ExtractContent(reader, nil, digest, conf); err != nil {
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
ErrorDetail: fmt.Sprintf("cannot extract PDF content: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
text := sb.String()
|
|
if len(text) > maxTextLength {
|
|
text = text[:maxTextLength] + "\n[... truncated]"
|
|
}
|
|
|
|
return agent.ResultJSON(
|
|
downloadPDFResult{
|
|
Text: text,
|
|
PageCount: pageCount,
|
|
},
|
|
), nil
|
|
},
|
|
)
|
|
}
|