Apply five style rules: convert iota string enums to typed string constants, replace errors.As with errors.AsType, merge three-group imports into two groups, fix multiline parameter/argument formatting, and replace fmt.Sprintf URL construction with net/url. Signed-off-by: Émile Ré <emile@probo.com>
123 lines
3.5 KiB
Go
123 lines
3.5 KiB
Go
// Copyright (c) 2026 Probo Inc <hello@getprobo.com>.
|
|
//
|
|
// Permission to use, copy, modify, and/or distribute this software for any
|
|
// purpose with or without fee is hereby granted, provided that the above
|
|
// copyright notice and this permission notice appear in all copies.
|
|
//
|
|
// THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH
|
|
// REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY
|
|
// AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY SPECIAL, DIRECT,
|
|
// INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM
|
|
// LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE OR
|
|
// OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR
|
|
// PERFORMANCE OF THIS SOFTWARE.
|
|
|
|
package browser
|
|
|
|
import (
|
|
"bufio"
|
|
"context"
|
|
"fmt"
|
|
"net/http"
|
|
"net/url"
|
|
"strings"
|
|
"time"
|
|
|
|
"go.probo.inc/probo/pkg/agent"
|
|
)
|
|
|
|
type (
|
|
robotsParams struct {
|
|
Domain string `json:"domain" jsonschema:"The domain to fetch robots.txt from (e.g. example.com)"`
|
|
}
|
|
|
|
robotsResult struct {
|
|
Found bool `json:"found"`
|
|
Sitemaps []string `json:"sitemaps,omitempty"`
|
|
Disallowed []string `json:"disallowed_paths,omitempty"`
|
|
ErrorDetail string `json:"error_detail,omitempty"`
|
|
}
|
|
)
|
|
|
|
func FetchRobotsTxtTool() agent.Tool {
|
|
client := &http.Client{Timeout: 10 * time.Second}
|
|
|
|
return agent.FunctionTool(
|
|
"fetch_robots_txt",
|
|
"Fetch and parse the robots.txt file for a domain. Returns sitemap URLs and disallowed paths, which can reveal hidden pages the crawler might miss.",
|
|
func(ctx context.Context, p robotsParams) (agent.ToolResult, error) {
|
|
if err := validatePublicDomain(p.Domain); err != nil {
|
|
return agent.ResultJSON(
|
|
robotsResult{
|
|
Found: false,
|
|
ErrorDetail: fmt.Sprintf("domain not allowed: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
u := &url.URL{
|
|
Scheme: "https",
|
|
Host: p.Domain,
|
|
Path: "/robots.txt",
|
|
}
|
|
|
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, u.String(), nil)
|
|
if err != nil {
|
|
return agent.ResultJSON(
|
|
robotsResult{
|
|
Found: false,
|
|
ErrorDetail: fmt.Sprintf("cannot create request: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
resp, err := client.Do(req)
|
|
if err != nil {
|
|
return agent.ResultJSON(
|
|
robotsResult{
|
|
Found: false,
|
|
ErrorDetail: fmt.Sprintf("cannot fetch robots.txt: %s", err),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
defer func() { _ = resp.Body.Close() }()
|
|
|
|
if resp.StatusCode != http.StatusOK {
|
|
return agent.ResultJSON(
|
|
robotsResult{
|
|
Found: false,
|
|
ErrorDetail: fmt.Sprintf("robots.txt returned status %d", resp.StatusCode),
|
|
},
|
|
), nil
|
|
}
|
|
|
|
var result robotsResult
|
|
|
|
result.Found = true
|
|
|
|
scanner := bufio.NewScanner(resp.Body)
|
|
for scanner.Scan() {
|
|
line := strings.TrimSpace(scanner.Text())
|
|
|
|
// Directive names are case-insensitive but values
|
|
// (URLs, paths) are case-sensitive, so extract the
|
|
// original-case suffix from the raw line rather than
|
|
// reading it off the lowercased copy.
|
|
if after, ok := strings.CutPrefix(strings.ToLower(line), "sitemap:"); ok {
|
|
result.Sitemaps = append(result.Sitemaps, strings.TrimSpace(line[len(line)-len(after):]))
|
|
}
|
|
|
|
if after, ok := strings.CutPrefix(strings.ToLower(line), "disallow:"); ok {
|
|
path := strings.TrimSpace(line[len(line)-len(after):])
|
|
if path != "" && len(result.Disallowed) < 50 {
|
|
result.Disallowed = append(result.Disallowed, path)
|
|
}
|
|
}
|
|
}
|
|
|
|
return agent.ResultJSON(result), nil
|
|
},
|
|
)
|
|
}
|