// Copyright (c) 2026 Probo Inc . // // Permission is hereby granted, free of charge, to any person obtaining a copy // of this software and associated documentation files (the "Software"), to deal // in the Software without restriction, including without limitation the rights // to use, copy, modify, merge, publish, distribute, sublicense, and/or sell // copies of the Software, and to permit persons to whom the Software is // furnished to do so, subject to the following conditions: // // The above copyright notice and this permission notice shall be included in // all copies or substantial portions of the Software. // // THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR // IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, // FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE // AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER // LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, // OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE // SOFTWARE. package browser import ( "context" "net/url" "github.com/chromedp/chromedp" "go.probo.inc/probo/pkg/agent" ) type ( extractLinksParams struct { URL string `json:"url" jsonschema:"The URL to extract links from"` } link struct { Href string `json:"href"` Text string `json:"text"` } ) func ExtractLinksTool(b *Browser) agent.Tool { return agent.FunctionTool( "extract_links", "Navigate to a URL and extract all links ( elements) with their href and text.", func(ctx context.Context, p extractLinksParams) (agent.ToolResult, error) { if r := b.checkAlive(); r != nil { return *r, nil } u, err := url.Parse(p.URL) if err != nil || (u.Scheme != "http" && u.Scheme != "https") { return agent.ResultError("invalid URL scheme: only http and https are allowed"), nil } if r := b.checkURL(p.URL); r != nil { return *r, nil } ctx, timeoutCancel := withToolTimeout(ctx) defer timeoutCancel() tabCtx, cancel := b.NewTab(ctx) defer cancel() var links []link err = chromedp.Run( tabCtx, chromedp.Navigate(p.URL), waitForPage(), chromedp.Evaluate( `Array.from(document.querySelectorAll("a[href]")).map(a => ({ href: a.href, text: a.innerText.trim().substring(0, 200) }))`, &links, ), ) if err != nil { return agent.ResultError(b.classifyError(ctx, p.URL, err)), nil } return agent.ResultJSON(links), nil }, ) }