mirror of
https://wget.la/https://github.com/leookun/cursor-byok
synced 2026-08-17 03:27:02 +08:00
feat: WebSearch 接入百度搜索,DuckDuckGo 作为兜底
百度搜索失败或无结果时自动回退到 DuckDuckGo,不引入 Bing。
This commit is contained in:
@@ -5,6 +5,7 @@ go 1.25.0
|
|||||||
require (
|
require (
|
||||||
codeberg.org/readeck/go-readability/v2 v2.1.1
|
codeberg.org/readeck/go-readability/v2 v2.1.1
|
||||||
connectrpc.com/connect v1.19.1
|
connectrpc.com/connect v1.19.1
|
||||||
|
github.com/PuerkitoBio/goquery v1.9.2
|
||||||
github.com/denisbrodbeck/machineid v1.0.1
|
github.com/denisbrodbeck/machineid v1.0.1
|
||||||
github.com/elazarl/goproxy v1.7.2
|
github.com/elazarl/goproxy v1.7.2
|
||||||
github.com/firecrawl/html-to-markdown v0.0.0-20260312013131-1af9901a5d61
|
github.com/firecrawl/html-to-markdown v0.0.0-20260312013131-1af9901a5d61
|
||||||
@@ -29,7 +30,6 @@ require (
|
|||||||
dario.cat/mergo v1.0.2 // indirect
|
dario.cat/mergo v1.0.2 // indirect
|
||||||
github.com/Microsoft/go-winio v0.6.2 // indirect
|
github.com/Microsoft/go-winio v0.6.2 // indirect
|
||||||
github.com/ProtonMail/go-crypto v1.3.0 // indirect
|
github.com/ProtonMail/go-crypto v1.3.0 // indirect
|
||||||
github.com/PuerkitoBio/goquery v1.9.2 // indirect
|
|
||||||
github.com/adrg/xdg v0.5.3 // indirect
|
github.com/adrg/xdg v0.5.3 // indirect
|
||||||
github.com/andybalholm/cascadia v1.3.3 // indirect
|
github.com/andybalholm/cascadia v1.3.3 // indirect
|
||||||
github.com/araddon/dateparse v0.0.0-20210429162001-6b43995a97de // indirect
|
github.com/araddon/dateparse v0.0.0-20210429162001-6b43995a97de // indirect
|
||||||
|
|||||||
@@ -0,0 +1,212 @@
|
|||||||
|
package interaction
|
||||||
|
|
||||||
|
import (
|
||||||
|
"net/http"
|
||||||
|
neturl "net/url"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/PuerkitoBio/goquery"
|
||||||
|
|
||||||
|
"cursor/gen/agentv1"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
baiduWebSearchBaseURL = "https://www.baidu.com/s?ie=utf-8&tn=baidu&wd="
|
||||||
|
baiduWebSearchHostURL = "https://www.baidu.com"
|
||||||
|
baiduSearchAbstractLimit = 300
|
||||||
|
baiduSearchReferenceLimit = 8
|
||||||
|
)
|
||||||
|
|
||||||
|
// extractBaiduWebSearchReferences 从百度搜索结果页 HTML 中解析出搜索结果列表。
|
||||||
|
func extractBaiduWebSearchReferences(body string) []*agentv1.WebSearchReference {
|
||||||
|
document, err := goquery.NewDocumentFromReader(strings.NewReader(body))
|
||||||
|
if err != nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
references := make([]*agentv1.WebSearchReference, 0, baiduSearchReferenceLimit)
|
||||||
|
document.Find("#content_left > div").EachWithBreak(func(_ int, selection *goquery.Selection) bool {
|
||||||
|
if len(references) >= baiduSearchReferenceLimit {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if !selection.HasClass("c-container") {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
title, resultURL, abstract := extractBaiduSearchResult(selection)
|
||||||
|
if title == "" || resultURL == "" {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
references = append(references, &agentv1.WebSearchReference{
|
||||||
|
Title: title,
|
||||||
|
Url: normalizeBaiduSearchURL(resultURL),
|
||||||
|
Chunk: truncateBaiduSearchAbstract(abstract),
|
||||||
|
})
|
||||||
|
return true
|
||||||
|
})
|
||||||
|
return references
|
||||||
|
}
|
||||||
|
|
||||||
|
// extractBaiduSearchResult 从单条百度搜索结果节点中提取标题、链接和摘要。
|
||||||
|
func extractBaiduSearchResult(selection *goquery.Selection) (string, string, string) {
|
||||||
|
title := cleanBaiduSearchText(selection.Find("h3").First().Text())
|
||||||
|
resultURL, _ := selection.Find("h3 a").First().Attr("href")
|
||||||
|
if title == "" {
|
||||||
|
title = firstBaiduSearchLine(selection.Text())
|
||||||
|
}
|
||||||
|
if resultURL == "" {
|
||||||
|
resultURL, _ = selection.Find("a").First().Attr("href")
|
||||||
|
}
|
||||||
|
abstract := cleanBaiduSearchText(selection.Find(".c-abstract").First().Text())
|
||||||
|
if abstract == "" {
|
||||||
|
abstract = cleanBaiduSearchText(selection.ChildrenFiltered("div").First().Text())
|
||||||
|
}
|
||||||
|
if abstract == "" {
|
||||||
|
abstract = baiduSearchTextAfterFirstLine(selection.Text())
|
||||||
|
}
|
||||||
|
return title, strings.TrimSpace(resultURL), abstract
|
||||||
|
}
|
||||||
|
|
||||||
|
// normalizeBaiduSearchURL 把百度返回的相对或协议省略链接归一化为绝对 URL。
|
||||||
|
func normalizeBaiduSearchURL(rawURL string) string {
|
||||||
|
rawURL = strings.TrimSpace(rawURL)
|
||||||
|
if strings.HasPrefix(rawURL, "//") {
|
||||||
|
return "https:" + rawURL
|
||||||
|
}
|
||||||
|
if strings.HasPrefix(rawURL, "/") {
|
||||||
|
return baiduWebSearchHostURL + rawURL
|
||||||
|
}
|
||||||
|
return rawURL
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveBaiduWebSearchRedirects 把百度跳转链接解析为最终目标地址,就地更新引用列表。
|
||||||
|
func resolveBaiduWebSearchRedirects(client *http.Client, references []*agentv1.WebSearchReference) {
|
||||||
|
for _, reference := range references {
|
||||||
|
if reference == nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
reference.Url = resolveBaiduRedirectURL(client, reference.GetUrl())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveBaiduRedirectURL 判断链接是否是百度跳转链接,并尝试解析出真实目标。
|
||||||
|
func resolveBaiduRedirectURL(client *http.Client, rawURL string) string {
|
||||||
|
resultURL := normalizeBaiduSearchURL(rawURL)
|
||||||
|
if !isBaiduRedirectURL(resultURL) {
|
||||||
|
return resultURL
|
||||||
|
}
|
||||||
|
redirectClient := baiduRedirectHTTPClient(client)
|
||||||
|
if location := requestBaiduRedirectLocation(redirectClient, http.MethodHead, resultURL); location != "" {
|
||||||
|
return location
|
||||||
|
}
|
||||||
|
if location := requestBaiduRedirectLocation(redirectClient, http.MethodGet, resultURL); location != "" {
|
||||||
|
return location
|
||||||
|
}
|
||||||
|
return resultURL
|
||||||
|
}
|
||||||
|
|
||||||
|
// baiduRedirectHTTPClient 基于基础 client 构造一个不自动跟随重定向的短超时客户端。
|
||||||
|
func baiduRedirectHTTPClient(base *http.Client) *http.Client {
|
||||||
|
if base == nil {
|
||||||
|
base = http.DefaultClient
|
||||||
|
}
|
||||||
|
client := *base
|
||||||
|
if client.Timeout == 0 || client.Timeout > 6*time.Second {
|
||||||
|
client.Timeout = 6 * time.Second
|
||||||
|
}
|
||||||
|
client.CheckRedirect = func(_ *http.Request, _ []*http.Request) error {
|
||||||
|
return http.ErrUseLastResponse
|
||||||
|
}
|
||||||
|
return &client
|
||||||
|
}
|
||||||
|
|
||||||
|
// requestBaiduRedirectLocation 发起一次请求并读取响应头中的重定向目标地址。
|
||||||
|
func requestBaiduRedirectLocation(client *http.Client, method string, rawURL string) string {
|
||||||
|
request, err := http.NewRequest(method, rawURL, nil)
|
||||||
|
if err != nil {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
request.Header.Set("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/120.0.0.0 Safari/537.36")
|
||||||
|
request.Header.Set("Accept-Language", "zh-CN,zh;q=0.9")
|
||||||
|
request.Header.Set("Referer", baiduWebSearchHostURL+"/")
|
||||||
|
response, err := client.Do(request)
|
||||||
|
if err != nil {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
defer response.Body.Close()
|
||||||
|
location := strings.TrimSpace(response.Header.Get("Location"))
|
||||||
|
if location == "" {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return resolveBaiduLocationURL(rawURL, location)
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveBaiduLocationURL 把响应头里的相对重定向地址解析为绝对地址。
|
||||||
|
func resolveBaiduLocationURL(baseURL string, location string) string {
|
||||||
|
parsedLocation, err := neturl.Parse(location)
|
||||||
|
if err != nil {
|
||||||
|
return location
|
||||||
|
}
|
||||||
|
if parsedLocation.IsAbs() {
|
||||||
|
return parsedLocation.String()
|
||||||
|
}
|
||||||
|
parsedBase, err := neturl.Parse(baseURL)
|
||||||
|
if err != nil {
|
||||||
|
return location
|
||||||
|
}
|
||||||
|
return parsedBase.ResolveReference(parsedLocation).String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// isBaiduRedirectURL 判断给定地址是否是百度域名下的跳转链接。
|
||||||
|
func isBaiduRedirectURL(rawURL string) bool {
|
||||||
|
parsedURL, err := neturl.Parse(strings.TrimSpace(rawURL))
|
||||||
|
if err != nil {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
host := strings.ToLower(parsedURL.Hostname())
|
||||||
|
path := strings.ToLower(parsedURL.EscapedPath())
|
||||||
|
return (host == "baidu.com" || strings.HasSuffix(host, ".baidu.com")) && strings.HasPrefix(path, "/link")
|
||||||
|
}
|
||||||
|
|
||||||
|
// truncateBaiduSearchAbstract 按字符数截断摘要文本,避免结果过长。
|
||||||
|
func truncateBaiduSearchAbstract(value string) string {
|
||||||
|
value = cleanBaiduSearchText(value)
|
||||||
|
if baiduSearchAbstractLimit <= 0 {
|
||||||
|
return value
|
||||||
|
}
|
||||||
|
runes := []rune(value)
|
||||||
|
if len(runes) <= baiduSearchAbstractLimit {
|
||||||
|
return value
|
||||||
|
}
|
||||||
|
return string(runes[:baiduSearchAbstractLimit])
|
||||||
|
}
|
||||||
|
|
||||||
|
// firstBaiduSearchLine 返回文本中第一个非空行。
|
||||||
|
func firstBaiduSearchLine(value string) string {
|
||||||
|
for _, line := range strings.Split(strings.ReplaceAll(value, "\r", "\n"), "\n") {
|
||||||
|
line = cleanBaiduSearchText(line)
|
||||||
|
if line != "" {
|
||||||
|
return line
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// baiduSearchTextAfterFirstLine 返回除第一个非空行外剩余文本的拼接结果。
|
||||||
|
func baiduSearchTextAfterFirstLine(value string) string {
|
||||||
|
nonEmpty := make([]string, 0, 8)
|
||||||
|
for _, line := range strings.Split(strings.ReplaceAll(value, "\r", "\n"), "\n") {
|
||||||
|
line = cleanBaiduSearchText(line)
|
||||||
|
if line != "" {
|
||||||
|
nonEmpty = append(nonEmpty, line)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(nonEmpty) <= 1 {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return cleanBaiduSearchText(strings.Join(nonEmpty[1:], " "))
|
||||||
|
}
|
||||||
|
|
||||||
|
// cleanBaiduSearchText 折叠多余空白并去除首尾空格。
|
||||||
|
func cleanBaiduSearchText(value string) string {
|
||||||
|
return strings.Join(strings.Fields(strings.TrimSpace(value)), " ")
|
||||||
|
}
|
||||||
@@ -577,13 +577,68 @@ const (
|
|||||||
)
|
)
|
||||||
|
|
||||||
func (bridge *Bridge) executeWebSearch(searchTerm string) ([]*agentv1.WebSearchReference, string, error) {
|
func (bridge *Bridge) executeWebSearch(searchTerm string) ([]*agentv1.WebSearchReference, string, error) {
|
||||||
if strings.TrimSpace(searchTerm) == "" {
|
searchTerm = strings.TrimSpace(searchTerm)
|
||||||
|
if searchTerm == "" {
|
||||||
return nil, "", fmt.Errorf("web search search_term is required")
|
return nil, "", fmt.Errorf("web search search_term is required")
|
||||||
}
|
}
|
||||||
client := bridge.httpClient
|
client := bridge.httpClient
|
||||||
if client == nil {
|
if client == nil {
|
||||||
client = netproxy.NewHTTPClient(15 * time.Second)
|
client = netproxy.NewHTTPClient(15 * time.Second)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// 先尝试百度搜索
|
||||||
|
baiduReferences, baiduPayload, baiduErr := bridge.tryBaiduWebSearch(client, searchTerm)
|
||||||
|
if baiduErr == nil && len(baiduReferences) > 0 {
|
||||||
|
return baiduReferences, baiduPayload, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// 百度失败,回退到 DuckDuckGo
|
||||||
|
duckReferences, duckPayload, duckErr := bridge.tryDuckDuckGoWebSearch(client, searchTerm)
|
||||||
|
if duckErr == nil && len(duckReferences) > 0 {
|
||||||
|
return duckReferences, duckPayload, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// 两者都失败,返回综合错误
|
||||||
|
if baiduErr != nil && duckErr != nil {
|
||||||
|
return nil, "", fmt.Errorf("web search failed: baidu=%v, duckduckgo=%v", baiduErr, duckErr)
|
||||||
|
}
|
||||||
|
return nil, "", fmt.Errorf("web search returned no parseable results")
|
||||||
|
}
|
||||||
|
|
||||||
|
func (bridge *Bridge) tryBaiduWebSearch(client *http.Client, searchTerm string) ([]*agentv1.WebSearchReference, string, error) {
|
||||||
|
requestURL := baiduWebSearchBaseURL + neturl.QueryEscape(searchTerm)
|
||||||
|
request, err := http.NewRequest(http.MethodGet, requestURL, nil)
|
||||||
|
if err != nil {
|
||||||
|
return nil, "", err
|
||||||
|
}
|
||||||
|
request.Header.Set("User-Agent", "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/68.0.3440.106 Safari/537.36")
|
||||||
|
request.Header.Set("Accept", "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,image/apng,*/*;q=0.8")
|
||||||
|
request.Header.Set("Accept-Language", "zh-CN,zh;q=0.9,en;q=0.8")
|
||||||
|
request.Header.Set("Referer", baiduWebSearchHostURL+"/")
|
||||||
|
response, err := client.Do(request)
|
||||||
|
if err != nil {
|
||||||
|
return nil, "", err
|
||||||
|
}
|
||||||
|
defer response.Body.Close()
|
||||||
|
if response.StatusCode < 200 || response.StatusCode >= 300 {
|
||||||
|
return nil, "", fmt.Errorf("baidu http status %d", response.StatusCode)
|
||||||
|
}
|
||||||
|
body, err := io.ReadAll(io.LimitReader(response.Body, 2*1024*1024))
|
||||||
|
if err != nil {
|
||||||
|
return nil, "", err
|
||||||
|
}
|
||||||
|
references := extractBaiduWebSearchReferences(string(body))
|
||||||
|
if len(references) == 0 {
|
||||||
|
return nil, "", fmt.Errorf("baidu returned no parseable results")
|
||||||
|
}
|
||||||
|
if len(references) > 5 {
|
||||||
|
references = references[:5]
|
||||||
|
}
|
||||||
|
resolveBaiduWebSearchRedirects(client, references)
|
||||||
|
return references, formatWebSearchPayload(searchTerm, references), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (bridge *Bridge) tryDuckDuckGoWebSearch(client *http.Client, searchTerm string) ([]*agentv1.WebSearchReference, string, error) {
|
||||||
requestURL := webSearchURLOverride + neturl.QueryEscape(searchTerm)
|
requestURL := webSearchURLOverride + neturl.QueryEscape(searchTerm)
|
||||||
request, err := http.NewRequest(http.MethodGet, requestURL, nil)
|
request, err := http.NewRequest(http.MethodGet, requestURL, nil)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
Reference in New Issue
Block a user