Files
param-auto-manager/services/search_service.py
T

393 lines
15 KiB
Python
Raw Normal View History

2026-07-12 01:07:26 +08:00
"""
搜索服务 - 从内容库和互联网搜索数据
"""
import requests
from bs4 import BeautifulSoup
import json
2026-07-13 11:55:33 +08:00
import subprocess
import os
import re
import urllib.parse
2026-07-12 01:07:26 +08:00
from datetime import datetime
from config import Config
from models.database import db
class SearchService:
def __init__(self):
self.timeout = Config.SEARCH_TIMEOUT
self.max_results = Config.SEARCH_MAX_RESULTS
2026-07-13 11:55:33 +08:00
def _run_browser(self, *args, timeout=30000):
"""运行 agent-browser 命令"""
env = os.environ.copy()
env['XDG_RUNTIME_DIR'] = '/tmp/agent-browser-runtime'
os.makedirs(env['XDG_RUNTIME_DIR'], exist_ok=True)
cmd = ['agent-browser'] + list(args)
result = subprocess.run(
cmd,
capture_output=True,
text=True,
env=env,
timeout=timeout // 1000 + 5
)
return result.stdout, result.stderr, result.returncode
2026-07-13 22:55:36 +08:00
def search_internet(self, keyword, max_results=None, engine='bing_cn', use_cache=True, cache_days=7):
2026-07-12 01:07:26 +08:00
"""
2026-07-13 11:55:33 +08:00
从互联网搜索(使用 agent-browser 浏览器自动化)
2026-07-13 17:03:26 +08:00
支持的搜索引擎:
- bing_cn: Bing 中国(默认)
- bing_global: Bing 国际版
- google: Google
- baidu: 百度
2026-07-13 22:55:36 +08:00
参数:
- use_cache: 是否使用缓存(默认True
- cache_days: 缓存有效天数(默认7天)
2026-07-12 01:07:26 +08:00
"""
max_results = max_results or self.max_results
results = []
2026-07-13 22:55:36 +08:00
# 优先查询缓存
if use_cache:
cached = db.get_search_cache(keyword, engine)
if cached:
print(f"使用缓存结果: {keyword} ({engine})")
return cached['results']
2026-07-13 17:03:26 +08:00
# 根据搜索引擎选择 URL
encoded_keyword = urllib.parse.quote(keyword)
search_urls = {
'bing_cn': f"https://cn.bing.com/search?q={encoded_keyword}",
'bing_global': f"https://www.bing.com/search?q={encoded_keyword}",
'google': f"https://www.google.com/search?q={encoded_keyword}",
'baidu': f"https://www.baidu.com/s?wd={encoded_keyword}"
}
search_url = search_urls.get(engine, search_urls['bing_cn'])
2026-07-13 11:55:33 +08:00
try:
2026-07-13 17:03:26 +08:00
# 1. 打开搜索引擎
2026-07-13 11:55:33 +08:00
stdout, stderr, code = self._run_browser('open', search_url, '--timeout', '20000')
if code != 0:
print(f"打开搜索页面失败: {stderr}")
return results
# 等待页面加载
stdout, stderr, code = self._run_browser('wait', '5000')
# 2. 获取搜索结果页面结构 (JSON 格式)
stdout, stderr, code = self._run_browser('snapshot', '--json', '--timeout', '30000')
if code != 0:
print(f"获取页面结构失败: {stderr}")
return results
# 3. 解析 JSON 提取搜索结果
2026-07-13 23:03:53 +08:00
try:
2026-07-13 11:55:33 +08:00
data = json.loads(stdout)
except json.JSONDecodeError:
print(f"解析 JSON 失败: {stdout[:500]}")
return results
2026-07-13 17:03:26 +08:00
# 4. 根据搜索引擎选择解析方法
if engine == 'baidu':
results = self._parse_baidu_results(data, max_results)
else:
results = self._parse_bing_results(data, max_results)
2026-07-13 11:55:33 +08:00
# 5. 关闭浏览器
self._run_browser('close')
2026-07-13 22:55:36 +08:00
# 6. 保存到缓存
if results and use_cache:
db.save_search_cache(keyword, engine, results, cache_days)
2026-07-13 11:55:33 +08:00
except subprocess.TimeoutExpired:
print(f"搜索超时: {keyword}")
except Exception as e:
print(f"搜索出错: {str(e)}")
# 尝试关闭浏览器
try:
self._run_browser('close')
except:
pass
2026-07-12 01:07:26 +08:00
return results
2026-07-13 11:55:33 +08:00
def _parse_bing_results(self, snapshot_data, max_results=10):
"""
从 Bing 搜索结果的 snapshot 中解析出标题和链接
snapshot_data 是 agent-browser snapshot --json 的输出
结构: {success, data: {snapshot: "文本格式的 accessibility tree"}, error}
"""
results = []
# 获取 snapshot 文本
snapshot = snapshot_data.get('data', {}).get('snapshot', '')
if not snapshot:
return results
# 解析 accessibility tree 文本
in_results = False
refs = [] # 存储 (title, ref) 元组
lines = snapshot.split('\n')
for i, line in enumerate(lines):
line = line.strip()
# 进入搜索结果区域
if 'main "搜索结果"' in line:
in_results = True
continue
# 离开搜索结果区域
if in_results and line.startswith('- ') and 'main' in line and '搜索结果' not in line:
break
if not in_results:
continue
# 匹配标题链接:link "标题文字" [ref=eXX]
2026-07-13 23:40:13 +08:00
# 过滤域名链接(如 "zhihu.com")和短链接
2026-07-13 11:55:33 +08:00
if 'link "' in line and '[ref=' in line:
match = re.search(r'link "([^"]+)" \[ref=(e\d+)\]', line)
if match:
title = match.group(1)
ref = match.group(2)
2026-07-13 23:40:13 +08:00
# 放宽过滤条件:只要不是纯域名格式就保留
if not (title.endswith('.com') or title.endswith('.cn') or title.endswith('.net')):
2026-07-13 11:55:33 +08:00
refs.append((title, ref))
# 获取每个结果的 URL
for title, ref in refs[:max_results]:
url = self._get_link_url(ref)
if url and 'bing.com/search' not in url: # 过滤搜索结果页本身的链接
results.append({
'title': title,
'url': url,
'snippet': '',
'source': 'bing'
})
return results
2026-07-13 17:03:26 +08:00
def _parse_baidu_results(self, snapshot_data, max_results=10):
"""从百度搜索结果中解析标题和链接"""
results = []
snapshot = snapshot_data.get('data', {}).get('snapshot', '')
if not snapshot:
return results
# 百度搜索结果解析
refs = []
lines = snapshot.split('\n')
for line in lines:
line = line.strip()
# 百度结果通常在 link 标签中
if 'link "' in line and '[ref=' in line:
match = re.search(r'link "([^"]+)" \[ref=(e\d+)\]', line)
if match:
title = match.group(1)
ref = match.group(2)
# 过滤百度内部链接和广告
if len(title) > 10 and '百度' not in title[:6]:
refs.append((title, ref))
# 获取每个结果的 URL
for title, ref in refs[:max_results]:
url = self._get_link_url(ref)
if url and 'baidu.com' not in url:
results.append({
'title': title,
'url': url,
'snippet': '',
'source': 'baidu'
})
return results
2026-07-13 11:55:33 +08:00
def _get_link_url(self, ref):
"""通过 agent-browser 获取链接的 URL"""
try:
stdout, stderr, code = self._run_browser('get', 'attr', f'@{ref}', 'href', '--json', '--timeout', '5000')
if code == 0 and stdout:
data = json.loads(stdout)
return data.get('data', {}).get('value', '')
except Exception as e:
print(f"获取 URL 失败 (ref={ref}): {e}")
return None
2026-07-12 01:07:26 +08:00
def fetch_url_content(self, url):
"""抓取网页内容(使用 agent-browser 浏览器方式,绕过反爬虫)"""
2026-07-13 23:40:13 +08:00
error_message = None
2026-07-12 01:07:26 +08:00
try:
# 使用浏览器方式抓取,增加超时时间到60秒
stdout, stderr, code = self._run_browser('open', url, '--timeout', '60000')
if code != 0:
2026-07-13 23:40:13 +08:00
error_message = stderr.strip() if stderr else '浏览器打开页面失败'
print(f"打开页面失败: {stderr}")
# 浏览器失败,尝试使用 requests 备用方案
2026-07-13 23:40:13 +08:00
result = self._fetch_with_requests(url)
if result:
return result
return {'success': False, 'error': error_message}
2026-07-12 01:07:26 +08:00
# 等待页面加载(增加到10秒)
self._run_browser('wait', '10000')
2026-07-12 01:07:26 +08:00
# 获取页面标题
stdout, stderr, code = self._run_browser('get', 'title', '--timeout', '5000')
title = stdout.strip().replace('[agent-browser] ', '').strip() if code == 0 else ''
2026-07-12 01:07:26 +08:00
# 获取页面内容(通过 snapshot 获取 accessibility tree
stdout, stderr, code = self._run_browser('snapshot', '--json', '--timeout', '15000')
text = ''
if code == 0 and stdout:
try:
data = json.loads(stdout)
snapshot = data.get('data', {}).get('snapshot', '')
# 从 snapshot 中提取所有 StaticText
text = self._extract_text_from_snapshot(snapshot)
except:
pass
2026-07-12 01:07:26 +08:00
# 获取 URL(可能被重定向)
stdout, stderr, code = self._run_browser('get', 'url', '--timeout', '5000')
actual_url = stdout.strip() if code == 0 else url
2026-07-12 01:07:26 +08:00
# 关闭浏览器
self._run_browser('close')
# 提取描述(从页面内容的前200字符)
description = text[:200].strip() if text else ''
2026-07-12 01:07:26 +08:00
return {
2026-07-13 23:40:13 +08:00
'success': True,
2026-07-12 01:07:26 +08:00
'title': title,
'description': description,
'content': text,
'url': actual_url,
2026-07-12 01:07:26 +08:00
'fetch_date': datetime.now().isoformat()
}
except Exception as e:
2026-07-13 23:40:13 +08:00
error_message = str(e)
print(f"抓取URL失败: {url}, 错误: {error_message}")
# 尝试关闭浏览器
try:
self._run_browser('close')
except:
pass
2026-07-13 23:40:13 +08:00
return {'success': False, 'error': error_message}
2026-07-12 01:07:26 +08:00
def _extract_text_from_snapshot(self, snapshot):
"""从 accessibility tree snapshot 中提取文本内容"""
texts = []
for line in snapshot.split('\n'):
line = line.strip()
if 'StaticText' in line and 'checkbox' not in line:
# 找到 StaticText 后的内容
idx = line.find('StaticText')
after = line[idx + 10:].strip() # 跳过 'StaticText'
# 去掉开头的引号
if after.startswith('"'):
after = after[1:]
# 如果以 JSON 开头(错误信息),跳过
if after.startswith('{'):
continue
# 提取文本内容
text = after.rstrip('"').strip()
if text and len(text) > 1:
texts.append(text)
result = '\n'.join(texts)
# 检测是否是反爬错误页面
if '请求存在异常' in result or '暂时限制本次访问' in result:
return '[该网站触发了反爬机制,无法抓取内容]'
return result
def _fetch_with_requests(self, url):
"""备用方案:使用 requests 抓取静态内容"""
try:
headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8'
}
response = requests.get(url, headers=headers, timeout=30)
response.raise_for_status()
soup = BeautifulSoup(response.text, 'html.parser')
# 获取标题
title = soup.title.string.strip() if soup.title else ''
# 移除不需要的标签
for tag in soup(['script', 'style', 'nav', 'footer', 'header', 'aside']):
tag.decompose()
# 获取主要内容
text = soup.get_text(separator='\n', strip=True)
# 清理多余空白行
lines = [line.strip() for line in text.split('\n') if line.strip()]
text = '\n'.join(lines)
# 提取描述(前200字符)
description = text[:200].strip() if text else ''
return {
2026-07-13 23:40:13 +08:00
'success': True,
'title': title,
'description': description,
'content': text,
'url': url,
'fetch_date': datetime.now().isoformat()
}
except Exception as e:
print(f"备用抓取失败: {url}, 错误: {str(e)}")
return None
2026-07-12 01:07:26 +08:00
def search_articles(self, keyword, category=None):
"""从内容库搜索"""
return db.search_articles(keyword, category)
def search_all(self, keyword, category=None, include_internet=True):
"""
综合搜索:内容库 + 互联网
"""
results = {
'articles': [],
'internet': [],
'total': 0
}
# 1. 从内容库搜索
articles = self.search_articles(keyword, category)
results['articles'] = articles
# 2. 从互联网搜索(如果启用)
if include_internet:
internet_results = self.search_internet(keyword)
results['internet'] = internet_results
results['total'] = len(articles) + len(results['internet'])
return results
def save_to_articles(self, product_names, category, keywords, summary, content, source, url=None):
"""保存搜索结果到内容库"""
return db.add_article(product_names, category, keywords, summary, content, source, url)
# 全局搜索服务实例
search_service = SearchService()