修复搜索和重复检查问题
1. 搜索文章时增加URL字段搜索,解决重复抓取问题 2. 优化搜索结果解析,多获取几个结果确保返回足够数量
This commit is contained in:
@@ -157,12 +157,17 @@ class SearchService:
|
||||
if match:
|
||||
title = match.group(1)
|
||||
ref = match.group(2)
|
||||
# 放宽过滤条件:只要不是纯域名格式就保留
|
||||
if not (title.endswith('.com') or title.endswith('.cn') or title.endswith('.net')):
|
||||
# 过滤纯域名格式
|
||||
is_domain = (title.endswith('.com') or title.endswith('.cn') or
|
||||
title.endswith('.net') or title.endswith('.org') or
|
||||
title.endswith('.edu') or title.endswith('.gov'))
|
||||
if not is_domain:
|
||||
refs.append((title, ref))
|
||||
|
||||
# 获取每个结果的 URL
|
||||
for title, ref in refs[:max_results]:
|
||||
# 获取每个结果的 URL,多获取几个以防解析失败
|
||||
for title, ref in refs[:max_results + 5]:
|
||||
if len(results) >= max_results:
|
||||
break
|
||||
url = self._get_link_url(ref)
|
||||
if url and 'bing.com/search' not in url: # 过滤搜索结果页本身的链接
|
||||
results.append({
|
||||
|
||||
Reference in New Issue
Block a user