From 4177acf75c8bf8e67f9d8776490ce3b0f6cde7e0 Mon Sep 17 00:00:00 2001 From: hz4th_coder Date: Wed, 15 Jul 2026 16:42:19 +0800 Subject: [PATCH] =?UTF-8?q?refactor:=20=E6=AD=A5=E9=AA=A45=E7=AE=80?= =?UTF-8?q?=E5=8C=96=E4=B8=BA=E6=95=B0=E6=8D=AE=E7=94=9F=E6=88=90+?= =?UTF-8?q?=E6=A0=BC=E5=BC=8F=E6=A3=80=E6=9F=A5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 步骤5不再提交到ParamHub,只生成产品数据 - 智能体任务模板去掉提交部分,增加格式检查要求 - 新增本地格式验证方法_validate_product_data - 验证字段类型、必填项、数值范围等 - 步骤6恢复为独立的提交审核步骤 - 更新步骤描述为'调用智能体生成产品数据并检查格式' --- config/agent_fill_fields_template.txt | 50 ++++---- services/process_monitor.py | 159 +++++++++++++++++--------- 2 files changed, 126 insertions(+), 83 deletions(-) diff --git a/config/agent_fill_fields_template.txt b/config/agent_fill_fields_template.txt index 2b4979a..89b3abf 100644 --- a/config/agent_fill_fields_template.txt +++ b/config/agent_fill_fields_template.txt @@ -17,9 +17,9 @@ 首先,请访问 ParamHub API 文档获取对应类别的字段定义: - API文档地址:http://192.168.2.8:12007/hz4th_coder/param-hub-python/src/branch/master/API.md - 根据产品类别({{category}})确定应该使用哪个API: - - AI模型 → `/api/models` - - GPU → `/api/gpus` - - CPU → `/api/cpus` + - AI模型 → `/api/models`,字段包括:name, organization, parameters, context_length, mmlu, publish_date, visible, is_pinned + - GPU → `/api/gpus`,字段包括:name, manufacturer, memory_gb, cuda_cores, tensor_cores, price_usd, release_year, visible, is_pinned + - CPU → `/api/cpus`,字段包括:name, manufacturer, cores, threads, base_clock, boost_clock, price_usd, visible, is_pinned - 其他动态分类 → `/api/items/{category_id}` ### 2. 从内容库获取数据内容 @@ -28,39 +28,32 @@ ### 3. 整理产品参数 根据获取到的内容,提取并整理产品的各项参数,严格按照API文档中定义的字段格式填充。 -### 4. 提交到审核系统 -使用以下命令提交到ParamHub审核系统: - -```bash -# 先登录获取cookie -curl -c /tmp/paramhub_cookie.txt -X POST "http://localhost:16041/login" \ - -H "Content-Type: application/json" \ - -d '{"password": "admin123"}' - -# 然后提交产品数据(根据类别选择对应的API) -curl -b /tmp/paramhub_cookie.txt -X POST "http://localhost:16041/api/{对应类别API}" \ - -H "Content-Type: application/json" \ - -d '{ - "name": "{{product_name}}", - "visible": true, - "is_pinned": false, - // ... 其他字段根据API文档填充 - }' -``` +### 4. 格式检查 +对生成的数据进行以下检查: +- 必填字段是否齐全(name必须有值) +- 字段类型是否正确(数字字段不能是字符串,布尔字段必须是true/false) +- 字段值是否合理(如参数量应为正数,价格应为正数等) +- 如果发现格式问题,请修正后重新输出 ### 5. 输出要求 -请以JSON格式输出执行结果: +请以JSON格式输出最终的产品数据(不要提交,只输出数据): ```json { "success": true, - "review_id": "审核ID(如果提交成功)", - "submitted_data": { + "product_data": { "name": "产品名称", "field1": "值1", - "field2": "值2" + "field2": "值2", + "visible": true, + "is_pinned": false }, "data_sources": [数据ID列表], - "message": "执行说明" + "format_check": { + "passed": true, + "issues_found": [], + "issues_fixed": [] + }, + "message": "数据生成说明" } ``` @@ -68,4 +61,5 @@ curl -b /tmp/paramhub_cookie.txt -X POST "http://localhost:16041/api/{对应类 - 严格按照API文档的字段定义填充数据 - 不要编造或推测任何参数,只使用内容库中实际存在的信息 - 如果某些字段无法从内容中提取,可以留空或填写默认值 -- 提交成功后记录返回的review_id +- 不要执行任何提交操作,只生成并输出数据 +- 确保输出的JSON格式正确,可以被程序解析 diff --git a/services/process_monitor.py b/services/process_monitor.py index 8663596..36a5bb3 100644 --- a/services/process_monitor.py +++ b/services/process_monitor.py @@ -21,7 +21,7 @@ PROCESS_STEPS = [ {'num': 2, 'name': '搜索互联网', 'description': '从互联网搜索最新数据'}, {'num': 3, 'name': '抓取网页内容', 'description': '抓取搜索结果网页的详细内容'}, {'num': 4, 'name': '提取产品数据(智能体)', 'description': '调用hz4th_editor智能体提取产品相关内容'}, - {'num': 5, 'name': '填充字段(智能体)', 'description': '调用hz4th_editor智能体整理产品数据并提交审核'}, + {'num': 5, 'name': '填充字段(智能体)', 'description': '调用智能体生成产品数据并检查格式'}, {'num': 6, 'name': '提交审核', 'description': '提交到ParamHub待审核区'}, ] @@ -227,9 +227,9 @@ class ProcessMonitor: except Exception as e: self._fail_step(session_id, 4, str(e)) - # 步骤5: 填充字段并提交审核(调用智能体执行) + # 步骤5: 填充字段(调用智能体生成数据并检查格式) if not self._check_pause(session_id) and all_data['extracted_data']: - self._start_step(session_id, product_name, 5, '填充字段并提交(智能体)') + self._start_step(session_id, product_name, 5, '填充字段(智能体)') try: # 构建任务文本 fill_task_text = self._build_fill_fields_task( @@ -243,60 +243,42 @@ class ProcessMonitor: fill_parsed = self._parse_fill_agent_response(fill_agent_result.get('output', '')) if fill_parsed and fill_parsed.get('success'): - review_id = fill_parsed.get('review_id') - submitted_data = fill_parsed.get('submitted_data', {}) + product_data = fill_parsed.get('product_data', {}) + format_check = fill_parsed.get('format_check', {}) - all_data['filled_data'] = submitted_data + # 本地格式验证 + validation_result = self._validate_product_data(product_data, category) - self._complete_step(session_id, 5, { - 'filled': True, - 'agent': 'hz4th_editor', - 'task_text': fill_task_text, - 'review_id': review_id, - 'submitted_data': submitted_data, - 'agent_output': fill_agent_result.get('output', '')[:2000] - }) - - # 如果智能体返回了review_id,直接标记步骤6完成 - if review_id: - result['success'] = True - result['review_id'] = review_id + if validation_result.get('valid'): + all_data['filled_data'] = product_data - db.update_session_status(session_id, 'completed', - review_id=review_id, - result=json.dumps(result, ensure_ascii=False)) - - db.add_process_history( - product_name=product_name, - category=category, - subcategory=subcategory, - status='submitted', - review_id=review_id, - details=all_data - ) - logger.info(f"[{session_id}] 步骤5完成: 智能体提交成功, review_id={review_id}") - else: self._complete_step(session_id, 5, { 'filled': True, 'agent': 'hz4th_editor', 'task_text': fill_task_text, - 'submitted_data': submitted_data, - 'message': '数据已整理但未获取到review_id', + 'product_data': product_data, + 'format_check': format_check, + 'validation': validation_result, 'agent_output': fill_agent_result.get('output', '')[:2000] }) - result['message'] = '数据已整理但提交状态未知' + logger.info(f"[{session_id}] 步骤5完成: 数据生成成功,格式验证通过") + else: + # 格式验证失败,记录问题 + self._fail_step(session_id, 5, f"数据格式验证失败: {validation_result.get('errors', [])}") + result['message'] = '数据格式验证失败' else: - self._fail_step(session_id, 5, f"智能体执行失败: {fill_parsed.get('message', '未知错误') if fill_parsed else '解析失败'}") - result['message'] = f'智能体执行失败' + error_msg = fill_parsed.get('message', '未知错误') if fill_parsed else '解析失败' + self._fail_step(session_id, 5, f"智能体执行失败: {error_msg}") + result['message'] = f'智能体执行失败: {error_msg}' else: self._fail_step(session_id, 5, f"智能体调用失败: {fill_agent_result.get('error', '未知错误')}") result['message'] = f'智能体调用失败: {fill_agent_result.get("error")}' except Exception as e: self._fail_step(session_id, 5, str(e)) - # 步骤6: 确认提交结果(如果步骤5未获取到review_id,尝试本地提交) - if not self._check_pause(session_id) and all_data['filled_data'] and not result.get('review_id'): - self._start_step(session_id, product_name, 6, '确认提交结果') + # 步骤6: 提交审核 + if not self._check_pause(session_id) and all_data['filled_data']: + self._start_step(session_id, product_name, 6, '提交审核') try: category_type = self._get_category_type(category) success, review_id_or_error = paramhub_client.submit_for_review( @@ -322,7 +304,7 @@ class ProcessMonitor: review_id=review_id_or_error, details=all_data ) - logger.info(f"[{session_id}] 步骤6完成: 本地提交成功") + logger.info(f"[{session_id}] 步骤6完成: 提交成功") else: self._fail_step(session_id, 6, review_id_or_error) db.update_session_status(session_id, 'failed') @@ -616,26 +598,93 @@ class ProcessMonitor: if parsed_data: return { 'success': parsed_data.get('success', False), - 'review_id': parsed_data.get('review_id'), - 'submitted_data': parsed_data.get('submitted_data', {}), + 'product_data': parsed_data.get('product_data', {}), 'data_sources': parsed_data.get('data_sources', []), + 'format_check': parsed_data.get('format_check', {}), 'message': parsed_data.get('message', ''), 'raw_output': output } - # 尝试从文本中提取review_id - review_match = re.search(r'review[_-]?id[\s:]*([\w-]+)', output, re.I) - if review_match: - return { - 'success': True, - 'review_id': review_match.group(1), - 'submitted_data': {}, - 'message': '从输出中提取到review_id', - 'raw_output': output - } - return None + def _validate_product_data(self, product_data, category): + """本地验证产品数据格式""" + errors = [] + warnings = [] + + if not product_data: + return {'valid': False, 'errors': ['数据为空'], 'warnings': []} + + # 检查必填字段 + if not product_data.get('name'): + errors.append('缺少必填字段: name') + + # 检查字段类型 + category_type = self._get_category_type(category) + + if category_type == 'model': + # AI模型字段验证 + if 'parameters' in product_data and product_data['parameters']: + params = product_data['parameters'] + if not isinstance(params, str) or not params.endswith('B'): + warnings.append('parameters应为字符串格式如"70B"') + + if 'context_length' in product_data and product_data['context_length']: + ctx = product_data['context_length'] + if not isinstance(ctx, int) or ctx <= 0: + errors.append('context_length应为正整数') + + if 'mmlu' in product_data and product_data['mmlu']: + mmlu = product_data['mmlu'] + if not isinstance(mmlu, (int, float)) or mmlu < 0 or mmlu > 100: + warnings.append('mmlu应为0-100之间的数值') + + elif category_type == 'gpu': + # GPU字段验证 + if 'memory_gb' in product_data and product_data['memory_gb']: + mem = product_data['memory_gb'] + if not isinstance(mem, (int, float)) or mem <= 0: + errors.append('memory_gb应为正数') + + if 'cuda_cores' in product_data and product_data['cuda_cores']: + cores = product_data['cuda_cores'] + if not isinstance(cores, int) or cores <= 0: + errors.append('cuda_cores应为正整数') + + if 'price_usd' in product_data and product_data['price_usd']: + price = product_data['price_usd'] + if not isinstance(price, (int, float)) or price <= 0: + warnings.append('price_usd应为正数') + + elif category_type == 'cpu': + # CPU字段验证 + if 'cores' in product_data and product_data['cores']: + cores = product_data['cores'] + if not isinstance(cores, int) or cores <= 0: + errors.append('cores应为正整数') + + if 'threads' in product_data and product_data['threads']: + threads = product_data['threads'] + if not isinstance(threads, int) or threads <= 0: + errors.append('threads应为正整数') + + if 'base_clock' in product_data and product_data['base_clock']: + clock = product_data['base_clock'] + if not isinstance(clock, (int, float)) or clock <= 0: + errors.append('base_clock应为正数') + + # 检查布尔字段 + for bool_field in ['visible', 'is_pinned']: + if bool_field in product_data: + if not isinstance(product_data[bool_field], bool): + warnings.append(f'{bool_field}应为布尔值') + + return { + 'valid': len(errors) == 0, + 'errors': errors, + 'warnings': warnings + } + def _extract_data(self, product_name, all_data): """提取产品数据(备用,已被智能体替代)""" all_content = []