| """ |
| JSON内容提取核心逻辑 |
| """ |
|
|
| import json |
| import logging |
| import os |
| import glob |
| from typing import List, Any |
|
|
| logger = logging.getLogger(__name__) |
|
|
|
|
| class JsonContentExtractorCore: |
| """从JSON文件中提取content字段的核心逻辑""" |
|
|
| def _is_conversation_format(self, json_data: Any) -> bool: |
| """ |
| 检测JSON是否为OpenAI对话格式 |
| |
| Args: |
| json_data: JSON数据 |
| |
| Returns: |
| 是否为对话格式 |
| """ |
| if not isinstance(json_data, dict): |
| return False |
|
|
| |
| if "messages" not in json_data: |
| return False |
|
|
| messages = json_data["messages"] |
| if not isinstance(messages, list) or len(messages) == 0: |
| return False |
|
|
| |
| first_msg = messages[0] |
| if isinstance(first_msg, dict) and "role" in first_msg and "content" in first_msg: |
| return True |
|
|
| return False |
|
|
| def _extract_conversation(self, json_data: dict) -> List[dict]: |
| """ |
| 从OpenAI对话格式中提取消息 |
| |
| Args: |
| json_data: 对话格式的JSON数据 |
| |
| Returns: |
| 消息列表,每个元素包含role和content |
| """ |
| messages = json_data.get("messages", []) |
| result = [] |
|
|
| for msg in messages: |
| if not isinstance(msg, dict): |
| continue |
|
|
| role = msg.get("role", "unknown") |
| content = msg.get("content", "") |
|
|
| |
| if isinstance(content, list): |
| formatted_parts = [] |
| for item in content: |
| if not isinstance(item, dict): |
| if isinstance(item, str): |
| formatted_parts.append(item) |
| continue |
|
|
| item_type = item.get("type", "") |
|
|
| |
| if item_type == "text": |
| text = item.get("text", "") |
| if text.strip(): |
| formatted_parts.append(text.strip()) |
|
|
| |
| elif item_type == "thinking": |
| thinking = item.get("thinking", "") |
| if thinking.strip(): |
| formatted_parts.append(f"[思考] {thinking.strip()}") |
|
|
| |
| elif item_type == "tool_use": |
| tool_name = item.get("name", "未知工具") |
| tool_input = item.get("input", {}) |
| |
| tool_str = f"[调用工具: {tool_name}]" |
| if tool_input and isinstance(tool_input, dict): |
| |
| params = [] |
| for k, v in tool_input.items(): |
| if isinstance(v, str): |
| |
| if "\n" in v or len(v) > 80: |
| params.append(f"{k}:\n {v}") |
| else: |
| params.append(f"{k}={v}") |
| else: |
| params.append(f"{k}={v}") |
| tool_str += "\n" + "\n".join(params) |
| formatted_parts.append(tool_str) |
|
|
| |
| elif item_type == "tool_result": |
| tool_result_content = item.get("content", "") |
| is_error = item.get("is_error", False) |
| if isinstance(tool_result_content, str): |
| if is_error: |
| formatted_parts.append(f"[工具错误] {tool_result_content}") |
| else: |
| |
| if len(tool_result_content) > 500: |
| tool_result_content = tool_result_content[:500] + "..." |
| formatted_parts.append(f"[工具结果] {tool_result_content}") |
|
|
| content = "\n".join(formatted_parts) |
| elif isinstance(content, str): |
| content = content.strip() |
| else: |
| content = "" |
|
|
| if content: |
| result.append({ |
| "role": role, |
| "content": content |
| }) |
|
|
| return result |
|
|
| def extract_content_from_json(self, json_data: Any) -> List[str]: |
| """ |
| 从JSON数据中递归提取所有content字段的内容 |
| |
| Args: |
| json_data: JSON数据 |
| |
| Returns: |
| 提取的content内容列表 |
| """ |
| contents = [] |
|
|
| if isinstance(json_data, dict): |
| |
| if "content" in json_data: |
| content = json_data["content"] |
| if isinstance(content, str): |
| contents.append(content) |
| elif isinstance(content, list): |
| |
| for item in content: |
| if isinstance(item, dict): |
| |
| if "text" in item: |
| contents.append(item["text"]) |
| |
| contents.extend(self.extract_content_from_json(item)) |
| elif isinstance(item, str): |
| contents.append(item) |
|
|
| |
| for key, value in json_data.items(): |
| if key != "content": |
| contents.extend(self.extract_content_from_json(value)) |
|
|
| elif isinstance(json_data, list): |
| for item in json_data: |
| contents.extend(self.extract_content_from_json(item)) |
|
|
| return contents |
|
|
| def format_content(self, content: str) -> str: |
| """ |
| 格式化content内容,处理换行等转义字符 |
| |
| Args: |
| content: 原始content内容 |
| |
| Returns: |
| 格式化后的内容 |
| """ |
| |
| formatted = content.replace("\n", "\n") |
| |
| formatted = formatted.replace("\t", "\t") |
| formatted = formatted.replace('\\"', '"') |
| return formatted |
|
|
| def extract_content_from_file(self, file_path: str) -> List[str]: |
| """ |
| 从JSON文件中提取所有content字段的内容 |
| |
| Args: |
| file_path: JSON文件路径 |
| |
| Returns: |
| 提取的content内容列表 |
| """ |
| try: |
| with open(file_path, 'r', encoding='utf-8') as f: |
| json_data = json.load(f) |
|
|
| return self.extract_content_from_json(json_data) |
|
|
| except Exception as e: |
| logger.error(f"读取文件 {file_path} 时出错: {str(e)}") |
| return [] |
|
|
| def process_json_file(self, file_path: str) -> str: |
| """ |
| 处理单个JSON文件,返回格式化的内容 |
| |
| Args: |
| file_path: JSON文件路径 |
| |
| Returns: |
| 格式化的内容字符串 |
| """ |
| try: |
| with open(file_path, 'r', encoding='utf-8') as f: |
| json_data = json.load(f) |
|
|
| |
| if self._is_conversation_format(json_data): |
| return self._format_conversation(json_data) |
|
|
| |
| contents = self.extract_content_from_json(json_data) |
|
|
| if not contents: |
| return f"文件 {file_path} 中未找到content内容\n" |
|
|
| result = [] |
| for i, content in enumerate(contents, 1): |
| formatted = self.format_content(content) |
| result.append(f"=== Content {i} ===\n{formatted}\n") |
|
|
| return "\n".join(result) |
|
|
| except Exception as e: |
| logger.error(f"处理文件 {file_path} 时出错: {str(e)}") |
| return f"处理文件 {file_path} 时出错: {str(e)}\n" |
|
|
| def _format_conversation(self, json_data: dict) -> str: |
| """ |
| 将对话格式的JSON格式化为可读的对话记录 |
| |
| Args: |
| json_data: 对话格式的JSON数据 |
| |
| Returns: |
| 格式化的对话记录 |
| """ |
| messages = self._extract_conversation(json_data) |
|
|
| if not messages: |
| return "未找到对话消息\n" |
|
|
| result = [] |
| for msg in messages: |
| role = msg["role"] |
| content = msg["content"] |
|
|
| |
| role_display = { |
| "system": "系统", |
| "user": "用户", |
| "assistant": "助手" |
| }.get(role, role) |
|
|
| |
| lines = content.split("\n") |
| formatted_lines = [] |
| for i, line in enumerate(lines): |
| if i == 0: |
| formatted_lines.append(line) |
| else: |
| |
| formatted_lines.append(" " + line) |
| formatted_content = "\n".join(formatted_lines) |
|
|
| result.append(f"[{role_display}] {formatted_content}\n") |
|
|
| return "\n".join(result) |
|
|
| def process_directory(self, dir_path: str, output_file: str = None) -> str: |
| """ |
| 处理目录中的所有JSON文件 |
| |
| Args: |
| dir_path: 目录路径 |
| output_file: 输出文件路径(可选) |
| |
| Returns: |
| 所有文件的提取结果 |
| """ |
| if not os.path.exists(dir_path): |
| return f"目录 {dir_path} 不存在\n" |
|
|
| |
| json_files = glob.glob(os.path.join(dir_path, "*.json")) |
|
|
| if not json_files: |
| return f"目录 {dir_path} 中未找到JSON文件\n" |
|
|
| all_results = [] |
| for json_file in sorted(json_files): |
| file_name = os.path.basename(json_file) |
| all_results.append(f"\n{'='*60}") |
| all_results.append(f"文件: {file_name}") |
| all_results.append(f"{'='*60}\n") |
|
|
| result = self.process_json_file(json_file) |
| all_results.append(result) |
|
|
| final_output = "\n".join(all_results) |
|
|
| |
| if output_file: |
| try: |
| with open(output_file, 'w', encoding='utf-8') as f: |
| f.write(final_output) |
| logger.info(f"结果已保存到: {output_file}") |
| except Exception as e: |
| logger.error(f"保存文件时出错: {str(e)}") |
|
|
| return final_output |
|
|
| def process_single_file(self, file_path: str, output_file: str = None) -> str: |
| """ |
| 处理单个文件 |
| |
| Args: |
| file_path: 文件路径 |
| output_file: 输出文件路径(可选) |
| |
| Returns: |
| 提取结果 |
| """ |
| if not os.path.exists(file_path): |
| return f"文件 {file_path} 不存在\n" |
|
|
| result = self.process_json_file(file_path) |
|
|
| |
| if output_file: |
| try: |
| with open(output_file, 'w', encoding='utf-8') as f: |
| f.write(result) |
| logger.info(f"结果已保存到: {output_file}") |
| except Exception as e: |
| logger.error(f"保存文件时出错: {str(e)}") |
|
|
| return result |
|
|
| def extract_conversation_from_json(self, json_data: Any) -> List[dict]: |
| """ |
| 从JSON数据中提取对话消息(结构化) |
| |
| Args: |
| json_data: JSON数据 |
| |
| Returns: |
| 对话消息列表,每个元素包含role和content |
| """ |
| if self._is_conversation_format(json_data): |
| return self._extract_conversation(json_data) |
| return [] |
|
|
| def extract_conversation_from_file(self, file_path: str) -> List[dict]: |
| """ |
| 从文件中提取对话消息(结构化) |
| |
| Args: |
| file_path: JSON文件路径 |
| |
| Returns: |
| 对话消息列表 |
| """ |
| try: |
| with open(file_path, 'r', encoding='utf-8') as f: |
| json_data = json.load(f) |
| return self.extract_conversation_from_json(json_data) |
| except Exception as e: |
| logger.error(f"读取文件 {file_path} 时出错: {str(e)}") |
| return [] |
|
|