import json
import fitz # PyMuPDF
import os
import fitz
from collections import Counter
import uuid
from openai import OpenAI
from kg_build.Prompt import Prompt_compose
from extract_util import safe_json_loads,highlight_pdf_file,get_content_bbox
import re
from typing import List, Dict, Any, Tuple
from modelsAPI.model_api import OpenaiAPI
import asyncio # <-- 新增导入
import re
def is_compose_table_by_first_tr(table_body: str) -> bool:
if not isinstance(table_body, str):
return False
# 非贪婪匹配第一个
...
match = re.search(r']*>(.*?)
', table_body, re.IGNORECASE | re.DOTALL)
if not match:
return False
first_tr_html = match.group(1)
# 去除所有 HTML 标签,只保留纯文本
text_in_first_tr = re.sub(r'<[^>]+>', '', first_tr_html)
# 统一转为小写便于匹配
text_lower = text_in_first_tr.lower()
# 定义多组关键词组合(任意一组满足即可)
keyword_groups = [
["序号", "组成部分", "功能", "数量"],
["序号", "组成部分", "功能"]
]
# 检查是否存在至少一组关键词全部出现在首行中
for group in keyword_groups:
if all(kw in text_lower for kw in group):
return True
return False
async def get_compose_chunk_bbox_and_removed_data(data):
"""
返回:
- 符合条件的组成表分组(每组 [前前, 前, 当前])
- 从原始 data 中移除了所有相关记录后的新 data 列表
"""
record_map = {item["id"]: item for item in data}
results = []
ids_to_remove = set()
for ins in data:
if ins.get("type") == "table":
table_body = ins.get("table_body", "")
# 使用关键词组合判断是否为“组成表”
if is_compose_table_by_first_tr(table_body):
current_id = ins["id"]
prev_ids = [current_id - 2, current_id - 1]
# 收集要移除的 ID(前两条 + 当前)
group_ids = [pid for pid in prev_ids if pid >= 1] + [current_id]
ids_to_remove.update(group_ids)
# 构建分组
group = [record_map.get(pid) for pid in prev_ids if pid >= 1] + [ins]
results.append(group)
# 过滤原始 data
filtered_data = [item for item in data if item["id"] not in ids_to_remove]
return results, filtered_data
# def openai_chat(query: str, model="Qwen3-32B"):
# """调用OpenAI API"""
# try:
# client = OpenAI(
# api_key="none",
# base_url="http://192.168.0.46:59800/v1"
# )
# response = client.chat.completions.create(
# model=model,
# messages=[
# {"role": "user", "content": query},
# ],
# temperature=0.1,
# stream=False,
# response_format={"type": "json_object"},
# extra_body={"chat_template_kwargs": {"enable_thinking": False}}
# )
# return response.choices[0].message.content
# except Exception as e:
# print(f"调用OpenAI API时出错: {e}")
# return None
async def extract_compose_table_entity(chunk):
"""
提取组成表的设备、零部件实体以及关系
"""
print("提取组成表实体和关系")
# 建议:将 all_results 改为复数形式以符合习惯
final_prompt = Prompt_compose.format(text=chunk)
raw_response = await OpenaiAPI.openai_chat_aysnc(final_prompt,timeout=480)
if raw_response is None:
print(f" ⚠️ 切片 API 调用失败")
# 尝试解析 JSON
try:
cleaned = raw_response.strip()
if cleaned.startswith("```json"):
cleaned = cleaned[7:].lstrip()
if cleaned.endswith("```"):
cleaned = cleaned[:-3].rstrip()
json_result = safe_json_loads(cleaned)
# 打印美化版
# print(json.dumps(json_result, ensure_ascii=False, indent=2))
except Exception as e:
print(f" ⚠️切片 JSON 解析失败: {e}")
print(f" 原始响应: {repr(raw_response[:500])}")
json_result = []
# 返回按切片分组的结果列表
return json_result
async def get_compose_node_relation(input_pdf_path,data,prefix_url = "http://192.168.0.46:59085"):
for idx, item in enumerate(data, start=1):
item["id"] = idx
results, filtered_data =await get_compose_chunk_bbox_and_removed_data(data=data)
print(f"抽取的组成表总数量为:{len(results)}")
if not results:
print("未找到符合条件的组成表格,跳过处理。")
all_entity_results,all_relation_results = [],[]
else:
all_entity_results,all_relation_results = [],[] # 用于收集所有成功提取的结果
for i, compose in enumerate(results):
try:
# 获取内容和高亮信息
print(f"开始处理第{i+1}个组成表")
content, highlight_list = get_content_bbox(compose)
print("组成表切片内容为:",content)
print(len(content))
# 提取实体
result =await extract_compose_table_entity(content)
if result is None:
print(f"第 {i+1} 组:实体提取失败,跳过")
continue
# 安全访问 entities 字段
entities = result.get("entities", [])
if not isinstance(entities, list):
print(f"第 {i+1} 组:entities 字段格式异常,跳过")
continue
relationships = result.get("relationships", [])
# 为每个 entity 添加 knowledge_source
highlight_filename, most_common_page,filename = highlight_pdf_file(input_pdf_path,highlight_list)
url = f"{prefix_url}/upload/{highlight_filename}#page={most_common_page}"
# for entity in entities:
# knowledge_source = [{
# "filename": filename,
# "info": content,
# "url": url
# }]
# # 确保 properties 存在
# if "properties" not in entity or not isinstance(entity["properties"], dict):
# entity["properties"] = {}
# entity["properties"]["knowledge_source"] = knowledge_source
for entity in entities:
# 构造原始数据对象 (List of Dict)
source_data_list = [{
"filename": filename,
"info": content,
"url": url
}]
# 2. 【关键修改】将列表序列化为 JSON 字符串
# ensure_ascii=False 确保中文内容不会被转义成 \uXXXX
knowledge_source_str = json.dumps(source_data_list, ensure_ascii=False)
# 确保 properties 存在且是字典
if "properties" not in entity or not isinstance(entity["properties"], dict):
entity["properties"] = {}
# 3. 赋值字符串而不是对象列表
entity["properties"]["knowledge_source"] = knowledge_source_str
# 更新 result 中的 entities
all_entity_results.extend(entities)
all_relation_results.extend(relationships)
except Exception as e:
print(f"处理第 {i+1} 组时发生未预期错误: {e}")
continue # 跳过当前组,继续下一轮
return all_entity_results,all_relation_results
# if __name__ == "__main__":
# file_path = '/storage01/home/hdf/project/wxkgrag/kgextract/output/output1/122-06A0014-B01001_发动机-维修手册_content_list.json'
# input_pdf_path = "/storage01/home/hdf/project/agent/测试.pdf"
# output_json_path = '/storage01/home/hdf/project/wxkgrag/kgextract/output/output2/组成表抽取结果.json' # 你可以自定义路径
# try:
# with open(file_path, 'r', encoding='utf-8') as file:
# data = json.load(file)
# except Exception as e:
# print(f"读取 JSON 文件时出错:{e}")
# data = []
# all_entity_results, all_relation_results = asyncio.run(
# get_compose_node_relation(input_pdf_path, data=data)
# )
# length_entity = len(all_entity_results)
# length_relation = len(all_relation_results)
# print(f"抽取{length_entity}个实体,抽取{length_relation}个关系")
# result = {
# "entities": all_entity_results,
# "relations": all_relation_results
# }
# # 保存为 JSON 文件
# try:
# with open(output_json_path, 'w', encoding='utf-8') as f:
# json.dump(result, f, ensure_ascii=False, indent=4)
# print(f"结果已保存至:{output_json_path}")
# except Exception as e:
# print(f"保存 JSON 文件时出错:{e}")