优化切片,增加章节图片信息
This commit is contained in:
parent
129846dc17
commit
325fed5b2d
107
chunk_text.py
107
chunk_text.py
@ -1255,7 +1255,7 @@ def find_ship_info_by_hull(json_file_path, data):
|
||||
except json.JSONDecodeError:
|
||||
print("错误:JSON 文件格式不正确")
|
||||
return None
|
||||
def merge_short_slices(slices, min_length=30,filename="122-06A0014-B01003_雷达-使用说明书.pdf"):
|
||||
# def merge_short_slices(slices, min_length=30,filename="122-06A0014-B01003_雷达-使用说明书.pdf"):
|
||||
"""
|
||||
合并过短的切片:
|
||||
- 如果某切片 content 长度 <= min_length,则将其合并到下一个切片的开头
|
||||
@ -1351,6 +1351,111 @@ def merge_short_slices(slices, min_length=30,filename="122-06A0014-B01003_雷达
|
||||
ins['content'] = content[:newline_idx] + suffix + content[newline_idx:]
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def merge_short_slices(slices, min_length=30, filename="122-06A0014-B01003_雷达-使用说明书.pdf"):
|
||||
"""
|
||||
合并过短的切片:
|
||||
- 如果某切片 content 长度 <= min_length,则将其合并到下一个切片的开头
|
||||
- 若处于末尾无下一个切片,则反向合并到上一个切片末尾
|
||||
- content 用换行拼接,positions 顺序拼接
|
||||
|
||||
Args:
|
||||
slices: [{"content": str, "positions": [...]}]
|
||||
min_length: 短切片的字符长度阈值(含)
|
||||
filename: 用于实体提取的文件名
|
||||
|
||||
Returns:
|
||||
合并后的 slices 列表
|
||||
"""
|
||||
if not slices:
|
||||
return slices
|
||||
|
||||
result = []
|
||||
pending_contents = [] # 缓存等待合并到"下一个"的短切片 content
|
||||
pending_positions = [] # 缓存对应的 positions
|
||||
|
||||
for ins in slices:
|
||||
content = ins.get("content", "") or ""
|
||||
positions = ins.get("positions", []) or []
|
||||
|
||||
if len(content) <= min_length:
|
||||
# 暂存,等到下一个正常长度的切片再合并
|
||||
pending_contents.append(content)
|
||||
pending_positions.extend(positions)
|
||||
else:
|
||||
# 正常切片:把暂存的短切片合并到它的开头
|
||||
if pending_contents:
|
||||
merged_prefix = "\n".join(pending_contents)
|
||||
content = merged_prefix + ("\n" if merged_prefix else "") + content
|
||||
positions = pending_positions + positions
|
||||
pending_contents = []
|
||||
pending_positions = []
|
||||
|
||||
result.append({
|
||||
"content": content,
|
||||
"positions": positions
|
||||
})
|
||||
|
||||
# 收尾:如果末尾还有未合并的短切片(后面没有正常切片可合并)
|
||||
# 则反向合并到上一个切片末尾
|
||||
if pending_contents:
|
||||
merged_suffix = "\n".join(pending_contents)
|
||||
if result:
|
||||
last = result[-1]
|
||||
last["content"] = (last["content"] or "") + ("\n" if last["content"] else "") + merged_suffix
|
||||
last["positions"] = (last["positions"] or []) + pending_positions
|
||||
else:
|
||||
# 极端情况:所有切片都很短,整体作为一个切片返回
|
||||
result.append({
|
||||
"content": merged_suffix,
|
||||
"positions": pending_positions
|
||||
})
|
||||
|
||||
try:
|
||||
final_result = get_entity(filename)
|
||||
xinghao = find_ship_info_by_hull(SHIP_MODEL_NAME, final_result)
|
||||
if xinghao is None: # 确保 xinghao 为 None 时不会报错
|
||||
xinghao = {"model_name": "", "ship_name": ""}
|
||||
print("未找到匹配的舰船信息")
|
||||
|
||||
# 安全处理 xinghao 为 None 的情况
|
||||
model_name = ""
|
||||
if xinghao and 'model_name' in xinghao:
|
||||
model_name = xinghao['model_name']
|
||||
|
||||
other_data = format_entity_text(final_result)
|
||||
logger.info(f"文件名实体提取成功: {other_data}")
|
||||
except Exception as e:
|
||||
logger.warning(f"文件名实体提取失败: {e}")
|
||||
other_data = ""
|
||||
model_name = "" # 确保异常时 model_name 有定义
|
||||
|
||||
# 将提取的信息(如舰艇名、型号)注入到每个切片的开头
|
||||
for ins in result:
|
||||
content = ins['content']
|
||||
# 如果没有提取到有效数据,则跳过注入
|
||||
|
||||
newline_idx = content.find('\n')
|
||||
info = other_data
|
||||
if model_name:
|
||||
info += f',型号为{model_name}'
|
||||
|
||||
# ===== 修改部分:拼接文件名 + 实体信息 =====
|
||||
# 构建前缀:#文件名为:xxx.pdf(实体信息)
|
||||
filename_prefix = f"#文件名为:{filename}"
|
||||
if info:
|
||||
filename_prefix += f"({info})"
|
||||
# 注意:这里使用全角右括号")"以匹配你的示例,如需半角请改为 ")"
|
||||
|
||||
# 将前缀插入到 content 的最开头
|
||||
if content:
|
||||
ins['content'] = filename_prefix + "\n" + content
|
||||
else:
|
||||
ins['content'] = filename_prefix
|
||||
# ==========================================
|
||||
|
||||
return result
|
||||
def find_and_read_content_list(directory, original_filename, encoding='utf-8'):
|
||||
"""
|
||||
根据原始文件名,在指定目录及其子目录下查找对应的 _content_list.json 文件并返回内容。
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user