From 325fed5b2db990dc029b95d1a0dff61fecd9d189 Mon Sep 17 00:00:00 2001 From: Defeng Date: Mon, 20 Jul 2026 21:11:16 +0800 Subject: [PATCH] =?UTF-8?q?=E4=BC=98=E5=8C=96=E5=88=87=E7=89=87=EF=BC=8C?= =?UTF-8?q?=E5=A2=9E=E5=8A=A0=E7=AB=A0=E8=8A=82=E5=9B=BE=E7=89=87=E4=BF=A1?= =?UTF-8?q?=E6=81=AF?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- chunk_text.py | 107 +++++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 106 insertions(+), 1 deletion(-) diff --git a/chunk_text.py b/chunk_text.py index 42f3e91..8a6239a 100644 --- a/chunk_text.py +++ b/chunk_text.py @@ -1255,7 +1255,7 @@ def find_ship_info_by_hull(json_file_path, data): except json.JSONDecodeError: print("错误:JSON 文件格式不正确") return None -def merge_short_slices(slices, min_length=30,filename="122-06A0014-B01003_雷达-使用说明书.pdf"): +# def merge_short_slices(slices, min_length=30,filename="122-06A0014-B01003_雷达-使用说明书.pdf"): """ 合并过短的切片: - 如果某切片 content 长度 <= min_length,则将其合并到下一个切片的开头 @@ -1351,6 +1351,111 @@ def merge_short_slices(slices, min_length=30,filename="122-06A0014-B01003_雷达 ins['content'] = content[:newline_idx] + suffix + content[newline_idx:] return result + + +def merge_short_slices(slices, min_length=30, filename="122-06A0014-B01003_雷达-使用说明书.pdf"): + """ + 合并过短的切片: + - 如果某切片 content 长度 <= min_length,则将其合并到下一个切片的开头 + - 若处于末尾无下一个切片,则反向合并到上一个切片末尾 + - content 用换行拼接,positions 顺序拼接 + + Args: + slices: [{"content": str, "positions": [...]}] + min_length: 短切片的字符长度阈值(含) + filename: 用于实体提取的文件名 + + Returns: + 合并后的 slices 列表 + """ + if not slices: + return slices + + result = [] + pending_contents = [] # 缓存等待合并到"下一个"的短切片 content + pending_positions = [] # 缓存对应的 positions + + for ins in slices: + content = ins.get("content", "") or "" + positions = ins.get("positions", []) or [] + + if len(content) <= min_length: + # 暂存,等到下一个正常长度的切片再合并 + pending_contents.append(content) + pending_positions.extend(positions) + else: + # 正常切片:把暂存的短切片合并到它的开头 + if pending_contents: + merged_prefix = "\n".join(pending_contents) + content = merged_prefix + ("\n" if merged_prefix else "") + content + positions = pending_positions + positions + pending_contents = [] + pending_positions = [] + + result.append({ + "content": content, + "positions": positions + }) + + # 收尾:如果末尾还有未合并的短切片(后面没有正常切片可合并) + # 则反向合并到上一个切片末尾 + if pending_contents: + merged_suffix = "\n".join(pending_contents) + if result: + last = result[-1] + last["content"] = (last["content"] or "") + ("\n" if last["content"] else "") + merged_suffix + last["positions"] = (last["positions"] or []) + pending_positions + else: + # 极端情况:所有切片都很短,整体作为一个切片返回 + result.append({ + "content": merged_suffix, + "positions": pending_positions + }) + + try: + final_result = get_entity(filename) + xinghao = find_ship_info_by_hull(SHIP_MODEL_NAME, final_result) + if xinghao is None: # 确保 xinghao 为 None 时不会报错 + xinghao = {"model_name": "", "ship_name": ""} + print("未找到匹配的舰船信息") + + # 安全处理 xinghao 为 None 的情况 + model_name = "" + if xinghao and 'model_name' in xinghao: + model_name = xinghao['model_name'] + + other_data = format_entity_text(final_result) + logger.info(f"文件名实体提取成功: {other_data}") + except Exception as e: + logger.warning(f"文件名实体提取失败: {e}") + other_data = "" + model_name = "" # 确保异常时 model_name 有定义 + + # 将提取的信息(如舰艇名、型号)注入到每个切片的开头 + for ins in result: + content = ins['content'] + # 如果没有提取到有效数据,则跳过注入 + + newline_idx = content.find('\n') + info = other_data + if model_name: + info += f',型号为{model_name}' + + # ===== 修改部分:拼接文件名 + 实体信息 ===== + # 构建前缀:#文件名为:xxx.pdf(实体信息) + filename_prefix = f"#文件名为:{filename}" + if info: + filename_prefix += f"({info})" + # 注意:这里使用全角右括号")"以匹配你的示例,如需半角请改为 ")" + + # 将前缀插入到 content 的最开头 + if content: + ins['content'] = filename_prefix + "\n" + content + else: + ins['content'] = filename_prefix + # ========================================== + + return result def find_and_read_content_list(directory, original_filename, encoding='utf-8'): """ 根据原始文件名,在指定目录及其子目录下查找对应的 _content_list.json 文件并返回内容。