From 47b3a27812fda501a34fbf1a9d0f6b429c5e668c Mon Sep 17 00:00:00 2001 From: Defeng Date: Tue, 7 Jul 2026 17:45:40 +0800 Subject: [PATCH] =?UTF-8?q?=E6=9B=B4=E6=96=B0=20chunk=5Ftext.py?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 增加型号切片的异常错误捕捉 --- chunk_text.py | 43 +++++++++++++++++++++++++++---------------- 1 file changed, 27 insertions(+), 16 deletions(-) diff --git a/chunk_text.py b/chunk_text.py index 9a8c9cb..a3406eb 100644 --- a/chunk_text.py +++ b/chunk_text.py @@ -898,8 +898,8 @@ def find_ship_info_by_hull(json_file_path, data): except json.JSONDecodeError: print("错误:JSON 文件格式不正确") return None -def merge_short_slices(slices, min_length=30,filename="163-06A0014-B01001_发动机-维修手册.pdf"): - """ +def merge_short_slices(slices, min_length=30,filename="122-06A0014-B01003_雷达-使用说明书.pdf"): + """ 合并过短的切片: - 如果某切片 content 长度 <= min_length,则将其合并到下一个切片的开头 - 若处于末尾无下一个切片,则反向合并到上一个切片末尾 @@ -908,6 +908,8 @@ def merge_short_slices(slices, min_length=30,filename="163-06A0014-B01001_发动 Args: slices: [{"content": str, "positions": [...]}] min_length: 短切片的字符长度阈值(含) + filename: 用于实体提取的文件名 + Returns: 合并后的 slices 列表 """ @@ -915,7 +917,7 @@ def merge_short_slices(slices, min_length=30,filename="163-06A0014-B01001_发动 return slices result = [] - pending_contents = [] # 缓存等待合并到"下一个"的短切片 content + pending_contents = [] # 缓存等待合并到"下一个"的短切片 content pending_positions = [] # 缓存对应的 positions for ins in slices: @@ -934,6 +936,7 @@ def merge_short_slices(slices, min_length=30,filename="163-06A0014-B01001_发动 positions = pending_positions + positions pending_contents = [] pending_positions = [] + result.append({ "content": content, "positions": positions @@ -953,35 +956,43 @@ def merge_short_slices(slices, min_length=30,filename="163-06A0014-B01001_发动 "content": merged_suffix, "positions": pending_positions }) + try: final_result = get_entity(filename) - print(final_result) - xinghao = find_ship_info_by_hull(SHIP_MODEL_NAME,final_result) - if xinghao.get('model_name'): + xinghao = find_ship_info_by_hull(SHIP_MODEL_NAME, final_result) + if xinghao is None: # 确保 xinghao 为 None 时不会报错 + xinghao = {"model_name": "", "ship_name": ""} + print("未找到匹配的舰船信息") + + # 安全处理 xinghao 为 None 的情况 + model_name = "" + if xinghao and 'model_name' in xinghao: model_name = xinghao['model_name'] - else: - model_name = "" - # lower_entity = final_result.get("low_level") - otehr_data = format_entity_text(final_result) - logger.info(f"文件名实体提取成功: {otehr_data}") + + other_data = format_entity_text(final_result) + logger.info(f"文件名实体提取成功: {other_data}") except Exception as e: logger.warning(f"文件名实体提取失败: {e}") - otehr_data = "" # 建议加个兜底,避免后面 NameError + other_data = "" + model_name = "" # 确保异常时 model_name 有定义 + # 将提取的信息(如舰艇名、型号)注入到每个切片的开头 for ins in result: content = ins['content'] - if not otehr_data: - continue + # 如果没有提取到有效数据,则跳过注入 + newline_idx = content.find('\n') - # 拼接成一个括号:otehr_data + 型号为xxx - info = otehr_data + info = other_data if model_name: info += f',型号为{model_name}' suffix = f'({info})' + + # 将信息插入到第一行末尾 if newline_idx == -1: ins['content'] = content + suffix else: ins['content'] = content[:newline_idx] + suffix + content[newline_idx:] + return result def find_and_read_content_list(directory, original_filename, encoding='utf-8'): """