更新 app.py
优化正词问题、10万文档的速度显示、改chunk为1000,且以句号未分割,不纯粹按字数
This commit is contained in:
parent
accd1e5ad2
commit
31a512c483
107
app.py
107
app.py
@ -87,6 +87,7 @@ FastAPI 主 Agent 接口 - 修复重复执行问题
|
|||||||
集成 LangGraph Checkpointer 实现状态持久化
|
集成 LangGraph Checkpointer 实现状态持久化
|
||||||
"""
|
"""
|
||||||
import logging
|
import logging
|
||||||
|
import unicodedata
|
||||||
from time import sleep
|
from time import sleep
|
||||||
import aiofiles
|
import aiofiles
|
||||||
|
|
||||||
@ -143,7 +144,7 @@ REWRITE_TYPE_MAP = {
|
|||||||
"condense": "总结",
|
"condense": "总结",
|
||||||
}
|
}
|
||||||
# /Review 流式并发配置:块大小 5000,最多 5 块并行
|
# /Review 流式并发配置:块大小 5000,最多 5 块并行
|
||||||
REVIEW_CHUNK_SIZE = 5000
|
REVIEW_CHUNK_SIZE = 1000
|
||||||
REVIEW_MAX_CONCURRENCY = 5
|
REVIEW_MAX_CONCURRENCY = 5
|
||||||
REVIEW_SYSTEM_PROMPT = (
|
REVIEW_SYSTEM_PROMPT = (
|
||||||
"你是一个严谨、穷尽式的专业文本校对专家,必须完整检查全文并返回所有符合规则的问题,不能只挑选部分问题。"
|
"你是一个严谨、穷尽式的专业文本校对专家,必须完整检查全文并返回所有符合规则的问题,不能只挑选部分问题。"
|
||||||
@ -456,11 +457,14 @@ def parse_review_output(text: str) -> List[dict]:
|
|||||||
result = []
|
result = []
|
||||||
|
|
||||||
for match in _REVIEW_BLOCK_RE.finditer(text):
|
for match in _REVIEW_BLOCK_RE.finditer(text):
|
||||||
result.append({
|
item = {
|
||||||
"error": match.group("error").strip(),
|
"error": match.group("error").strip(),
|
||||||
"original": match.group("original").strip(),
|
"original": match.group("original").strip(),
|
||||||
"suggestion": match.group("suggestion").strip(),
|
"suggestion": match.group("suggestion").strip(),
|
||||||
})
|
}
|
||||||
|
# 原文字段为空或仅包含空白时,不作为有效校对记录。
|
||||||
|
if item["original"]:
|
||||||
|
result.append(item)
|
||||||
|
|
||||||
return result
|
return result
|
||||||
|
|
||||||
@ -475,7 +479,26 @@ def _split_review_error_types(error: str) -> List[str]:
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
def repair_review_output(model_output: str, source_text: str) -> str:
|
def _is_whitespace_only_change(original: str, suggestion: str) -> bool:
|
||||||
|
"""判断原文与建议是否仅存在空白或末尾标点差异。"""
|
||||||
|
if not original or not suggestion or original == suggestion:
|
||||||
|
return False
|
||||||
|
def normalize(value: str) -> str:
|
||||||
|
# 统一全角/半角字符,避免“(4)”与“(4)”被误判为内容修改。
|
||||||
|
value = unicodedata.normalize("NFKC", str(value))
|
||||||
|
value = re.sub(r"\s+", "", value)
|
||||||
|
# 校对模型有时会给原文补充句末冒号、句号等格式标点;
|
||||||
|
# 若除此之外内容完全一致,则不作为实际错误返回。
|
||||||
|
return re.sub(r"[,。;:、,.!?!?:;]+$", "", value)
|
||||||
|
|
||||||
|
return normalize(original) == normalize(suggestion)
|
||||||
|
|
||||||
|
|
||||||
|
def repair_review_output(
|
||||||
|
model_output: str,
|
||||||
|
source_text: str,
|
||||||
|
positive_words: Optional[Set[str]] = None,
|
||||||
|
) -> str:
|
||||||
"""修复提前改写原文以及修改范围重叠的校对结果。"""
|
"""修复提前改写原文以及修改范围重叠的校对结果。"""
|
||||||
items = parse_review_output(model_output)
|
items = parse_review_output(model_output)
|
||||||
if not items:
|
if not items:
|
||||||
@ -486,6 +509,7 @@ def repair_review_output(model_output: str, source_text: str) -> str:
|
|||||||
if item.get("original")
|
if item.get("original")
|
||||||
and item.get("suggestion")
|
and item.get("suggestion")
|
||||||
and item["original"] != item["suggestion"]
|
and item["original"] != item["suggestion"]
|
||||||
|
and not _is_whitespace_only_change(item["original"], item["suggestion"])
|
||||||
]
|
]
|
||||||
corrections.sort(key=lambda item: len(item["suggestion"]), reverse=True)
|
corrections.sort(key=lambda item: len(item["suggestion"]), reverse=True)
|
||||||
|
|
||||||
@ -560,8 +584,20 @@ def repair_review_output(model_output: str, source_text: str) -> str:
|
|||||||
final_items = [
|
final_items = [
|
||||||
item for index, item in enumerate(repaired_items)
|
item for index, item in enumerate(repaired_items)
|
||||||
if index not in covered_indexes
|
if index not in covered_indexes
|
||||||
|
and item.get("original", "").strip()
|
||||||
|
and item.get("error", "").strip()
|
||||||
|
and not _is_whitespace_only_change(
|
||||||
|
item.get("original", ""), item.get("suggestion", "")
|
||||||
|
)
|
||||||
]
|
]
|
||||||
|
|
||||||
|
# 正词命中的“错别字”记录不视为错误,例如“维休”被定义为正词时,
|
||||||
|
# 不返回“维休”→“维修”的错别字建议。
|
||||||
|
final_items = filter_positive_word_false_positives(
|
||||||
|
final_items,
|
||||||
|
positive_words or set(),
|
||||||
|
)
|
||||||
|
|
||||||
blocks = [
|
blocks = [
|
||||||
"\n".join([
|
"\n".join([
|
||||||
f"错误:{item.get('error', '').strip()}",
|
f"错误:{item.get('error', '').strip()}",
|
||||||
@ -626,6 +662,35 @@ def build_review_stream_require(
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
def split_review_chunks(content: str, max_size: int = REVIEW_CHUNK_SIZE) -> List[str]:
|
||||||
|
"""优先按段落、其次按句末标点切分,单块不超过 max_size。"""
|
||||||
|
if not content:
|
||||||
|
return []
|
||||||
|
if max_size <= 0:
|
||||||
|
raise ValueError("max_size must be greater than zero")
|
||||||
|
|
||||||
|
chunks: List[str] = []
|
||||||
|
start = 0
|
||||||
|
while start < len(content):
|
||||||
|
if len(content) - start <= max_size:
|
||||||
|
chunks.append(content[start:])
|
||||||
|
break
|
||||||
|
|
||||||
|
candidate = content[start:start + max_size]
|
||||||
|
paragraph_matches = list(re.finditer(r"\n\s*", candidate))
|
||||||
|
if paragraph_matches:
|
||||||
|
cut = paragraph_matches[-1].end()
|
||||||
|
else:
|
||||||
|
sentence_matches = list(re.finditer(
|
||||||
|
r"[。!?;.!?;](?:[\"'”’))】》])?\s*", candidate))
|
||||||
|
cut = sentence_matches[-1].end() if sentence_matches else max_size
|
||||||
|
|
||||||
|
chunks.append(content[start:start + cut])
|
||||||
|
start += cut
|
||||||
|
|
||||||
|
return chunks
|
||||||
|
|
||||||
|
|
||||||
async def stream_review_content(
|
async def stream_review_content(
|
||||||
content: str,
|
content: str,
|
||||||
types: List[str],
|
types: List[str],
|
||||||
@ -641,10 +706,7 @@ async def stream_review_content(
|
|||||||
positive_words=positive_words,
|
positive_words=positive_words,
|
||||||
)
|
)
|
||||||
|
|
||||||
chunks = [
|
chunks = split_review_chunks(content, REVIEW_CHUNK_SIZE)
|
||||||
content[i:i + REVIEW_CHUNK_SIZE]
|
|
||||||
for i in range(0, len(content), REVIEW_CHUNK_SIZE)
|
|
||||||
]
|
|
||||||
if not chunks:
|
if not chunks:
|
||||||
return
|
return
|
||||||
|
|
||||||
@ -660,10 +722,14 @@ async def stream_review_content(
|
|||||||
# 累积单块完整输出后再返回:并发块之间不能交错 token,
|
# 累积单块完整输出后再返回:并发块之间不能交错 token,
|
||||||
# 否则会破坏“错误/原文/建议”块结构,导致后端解析错乱。
|
# 否则会破坏“错误/原文/建议”块结构,导致后端解析错乱。
|
||||||
model_output = await call_review_model(prompt)
|
model_output = await call_review_model(prompt)
|
||||||
logger.info("[公文校对] 模型原始完整输出:\n%s", model_output)
|
repaired_output = repair_review_output(
|
||||||
repaired_output = repair_review_output(model_output, chunk)
|
model_output,
|
||||||
if repaired_output != model_output:
|
chunk,
|
||||||
logger.info("[公文校对] 修复后的完整输出:\n%s", repaired_output)
|
positive_words=normalize_words(positive_words),
|
||||||
|
)
|
||||||
|
# 仅打印过滤后的最终保留记录,不打印模型原始输出或中间结果。
|
||||||
|
if repaired_output.strip() and repaired_output.strip() != "###\n###":
|
||||||
|
logger.info("[公文校对] 最终保留记录:\n%s", repaired_output)
|
||||||
return repaired_output
|
return repaired_output
|
||||||
|
|
||||||
tasks = [asyncio.create_task(process_chunk(chunk)) for chunk in chunks]
|
tasks = [asyncio.create_task(process_chunk(chunk)) for chunk in chunks]
|
||||||
@ -704,11 +770,7 @@ async def review_text(
|
|||||||
negative_word_set = normalize_words(negative_words)
|
negative_word_set = normalize_words(negative_words)
|
||||||
positive_word_set = normalize_words(positive_words)
|
positive_word_set = normalize_words(positive_words)
|
||||||
|
|
||||||
chunk_size = REVIEW_CHUNK_SIZE
|
chunks = split_review_chunks(content, REVIEW_CHUNK_SIZE)
|
||||||
chunks = [
|
|
||||||
content[i:i + chunk_size]
|
|
||||||
for i in range(0, len(content), chunk_size)
|
|
||||||
]
|
|
||||||
|
|
||||||
all_items = []
|
all_items = []
|
||||||
|
|
||||||
@ -731,7 +793,14 @@ async def review_text(
|
|||||||
negative_words=negative_word_set,
|
negative_words=negative_word_set,
|
||||||
)
|
)
|
||||||
|
|
||||||
all_items.extend(llm_items)
|
all_items.extend(
|
||||||
|
item for item in llm_items
|
||||||
|
if item.get("original", "").strip()
|
||||||
|
and item.get("error", "").strip()
|
||||||
|
and not _is_whitespace_only_change(
|
||||||
|
item.get("original", ""), item.get("suggestion", "")
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
return dedupe_and_filter(all_items, content)
|
return dedupe_and_filter(all_items, content)
|
||||||
|
|
||||||
@ -1699,4 +1768,4 @@ async def root():
|
|||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
import uvicorn
|
import uvicorn
|
||||||
|
|
||||||
uvicorn.run(app, host="0.0.0.0", port=9090)
|
uvicorn.run(app, host="0.0.0.0", port=19092)
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user