khs commited on
Commit
bfa2b47
·
1 Parent(s): f127c23

Replace table with risk-level filter and persist history view

Browse files
README.md CHANGED
@@ -26,29 +26,15 @@ short_description: 'three-switchable AIGC detectors for Chinese papers'
26
  - 默认全文检测(不再区分快速模式)
27
  - 三模型可切换(默认 `paperpass-v3`)
28
  - 推理进度百分比显示
29
- - 页眉页脚、页码、重复行基础清洗(PDF)
30
- - 段落详情显示完整段落(不再截断到 380 字)
 
31
  - 可选实验模式:`mba-aigc-detector`(需本地模型包)
32
 
33
  ## 关于 mba-aigc-detector
34
 
35
  该模型不是单个 `transformers` 分类模型,而是 `RoBERTa 特征 + 多个树模型` 的融合方案。
36
- 因此需要额外的本地模型文件(`models/mba/*.pkl`)才能运行,Hugging Face Space 默认不会自带这些大文件。
37
-
38
- ### 你选的方案1(仓库内置模型包)
39
-
40
- 把以下文件提交到仓库目录 `models/mba/`:
41
-
42
- - `select5_tree_d2_model.pkl`
43
- - `select10_tree_d2_model.pkl`
44
- - `select15_tree_d3_model.pkl`
45
- - `select20_tree_d2_model.pkl`
46
- - `bert_tree_d1_model.pkl`
47
-
48
- 备注:
49
-
50
- - 这些文件通常较大,推荐用 Hugging Face 网页端直接上传到 Space 仓库,避免本地 `git-lfs` 环境问题。
51
- - 上传完成后无需改代码,页面中直接切换到 `mba-aigc-detector(实验版,需本地模型包)` 即可。
52
 
53
  ## 校准
54
 
 
26
  - 默认全文检测(不再区分快速模式)
27
  - 三模型可切换(默认 `paperpass-v3`)
28
  - 推理进度百分比显示
29
+ - PDF 页眉页脚、页码、重复行基础清洗
30
+ - 风险筛选器:`全部 / 高风险 / 中风险 / 低风险`
31
+ - 历史记录保存与查看
32
  - 可选实验模式:`mba-aigc-detector`(需本地模型包)
33
 
34
  ## 关于 mba-aigc-detector
35
 
36
  该模型不是单个 `transformers` 分类模型,而是 `RoBERTa 特征 + 多个树模型` 的融合方案。
37
+ 需要 `models/mba/` 下的真实模型文件;如果上传的是 Git LFS 指针文件,系统会自动提示。
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38
 
39
  ## 校准
40
 
app.py CHANGED
@@ -1,5 +1,7 @@
1
  import json
2
  import re
 
 
3
  from pathlib import Path
4
  from typing import Dict, List, Tuple
5
 
@@ -29,10 +31,12 @@ DEFAULT_MODEL_LABEL = "paperpass-v3(默认,论文场景优先)"
29
 
30
  MIN_PARAGRAPH_CHARS = 80
31
  WINDOW_MAX_LENGTH = 512
32
- WINDOW_STRIDE = 128
 
33
 
34
  CALIBRATION_PATH = Path("calibration/model.json")
35
  MBA_MODELS_DIR = Path("models/mba")
 
36
 
37
  CURRENT_MODEL_NAME = None
38
  CURRENT_TOKENIZER = None
@@ -45,6 +49,12 @@ MBA_STATE = {
45
  "tree_models": {},
46
  }
47
 
 
 
 
 
 
 
48
 
49
  def load_calibration_model() -> Dict:
50
  if not CALIBRATION_PATH.exists():
@@ -62,13 +72,52 @@ def load_calibration_model() -> Dict:
62
  CALIBRATION_MODEL = load_calibration_model()
63
 
64
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
65
  def get_or_load_model(model_name: str):
66
  global CURRENT_MODEL_NAME, CURRENT_TOKENIZER, CURRENT_MODEL
67
  if CURRENT_MODEL_NAME == model_name and CURRENT_TOKENIZER is not None and CURRENT_MODEL is not None:
68
  return CURRENT_TOKENIZER, CURRENT_MODEL
 
69
  tokenizer = AutoTokenizer.from_pretrained(model_name)
70
  model = AutoModelForSequenceClassification.from_pretrained(model_name)
71
  model.eval()
 
72
  CURRENT_MODEL_NAME = model_name
73
  CURRENT_TOKENIZER = tokenizer
74
  CURRENT_MODEL = model
@@ -94,17 +143,15 @@ def is_probable_page_number(line: str) -> bool:
94
 
95
  def clean_common_noise(line: str) -> str:
96
  line = normalize_text(line)
97
- line = re.sub(r"[ ]+", " ", line)
98
- return line
99
 
100
 
101
  def extract_pdf_text(file_path: str) -> Tuple[str, Dict]:
102
  doc = fitz.open(file_path)
103
  all_pages = len(doc)
104
- page_limit = all_pages
105
  page_lines: List[List[str]] = []
106
 
107
- for idx in range(page_limit):
108
  page = doc[idx]
109
  rect = page.rect
110
  top_cut = rect.height * 0.06
@@ -113,42 +160,31 @@ def extract_pdf_text(file_path: str) -> Tuple[str, Dict]:
113
 
114
  lines = []
115
  for b in sorted(blocks, key=lambda x: (round(x[1], 1), round(x[0], 1))):
116
- x0, y0, x1, y1, text, *_ = b
117
  if y1 <= top_cut or y0 >= rect.height - bottom_cut:
118
  continue
119
  for raw in text.splitlines():
120
  line = clean_common_noise(raw)
121
- if not line:
122
- continue
123
- if is_probable_page_number(line):
124
  continue
125
  lines.append(line)
126
  page_lines.append(lines)
127
 
128
- # Remove repeating header/footer lines across many pages.
129
  freq = {}
130
  for lines in page_lines:
131
  if not lines:
132
  continue
133
- cand = set(lines[:2] + lines[-2:])
134
- for c in cand:
135
  if len(c) >= 4:
136
  freq[c] = freq.get(c, 0) + 1
137
-
138
- repeat_lines = {k for k, v in freq.items() if v >= max(3, int(0.4 * page_limit))}
139
 
140
  merged_pages = []
141
  for lines in page_lines:
142
  cleaned = [ln for ln in lines if ln not in repeat_lines and not is_probable_page_number(ln)]
143
  merged_pages.append("\n".join(cleaned))
144
 
145
- text = "\n\n".join(merged_pages)
146
- meta = {
147
- "total_pages": all_pages,
148
- "used_pages": page_limit,
149
- "page_truncated": page_limit < all_pages,
150
- }
151
- return text, meta
152
 
153
 
154
  def extract_docx_text(file_path: str) -> Tuple[str, Dict]:
@@ -218,7 +254,7 @@ def detector_score_transformer(text: str, model_name: str) -> float:
218
  padding=True,
219
  return_tensors="pt",
220
  )
221
- with torch.no_grad():
222
  outputs = model(**inputs)
223
  probs = torch.softmax(outputs.logits, dim=-1)[:, 1].cpu().numpy()
224
  return float(0.75 * np.mean(probs) + 0.25 * np.max(probs))
@@ -261,11 +297,20 @@ def _extract_stat_features(text: str) -> np.ndarray:
261
  ])
262
 
263
 
 
 
 
 
 
 
 
 
264
  def init_mba_pack() -> Tuple[bool, str]:
265
  if MBA_STATE["ready"]:
266
  return True, ""
267
  if joblib is None:
268
  return False, "当前环境缺少 joblib,无法加载 mba-aigc-detector 本地模型包。"
 
269
  needed = [
270
  "select5_tree_d2_model.pkl",
271
  "select10_tree_d2_model.pkl",
@@ -275,12 +320,13 @@ def init_mba_pack() -> Tuple[bool, str]:
275
  ]
276
  missing = [f for f in needed if not (MBA_MODELS_DIR / f).exists()]
277
  if missing:
278
- return (
279
- False,
280
- "缺少 mba 模型文件。请将以下文件放入 models/mba/ : "
281
- + ", ".join(needed)
282
- + f"(当前缺失示例: {', '.join(missing[:2])})",
283
- )
 
284
 
285
  try:
286
  tok = AutoTokenizer.from_pretrained("hfl/chinese-roberta-wwm-ext")
@@ -297,11 +343,12 @@ def detector_score_mba(text: str) -> float:
297
  ok, msg = init_mba_pack()
298
  if not ok:
299
  raise RuntimeError(msg)
 
300
  tok = MBA_STATE["extractor_tokenizer"]
301
  mdl = MBA_STATE["extractor_model"]
302
-
303
  inputs = tok(text[:512], return_tensors="pt", max_length=512, truncation=True, padding=True)
304
- with torch.no_grad():
 
305
  bert_feat = mdl(**inputs).last_hidden_state[:, 0, :].cpu().numpy()[0]
306
 
307
  stat_feat = _extract_stat_features(text)
@@ -320,6 +367,7 @@ def analyze_paragraph(text: str, model_name: str) -> Dict[str, float]:
320
  detector = float(min(max(detector_score_mba(text) * 0.3, 0.0), 1.0))
321
  else:
322
  detector = detector_score_transformer(text, model_name)
 
323
  repetition = calc_repetition(text)
324
  variance = calc_sentence_variance(text)
325
  risk = float(min(max(detector * 0.78 + repetition * 0.12 + (1 - variance) * 0.10, 0.0), 1.0))
@@ -349,33 +397,41 @@ def predict_kn_like_rate(features: Dict[str, float]) -> float:
349
  return clip01(y)
350
 
351
 
352
- def analyze_document(upload_file, pasted_text, model_label, risk_threshold, topk, progress=gr.Progress()):
 
 
 
 
 
 
 
 
 
 
353
  if upload_file is None and not normalize_text(pasted_text or ""):
354
- return "请先上传文件,或粘贴文本。", [], ""
355
 
356
  model_name = MODEL_CHOICES.get(model_label, MODEL_CHOICES[DEFAULT_MODEL_LABEL])
357
  if model_name == "mba_local_pack":
358
  ok, msg = init_mba_pack()
359
  if not ok:
360
- return f"# 当前模型: {model_label}\n\n{msg}\n\n请切回 paperpass-v3 / zhv3 / zhv2。", [], ""
361
 
 
362
  if normalize_text(pasted_text or ""):
363
  raw_text = pasted_text
364
  extract_meta = {"total_pages": None, "used_pages": None, "page_truncated": False}
365
- # Textbox mode: respect user text directly, avoid extra document cleaning.
366
  paragraphs = split_paragraphs(raw_text)
367
  else:
 
368
  raw_text, extract_meta = extract_document_text(upload_file)
369
  paragraphs = [p for p in split_paragraphs(raw_text) if not should_skip_paragraph(p)]
370
- original_count = len(paragraphs)
371
- para_truncated = False
372
 
373
  if not paragraphs:
374
- return "未提取到可分析正文。请尝试文本层可复制的文件,或调整排版后再试。", [], ""
375
 
376
- rows = []
377
- risks = []
378
- details = []
379
  total = len(paragraphs)
380
  for i, p in enumerate(paragraphs, 1):
381
  progress(i / total, desc=f"推理进度: {int(i * 100 / total)}%")
@@ -387,9 +443,16 @@ def analyze_document(upload_file, pasted_text, model_label, risk_threshold, topk
387
  elif score["risk"] > max(0.55, risk_threshold - 0.15):
388
  level = "🟡"
389
 
390
- rows.append([i, f"{score['risk']:.2%}", f"{score['detector']:.2%}", f"{score['repetition']:.2%}", p[:120]])
 
 
 
 
 
391
  details.append(
392
- f"""
 
 
393
  {level} 段落 {i} AI风险: {score['risk']:.2%}
394
 
395
  Detector: {score['detector']:.2%}
@@ -397,44 +460,63 @@ Detector: {score['detector']:.2%}
397
  句式稳定性: {1 - score['variance']:.2%}
398
 
399
  {p}
400
- """
 
401
  )
402
 
403
  f = build_doc_features(risks)
404
- mode_line = "当前模式: 原始风险率(未加载校准模型)" if not CALIBRATION_MODEL else "当前模式: 知网对齐预测率(已加载校准模型)"
 
 
405
 
406
  trunc_info = []
407
- if extract_meta.get("page_truncated"):
408
- trunc_info.append(f"页面截断: 是({extract_meta.get('used_pages')}/{extract_meta.get('total_pages')} 页)")
409
- elif extract_meta.get("total_pages") is not None:
410
  trunc_info.append(f"页面截断: 否({extract_meta.get('used_pages')}/{extract_meta.get('total_pages')} 页)")
411
- trunc_info.append(f"段落截断: {'是' if para_truncated else '否'}(分析 {len(paragraphs)}/{original_count} 段)")
412
 
 
413
  summary = f"""
414
  # 当前模型: {model_label}
415
  # 综合AI风险率: {f['overall']:.2%}
416
- # 预测知网AIGC率: {predict_kn_like_rate(f):.2%}
417
  高风险段落占比: {f['high_ratio']:.2%}
418
  中风险段落占比: {f['mid_ratio']:.2%}
419
  有效段落数: {len(paragraphs)}
420
 
421
- 判定阈值: {risk_threshold:.2f}
422
  {mode_line}
423
  {' | '.join(trunc_info)}
424
 
425
  (说明:该结果为“风险分析与校准预测”,并非官方系统结果)
426
  """
427
 
428
- rows_sorted = sorted(rows, key=lambda r: float(r[1].strip("%")), reverse=True)
429
- top_rows = rows_sorted[: int(topk)]
430
- return summary, top_rows, "\n\n---\n\n".join(details)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
431
 
432
 
433
  with gr.Blocks(theme=gr.themes.Soft(primary_hue="emerald", secondary_hue="slate"), title="论文AIGC风险检测系统") as demo:
434
- gr.Markdown("""
 
435
  # 论文AIGC风险检测系统
436
  支持 `PDF / Word(.docx) / 文本(.txt, .md)`,默认全文检测,支持直接粘贴文本。
437
- """)
 
438
 
439
  with gr.Row():
440
  with gr.Column(scale=1):
@@ -443,24 +525,19 @@ with gr.Blocks(theme=gr.themes.Soft(primary_hue="emerald", secondary_hue="slate"
443
  model = gr.Dropdown(list(MODEL_CHOICES.keys()), value=DEFAULT_MODEL_LABEL, label="选择检测模型")
444
  with gr.Accordion("高级参数", open=False):
445
  risk_threshold = gr.Slider(0.5, 0.9, value=0.75, step=0.01, label="高风险阈值")
446
- topk = gr.Slider(5, 60, value=15, step=1, label="表格显示 Top-K 高风险段落")
447
  run_btn = gr.Button("开始分析", variant="primary")
448
 
449
  with gr.Column(scale=2):
450
  summary_out = gr.Markdown(label="总览")
451
- table_out = gr.Dataframe(
452
- headers=["段落", "风险", "Detector", "重复度", "片段预览"],
453
- datatype=["number", "str", "str", "str", "str"],
454
- label="高风险段落列表",
455
- wrap=True,
456
- )
457
 
458
  details_out = gr.Markdown(label="段落详情")
 
459
 
460
  run_btn.click(
461
  fn=analyze_document,
462
- inputs=[file_input, pasted_text, model, risk_threshold, topk],
463
- outputs=[summary_out, table_out, details_out],
464
  )
465
 
466
 
 
1
  import json
2
  import re
3
+ import time
4
+ from datetime import datetime
5
  from pathlib import Path
6
  from typing import Dict, List, Tuple
7
 
 
31
 
32
  MIN_PARAGRAPH_CHARS = 80
33
  WINDOW_MAX_LENGTH = 512
34
+ WINDOW_STRIDE = 192
35
+ MAX_HISTORY_ITEMS = 30
36
 
37
  CALIBRATION_PATH = Path("calibration/model.json")
38
  MBA_MODELS_DIR = Path("models/mba")
39
+ HISTORY_PATH = Path("history/analysis_records.json")
40
 
41
  CURRENT_MODEL_NAME = None
42
  CURRENT_TOKENIZER = None
 
49
  "tree_models": {},
50
  }
51
 
52
+ # Better CPU utilization for Torch inference.
53
+ try:
54
+ torch.set_num_threads(max(1, (torch.get_num_threads() or 4)))
55
+ except Exception:
56
+ pass
57
+
58
 
59
  def load_calibration_model() -> Dict:
60
  if not CALIBRATION_PATH.exists():
 
72
  CALIBRATION_MODEL = load_calibration_model()
73
 
74
 
75
+ def ensure_history_file():
76
+ HISTORY_PATH.parent.mkdir(parents=True, exist_ok=True)
77
+ if not HISTORY_PATH.exists():
78
+ HISTORY_PATH.write_text("[]", encoding="utf-8")
79
+
80
+
81
+ def load_history() -> List[Dict]:
82
+ ensure_history_file()
83
+ try:
84
+ data = json.loads(HISTORY_PATH.read_text(encoding="utf-8"))
85
+ if isinstance(data, list):
86
+ return data
87
+ except Exception:
88
+ pass
89
+ return []
90
+
91
+
92
+ def save_history_item(item: Dict):
93
+ items = load_history()
94
+ items.insert(0, item)
95
+ items = items[:MAX_HISTORY_ITEMS]
96
+ HISTORY_PATH.write_text(json.dumps(items, ensure_ascii=False, indent=2), encoding="utf-8")
97
+
98
+
99
+ def format_history_markdown() -> str:
100
+ items = load_history()
101
+ if not items:
102
+ return "暂无历史记录。"
103
+ lines = ["# 历史分析记录"]
104
+ for i, x in enumerate(items, 1):
105
+ lines.append(
106
+ f"{i}. `{x.get('time')}` | 文件: {x.get('source')} | 模型: {x.get('model')} | "
107
+ f"综合风险: {x.get('overall', 0):.2%} | 预测知网率: {x.get('kn_like', 0):.2%} | 段落数: {x.get('paragraphs', 0)}"
108
+ )
109
+ return "\n".join(lines)
110
+
111
+
112
  def get_or_load_model(model_name: str):
113
  global CURRENT_MODEL_NAME, CURRENT_TOKENIZER, CURRENT_MODEL
114
  if CURRENT_MODEL_NAME == model_name and CURRENT_TOKENIZER is not None and CURRENT_MODEL is not None:
115
  return CURRENT_TOKENIZER, CURRENT_MODEL
116
+
117
  tokenizer = AutoTokenizer.from_pretrained(model_name)
118
  model = AutoModelForSequenceClassification.from_pretrained(model_name)
119
  model.eval()
120
+
121
  CURRENT_MODEL_NAME = model_name
122
  CURRENT_TOKENIZER = tokenizer
123
  CURRENT_MODEL = model
 
143
 
144
  def clean_common_noise(line: str) -> str:
145
  line = normalize_text(line)
146
+ return re.sub(r"[ \t]+", " ", line)
 
147
 
148
 
149
  def extract_pdf_text(file_path: str) -> Tuple[str, Dict]:
150
  doc = fitz.open(file_path)
151
  all_pages = len(doc)
 
152
  page_lines: List[List[str]] = []
153
 
154
+ for idx in range(all_pages):
155
  page = doc[idx]
156
  rect = page.rect
157
  top_cut = rect.height * 0.06
 
160
 
161
  lines = []
162
  for b in sorted(blocks, key=lambda x: (round(x[1], 1), round(x[0], 1))):
163
+ _, y0, _, y1, text, *_ = b
164
  if y1 <= top_cut or y0 >= rect.height - bottom_cut:
165
  continue
166
  for raw in text.splitlines():
167
  line = clean_common_noise(raw)
168
+ if not line or is_probable_page_number(line):
 
 
169
  continue
170
  lines.append(line)
171
  page_lines.append(lines)
172
 
 
173
  freq = {}
174
  for lines in page_lines:
175
  if not lines:
176
  continue
177
+ for c in set(lines[:2] + lines[-2:]):
 
178
  if len(c) >= 4:
179
  freq[c] = freq.get(c, 0) + 1
180
+ repeat_lines = {k for k, v in freq.items() if v >= max(3, int(0.4 * all_pages))}
 
181
 
182
  merged_pages = []
183
  for lines in page_lines:
184
  cleaned = [ln for ln in lines if ln not in repeat_lines and not is_probable_page_number(ln)]
185
  merged_pages.append("\n".join(cleaned))
186
 
187
+ return "\n\n".join(merged_pages), {"total_pages": all_pages, "used_pages": all_pages, "page_truncated": False}
 
 
 
 
 
 
188
 
189
 
190
  def extract_docx_text(file_path: str) -> Tuple[str, Dict]:
 
254
  padding=True,
255
  return_tensors="pt",
256
  )
257
+ with torch.inference_mode():
258
  outputs = model(**inputs)
259
  probs = torch.softmax(outputs.logits, dim=-1)[:, 1].cpu().numpy()
260
  return float(0.75 * np.mean(probs) + 0.25 * np.max(probs))
 
297
  ])
298
 
299
 
300
+ def is_lfs_pointer(path: Path) -> bool:
301
+ try:
302
+ txt = path.read_text(encoding="utf-8", errors="ignore")
303
+ return txt.startswith("version https://git-lfs.github.com/spec/v1")
304
+ except Exception:
305
+ return False
306
+
307
+
308
  def init_mba_pack() -> Tuple[bool, str]:
309
  if MBA_STATE["ready"]:
310
  return True, ""
311
  if joblib is None:
312
  return False, "当前环境缺少 joblib,无法加载 mba-aigc-detector 本地模型包。"
313
+
314
  needed = [
315
  "select5_tree_d2_model.pkl",
316
  "select10_tree_d2_model.pkl",
 
320
  ]
321
  missing = [f for f in needed if not (MBA_MODELS_DIR / f).exists()]
322
  if missing:
323
+ return False, "缺少 mba 模型文件,请将模型文件放入 models/mba/。"
324
+
325
+ # Detect git-lfs pointer files early.
326
+ for f in needed:
327
+ p = MBA_MODELS_DIR / f
328
+ if is_lfs_pointer(p):
329
+ return False, "检测到 mba 模型文件是 Git LFS 指针,不是真实权重。请在仓库中上传真实模型二进制文件。"
330
 
331
  try:
332
  tok = AutoTokenizer.from_pretrained("hfl/chinese-roberta-wwm-ext")
 
343
  ok, msg = init_mba_pack()
344
  if not ok:
345
  raise RuntimeError(msg)
346
+
347
  tok = MBA_STATE["extractor_tokenizer"]
348
  mdl = MBA_STATE["extractor_model"]
 
349
  inputs = tok(text[:512], return_tensors="pt", max_length=512, truncation=True, padding=True)
350
+
351
+ with torch.inference_mode():
352
  bert_feat = mdl(**inputs).last_hidden_state[:, 0, :].cpu().numpy()[0]
353
 
354
  stat_feat = _extract_stat_features(text)
 
367
  detector = float(min(max(detector_score_mba(text) * 0.3, 0.0), 1.0))
368
  else:
369
  detector = detector_score_transformer(text, model_name)
370
+
371
  repetition = calc_repetition(text)
372
  variance = calc_sentence_variance(text)
373
  risk = float(min(max(detector * 0.78 + repetition * 0.12 + (1 - variance) * 0.10, 0.0), 1.0))
 
397
  return clip01(y)
398
 
399
 
400
+ def build_filtered_details(blocks: List[Dict], level_filter: str) -> str:
401
+ if level_filter == "全部":
402
+ selected = blocks
403
+ else:
404
+ selected = [b for b in blocks if b["risk_level"] == level_filter]
405
+ if not selected:
406
+ return f"当前筛选 `{level_filter}` 下暂无段落。"
407
+ return "\n\n---\n\n".join([b["content"] for b in selected])
408
+
409
+
410
+ def analyze_document(upload_file, pasted_text, model_label, risk_threshold, risk_filter, progress=gr.Progress()):
411
  if upload_file is None and not normalize_text(pasted_text or ""):
412
+ return "请先上传文件,或粘贴文本。", "", format_history_markdown()
413
 
414
  model_name = MODEL_CHOICES.get(model_label, MODEL_CHOICES[DEFAULT_MODEL_LABEL])
415
  if model_name == "mba_local_pack":
416
  ok, msg = init_mba_pack()
417
  if not ok:
418
+ return f"# 当前模型: {model_label}\n\n{msg}\n\n请切回其他模型。", "", format_history_markdown()
419
 
420
+ source = "pasted_text"
421
  if normalize_text(pasted_text or ""):
422
  raw_text = pasted_text
423
  extract_meta = {"total_pages": None, "used_pages": None, "page_truncated": False}
 
424
  paragraphs = split_paragraphs(raw_text)
425
  else:
426
+ source = Path(upload_file.name).name
427
  raw_text, extract_meta = extract_document_text(upload_file)
428
  paragraphs = [p for p in split_paragraphs(raw_text) if not should_skip_paragraph(p)]
 
 
429
 
430
  if not paragraphs:
431
+ return "未提取到可分析正文。", "", format_history_markdown()
432
 
433
+ t0 = time.time()
434
+ risks, details = [], []
 
435
  total = len(paragraphs)
436
  for i, p in enumerate(paragraphs, 1):
437
  progress(i / total, desc=f"推理进度: {int(i * 100 / total)}%")
 
443
  elif score["risk"] > max(0.55, risk_threshold - 0.15):
444
  level = "🟡"
445
 
446
+ risk_level = "低风险"
447
+ if score["risk"] > risk_threshold:
448
+ risk_level = "高风险"
449
+ elif score["risk"] > max(0.55, risk_threshold - 0.15):
450
+ risk_level = "中风险"
451
+
452
  details.append(
453
+ {
454
+ "risk_level": risk_level,
455
+ "content": f"""
456
  {level} 段落 {i} AI风险: {score['risk']:.2%}
457
 
458
  Detector: {score['detector']:.2%}
 
460
  句式稳定性: {1 - score['variance']:.2%}
461
 
462
  {p}
463
+ """,
464
+ }
465
  )
466
 
467
  f = build_doc_features(risks)
468
+ kn_like = predict_kn_like_rate(f)
469
+ elapsed = time.time() - t0
470
+ speed = len(paragraphs) / max(elapsed, 1e-6)
471
 
472
  trunc_info = []
473
+ if extract_meta.get("total_pages") is not None:
 
 
474
  trunc_info.append(f"页面截断: 否({extract_meta.get('used_pages')}/{extract_meta.get('total_pages')} 页)")
475
+ trunc_info.append(f"段落截断: 否(分析 {len(paragraphs)}/{len(paragraphs)} 段)")
476
 
477
+ mode_line = "当前模式: 原始风险率(未加载校准模型)" if not CALIBRATION_MODEL else "当前模式: 知网对齐预测率(已加载校准模型)"
478
  summary = f"""
479
  # 当前模型: {model_label}
480
  # 综合AI风险率: {f['overall']:.2%}
481
+ # 预测知网AIGC率: {kn_like:.2%}
482
  高风险段落占比: {f['high_ratio']:.2%}
483
  中风险段落占比: {f['mid_ratio']:.2%}
484
  有效段落数: {len(paragraphs)}
485
 
486
+ 平均速度: {speed:.2f} 段/秒
487
  {mode_line}
488
  {' | '.join(trunc_info)}
489
 
490
  (说明:该结果为“风险分析与校准预测”,并非官方系统结果)
491
  """
492
 
493
+ filtered_details = build_filtered_details(details, risk_filter)
494
+
495
+ save_history_item(
496
+ {
497
+ "time": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
498
+ "source": source,
499
+ "model": model_label,
500
+ "overall": f["overall"],
501
+ "kn_like": kn_like,
502
+ "paragraphs": len(paragraphs),
503
+ "high_ratio": f["high_ratio"],
504
+ "mid_ratio": f["mid_ratio"],
505
+ "elapsed_sec": elapsed,
506
+ "speed_para_per_sec": speed,
507
+ }
508
+ )
509
+
510
+ return summary, filtered_details, format_history_markdown()
511
 
512
 
513
  with gr.Blocks(theme=gr.themes.Soft(primary_hue="emerald", secondary_hue="slate"), title="论文AIGC风险检测系统") as demo:
514
+ gr.Markdown(
515
+ """
516
  # 论文AIGC风险检测系统
517
  支持 `PDF / Word(.docx) / 文本(.txt, .md)`,默认全文检测,支持直接粘贴文本。
518
+ """
519
+ )
520
 
521
  with gr.Row():
522
  with gr.Column(scale=1):
 
525
  model = gr.Dropdown(list(MODEL_CHOICES.keys()), value=DEFAULT_MODEL_LABEL, label="选择检测模型")
526
  with gr.Accordion("高级参数", open=False):
527
  risk_threshold = gr.Slider(0.5, 0.9, value=0.75, step=0.01, label="高风险阈值")
528
+ risk_filter = gr.Radio(["全部", "高风险", "中风险", "低风险"], value="全部", label="风险筛选")
529
  run_btn = gr.Button("开始分析", variant="primary")
530
 
531
  with gr.Column(scale=2):
532
  summary_out = gr.Markdown(label="总览")
 
 
 
 
 
 
533
 
534
  details_out = gr.Markdown(label="段落详情")
535
+ history_out = gr.Markdown(label="历史记录", value=format_history_markdown())
536
 
537
  run_btn.click(
538
  fn=analyze_document,
539
+ inputs=[file_input, pasted_text, model, risk_threshold, risk_filter],
540
+ outputs=[summary_out, details_out, history_out],
541
  )
542
 
543
 
models/mba/bert_tree_d1_meta.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "bert_tree_d1",
3
+ "depth": 1,
4
+ "feature_dim": 768,
5
+ "has_selector": false,
6
+ "use_bert": true
7
+ }
models/mba/model_manifest.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "models": [
3
+ {
4
+ "name": "select5_tree_d2",
5
+ "type": "feature_tree"
6
+ },
7
+ {
8
+ "name": "select10_tree_d2",
9
+ "type": "feature_tree"
10
+ },
11
+ {
12
+ "name": "select15_tree_d3",
13
+ "type": "feature_tree"
14
+ },
15
+ {
16
+ "name": "select20_tree_d2",
17
+ "type": "feature_tree"
18
+ },
19
+ {
20
+ "name": "bert_tree_d1",
21
+ "type": "bert_tree"
22
+ }
23
+ ],
24
+ "training_info": {
25
+ "original_samples": 15770,
26
+ "variant_samples": 150,
27
+ "total_samples": 15920,
28
+ "human_samples": 13191,
29
+ "ai_samples": 2729
30
+ }
31
+ }
models/mba/select10_tree_d2_meta.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "select10_tree_d2",
3
+ "k": 10,
4
+ "depth": 2,
5
+ "feature_dim": 10,
6
+ "has_selector": true,
7
+ "use_bert": false
8
+ }
models/mba/select15_tree_d3_meta.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "select15_tree_d3",
3
+ "k": 15,
4
+ "depth": 3,
5
+ "feature_dim": 15,
6
+ "has_selector": true,
7
+ "use_bert": false
8
+ }
models/mba/select20_tree_d2_meta.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "select20_tree_d2",
3
+ "k": 20,
4
+ "depth": 2,
5
+ "feature_dim": 20,
6
+ "has_selector": true,
7
+ "use_bert": false
8
+ }
models/mba/select5_tree_d2_meta.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "name": "select5_tree_d2",
3
+ "k": 5,
4
+ "depth": 2,
5
+ "feature_dim": 5,
6
+ "has_selector": true,
7
+ "use_bert": false
8
+ }