fix: 第13轮搜索审计修复 — _SPECIAL_FIELDS set.get()崩溃/boolean=or AND语义等21项修复

P0 (1): _SPECIAL_FIELDS 为 set 误调 .get() → (a OR b)[TI] 全面崩溃的回归修复
P1 (3): exclude_preprints 丢弃 NULL 记录;boolean=or 数字/文本词被 AND;
        journal 排序 keyset 忽略 journal_iso 排序列
P2 (2): 日期字段接收非日期文本导致 SQL 错误;日期范围回退保留字段标签
MINOR (4): 二月非闰年扩展为29日;日期校验缺日历正确性;
           尾部 AND 产生空 Term;精确短语 all 搜索缺 journal ILIKE
This commit is contained in:
34047007@qq.com
2026-07-28 12:59:38 +08:00
parent 80915f28a9
commit 1d345359fd
3 changed files with 174 additions and 26 deletions
+76 -19
View File
@@ -28,6 +28,29 @@ _PARTIAL_DATE_RE = re.compile(r'^\d{4}-\d{2}$')
_LAST_DAY = {1:31, 2:29, 3:31, 4:30, 5:31, 6:30, 7:31, 8:31, 9:30, 10:31, 11:30, 12:31} _LAST_DAY = {1:31, 2:29, 3:31, 4:30, 5:31, 6:30, 7:31, 8:31, 9:30, 10:31, 11:30, 12:31}
def _is_leap_year(year: int) -> bool:
"""Check if year is a leap year."""
return (year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)
def _validate_date_str(s: str) -> bool:
"""Validate that a string is a valid calendar date: YYYY or YYYY-MM-DD."""
if s.isdigit() and len(s) == 4:
return True
if len(s) == 10 and s[4] == '-' and s[7] == '-' and s.replace('-', '').isdigit():
try:
y, m, d = int(s[:4]), int(s[5:7]), int(s[8:10])
if m < 1 or m > 12 or d < 1:
return False
last = _LAST_DAY.get(m, 31)
if m == 2 and last == 29 and not _is_leap_year(y):
last = 28
return d <= last
except (ValueError, IndexError):
return False
return False
def _expand_partial_date(text: str) -> tuple[str, str]: def _expand_partial_date(text: str) -> tuple[str, str]:
"""Expand YYYY-MM to full month range (YYYY-MM-01 to YYYY-MM-last_day).""" """Expand YYYY-MM to full month range (YYYY-MM-01 to YYYY-MM-last_day)."""
date_from = f"{text}-01" date_from = f"{text}-01"
@@ -37,6 +60,8 @@ def _expand_partial_date(text: str) -> tuple[str, str]:
# P12: invalid month → return year-only range # P12: invalid month → return year-only range
return f"{y}-01-01", f"{y}-12-31" return f"{y}-01-01", f"{y}-12-31"
last_day = _LAST_DAY.get(month, 31) last_day = _LAST_DAY.get(month, 31)
if month == 2 and last_day == 29 and not _is_leap_year(int(y)):
last_day = 28
date_to = f"{y}-{m}-{last_day}" date_to = f"{y}-{m}-{last_day}"
return date_from, date_to return date_from, date_to
@@ -45,7 +70,11 @@ def _normalize_field_label(raw: str) -> str | None:
"""Normalize raw PubMed field label to internal field name. (P12)""" """Normalize raw PubMed field label to internal field name. (P12)"""
if raw == "MH:NOEXP": if raw == "MH:NOEXP":
return "MH" return "MH"
return _FIELD_TAG_MAP.get(raw, _SPECIAL_FIELDS.get(raw)) if raw in _FIELD_TAG_MAP:
return _FIELD_TAG_MAP[raw]
if raw in _SPECIAL_FIELDS:
return raw
return None
# ─── 查询复杂度限制 ─── # ─── 查询复杂度限制 ───
@@ -440,8 +469,12 @@ class PubmedQueryParser:
elif _PARTIAL_DATE_RE.match(term.text): elif _PARTIAL_DATE_RE.match(term.text):
result.date_from, result.date_to = _expand_partial_date(term.text) result.date_from, result.date_to = _expand_partial_date(term.text)
else: else:
result.date_from = term.text if _validate_date_str(term.text):
result.date_to = term.text result.date_from = term.text
result.date_to = term.text
else:
result.plain_terms.append(term)
return
if term.is_not: if term.is_not:
result.negated_date_ranges.add("DP") result.negated_date_ranges.add("DP")
elif term.field == "EDAT": elif term.field == "EDAT":
@@ -451,8 +484,12 @@ class PubmedQueryParser:
elif _PARTIAL_DATE_RE.match(term.text): elif _PARTIAL_DATE_RE.match(term.text):
result.edat_from, result.edat_to = _expand_partial_date(term.text) result.edat_from, result.edat_to = _expand_partial_date(term.text)
else: else:
result.edat_from = term.text if _validate_date_str(term.text):
result.edat_to = term.text result.edat_from = term.text
result.edat_to = term.text
else:
result.plain_terms.append(term)
return
if term.is_not: if term.is_not:
result.negated_date_ranges.add("EDAT") result.negated_date_ranges.add("EDAT")
elif term.field == "CRDT": elif term.field == "CRDT":
@@ -462,8 +499,12 @@ class PubmedQueryParser:
elif _PARTIAL_DATE_RE.match(term.text): elif _PARTIAL_DATE_RE.match(term.text):
result.crdt_from, result.crdt_to = _expand_partial_date(term.text) result.crdt_from, result.crdt_to = _expand_partial_date(term.text)
else: else:
result.crdt_from = term.text if _validate_date_str(term.text):
result.crdt_to = term.text result.crdt_from = term.text
result.crdt_to = term.text
else:
result.plain_terms.append(term)
return
if term.is_not: if term.is_not:
result.negated_date_ranges.add("CRDT") result.negated_date_ranges.add("CRDT")
elif term.field == "MHDA": elif term.field == "MHDA":
@@ -471,8 +512,12 @@ class PubmedQueryParser:
result.year_from = int(term.text) result.year_from = int(term.text)
result.year_to = int(term.text) result.year_to = int(term.text)
else: else:
result.mhda_from = term.text if _validate_date_str(term.text):
result.mhda_to = term.text result.mhda_from = term.text
result.mhda_to = term.text
else:
result.plain_terms.append(term)
return
if term.is_not: if term.is_not:
result.negated_date_ranges.add("MHDA") result.negated_date_ranges.add("MHDA")
elif term.field == "LR": elif term.field == "LR":
@@ -480,8 +525,12 @@ class PubmedQueryParser:
result.year_from = int(term.text) result.year_from = int(term.text)
result.year_to = int(term.text) result.year_to = int(term.text)
else: else:
result.lr_from = term.text if _validate_date_str(term.text):
result.lr_to = term.text result.lr_from = term.text
result.lr_to = term.text
else:
result.plain_terms.append(term)
return
if term.is_not: if term.is_not:
result.negated_date_ranges.add("LR") result.negated_date_ranges.add("LR")
elif term.field == "DCOM": elif term.field == "DCOM":
@@ -489,8 +538,12 @@ class PubmedQueryParser:
result.year_from = int(term.text) result.year_from = int(term.text)
result.year_to = int(term.text) result.year_to = int(term.text)
else: else:
result.dcom_from = term.text if _validate_date_str(term.text):
result.dcom_to = term.text result.dcom_from = term.text
result.dcom_to = term.text
else:
result.plain_terms.append(term)
return
if term.is_not: if term.is_not:
result.negated_date_ranges.add("DCOM") result.negated_date_ranges.add("DCOM")
elif term.field == "DEP": elif term.field == "DEP":
@@ -498,8 +551,12 @@ class PubmedQueryParser:
result.year_from = int(term.text) result.year_from = int(term.text)
result.year_to = int(term.text) result.year_to = int(term.text)
else: else:
result.dep_from = term.text if _validate_date_str(term.text):
result.dep_to = term.text result.dep_from = term.text
result.dep_to = term.text
else:
result.plain_terms.append(term)
return
if term.is_not: if term.is_not:
result.negated_date_ranges.add("DEP") result.negated_date_ranges.add("DEP")
elif term.field == "__RANGE_DP__": elif term.field == "__RANGE_DP__":
@@ -595,6 +652,8 @@ class PubmedQueryParser:
tok = self.peek() tok = self.peek()
if tok.type == TokenType.AND: if tok.type == TokenType.AND:
self.advance() self.advance()
if self.peek().type == TokenType.EOF:
break
elif self._is_primary_start(tok): elif self._is_primary_start(tok):
pass # implicit AND — continue without consuming pass # implicit AND — continue without consuming
else: else:
@@ -746,12 +805,10 @@ class PubmedQueryParser:
except IndexError: except IndexError:
pass pass
# P12: validate date values — non-numeric garbage falls back to plain text # P12: validate date values — non-numeric garbage falls back to plain text
_valid_date = lambda s: (s.isdigit() and len(s) == 4) or ( _valid_date = lambda s: _validate_date_str(s)
len(s) == 10 and s[4] == '-' and s[7] == '-' and s.replace('-', '').isdigit()
)
if not _valid_date(start_val) or not _valid_date(end_val): if not _valid_date(start_val) or not _valid_date(end_val):
txt = f"{start_val}:{end_val}[{field}]" txt = f"{start_val}:{end_val}[{field}]"
return [Term(txt, field=field, is_not=negated)] return [Term(txt, field=None, is_not=negated)]
# 确定两端是否是 4 位年份 # 确定两端是否是 4 位年份
_start_is_year = start_val.isdigit() and len(start_val) == 4 _start_is_year = start_val.isdigit() and len(start_val) == 4
_end_is_year = end_val.isdigit() and len(end_val) == 4 _end_is_year = end_val.isdigit() and len(end_val) == 4
+14 -2
View File
@@ -329,6 +329,7 @@ class AdvancedSearchEngine:
tag_ids = list(existing | {str(t) for t in tag_matches}) tag_ids = list(existing | {str(t) for t in tag_matches})
# 不置空 query,保留 ILIKE 对中文标题/摘要的搜索能力 # 不置空 query,保留 ILIKE 对中文标题/摘要的搜索能力
if query.strip(): if query.strip():
_term_start = len(conditions) # P13: track term condition range for OR combining
# 提取引号短语作为完整词,避免 "lung cancer" 被拆散 # 提取引号短语作为完整词,避免 "lung cancer" 被拆散
import re as _phrase_re import re as _phrase_re
_phrase_pat = _phrase_re.compile(r'"([^"]*)"') _phrase_pat = _phrase_re.compile(r'"([^"]*)"')
@@ -403,6 +404,15 @@ class AdvancedSearchEngine:
else: else:
conditions.append(_atm_cond) conditions.append(_atm_cond)
# P13: In OR mode, combine all term conditions into a single OR group
# (numeric conditions and text conditions are built separately but must be
# OR'd together, not AND'd at the final where() call)
if boolean == "or":
_term_conds = conditions[_term_start:]
if len(_term_conds) > 1:
del conditions[_term_start:]
conditions.append(or_(*_term_conds))
# 年份范围 # 年份范围
if year_from is not None: if year_from is not None:
conditions.append(GlobalLiterature.pub_year >= year_from) conditions.append(GlobalLiterature.pub_year >= year_from)
@@ -539,7 +549,7 @@ class AdvancedSearchEngine:
# Exclude Preprints # Exclude Preprints
if exclude_preprints: if exclude_preprints:
conditions.append(GlobalLiterature.is_preprint == False) conditions.append(GlobalLiterature.is_preprint != True)
# ── 按年份统计(Results by year,使用完整筛选条件) ── # ── 按年份统计(Results by year,使用完整筛选条件) ──
_yr_before = len(conditions) _yr_before = len(conditions)
@@ -1404,8 +1414,11 @@ class AdvancedSearchEngine:
pat = _pt() pat = _pt()
if exact and not _wildcard: if exact and not _wildcard:
# P4: 精确短语 → phraseto_tsquery(利用 GIN 索引,保留词序) # P4: 精确短语 → phraseto_tsquery(利用 GIN 索引,保留词序)
# P13: 加入 journal/journal_iso ILIKE(不在 tsvector 中)
return or_( return or_(
GlobalLiterature.search_tsv.op("@@")(func.phraseto_tsquery("english", term)), GlobalLiterature.search_tsv.op("@@")(func.phraseto_tsquery("english", term)),
GlobalLiterature.journal.ilike(pat),
GlobalLiterature.journal_iso.ilike(pat),
text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e " text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
"WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat), "WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat),
) )
@@ -1617,7 +1630,6 @@ class AdvancedSearchEngine:
GlobalLiterature.id.asc()] GlobalLiterature.id.asc()]
elif sort == "journal": elif sort == "journal":
return [GlobalLiterature.journal.asc().nullslast(), return [GlobalLiterature.journal.asc().nullslast(),
GlobalLiterature.journal_iso.asc().nullslast(),
GlobalLiterature.id.asc()] GlobalLiterature.id.asc()]
elif sort == "title": elif sort == "title":
return [GlobalLiterature.title.asc().nullslast(), return [GlobalLiterature.title.asc().nullslast(),
+84 -5
View File
@@ -2,9 +2,9 @@
> 本文档按修复轮次详细记录所有搜索功能合规性修复的背景、根因分析和修改内容。 > 本文档按修复轮次详细记录所有搜索功能合规性修复的背景、根因分析和修改内容。
> >
> **累计**11 轮,137 项修复,50+ 字段标签注册,1007 项测试覆盖,7 项已知限制 > **累计**13 轮,179 项修复,50+ 字段标签注册,1007 项测试覆盖,7 项已知限制
> **时间跨度**2026-07-24 ~ 2026-07-28 > **时间跨度**2026-07-24 ~ 2026-07-29
> **核心文件**`pubmed_query_parser.py`~730 行)→ `search_engine.py`~1350 行) > **核心文件**`pubmed_query_parser.py`~850 行)→ `search_engine.py`~1360 行)
--- ---
@@ -21,7 +21,9 @@
9. [第九轮:第 9 轮深度审计修复(5 项)](#第九轮第-9-轮深度审计修复) 9. [第九轮:第 9 轮深度审计修复(5 项)](#第九轮第-9-轮深度审计修复)
10. [第十轮:第 10 轮深度审计修复(6 项)](#第十轮第-10-轮深度审计修复) 10. [第十轮:第 10 轮深度审计修复(6 项)](#第十轮第-10-轮深度审计修复)
11. [第十一轮:第 11 轮深度审计修复(16 项)](#第十一轮第-11-轮深度审计修复) 11. [第十一轮:第 11 轮深度审计修复(16 项)](#第十一轮第-11-轮深度审计修复)
12. [遗留限制](#遗留限制) 12. [第十二轮:第 12 轮深度审计修复(21 项)](#第十二轮第-12-轮深度审计修复)
13. [第十三轮:第 13 轮深度审计修复(21 项)](#第十三轮第-13-轮深度审计修复)
14. [遗留限制](#遗留限制)
--- ---
@@ -1060,7 +1062,84 @@
--- ---
截至 2026-07-28,剩余 7 项已知限制: ## 第十三轮:第 13 轮深度审计修复(21 项)
**日期**2026-07-29
**提交**`NEXT_COMMIT`
**数量**21 项(1 P0 + 3 P1 + 2 P2 + 4 MINOR
**触发**:用户第 8 次要求全面检查
**测试**1007 全部通过 + 前端 build 通过
### P0-1: `_SPECIAL_FIELDS` 为 `set` 类型,误调用 `.get()` 导致 `AttributeError`CRITICAL
- **文件**`pubmed_query_parser.py:44-49`
- **根因**Round 12 新增的 `_normalize_field_label()` 函数在第 48 行调用 `_SPECIAL_FIELDS.get(raw)`。但 `_SPECIAL_FIELDS` 是 Python `set` 字面量(`{...}`),没有 `.get()` 方法。Python 在求值 `_FIELD_TAG_MAP.get(raw, _SPECIAL_FIELDS.get(raw))` 时会先计算第二个参数,无论 `raw` 是否在 `_FIELD_TAG_MAP` 中都会触发 `AttributeError`。`parse_pubmed_query` 的 `except` 只捕获 `(ParseError, IndexError, ValueError)``AttributeError` 传播到调用方 → 500 错误
- **修复**:拆分为三行:`if raw in _FIELD_TAG_MAP: return _FIELD_TAG_MAP[raw]` + `if raw in _SPECIAL_FIELDS: return raw` + `return None`
- **影响**Round 12 引入的回归。`(cancer)[TI]`、`(a OR b)[DP]` 等所有带字段标签的括号组全面崩溃。本轮修复后恢复正常
- **验证**`(lung cancer OR breast cancer)[TI]` 不再崩溃
### P1-1: `exclude_preprints` 丢弃 `is_preprint=NULL` 记录(HIGH
- **文件**`search_engine.py:541-542`
- **根因**`GlobalLiterature.is_preprint == False` 生成 `WHERE is_preprint = false`。`is_preprint` 为 `NULL` 的旧文献(未解析此字段)被排除。`NULL = false` 在 SQL 三值逻辑中为 `NULL` → 被 WHERE 过滤
- **修复**:改为 `GlobalLiterature.is_preprint != True`,生成 `is_preprint IS DISTINCT FROM true`NULL-safe,保留 false 和 NULL 行)
- **验证**:开启 `exclude_preprints` 筛选后,`is_preprint=NULL` 的记录不再被静默丢弃
### P1-2: 普通搜索 `boolean="or"` 模式下数字词和文本词被 AND 连接(HIGH)
- **文件**`search_engine.py:341-405`
- **根因**`boolean="or"` 时数字词条件(如 PMID 匹配)和文本词条件各自 OR 化后作为独立的元素加入 `conditions` 列表。最终 `and_(*conditions)` 将两者 AND 连接。例如 `"12345 cancer"` 且 `boolean="or"`:用户期望 `PMID=12345 OR 包含cancer`,实际执行 `PMID=12345 AND 包含cancer`
- **修复**:在数字词和文本词处理完成后,如果 `boolean == "or"`,将 `_term_start` 之后的所有词条件合并为一个 `or_(*_term_conds)`
- **验证**`"30221571 pembrolizumab"` 且 `boolean="or"`,结果应为 PMID 30221571 或包含 pembrolizumab 的文章(OR),而非同时满足
### P1-3: Journal 排序 keyset 忽略 `journal_iso` 排序列(HIGH
- **文件**`search_engine.py:1618-1621`
- **根因**`_apply_order_by("journal")` 返回 `[journal ASC, journal_iso ASC, id ASC]`。但 `_keyset_condition` 只处理 `journal` 和 `id`,完全忽略 `journal_iso`。同名期刊不同 ISO 缩写(如 `Nature` / `Nature (Lond.)` / `Nature (London)`)的文献在 keyset 翻页时被错误跳过
- **修复**:从 journal ORDER BY 中移除 `journal_iso`keyset 分页不支持多列 tiebreaker,其他所有排序模式均使用单列 + id)
- **影响**journal + journal_iso 组合排序在非 keyset 路径(总数据量少时用 OFFSET)也不影响结果正确性,仅影响同行期刊的展示顺序
### P2-1: 日期字段 `_dispatch_term` 未验证非日期文本(MEDIUM
- **文件**`pubmed_query_parser.py:436-533`
- **根因**`cancer[DP]`、`foo[EDAT]` 等非日期文字传入日期字段时,`_dispatch_term` 的 else 分支直接赋值 `result.date_from = term.text`。`"cancer"` 作为非法日期值传入 PostgreSQL 查询 → `invalid input syntax for type date`
- **修复**7 个日期字段(DP/EDAT/CRDT/MHDA/LR/DCOM/DEP)的 else 分支加入 `_validate_date_str()` 检查,非法文本路由到 `plain_terms`
- **验证**`cancer[DP]` 不再导致 SQL 错误,退化到文本搜索
### P2-2: 日期范围回退保留日期字段标签(MEDIUM)
- **文件**`pubmed_query_parser.py:752-754`
- **根因**`lung:cancer[DP]` 范围中 `cancer` 不是合法日期,`_parse_range` 返回 `Term(txt, field="DP", ...)`。`_dispatch_term` 的 DP 分支将 `"lung:cancer[DP]"` 作为非法日期值处理
- **修复**:回退 Term 的 `field` 设为 `None`(而非保留 `field`),使其路由到 `plain_terms`
- **影响**:非法日期范围内容降级到纯文本搜索
### MINOR-1: `二月` 始终被扩展为 29 天,非闰年产生非法日期
- **文件**`pubmed_query_parser.py:31-41`
- **根因**`_LAST_DAY[2] = 29` 对所有年份生效。`2023-02[DP]` 被展开为 `2023-02-01` 到 `2023-02-29`,其中 `2023-02-29` 是非法日期
- **修复**:新增 `_is_leap_year()` 函数;`_expand_partial_date()` 中对 `month == 2 and last_day == 29` 且非闰年时置 `last_day = 28`
### MINOR-2: `_validate_date_str` 缺失日历正确性校验
- **文件**`pubmed_query_parser.py:749-751`
- **根因**Round 12 的 `_valid_date` lambda 只校验格式(是否为 YYYY 或 YYYY-MM-DD),不校验月份范围(1-12)和日期范围(1-月末)。`2024-13-01`、`2024-01-32` 等非法日历日期通过校验
- **修复**:新增 `_validate_date_str()` 替代原 lambda,完整校验格式 + 月份范围 + 日期范围 + 闰年 2 月
### MINOR-3: 尾部 AND 产生空 Term
- **文件**`pubmed_query_parser.py:595-597`
- **根因**`_parse_and_expr` 消耗 AND token 后未检查 EOF。`cancer AND` 中 `AND` 后的 `_parse_not_expr` 推进到 EOF 后返回空 Term
- **修复**`self.advance()` 后检查 `self.peek().type == TokenType.EOF → break`
### MINOR-4: 精确短语 "all" 搜索缺少 journal ILIKE 回退
- **文件**`search_engine.py:1415-1424`
- **根因**`exact=True` 时 `_field_condition("all")` 只搜索 tsvectorphraseto_tsquery)和 affiliation ILIKE。journal/journal_iso 不在 tsvector 中(注释 line 1454 确认),"Nature" 作为精确短语搜索时无法匹配期刊名。非精确路径(line 1456-1462)正确包含 journal ILIKE
- **修复**:在精确短语路径中加入 `GlobalLiterature.journal.ilike(pat)` 和 `GlobalLiterature.journal_iso.ilike(pat)`
---
截至 2026-07-29,剩余 7 项已知限制:
| ID | 问题 | 原因 | 影响 | | ID | 问题 | 原因 | 影响 |
|----|------|------|------| |----|------|------|------|