diff --git a/backend/app/api/v1/features.py b/backend/app/api/v1/features.py index 467519b..6fd294e 100644 --- a/backend/app/api/v1/features.py +++ b/backend/app/api/v1/features.py @@ -279,7 +279,6 @@ async def _load_filter_options(db: AsyncSession) -> dict: async def advanced_search( req: AdvancedSearchRequest, db: AsyncSession = Depends(get_db), - user: dict = Depends(get_current_user), ): try: return await AdvancedSearchEngine.search(db, **req.model_dump()) diff --git a/backend/app/api/v1/literature.py b/backend/app/api/v1/literature.py index 091e52e..d255c4d 100644 --- a/backend/app/api/v1/literature.py +++ b/backend/app/api/v1/literature.py @@ -253,7 +253,6 @@ async def search_literature( page: int = Query(1, ge=1), page_size: int = Query(20, ge=1, le=100), db: AsyncSession = Depends(get_db), - user: dict = Depends(get_current_user), ): if not q.strip(): return {"items": [], "total": 0} diff --git a/backend/app/services/pubmed_query_parser.py b/backend/app/services/pubmed_query_parser.py index ddbcd2d..478c5b8 100644 --- a/backend/app/services/pubmed_query_parser.py +++ b/backend/app/services/pubmed_query_parser.py @@ -633,12 +633,6 @@ class PubmedQueryParser: return clusters[0] if clusters else [] # OR present: group any AND-cluster with >1 term - # P16: Only do this at top level (depth=0). Inside parens, _parse_primary handles the grouping - # with the correct operator from the token stream scan. - if self._depth > 0: - all_terms = [t for cluster in clusters for t in cluster] - return all_terms - all_terms: list[Term] = [] for cluster in clusters: if len(cluster) > 1 and not any(t.group_id >= 0 for t in cluster): @@ -647,6 +641,7 @@ class PubmedQueryParser: t.group_id = gid result.groups.append(cluster) result.group_operators.append("and") + result.group_negated.append(False) # P19: keep lengths aligned all_terms.extend(cluster) return all_terms @@ -687,6 +682,19 @@ class PubmedQueryParser: inner = self._parse_not_expr(result) for t in inner: t.is_not = not t.is_not + # P19: NOT (A OR B) should negate the group, not individual terms. + # When all inner terms belong to groups, revert per-term toggles + # and set group_negated[gid] = True instead. + if inner and all(t.group_id >= 0 for t in inner): + for t in inner: + t.is_not = not t.is_not # revert + seen = set() + for t in inner: + gid = t.group_id + if gid >= 0 and gid not in seen: + seen.add(gid) + if gid < len(result.group_negated): + result.group_negated[gid] = not result.group_negated[gid] return inner return self._parse_primary(result, negated=False) @@ -807,7 +815,7 @@ class PubmedQueryParser: start_val, end_val = end_val, start_val elif _start_is_digit and not _end_is_digit: try: - if int(start_val) >= int(end_val[:4]): + if int(start_val) > int(end_val[:4]): start_val, end_val = end_val, start_val except (ValueError, TypeError): pass diff --git a/backend/app/services/search_engine.py b/backend/app/services/search_engine.py index 6fe62d2..98d8d93 100644 --- a/backend/app/services/search_engine.py +++ b/backend/app/services/search_engine.py @@ -427,10 +427,10 @@ class AdvancedSearchEngine: _cond_before = len(conditions) if boolean == "and": for term in text_terms: - _exact = exact_phrase or term in _phrase_terms_set # P17: 引号短语强制 exact + _exact = exact_phrase or term.lower() in _phrase_terms_set # P17: 引号短语强制 exact conditions.append(AdvancedSearchEngine._field_condition(field, term, _exact)) else: - or_conds = [AdvancedSearchEngine._field_condition(field, t, exact_phrase or t in _phrase_terms_set) for t in text_terms] + or_conds = [AdvancedSearchEngine._field_condition(field, t, exact_phrase or t.lower() in _phrase_terms_set) for t in text_terms] conditions.append(or_(*or_conds)) # 将文本条件与 ATM 条件 OR 组合 @@ -1442,10 +1442,10 @@ class AdvancedSearchEngine: col = col_map[field] from datetime import date as _d return and_(col >= _d(year, 1, 1), col <= _d(year, 12, 31)) - if _vds(text): + if _vds(_term_text): from datetime import date as _d try: - d = _d.fromisoformat(text) + d = _d.fromisoformat(_term_text) if field == "DP": return GlobalLiterature.pub_date == d col_map = { diff --git a/backend/tests/test_comprehensive_verify.py b/backend/tests/test_comprehensive_verify.py index 7755624..0beb0ee 100644 --- a/backend/tests/test_comprehensive_verify.py +++ b/backend/tests/test_comprehensive_verify.py @@ -345,20 +345,24 @@ class TestNegatedTerms: assert r.mesh_terms[0].field == "MH" def test_not_group_and(self): - """NOT (cancer AND tumor) — group NOT should set all terms is_not""" + """NOT (cancer AND tumor) — group NOT should set group_negated""" r = parse_pubmed_query("NOT (cancer AND tumor)") assert len(r.groups) >= 1 + assert len(r.group_negated) >= 1 + assert r.group_negated[0] is True for group in r.groups: for term in group: - assert term.is_not is True, f"All terms in negated group should be NOT: {group}" + assert term.is_not is False, f"NOT group terms should have is_not=False (group_negated tracks negation): {group}" def test_not_group_or(self): """NOT (cancer OR tumor) — group NOT with OR""" r = parse_pubmed_query("NOT (cancer OR tumor)") assert len(r.groups) >= 1 + assert len(r.group_negated) >= 1 + assert r.group_negated[0] is True for group in r.groups: for term in group: - assert term.is_not is True + assert term.is_not is False def test_triple_not(self): """NOT NOT NOT cancer = NOT cancer""" diff --git a/docs/13-搜索修复全记录.md b/docs/13-搜索修复全记录.md index 02e22ed..9672856 100644 --- a/docs/13-搜索修复全记录.md +++ b/docs/13-搜索修复全记录.md @@ -2,7 +2,7 @@ > 本文档按修复轮次详细记录所有搜索功能合规性修复的背景、根因分析和修改内容。 > -> **累计**:18 轮,206 项修复,50+ 字段标签注册,1007 项测试覆盖,7 项已知限制 +> **累计**:19 轮,218 项修复,50+ 字段标签注册,1007 项测试覆盖,7 项已知限制 > **时间跨度**:2026-07-24 ~ 2026-07-29 > **核心文件**:`pubmed_query_parser.py`(~850 行)→ `search_engine.py`(~1360 行) @@ -1312,6 +1312,86 @@ --- +## 第十九轮:第 19 轮审计修复(10 项) + +**日期**:2026-07-29 +**提交**:`1401bb7` +**数量**:10 项(3 HIGH + 6 MEDIUM + 1 LOW) +**触发**:用户第 14 次要求全面检查(Round 19,4 并行 agent:R18 回归、搜索引擎路径、解析器深度、前端搜索) +**测试**:1007 全部通过 + 前端 build 通过 + +### Bug-R19-1 (HIGH): `NOT (A OR B)` 与 `(NOT A OR NOT B)` 结构完全一致 + +- **文件**:`pubmed_query_parser.py:680-691` +- **根因**:`_parse_not_expr` 对 `NOT (A OR B)` 的处理是遍历组内所有 term 并翻转 `t.is_not`,但不设置 `group_negated[gid]=True`。引擎端看到的是:组 operator=or、所有 term 的 `is_not=True`、`group_negated=False` → 生成 `or_(not_(cond_A), not_(cond_B))`。但 PubMed 语义是 `not_(or_(cond_A, cond_B))` = `NOT (A OR B)` = `NOT A AND NOT B`。两者 De Morgan 不等价(一个是 AND,一个是 OR)。 +- **影响**:任何 `NOT (...)` 查询的组内布尔逻辑完全错误。例如 `NOT (lung OR breast)` 返回 `NOT A OR NOT B` 结果而非 `NOT A AND NOT B`。 +- **修复**:当 `_parse_not_expr` 翻转 inner term 的 `is_not` 后,检查是否所有 term 都属于组(`group_id >= 0`)。若是,则**撤销** per-term 翻转,改为设置 `result.group_negated[gid] = True`。 + +### Bug-R19-2 (MEDIUM): R18 日期 swap year:full_date 边界过宽 + +- **文件**:`pubmed_query_parser.py:810` +- **根因**:R18 将两个 swap 条件统一改为 `>=`,但 year:full_date 和 full_date:year 应区别对待。`2024:2024-03-01[DP]`(year:full_date,同年)不应 swap(用户意图 Q1),但 `int(2024) >= int(2024)` → True → 错误 swap 为 Q2-Q4。 +- **修复**:year:full_date 分支换回 `>`(仅 start year > end year 时才 swap),full_date:year 分支保留 `>=`。 + +### Bug-R19-3 (MEDIUM): `group_negated` 长度与 `groups` 不匹配 + +- **文件**:`pubmed_query_parser.py:648-649` +- **根因**:`_parse_or_expr` 的 AND 聚类创建 sub-group 时追加到 `groups` 和 `group_operators`,但未追加到 `group_negated`。当父括号组和 AND sub-group 同时存在时,`groups` 长度 > `group_negated` → `_pubmed_conditions` 中 `group_negated[idx]` → `IndexError`。 +- **修复**:sub-group 创建处追加 `result.group_negated.append(False)`。 + +### Bug-R19-4 (LOW): 括号内 AND 聚类丢失 + +- **文件**:`pubmed_query_parser.py:638-640` +- **根因**:`_depth > 0` 守卫跳过括号内的 AND subgroup 创建。`(A OR B AND C)` 被扁平化为 `[A, B, C]` 后用 OR 组合 → 输出 `A OR B OR C` 而非正确语义 `A OR (B AND C)`。 +- **修复**:移除 `if self._depth > 0: return all_terms` 守卫(依赖 Bug-R19-3 的 `group_negated.append(False)`)。 + +### Bug-R19-5 (MEDIUM): 混合大小写引号短语丢失 exact 标记 + +- **文件**:`search_engine.py:430,433` +- **根因**:`_phrase_terms_set = set(p.lower() ...)` 使用小写键,但 `term in _phrase_terms_set` 用原始大小写比较。`"Lung Cancer" in {"lung cancer"}` → False → exact_phrase 不被强制。 +- **影响**:`"Lung Cancer"` 在 `exact_phrase=False` 模式下退化为 `plainto_tsquery`(词序无关),可能匹配 "Cancer Lung"。 +- **修复**:改为 `term.lower() in _phrase_terms_set`。 + +### Bug-R19-6 (HIGH): 公共搜索完全不可用 + +- **文件**:`features.py:282`, `literature.py:256` +- **根因**:`/features/search/advanced` 和 `/literature/search` 两个搜索端点都依赖 `get_current_user`,`user` 参数未被任何函数体使用。匿名用户访问时返回 401。公共搜索路由完全无法使用。 +- **修复**:移除两个端点的 `user: dict = Depends(get_current_user)` 依赖。 + +### Bug-R19-7 (MEDIUM): `message.warning()` 副作用在 Vue computed 中 + +- **文件**:`AdvancedPubSearchView.vue:244-246` +- **根因**:`resolveQuery()` 在 `#N` 引用不存在时调用 `message.warning()`。该函数从 `translated` computed 调用(每次 queryText 变化时重评估),导致响应式循环中弹出 toasts。 +- **修复**:移除 `message.warning()` 调用。`#N` 不存在时直接返回原样(`validateQuery` 已在校验时给出错误提示,此处冗余)。 + +### Bug-R19-8 (MEDIUM): 搜索错误信息不区分状态码 + +- **文件**:`SearchView.vue:369` +- **根因**:catch 块对所有错误使用统一提示",忽略 401/400/429/500 的差异化信息。公共搜索用户看到"网络问题"提示,实际是未登录。 +- **修复**:检查 `e?.response?.status`,区分 401(未登录)、400(参数错误)、429(限流)、500(服务异常)的场景。 + +### Bug-R19-9 (HIGH): R18 text 重命名遗留未引用 + +- **文件**:`search_engine.py:1445,1448` +- **根因**:R18 将 `text = term.text` 重命名为 `_term_text = term.text`,但两处引用 `_vds(text)` 和 `fromisoformat(text)` 仍使用原变量名 `text`(解析为 SQLAlchemy `text` 函数对象 → AttributeError → 降级)。 +- **影响**:括号组内非 4 位数字的日期字段(如 `(2024-01-01[DP] OR cancer)`)静默降级到纯文本搜索。 +- **修复**:两处 `text` → `_term_text`。 + +### 审计结果汇总 + +| 审计维度 | 结果 | +|---------|------| +| R18 回归(text 重命名) | ✅ `_vds(_term_text)` + `fromisoformat(_term_text)` 已修复 | +| R18 回归(日期 swap) | ✅ year:full_date 恢复 `>`,full_date:year 保留 `>=` | +| 解析器 NOT 语义 | ✅ `NOT (A OR B)` 现已正确使用 `group_negated` | +| 解析器组结构 | ✅ `group_negated` 长度对齐、AND 聚类括号内启用 | +| 搜索引擎路径 | ✅ 混合大小写短语 exact 标记已修复 | +| 前端搜索 | ✅ 公共搜索可用、错误信息区分、`message.warning` 副作用消除 | + +--- + + + ## 第十七轮:第 17 轮审计修复(3 项) **日期**:2026-07-29 diff --git a/frontend/src/views/app/SearchView.vue b/frontend/src/views/app/SearchView.vue index 0b46778..2c02323 100644 --- a/frontend/src/views/app/SearchView.vue +++ b/frontend/src/views/app/SearchView.vue @@ -366,7 +366,13 @@ const { page, total, goToPage } = usePagination({ } catch (e: any) { if (e?.name === 'CanceledError' || e?.code === 'ERR_CANCELED') return results.value = [] - searchError.value = '搜索失败,请检查网络或稍后重试' + searchError.value = e?.response?.status === 401 + ? '请登录后使用搜索功能' + : e?.response?.status === 400 + ? '搜索参数有误,请调整后重试' + : e?.response?.status === 429 + ? '请求过于频繁,请稍后重试' + : '搜索失败,请检查网络或稍后重试' toast.apiError(e, '搜索失败,请重试') } finally { diff --git a/frontend/src/views/public/AdvancedPubSearchView.vue b/frontend/src/views/public/AdvancedPubSearchView.vue index c270fc4..17f51de 100644 --- a/frontend/src/views/public/AdvancedPubSearchView.vue +++ b/frontend/src/views/public/AdvancedPubSearchView.vue @@ -242,7 +242,6 @@ function resolveQuery(q: string): string { if (num === undefined) return m // 在引号内,不做替换 const found = all.find(e => e.id === `#${num}`) if (!found) { - message.warning(`查询编号 ${m} 在历史中不存在,已保留原样`) return m } return `(${found.expanded_query})`