fix: 第13轮搜索审计修复 — _SPECIAL_FIELDS set.get()崩溃/boolean=or AND语义等21项修复
P0 (1): _SPECIAL_FIELDS 为 set 误调 .get() → (a OR b)[TI] 全面崩溃的回归修复
P1 (3): exclude_preprints 丢弃 NULL 记录;boolean=or 数字/文本词被 AND;
journal 排序 keyset 忽略 journal_iso 排序列
P2 (2): 日期字段接收非日期文本导致 SQL 错误;日期范围回退保留字段标签
MINOR (4): 二月非闰年扩展为29日;日期校验缺日历正确性;
尾部 AND 产生空 Term;精确短语 all 搜索缺 journal ILIKE
This commit is contained in:
@@ -28,6 +28,29 @@ _PARTIAL_DATE_RE = re.compile(r'^\d{4}-\d{2}$')
|
|||||||
_LAST_DAY = {1:31, 2:29, 3:31, 4:30, 5:31, 6:30, 7:31, 8:31, 9:30, 10:31, 11:30, 12:31}
|
_LAST_DAY = {1:31, 2:29, 3:31, 4:30, 5:31, 6:30, 7:31, 8:31, 9:30, 10:31, 11:30, 12:31}
|
||||||
|
|
||||||
|
|
||||||
|
def _is_leap_year(year: int) -> bool:
|
||||||
|
"""Check if year is a leap year."""
|
||||||
|
return (year % 4 == 0 and year % 100 != 0) or (year % 400 == 0)
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_date_str(s: str) -> bool:
|
||||||
|
"""Validate that a string is a valid calendar date: YYYY or YYYY-MM-DD."""
|
||||||
|
if s.isdigit() and len(s) == 4:
|
||||||
|
return True
|
||||||
|
if len(s) == 10 and s[4] == '-' and s[7] == '-' and s.replace('-', '').isdigit():
|
||||||
|
try:
|
||||||
|
y, m, d = int(s[:4]), int(s[5:7]), int(s[8:10])
|
||||||
|
if m < 1 or m > 12 or d < 1:
|
||||||
|
return False
|
||||||
|
last = _LAST_DAY.get(m, 31)
|
||||||
|
if m == 2 and last == 29 and not _is_leap_year(y):
|
||||||
|
last = 28
|
||||||
|
return d <= last
|
||||||
|
except (ValueError, IndexError):
|
||||||
|
return False
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def _expand_partial_date(text: str) -> tuple[str, str]:
|
def _expand_partial_date(text: str) -> tuple[str, str]:
|
||||||
"""Expand YYYY-MM to full month range (YYYY-MM-01 to YYYY-MM-last_day)."""
|
"""Expand YYYY-MM to full month range (YYYY-MM-01 to YYYY-MM-last_day)."""
|
||||||
date_from = f"{text}-01"
|
date_from = f"{text}-01"
|
||||||
@@ -37,6 +60,8 @@ def _expand_partial_date(text: str) -> tuple[str, str]:
|
|||||||
# P12: invalid month → return year-only range
|
# P12: invalid month → return year-only range
|
||||||
return f"{y}-01-01", f"{y}-12-31"
|
return f"{y}-01-01", f"{y}-12-31"
|
||||||
last_day = _LAST_DAY.get(month, 31)
|
last_day = _LAST_DAY.get(month, 31)
|
||||||
|
if month == 2 and last_day == 29 and not _is_leap_year(int(y)):
|
||||||
|
last_day = 28
|
||||||
date_to = f"{y}-{m}-{last_day}"
|
date_to = f"{y}-{m}-{last_day}"
|
||||||
return date_from, date_to
|
return date_from, date_to
|
||||||
|
|
||||||
@@ -45,7 +70,11 @@ def _normalize_field_label(raw: str) -> str | None:
|
|||||||
"""Normalize raw PubMed field label to internal field name. (P12)"""
|
"""Normalize raw PubMed field label to internal field name. (P12)"""
|
||||||
if raw == "MH:NOEXP":
|
if raw == "MH:NOEXP":
|
||||||
return "MH"
|
return "MH"
|
||||||
return _FIELD_TAG_MAP.get(raw, _SPECIAL_FIELDS.get(raw))
|
if raw in _FIELD_TAG_MAP:
|
||||||
|
return _FIELD_TAG_MAP[raw]
|
||||||
|
if raw in _SPECIAL_FIELDS:
|
||||||
|
return raw
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
# ─── 查询复杂度限制 ───
|
# ─── 查询复杂度限制 ───
|
||||||
@@ -440,8 +469,12 @@ class PubmedQueryParser:
|
|||||||
elif _PARTIAL_DATE_RE.match(term.text):
|
elif _PARTIAL_DATE_RE.match(term.text):
|
||||||
result.date_from, result.date_to = _expand_partial_date(term.text)
|
result.date_from, result.date_to = _expand_partial_date(term.text)
|
||||||
else:
|
else:
|
||||||
|
if _validate_date_str(term.text):
|
||||||
result.date_from = term.text
|
result.date_from = term.text
|
||||||
result.date_to = term.text
|
result.date_to = term.text
|
||||||
|
else:
|
||||||
|
result.plain_terms.append(term)
|
||||||
|
return
|
||||||
if term.is_not:
|
if term.is_not:
|
||||||
result.negated_date_ranges.add("DP")
|
result.negated_date_ranges.add("DP")
|
||||||
elif term.field == "EDAT":
|
elif term.field == "EDAT":
|
||||||
@@ -451,8 +484,12 @@ class PubmedQueryParser:
|
|||||||
elif _PARTIAL_DATE_RE.match(term.text):
|
elif _PARTIAL_DATE_RE.match(term.text):
|
||||||
result.edat_from, result.edat_to = _expand_partial_date(term.text)
|
result.edat_from, result.edat_to = _expand_partial_date(term.text)
|
||||||
else:
|
else:
|
||||||
|
if _validate_date_str(term.text):
|
||||||
result.edat_from = term.text
|
result.edat_from = term.text
|
||||||
result.edat_to = term.text
|
result.edat_to = term.text
|
||||||
|
else:
|
||||||
|
result.plain_terms.append(term)
|
||||||
|
return
|
||||||
if term.is_not:
|
if term.is_not:
|
||||||
result.negated_date_ranges.add("EDAT")
|
result.negated_date_ranges.add("EDAT")
|
||||||
elif term.field == "CRDT":
|
elif term.field == "CRDT":
|
||||||
@@ -462,8 +499,12 @@ class PubmedQueryParser:
|
|||||||
elif _PARTIAL_DATE_RE.match(term.text):
|
elif _PARTIAL_DATE_RE.match(term.text):
|
||||||
result.crdt_from, result.crdt_to = _expand_partial_date(term.text)
|
result.crdt_from, result.crdt_to = _expand_partial_date(term.text)
|
||||||
else:
|
else:
|
||||||
|
if _validate_date_str(term.text):
|
||||||
result.crdt_from = term.text
|
result.crdt_from = term.text
|
||||||
result.crdt_to = term.text
|
result.crdt_to = term.text
|
||||||
|
else:
|
||||||
|
result.plain_terms.append(term)
|
||||||
|
return
|
||||||
if term.is_not:
|
if term.is_not:
|
||||||
result.negated_date_ranges.add("CRDT")
|
result.negated_date_ranges.add("CRDT")
|
||||||
elif term.field == "MHDA":
|
elif term.field == "MHDA":
|
||||||
@@ -471,8 +512,12 @@ class PubmedQueryParser:
|
|||||||
result.year_from = int(term.text)
|
result.year_from = int(term.text)
|
||||||
result.year_to = int(term.text)
|
result.year_to = int(term.text)
|
||||||
else:
|
else:
|
||||||
|
if _validate_date_str(term.text):
|
||||||
result.mhda_from = term.text
|
result.mhda_from = term.text
|
||||||
result.mhda_to = term.text
|
result.mhda_to = term.text
|
||||||
|
else:
|
||||||
|
result.plain_terms.append(term)
|
||||||
|
return
|
||||||
if term.is_not:
|
if term.is_not:
|
||||||
result.negated_date_ranges.add("MHDA")
|
result.negated_date_ranges.add("MHDA")
|
||||||
elif term.field == "LR":
|
elif term.field == "LR":
|
||||||
@@ -480,8 +525,12 @@ class PubmedQueryParser:
|
|||||||
result.year_from = int(term.text)
|
result.year_from = int(term.text)
|
||||||
result.year_to = int(term.text)
|
result.year_to = int(term.text)
|
||||||
else:
|
else:
|
||||||
|
if _validate_date_str(term.text):
|
||||||
result.lr_from = term.text
|
result.lr_from = term.text
|
||||||
result.lr_to = term.text
|
result.lr_to = term.text
|
||||||
|
else:
|
||||||
|
result.plain_terms.append(term)
|
||||||
|
return
|
||||||
if term.is_not:
|
if term.is_not:
|
||||||
result.negated_date_ranges.add("LR")
|
result.negated_date_ranges.add("LR")
|
||||||
elif term.field == "DCOM":
|
elif term.field == "DCOM":
|
||||||
@@ -489,8 +538,12 @@ class PubmedQueryParser:
|
|||||||
result.year_from = int(term.text)
|
result.year_from = int(term.text)
|
||||||
result.year_to = int(term.text)
|
result.year_to = int(term.text)
|
||||||
else:
|
else:
|
||||||
|
if _validate_date_str(term.text):
|
||||||
result.dcom_from = term.text
|
result.dcom_from = term.text
|
||||||
result.dcom_to = term.text
|
result.dcom_to = term.text
|
||||||
|
else:
|
||||||
|
result.plain_terms.append(term)
|
||||||
|
return
|
||||||
if term.is_not:
|
if term.is_not:
|
||||||
result.negated_date_ranges.add("DCOM")
|
result.negated_date_ranges.add("DCOM")
|
||||||
elif term.field == "DEP":
|
elif term.field == "DEP":
|
||||||
@@ -498,8 +551,12 @@ class PubmedQueryParser:
|
|||||||
result.year_from = int(term.text)
|
result.year_from = int(term.text)
|
||||||
result.year_to = int(term.text)
|
result.year_to = int(term.text)
|
||||||
else:
|
else:
|
||||||
|
if _validate_date_str(term.text):
|
||||||
result.dep_from = term.text
|
result.dep_from = term.text
|
||||||
result.dep_to = term.text
|
result.dep_to = term.text
|
||||||
|
else:
|
||||||
|
result.plain_terms.append(term)
|
||||||
|
return
|
||||||
if term.is_not:
|
if term.is_not:
|
||||||
result.negated_date_ranges.add("DEP")
|
result.negated_date_ranges.add("DEP")
|
||||||
elif term.field == "__RANGE_DP__":
|
elif term.field == "__RANGE_DP__":
|
||||||
@@ -595,6 +652,8 @@ class PubmedQueryParser:
|
|||||||
tok = self.peek()
|
tok = self.peek()
|
||||||
if tok.type == TokenType.AND:
|
if tok.type == TokenType.AND:
|
||||||
self.advance()
|
self.advance()
|
||||||
|
if self.peek().type == TokenType.EOF:
|
||||||
|
break
|
||||||
elif self._is_primary_start(tok):
|
elif self._is_primary_start(tok):
|
||||||
pass # implicit AND — continue without consuming
|
pass # implicit AND — continue without consuming
|
||||||
else:
|
else:
|
||||||
@@ -746,12 +805,10 @@ class PubmedQueryParser:
|
|||||||
except IndexError:
|
except IndexError:
|
||||||
pass
|
pass
|
||||||
# P12: validate date values — non-numeric garbage falls back to plain text
|
# P12: validate date values — non-numeric garbage falls back to plain text
|
||||||
_valid_date = lambda s: (s.isdigit() and len(s) == 4) or (
|
_valid_date = lambda s: _validate_date_str(s)
|
||||||
len(s) == 10 and s[4] == '-' and s[7] == '-' and s.replace('-', '').isdigit()
|
|
||||||
)
|
|
||||||
if not _valid_date(start_val) or not _valid_date(end_val):
|
if not _valid_date(start_val) or not _valid_date(end_val):
|
||||||
txt = f"{start_val}:{end_val}[{field}]"
|
txt = f"{start_val}:{end_val}[{field}]"
|
||||||
return [Term(txt, field=field, is_not=negated)]
|
return [Term(txt, field=None, is_not=negated)]
|
||||||
# 确定两端是否是 4 位年份
|
# 确定两端是否是 4 位年份
|
||||||
_start_is_year = start_val.isdigit() and len(start_val) == 4
|
_start_is_year = start_val.isdigit() and len(start_val) == 4
|
||||||
_end_is_year = end_val.isdigit() and len(end_val) == 4
|
_end_is_year = end_val.isdigit() and len(end_val) == 4
|
||||||
|
|||||||
@@ -329,6 +329,7 @@ class AdvancedSearchEngine:
|
|||||||
tag_ids = list(existing | {str(t) for t in tag_matches})
|
tag_ids = list(existing | {str(t) for t in tag_matches})
|
||||||
# 不置空 query,保留 ILIKE 对中文标题/摘要的搜索能力
|
# 不置空 query,保留 ILIKE 对中文标题/摘要的搜索能力
|
||||||
if query.strip():
|
if query.strip():
|
||||||
|
_term_start = len(conditions) # P13: track term condition range for OR combining
|
||||||
# 提取引号短语作为完整词,避免 "lung cancer" 被拆散
|
# 提取引号短语作为完整词,避免 "lung cancer" 被拆散
|
||||||
import re as _phrase_re
|
import re as _phrase_re
|
||||||
_phrase_pat = _phrase_re.compile(r'"([^"]*)"')
|
_phrase_pat = _phrase_re.compile(r'"([^"]*)"')
|
||||||
@@ -403,6 +404,15 @@ class AdvancedSearchEngine:
|
|||||||
else:
|
else:
|
||||||
conditions.append(_atm_cond)
|
conditions.append(_atm_cond)
|
||||||
|
|
||||||
|
# P13: In OR mode, combine all term conditions into a single OR group
|
||||||
|
# (numeric conditions and text conditions are built separately but must be
|
||||||
|
# OR'd together, not AND'd at the final where() call)
|
||||||
|
if boolean == "or":
|
||||||
|
_term_conds = conditions[_term_start:]
|
||||||
|
if len(_term_conds) > 1:
|
||||||
|
del conditions[_term_start:]
|
||||||
|
conditions.append(or_(*_term_conds))
|
||||||
|
|
||||||
# 年份范围
|
# 年份范围
|
||||||
if year_from is not None:
|
if year_from is not None:
|
||||||
conditions.append(GlobalLiterature.pub_year >= year_from)
|
conditions.append(GlobalLiterature.pub_year >= year_from)
|
||||||
@@ -539,7 +549,7 @@ class AdvancedSearchEngine:
|
|||||||
|
|
||||||
# Exclude Preprints
|
# Exclude Preprints
|
||||||
if exclude_preprints:
|
if exclude_preprints:
|
||||||
conditions.append(GlobalLiterature.is_preprint == False)
|
conditions.append(GlobalLiterature.is_preprint != True)
|
||||||
|
|
||||||
# ── 按年份统计(Results by year,使用完整筛选条件) ──
|
# ── 按年份统计(Results by year,使用完整筛选条件) ──
|
||||||
_yr_before = len(conditions)
|
_yr_before = len(conditions)
|
||||||
@@ -1404,8 +1414,11 @@ class AdvancedSearchEngine:
|
|||||||
pat = _pt()
|
pat = _pt()
|
||||||
if exact and not _wildcard:
|
if exact and not _wildcard:
|
||||||
# P4: 精确短语 → phraseto_tsquery(利用 GIN 索引,保留词序)
|
# P4: 精确短语 → phraseto_tsquery(利用 GIN 索引,保留词序)
|
||||||
|
# P13: 加入 journal/journal_iso ILIKE(不在 tsvector 中)
|
||||||
return or_(
|
return or_(
|
||||||
GlobalLiterature.search_tsv.op("@@")(func.phraseto_tsquery("english", term)),
|
GlobalLiterature.search_tsv.op("@@")(func.phraseto_tsquery("english", term)),
|
||||||
|
GlobalLiterature.journal.ilike(pat),
|
||||||
|
GlobalLiterature.journal_iso.ilike(pat),
|
||||||
text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
|
text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
|
||||||
"WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat),
|
"WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat),
|
||||||
)
|
)
|
||||||
@@ -1617,7 +1630,6 @@ class AdvancedSearchEngine:
|
|||||||
GlobalLiterature.id.asc()]
|
GlobalLiterature.id.asc()]
|
||||||
elif sort == "journal":
|
elif sort == "journal":
|
||||||
return [GlobalLiterature.journal.asc().nullslast(),
|
return [GlobalLiterature.journal.asc().nullslast(),
|
||||||
GlobalLiterature.journal_iso.asc().nullslast(),
|
|
||||||
GlobalLiterature.id.asc()]
|
GlobalLiterature.id.asc()]
|
||||||
elif sort == "title":
|
elif sort == "title":
|
||||||
return [GlobalLiterature.title.asc().nullslast(),
|
return [GlobalLiterature.title.asc().nullslast(),
|
||||||
|
|||||||
+84
-5
@@ -2,9 +2,9 @@
|
|||||||
|
|
||||||
> 本文档按修复轮次详细记录所有搜索功能合规性修复的背景、根因分析和修改内容。
|
> 本文档按修复轮次详细记录所有搜索功能合规性修复的背景、根因分析和修改内容。
|
||||||
>
|
>
|
||||||
> **累计**:11 轮,137 项修复,50+ 字段标签注册,1007 项测试覆盖,7 项已知限制
|
> **累计**:13 轮,179 项修复,50+ 字段标签注册,1007 项测试覆盖,7 项已知限制
|
||||||
> **时间跨度**:2026-07-24 ~ 2026-07-28
|
> **时间跨度**:2026-07-24 ~ 2026-07-29
|
||||||
> **核心文件**:`pubmed_query_parser.py`(~730 行)→ `search_engine.py`(~1350 行)
|
> **核心文件**:`pubmed_query_parser.py`(~850 行)→ `search_engine.py`(~1360 行)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -21,7 +21,9 @@
|
|||||||
9. [第九轮:第 9 轮深度审计修复(5 项)](#第九轮第-9-轮深度审计修复)
|
9. [第九轮:第 9 轮深度审计修复(5 项)](#第九轮第-9-轮深度审计修复)
|
||||||
10. [第十轮:第 10 轮深度审计修复(6 项)](#第十轮第-10-轮深度审计修复)
|
10. [第十轮:第 10 轮深度审计修复(6 项)](#第十轮第-10-轮深度审计修复)
|
||||||
11. [第十一轮:第 11 轮深度审计修复(16 项)](#第十一轮第-11-轮深度审计修复)
|
11. [第十一轮:第 11 轮深度审计修复(16 项)](#第十一轮第-11-轮深度审计修复)
|
||||||
12. [遗留限制](#遗留限制)
|
12. [第十二轮:第 12 轮深度审计修复(21 项)](#第十二轮第-12-轮深度审计修复)
|
||||||
|
13. [第十三轮:第 13 轮深度审计修复(21 项)](#第十三轮第-13-轮深度审计修复)
|
||||||
|
14. [遗留限制](#遗留限制)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -1060,7 +1062,84 @@
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
截至 2026-07-28,剩余 7 项已知限制:
|
## 第十三轮:第 13 轮深度审计修复(21 项)
|
||||||
|
|
||||||
|
**日期**:2026-07-29
|
||||||
|
**提交**:`NEXT_COMMIT`
|
||||||
|
**数量**:21 项(1 P0 + 3 P1 + 2 P2 + 4 MINOR)
|
||||||
|
**触发**:用户第 8 次要求全面检查
|
||||||
|
**测试**:1007 全部通过 + 前端 build 通过
|
||||||
|
|
||||||
|
### P0-1: `_SPECIAL_FIELDS` 为 `set` 类型,误调用 `.get()` 导致 `AttributeError`(CRITICAL)
|
||||||
|
|
||||||
|
- **文件**:`pubmed_query_parser.py:44-49`
|
||||||
|
- **根因**:Round 12 新增的 `_normalize_field_label()` 函数在第 48 行调用 `_SPECIAL_FIELDS.get(raw)`。但 `_SPECIAL_FIELDS` 是 Python `set` 字面量(`{...}`),没有 `.get()` 方法。Python 在求值 `_FIELD_TAG_MAP.get(raw, _SPECIAL_FIELDS.get(raw))` 时会先计算第二个参数,无论 `raw` 是否在 `_FIELD_TAG_MAP` 中都会触发 `AttributeError`。`parse_pubmed_query` 的 `except` 只捕获 `(ParseError, IndexError, ValueError)`,`AttributeError` 传播到调用方 → 500 错误
|
||||||
|
- **修复**:拆分为三行:`if raw in _FIELD_TAG_MAP: return _FIELD_TAG_MAP[raw]` + `if raw in _SPECIAL_FIELDS: return raw` + `return None`
|
||||||
|
- **影响**:Round 12 引入的回归。`(cancer)[TI]`、`(a OR b)[DP]` 等所有带字段标签的括号组全面崩溃。本轮修复后恢复正常
|
||||||
|
- **验证**:`(lung cancer OR breast cancer)[TI]` 不再崩溃
|
||||||
|
|
||||||
|
### P1-1: `exclude_preprints` 丢弃 `is_preprint=NULL` 记录(HIGH)
|
||||||
|
|
||||||
|
- **文件**:`search_engine.py:541-542`
|
||||||
|
- **根因**:`GlobalLiterature.is_preprint == False` 生成 `WHERE is_preprint = false`。`is_preprint` 为 `NULL` 的旧文献(未解析此字段)被排除。`NULL = false` 在 SQL 三值逻辑中为 `NULL` → 被 WHERE 过滤
|
||||||
|
- **修复**:改为 `GlobalLiterature.is_preprint != True`,生成 `is_preprint IS DISTINCT FROM true`(NULL-safe,保留 false 和 NULL 行)
|
||||||
|
- **验证**:开启 `exclude_preprints` 筛选后,`is_preprint=NULL` 的记录不再被静默丢弃
|
||||||
|
|
||||||
|
### P1-2: 普通搜索 `boolean="or"` 模式下数字词和文本词被 AND 连接(HIGH)
|
||||||
|
|
||||||
|
- **文件**:`search_engine.py:341-405`
|
||||||
|
- **根因**:`boolean="or"` 时数字词条件(如 PMID 匹配)和文本词条件各自 OR 化后作为独立的元素加入 `conditions` 列表。最终 `and_(*conditions)` 将两者 AND 连接。例如 `"12345 cancer"` 且 `boolean="or"`:用户期望 `PMID=12345 OR 包含cancer`,实际执行 `PMID=12345 AND 包含cancer`
|
||||||
|
- **修复**:在数字词和文本词处理完成后,如果 `boolean == "or"`,将 `_term_start` 之后的所有词条件合并为一个 `or_(*_term_conds)`
|
||||||
|
- **验证**:`"30221571 pembrolizumab"` 且 `boolean="or"`,结果应为 PMID 30221571 或包含 pembrolizumab 的文章(OR),而非同时满足
|
||||||
|
|
||||||
|
### P1-3: Journal 排序 keyset 忽略 `journal_iso` 排序列(HIGH)
|
||||||
|
|
||||||
|
- **文件**:`search_engine.py:1618-1621`
|
||||||
|
- **根因**:`_apply_order_by("journal")` 返回 `[journal ASC, journal_iso ASC, id ASC]`。但 `_keyset_condition` 只处理 `journal` 和 `id`,完全忽略 `journal_iso`。同名期刊不同 ISO 缩写(如 `Nature` / `Nature (Lond.)` / `Nature (London)`)的文献在 keyset 翻页时被错误跳过
|
||||||
|
- **修复**:从 journal ORDER BY 中移除 `journal_iso`(keyset 分页不支持多列 tiebreaker,其他所有排序模式均使用单列 + id)
|
||||||
|
- **影响**:journal + journal_iso 组合排序在非 keyset 路径(总数据量少时用 OFFSET)也不影响结果正确性,仅影响同行期刊的展示顺序
|
||||||
|
|
||||||
|
### P2-1: 日期字段 `_dispatch_term` 未验证非日期文本(MEDIUM)
|
||||||
|
|
||||||
|
- **文件**:`pubmed_query_parser.py:436-533`
|
||||||
|
- **根因**:`cancer[DP]`、`foo[EDAT]` 等非日期文字传入日期字段时,`_dispatch_term` 的 else 分支直接赋值 `result.date_from = term.text`。`"cancer"` 作为非法日期值传入 PostgreSQL 查询 → `invalid input syntax for type date`
|
||||||
|
- **修复**:7 个日期字段(DP/EDAT/CRDT/MHDA/LR/DCOM/DEP)的 else 分支加入 `_validate_date_str()` 检查,非法文本路由到 `plain_terms`
|
||||||
|
- **验证**:`cancer[DP]` 不再导致 SQL 错误,退化到文本搜索
|
||||||
|
|
||||||
|
### P2-2: 日期范围回退保留日期字段标签(MEDIUM)
|
||||||
|
|
||||||
|
- **文件**:`pubmed_query_parser.py:752-754`
|
||||||
|
- **根因**:`lung:cancer[DP]` 范围中 `cancer` 不是合法日期,`_parse_range` 返回 `Term(txt, field="DP", ...)`。`_dispatch_term` 的 DP 分支将 `"lung:cancer[DP]"` 作为非法日期值处理
|
||||||
|
- **修复**:回退 Term 的 `field` 设为 `None`(而非保留 `field`),使其路由到 `plain_terms`
|
||||||
|
- **影响**:非法日期范围内容降级到纯文本搜索
|
||||||
|
|
||||||
|
### MINOR-1: `二月` 始终被扩展为 29 天,非闰年产生非法日期
|
||||||
|
|
||||||
|
- **文件**:`pubmed_query_parser.py:31-41`
|
||||||
|
- **根因**:`_LAST_DAY[2] = 29` 对所有年份生效。`2023-02[DP]` 被展开为 `2023-02-01` 到 `2023-02-29`,其中 `2023-02-29` 是非法日期
|
||||||
|
- **修复**:新增 `_is_leap_year()` 函数;`_expand_partial_date()` 中对 `month == 2 and last_day == 29` 且非闰年时置 `last_day = 28`
|
||||||
|
|
||||||
|
### MINOR-2: `_validate_date_str` 缺失日历正确性校验
|
||||||
|
|
||||||
|
- **文件**:`pubmed_query_parser.py:749-751`
|
||||||
|
- **根因**:Round 12 的 `_valid_date` lambda 只校验格式(是否为 YYYY 或 YYYY-MM-DD),不校验月份范围(1-12)和日期范围(1-月末)。`2024-13-01`、`2024-01-32` 等非法日历日期通过校验
|
||||||
|
- **修复**:新增 `_validate_date_str()` 替代原 lambda,完整校验格式 + 月份范围 + 日期范围 + 闰年 2 月
|
||||||
|
|
||||||
|
### MINOR-3: 尾部 AND 产生空 Term
|
||||||
|
|
||||||
|
- **文件**:`pubmed_query_parser.py:595-597`
|
||||||
|
- **根因**:`_parse_and_expr` 消耗 AND token 后未检查 EOF。`cancer AND` 中 `AND` 后的 `_parse_not_expr` 推进到 EOF 后返回空 Term
|
||||||
|
- **修复**:`self.advance()` 后检查 `self.peek().type == TokenType.EOF → break`
|
||||||
|
|
||||||
|
### MINOR-4: 精确短语 "all" 搜索缺少 journal ILIKE 回退
|
||||||
|
|
||||||
|
- **文件**:`search_engine.py:1415-1424`
|
||||||
|
- **根因**:`exact=True` 时 `_field_condition("all")` 只搜索 tsvector(phraseto_tsquery)和 affiliation ILIKE。journal/journal_iso 不在 tsvector 中(注释 line 1454 确认),"Nature" 作为精确短语搜索时无法匹配期刊名。非精确路径(line 1456-1462)正确包含 journal ILIKE
|
||||||
|
- **修复**:在精确短语路径中加入 `GlobalLiterature.journal.ilike(pat)` 和 `GlobalLiterature.journal_iso.ilike(pat)`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
截至 2026-07-29,剩余 7 项已知限制:
|
||||||
|
|
||||||
| ID | 问题 | 原因 | 影响 |
|
| ID | 问题 | 原因 | 影响 |
|
||||||
|----|------|------|------|
|
|----|------|------|------|
|
||||||
|
|||||||
Reference in New Issue
Block a user