2270 lines
111 KiB
Python
2270 lines
111 KiB
Python
"""高级搜索服务:布尔运算 + 字段限定 + PubMed 查询语法"""
|
||
|
||
import logging
|
||
import re
|
||
from datetime import datetime
|
||
|
||
from sqlalchemy import String, and_, case, cast, exists, func, literal_column, not_, or_, select, text
|
||
|
||
logger = logging.getLogger(__name__)
|
||
from sqlalchemy.ext.asyncio import AsyncSession
|
||
|
||
from app.core.cache import cache as _cache
|
||
from app.models.literature import GlobalJournal, GlobalLiterature, GlobalLiteratureTag, GlobalTag, GlobalTagTreeNumber
|
||
from app.schemas.literature import cap_pub_date
|
||
from app.services.pubmed_query_parser import is_pubmed_syntax, parse_pubmed_query
|
||
from app.services.query_expansion import expand_atm as _expand_atm
|
||
from app.services.tag_loader import load_tags_for_literature
|
||
|
||
|
||
def _escape_ilike(s: str) -> str:
|
||
"""转义 ILIKE 模式中的通配符 _ 和 %,防止用户输入 'EGFR_mutation' 误匹配 'EGFR mutation'
|
||
|
||
注意:先反转义已转义的 \\% 和 \\_ → % 和 _,再整体转义,避免双重转义。
|
||
"""
|
||
if not s:
|
||
return s
|
||
return s.replace('\\%', '%').replace('\\_', '_').replace('\\', '\\\\').replace('%', '\\%').replace('_', '\\_')
|
||
|
||
|
||
class AdvancedSearchEngine:
|
||
"""PG 高级搜索(ES 就绪后切换 search_service.py)"""
|
||
|
||
SEARCH_CACHE_TTL = 300 # 秒(夜间流水线更新,5分钟缓存很安全)
|
||
|
||
@staticmethod
|
||
def _search_cache_key(
|
||
query: str, field: str, boolean: str, exact_phrase: bool,
|
||
year_from: int | None, year_to: int | None,
|
||
date_from: str | None, date_to: str | None,
|
||
journal_tiers: list[str] | None, pub_types: list[str] | None,
|
||
tag_ids: list[str] | None,
|
||
retracted: str, negative_result: str,
|
||
is_oa: bool | None, language: str | None, languages: list[str] | None, nlm_subsets: list[str] | None,
|
||
page_size: int, sort: str, page: int = 1,
|
||
# PubMed filter params
|
||
has_abstract: bool | None = None,
|
||
is_free_full_text: bool | None = None,
|
||
has_full_text: bool | None = None,
|
||
has_associated_data: bool | None = None,
|
||
species: list[str] | None = None,
|
||
sex: list[str] | None = None,
|
||
age: list[str] | None = None,
|
||
medline_only: bool = False,
|
||
exclude_preprints: bool = False,
|
||
# Generic keyset cursor
|
||
cursor_val: str | None = None,
|
||
cursor_id: str | None = None,
|
||
) -> str:
|
||
"""归一化查询参数 → 确定性缓存 key(所有 list 排序后参与哈希)"""
|
||
import hashlib, json
|
||
from app.services.pubmed_query_parser import is_pubmed_syntax as _is_pm
|
||
norm = {
|
||
"q": query.strip().lower(),
|
||
"pm": _is_pm(query),
|
||
"f": field, "b": boolean, "ep": exact_phrase,
|
||
"yf": year_from, "yt": year_to,
|
||
"df": date_from, "dt": date_to,
|
||
"jt": sorted(journal_tiers) if journal_tiers else [],
|
||
"pt": sorted(pub_types) if pub_types else [],
|
||
"tid": sorted(tag_ids) if tag_ids else [],
|
||
"r": retracted, "nr": negative_result,
|
||
"oa": is_oa, "lang": language, "lgs": sorted(languages) if languages else [],
|
||
"ns": sorted(nlm_subsets) if nlm_subsets else [],
|
||
"ps": page_size, "s": sort, "p": page,
|
||
"ha": has_abstract,
|
||
"fft": is_free_full_text,
|
||
"hft": has_full_text,
|
||
"had": has_associated_data,
|
||
"sp": sorted(species) if species else [],
|
||
"sx": sorted(sex) if sex else [],
|
||
"ag": sorted(age) if age else [],
|
||
"mo": medline_only,
|
||
"epr": exclude_preprints,
|
||
# keyset cursor 唯一标识翻页位置(不含时为第 1 页)
|
||
"cv": cursor_val,
|
||
"ci": cursor_id,
|
||
}
|
||
raw = json.dumps(norm, sort_keys=True, ensure_ascii=False, default=str)
|
||
return f"search:advanced:{hashlib.md5(raw.encode()).hexdigest()}"
|
||
|
||
@staticmethod
|
||
def _facet_cache_key(
|
||
query: str, field: str, boolean: str, exact_phrase: bool,
|
||
year_from: int | None, year_to: int | None,
|
||
date_from: str | None, date_to: str | None,
|
||
journal_tiers: list[str] | None, pub_types: list[str] | None,
|
||
tag_ids: list[str] | None,
|
||
retracted: str, negative_result: str,
|
||
is_oa: bool | None, language: str | None, languages: list[str] | None, nlm_subsets: list[str] | None,
|
||
# PubMed filter params (same as _search_cache_key minus page/cursor/sort)
|
||
has_abstract: bool | None = None,
|
||
is_free_full_text: bool | None = None,
|
||
has_full_text: bool | None = None,
|
||
has_associated_data: bool | None = None,
|
||
species: list[str] | None = None,
|
||
sex: list[str] | None = None,
|
||
age: list[str] | None = None,
|
||
medline_only: bool = False,
|
||
exclude_preprints: bool = False,
|
||
) -> str:
|
||
"""归一化筛选条件 → year_counts/facet 缓存 key(不含 page/cursor/sort)
|
||
|
||
year_counts 只依赖筛选条件,分页/排序不影响分布。
|
||
与 _search_cache_key 的区别:不含 p/ps/s/cd/ci。
|
||
"""
|
||
import hashlib, json
|
||
from app.services.pubmed_query_parser import is_pubmed_syntax as _is_pm
|
||
norm = {
|
||
"q": query.strip().lower(),
|
||
"pm": _is_pm(query),
|
||
"f": field, "b": boolean, "ep": exact_phrase,
|
||
"yf": year_from, "yt": year_to,
|
||
"df": date_from, "dt": date_to,
|
||
"jt": sorted(journal_tiers) if journal_tiers else [],
|
||
"pt": sorted(pub_types) if pub_types else [],
|
||
"tid": sorted(tag_ids) if tag_ids else [],
|
||
"r": retracted, "nr": negative_result,
|
||
"oa": is_oa, "lang": language, "lgs": sorted(languages) if languages else [],
|
||
"ns": sorted(nlm_subsets) if nlm_subsets else [],
|
||
"ha": has_abstract,
|
||
"fft": is_free_full_text,
|
||
"hft": has_full_text,
|
||
"had": has_associated_data,
|
||
"sp": sorted(species) if species else [],
|
||
"sx": sorted(sex) if sex else [],
|
||
"ag": sorted(age) if age else [],
|
||
"mo": medline_only,
|
||
"epr": exclude_preprints,
|
||
}
|
||
raw = json.dumps(norm, sort_keys=True, ensure_ascii=False, default=str)
|
||
return f"search:year_counts:{hashlib.md5(raw.encode()).hexdigest()}"
|
||
|
||
@staticmethod
|
||
async def search(
|
||
db: AsyncSession,
|
||
query: str = "",
|
||
field: str = "all",
|
||
boolean: str = "and",
|
||
exact_phrase: bool = False,
|
||
year_from: int | None = None,
|
||
year_to: int | None = None,
|
||
date_from: str | None = None, # YYYY-MM-DD
|
||
date_to: str | None = None, # YYYY-MM-DD
|
||
journal_tiers: list[str] | None = None,
|
||
pub_types: list[str] | None = None,
|
||
tag_ids: list[str] | None = None,
|
||
retracted: str = "", # "no", "only"
|
||
negative_result: str = "", # "yes", "no", "only"
|
||
is_oa: bool | None = None, # True = 仅开放获取
|
||
language: str | None = None, # 语言代码(en/zh/fr 等,向后兼容)
|
||
languages: list[str] | None = None, # 语言代码列表(多选)
|
||
nlm_subsets: list[str] | None = None, # NLM 期刊子集(AIM/M/S 等)
|
||
page: int = 1,
|
||
page_size: int = 20,
|
||
sort: str = "date",
|
||
cursor_val: str | None = None, # 通用 keyset 游标值(sort 模式对应列的值)
|
||
cursor_id: str | None = None, # keyset 游标:上一页最后一条的 id(UUID)
|
||
cursor_date: str | None = None, # 向后兼容:旧 date 游标,映射到 cursor_val
|
||
# ── PubMed 筛选器参数 ──
|
||
has_abstract: bool | None = None,
|
||
is_free_full_text: bool | None = None,
|
||
has_full_text: bool | None = None,
|
||
has_associated_data: bool | None = None,
|
||
species: list[str] | None = None,
|
||
sex: list[str] | None = None,
|
||
age: list[str] | None = None,
|
||
medline_only: bool = False,
|
||
exclude_preprints: bool = False,
|
||
) -> dict:
|
||
"""执行高级搜索"""
|
||
from sqlalchemy.dialects.postgresql import JSONB
|
||
# R24: cap page_size and tag_ids to prevent abuse
|
||
page_size = min(page_size, 100)
|
||
if tag_ids:
|
||
tag_ids = list(set(str(t) for t in tag_ids))[:200]
|
||
conditions = []
|
||
|
||
# 向后兼容:cursor_date → cursor_val(旧前端发 cursor_date)
|
||
if cursor_val is None and cursor_date is not None:
|
||
cursor_val = cursor_date
|
||
|
||
is_first_page = (not cursor_val and cursor_id is None)
|
||
_search_cache_key = AdvancedSearchEngine._search_cache_key(
|
||
query, field, boolean, exact_phrase,
|
||
year_from, year_to, date_from, date_to,
|
||
journal_tiers, pub_types, tag_ids,
|
||
retracted, negative_result,
|
||
is_oa, language, languages, nlm_subsets,
|
||
page_size, sort, page,
|
||
has_abstract=has_abstract,
|
||
is_free_full_text=is_free_full_text,
|
||
has_full_text=has_full_text,
|
||
has_associated_data=has_associated_data,
|
||
species=species, sex=sex, age=age,
|
||
medline_only=medline_only,
|
||
exclude_preprints=exclude_preprints,
|
||
cursor_val=cursor_val,
|
||
cursor_id=cursor_id,
|
||
)
|
||
# year_counts 缓存键(不含 page/cursor/sort,所有页共享一份)
|
||
_facet_cache_key = AdvancedSearchEngine._facet_cache_key(
|
||
query, field, boolean, exact_phrase,
|
||
year_from, year_to, date_from, date_to,
|
||
journal_tiers, pub_types, tag_ids,
|
||
retracted, negative_result,
|
||
is_oa, language, languages, nlm_subsets,
|
||
has_abstract=has_abstract,
|
||
is_free_full_text=is_free_full_text,
|
||
has_full_text=has_full_text,
|
||
has_associated_data=has_associated_data,
|
||
species=species, sex=sex, age=age,
|
||
medline_only=medline_only,
|
||
exclude_preprints=exclude_preprints,
|
||
)
|
||
cached = await _cache.get(_search_cache_key)
|
||
if cached is not None:
|
||
# 谢缓存:只存了 lit_ids + 聚合数据,需从 ②③ 缓存回填 tags/journal
|
||
tier_map, name_map = await AdvancedSearchEngine._get_journal_map(db)
|
||
items = await AdvancedSearchEngine._hydrate_items(db, cached["lit_ids"], tier_map, name_map)
|
||
# year_counts 从独立 facet 缓存取(兼容 list 和 dict 两种格式)
|
||
yc_cached = await _cache.get(_facet_cache_key)
|
||
if isinstance(yc_cached, dict):
|
||
year_counts = yc_cached.get("year_counts", [])
|
||
elif isinstance(yc_cached, list):
|
||
year_counts = yc_cached
|
||
else:
|
||
year_counts = []
|
||
return {
|
||
"items": items,
|
||
"total": cached["total"],
|
||
"page": cached["page"],
|
||
"page_size": cached["page_size"],
|
||
"has_more": cached["has_more"],
|
||
"year_counts": year_counts,
|
||
"cursor_val": cached.get("cursor_val"),
|
||
"cursor_id": cached.get("cursor_id"),
|
||
}
|
||
|
||
# 30 秒查询超时(放在缓存检查之后,缓存命中不执行)
|
||
await db.execute(text("SET LOCAL statement_timeout = '30s'"))
|
||
|
||
# ─── PubMed 语法检测与解析 ───
|
||
_pubmed_parsed = None
|
||
_pubmed_failed = False
|
||
_is_flat_text = True #是否走传统 tsvector + ILIKE 路径
|
||
if query.strip():
|
||
|
||
if is_pubmed_syntax(query):
|
||
pp = parse_pubmed_query(query)
|
||
_pubmed_parsed = pp
|
||
|
||
# 只要解析器产出了有效结构,就走 PubMed 路径
|
||
has_pubmed_terms = bool(
|
||
pp.title_terms or pp.abstract_terms or pp.tiab_terms
|
||
or pp.author_terms or pp.journal_terms
|
||
or pp.mesh_terms or pp.majr_terms
|
||
or pp.pub_types or pp.doi_terms or pp.pmid_terms
|
||
or pp.affiliation_terms
|
||
or pp.language_terms or pp.volume_terms or pp.issue_terms
|
||
or pp.pages_terms or pp.lid_terms
|
||
or pp.grant_terms or pp.subheading_terms
|
||
or pp.registry_terms or pp.substance_terms
|
||
or pp.databank_terms or pp.pharmaco_terms
|
||
or pp.ed_terms or pp.investigator_terms or pp.personal_name_terms
|
||
or pp.pubnote_terms or pp.auid_terms or pp.cois_terms or pp.tt_terms
|
||
or pp.sb_terms or pp.stat_terms or pp.uid_terms
|
||
or pp.ot_terms or pp.gene_terms or pp.pmc_terms
|
||
or pp.edat_from or pp.crdt_from or pp.mhda_from
|
||
or pp.lr_from or pp.dcom_from or pp.dep_from
|
||
or pp.edat_to or pp.crdt_to or pp.mhda_to
|
||
or pp.lr_to or pp.dcom_to or pp.dep_to
|
||
or pp.date_from or pp.date_to
|
||
or pp.negated_date_ranges
|
||
or pp.groups
|
||
or pp.plain_terms or pp.has_not
|
||
or pp.year_from or pp.year_to
|
||
)
|
||
if has_pubmed_terms:
|
||
_is_flat_text = False
|
||
# P17: 异常保护 — _pubmed_conditions 内部异常时降级到 flat text
|
||
try:
|
||
conditions = await AdvancedSearchEngine._pubmed_conditions(
|
||
db, pp, conditions,
|
||
)
|
||
except Exception:
|
||
logger.exception("_pubmed_conditions failed, falling back to flat text")
|
||
_is_flat_text = True
|
||
_pubmed_failed = True
|
||
conditions = []
|
||
# R28: force query cleaning in flat text fallback
|
||
_pubmed_parsed = pp # keep parsed object, flag below triggers cleanup
|
||
|
||
# ─── 传统搜索路径(纯文本 / PubMed 退化) ───
|
||
if _is_flat_text and query.strip():
|
||
# 如果 PubMed 语法检测成功但解析后无有效词(如语法错误),剥离括号和操作符再搜索
|
||
if _pubmed_parsed is not None and (not any([
|
||
bool(_pubmed_parsed.title_terms or _pubmed_parsed.abstract_terms or _pubmed_parsed.tiab_terms
|
||
or _pubmed_parsed.author_terms or _pubmed_parsed.journal_terms
|
||
or _pubmed_parsed.mesh_terms or _pubmed_parsed.majr_terms
|
||
or _pubmed_parsed.pub_types or _pubmed_parsed.doi_terms or _pubmed_parsed.pmid_terms
|
||
or _pubmed_parsed.affiliation_terms
|
||
or _pubmed_parsed.language_terms or _pubmed_parsed.volume_terms or _pubmed_parsed.issue_terms
|
||
or _pubmed_parsed.pages_terms or _pubmed_parsed.lid_terms
|
||
or _pubmed_parsed.grant_terms or _pubmed_parsed.subheading_terms
|
||
or _pubmed_parsed.registry_terms or _pubmed_parsed.substance_terms
|
||
or _pubmed_parsed.databank_terms or _pubmed_parsed.pharmaco_terms
|
||
or _pubmed_parsed.ed_terms or _pubmed_parsed.investigator_terms or _pubmed_parsed.personal_name_terms
|
||
or _pubmed_parsed.pubnote_terms or _pubmed_parsed.auid_terms or _pubmed_parsed.cois_terms or _pubmed_parsed.tt_terms
|
||
or _pubmed_parsed.sb_terms or _pubmed_parsed.stat_terms or _pubmed_parsed.uid_terms
|
||
or _pubmed_parsed.ot_terms or _pubmed_parsed.gene_terms or _pubmed_parsed.pmc_terms
|
||
or _pubmed_parsed.edat_from or _pubmed_parsed.crdt_from or _pubmed_parsed.mhda_from
|
||
or _pubmed_parsed.lr_from or _pubmed_parsed.dcom_from or _pubmed_parsed.dep_from
|
||
or _pubmed_parsed.edat_to or _pubmed_parsed.crdt_to or _pubmed_parsed.mhda_to
|
||
or _pubmed_parsed.lr_to or _pubmed_parsed.dcom_to or _pubmed_parsed.dep_to
|
||
or _pubmed_parsed.date_from or _pubmed_parsed.date_to
|
||
or _pubmed_parsed.negated_date_ranges
|
||
or _pubmed_parsed.groups
|
||
or _pubmed_parsed.plain_terms or _pubmed_parsed.has_not
|
||
or _pubmed_parsed.year_from or _pubmed_parsed.year_to)
|
||
]) or _pubmed_failed):
|
||
# 解析失败但检测到 PubMed 语法 — 擦除 [field] 标签、布尔符、引号
|
||
query = re.sub(r'\[[\w/: -]+\]', '', query) # P5: [\w/: -] 覆盖 [Title/Abstract] 和 [MH:noexp]
|
||
query = re.sub(r'\b(AND|OR|NOT)\b', '', query)
|
||
query = query.replace('"', '').replace('(', '').replace(')', '')
|
||
query = ' '.join(query.split())
|
||
# P17: query 被原地修改后,缓存键指向脏数据 → 重新计算
|
||
_search_cache_key = AdvancedSearchEngine._search_cache_key(
|
||
query, field, boolean, exact_phrase,
|
||
year_from, year_to, date_from, date_to,
|
||
journal_tiers, pub_types, tag_ids,
|
||
retracted, negative_result,
|
||
is_oa, language, languages, nlm_subsets,
|
||
page_size, sort, page,
|
||
has_abstract=has_abstract,
|
||
is_free_full_text=is_free_full_text,
|
||
has_full_text=has_full_text,
|
||
has_associated_data=has_associated_data,
|
||
species=species, sex=sex, age=age,
|
||
medline_only=medline_only,
|
||
exclude_preprints=exclude_preprints,
|
||
cursor_val=cursor_val,
|
||
cursor_id=cursor_id,
|
||
)
|
||
_facet_cache_key = AdvancedSearchEngine._facet_cache_key(
|
||
query, field, boolean, exact_phrase,
|
||
year_from, year_to, date_from, date_to,
|
||
journal_tiers, pub_types, tag_ids,
|
||
retracted, negative_result,
|
||
is_oa, language, languages, nlm_subsets,
|
||
has_abstract=has_abstract,
|
||
is_free_full_text=is_free_full_text,
|
||
has_full_text=has_full_text,
|
||
has_associated_data=has_associated_data,
|
||
species=species, sex=sex, age=age,
|
||
medline_only=medline_only,
|
||
exclude_preprints=exclude_preprints,
|
||
)
|
||
# 中文搜索:自动匹配 GlobalTag.name_zh → 注入 tag_ids,跳过 ILIKE
|
||
_CHINESE_RE = re.compile(r'[一-鿿㐀-䶿豈-]')
|
||
if _CHINESE_RE.search(query):
|
||
tag_matches = (await db.execute(
|
||
select(GlobalTag.id).where(GlobalTag.name_zh.ilike(f'%{_escape_ilike(query.strip())}%'))
|
||
.limit(100) # P5: limit to prevent oversized subquery
|
||
)).scalars().all()
|
||
if tag_matches:
|
||
existing = set(tag_ids or [])
|
||
tag_ids = list(existing | {str(t) for t in tag_matches})
|
||
# 不置空 query,保留 ILIKE 对中文标题/摘要的搜索能力
|
||
if query.strip():
|
||
_term_start = len(conditions) # P13: track term condition range for OR combining
|
||
# 提取引号短语作为完整词,避免 "lung cancer" 被拆散
|
||
_phrase_pat = re.compile(r'"([^"]*)"')
|
||
_phrases = _phrase_pat.findall(query)
|
||
_query_no_quotes = _phrase_pat.sub(' ', query)
|
||
_rest = [t.strip().strip('"').strip("'") for t in _query_no_quotes.split() if t.strip()]
|
||
# P17: 记录引号短语,后续强制 exact=True (MUST be before terms to avoid closure NameError)
|
||
_phrase_terms_set = set(p.lower() for p in _phrases if p.strip())
|
||
terms = [p for p in _phrases if p.strip()] + [t for t in _rest if t.lower() not in _phrase_terms_set]
|
||
# 单数字词:优先 PMID 精确匹配(unique index 5ms 返回)
|
||
# 不是 PMID 时才回退到 ILIKE 兜底(DOI 片段等),不做 tsquery 避免 seq scan
|
||
numeric_terms = [t for t in terms if re.match(r'^\d{1,15}$', t)]
|
||
text_terms = [t for t in terms if not re.match(r'^\d{1,15}$', t)]
|
||
if numeric_terms:
|
||
num_conds = []
|
||
if exact_phrase:
|
||
# R24: always include PMID check even in exact_phrase mode
|
||
for t in numeric_terms:
|
||
num_conds.append(or_(
|
||
GlobalLiterature.pmid == int(t),
|
||
GlobalLiterature.title.ilike(t),
|
||
GlobalLiterature.doi.ilike(t),
|
||
))
|
||
else:
|
||
# P1-1: 批量检查所有 PMID(单次 IN 查询替代 N+1 次 SELECT)
|
||
numeric_ids = [int(t) for t in numeric_terms]
|
||
db_pmids: set[int] = set()
|
||
rows = await db.execute(
|
||
select(GlobalLiterature.pmid).where(GlobalLiterature.pmid.in_(numeric_ids))
|
||
)
|
||
for (pmid,) in rows:
|
||
db_pmids.add(pmid)
|
||
for t in numeric_terms:
|
||
p = int(t)
|
||
if p in db_pmids:
|
||
num_conds.append(GlobalLiterature.pmid == p)
|
||
else:
|
||
p_like = f"%{_escape_ilike(t)}%"
|
||
num_conds.append(or_(
|
||
GlobalLiterature.title.ilike(p_like),
|
||
GlobalLiterature.doi.ilike(p_like),
|
||
))
|
||
if boolean == "or" and len(num_conds) > 1:
|
||
conditions.append(or_(*num_conds))
|
||
else:
|
||
conditions.extend(num_conds)
|
||
if text_terms:
|
||
# ATM 展开(仅 field="all" 时,字段搜索不应自动扩到 MeSH)
|
||
_atm_cond = None
|
||
_atm_query = query.replace('"', '').replace("'", '').replace('(', '').replace(')', '').strip()
|
||
if _atm_query and field == "all":
|
||
try:
|
||
_atm_cond = await _expand_atm(db, _atm_query)
|
||
except Exception:
|
||
logger.exception("ATM expansion failed (flat text): %s", _atm_query[:100])
|
||
_atm_cond = None
|
||
|
||
_cond_before = len(conditions)
|
||
if boolean == "and":
|
||
for term in text_terms:
|
||
_exact = exact_phrase or term.lower() in _phrase_terms_set # P17: 引号短语强制 exact
|
||
conditions.append(AdvancedSearchEngine._field_condition(field, term, _exact))
|
||
else:
|
||
or_conds = [AdvancedSearchEngine._field_condition(field, t, exact_phrase or t.lower() in _phrase_terms_set) for t in text_terms]
|
||
conditions.append(or_(*or_conds))
|
||
|
||
# 将文本条件与 ATM 条件 OR 组合
|
||
if _atm_cond is not None:
|
||
_text_conds = conditions[_cond_before:]
|
||
del conditions[_cond_before:]
|
||
if _text_conds:
|
||
if len(_text_conds) == 1:
|
||
conditions.append(or_(_atm_cond, _text_conds[0]))
|
||
else:
|
||
conditions.append(or_(_atm_cond, and_(*_text_conds)))
|
||
else:
|
||
conditions.append(_atm_cond)
|
||
|
||
# P13: In OR mode, combine all term conditions into a single OR group
|
||
# (numeric conditions and text conditions are built separately but must be
|
||
# OR'd together, not AND'd at the final where() call)
|
||
if boolean == "or":
|
||
_term_conds = conditions[_term_start:]
|
||
if len(_term_conds) > 1:
|
||
del conditions[_term_start:]
|
||
conditions.append(or_(*_term_conds))
|
||
|
||
# 年份范围
|
||
if year_from is not None:
|
||
conditions.append(GlobalLiterature.pub_year >= year_from)
|
||
if year_to is not None:
|
||
conditions.append(GlobalLiterature.pub_year <= year_to)
|
||
|
||
# 具体日期范围(按天搜索)
|
||
from datetime import date as dt_date
|
||
if date_from:
|
||
try:
|
||
df = dt_date.fromisoformat(date_from)
|
||
conditions.append(GlobalLiterature.pub_date >= df)
|
||
except ValueError:
|
||
raise ValueError(f"Invalid date_from format: {date_from}")
|
||
if date_to:
|
||
try:
|
||
dt = dt_date.fromisoformat(date_to)
|
||
conditions.append(GlobalLiterature.pub_date <= dt)
|
||
except ValueError:
|
||
raise ValueError(f"Invalid date_to format: {date_to}")
|
||
|
||
# 期刊等级
|
||
if journal_tiers:
|
||
subq = select(GlobalJournal.issn).where(GlobalJournal.tier.in_(journal_tiers))
|
||
result = await db.execute(subq)
|
||
issns = [r for (r,) in result.all()]
|
||
if not issns:
|
||
logger.warning("journal_tiers filter matched zero journals: %s", journal_tiers)
|
||
conditions.append(GlobalLiterature.journal_issn.in_(issns))
|
||
|
||
# 标签筛选(含子标签递归)
|
||
if tag_ids:
|
||
import uuid as _uuid
|
||
tag_uuids = [_uuid.UUID(tid) if isinstance(tid, str) else tid for tid in tag_ids]
|
||
all_tags = (await db.execute(select(GlobalTag).where(GlobalTag.id.in_(tag_uuids)))).scalars().all()
|
||
all_tag_ids = set(t.id for t in all_tags if t)
|
||
# Batch child tag lookup (1 query instead of N)
|
||
paths = [t.path + "::" for t in all_tags if t]
|
||
if paths:
|
||
child_conds = [GlobalTag.path.like(f"{p}%") for p in paths]
|
||
children = (await db.execute(select(GlobalTag.id).where(or_(*child_conds)))).scalars().all()
|
||
all_tag_ids.update(children)
|
||
uids = list(all_tag_ids)
|
||
from sqlalchemy.dialects.postgresql import array as _pg_array
|
||
conditions.append(GlobalLiterature.tag_ids.overlap(_pg_array(uids)))
|
||
|
||
# 发表类型(PG JSONB contains)
|
||
if pub_types:
|
||
type_conds = [GlobalLiterature.pub_types.cast(JSONB).contains([pt]) for pt in pub_types]
|
||
conditions.append(or_(*type_conds))
|
||
|
||
# 撤稿过滤
|
||
if retracted == "no":
|
||
conditions.append(or_(
|
||
GlobalLiterature.retracted == False,
|
||
GlobalLiterature.retracted.is_(None),
|
||
))
|
||
elif retracted in ("only", "yes"):
|
||
conditions.append(GlobalLiterature.retracted == True)
|
||
|
||
# 阴性结果过滤
|
||
if negative_result == "no":
|
||
conditions.append(or_(
|
||
GlobalLiterature.is_negative_result == False,
|
||
GlobalLiterature.is_negative_result.is_(None),
|
||
))
|
||
elif negative_result in ("only", "yes"):
|
||
conditions.append(GlobalLiterature.is_negative_result == True)
|
||
|
||
# 开放获取
|
||
if is_oa is not None:
|
||
conditions.append(GlobalLiterature.is_oa == is_oa)
|
||
|
||
# 语言(多选优先,向后兼容单语言)
|
||
if languages:
|
||
conditions.append(GlobalLiterature.language.in_(languages))
|
||
elif language:
|
||
conditions.append(GlobalLiterature.language == language)
|
||
|
||
# NLM 期刊子集(如 AIM / Core Clinical Journals)
|
||
if nlm_subsets:
|
||
subq = select(GlobalJournal.issn).where(GlobalJournal.nlm_subsets.overlap(nlm_subsets))
|
||
result = await db.execute(subq)
|
||
issns = [r for (r,) in result.all()]
|
||
if not issns:
|
||
logger.warning("nlm_subsets filter matched zero journals: %s", nlm_subsets)
|
||
conditions.append(GlobalLiterature.journal_issn.in_(issns))
|
||
|
||
# ── PubMed 筛选器 ──
|
||
|
||
# Text Availability
|
||
if has_abstract:
|
||
conditions.append(and_(
|
||
GlobalLiterature.abstract.isnot(None),
|
||
GlobalLiterature.abstract != '',
|
||
))
|
||
if is_free_full_text:
|
||
conditions.append(GlobalLiterature.is_oa == True)
|
||
if has_full_text:
|
||
conditions.append(GlobalLiterature.pmc_id.isnot(None))
|
||
|
||
# Article Attribute: Associated data
|
||
if has_associated_data:
|
||
conditions.append(GlobalLiterature.databank_list.cast(JSONB) != '[]')
|
||
|
||
# Species / Sex / Age — mesh_headings JSONB @> UI match
|
||
if species:
|
||
conditions.append(or_(*[
|
||
GlobalLiterature.mesh_headings.contains([{"ui": ui}])
|
||
for ui in species
|
||
]))
|
||
if sex:
|
||
conditions.append(or_(*[
|
||
GlobalLiterature.mesh_headings.contains([{"ui": ui}])
|
||
for ui in sex
|
||
]))
|
||
if age:
|
||
from app.core.constants import AGE_GROUP_UI_MAP
|
||
age_uis = []
|
||
for val in age:
|
||
if val in AGE_GROUP_UI_MAP:
|
||
age_uis.extend(AGE_GROUP_UI_MAP[val])
|
||
else:
|
||
age_uis.append(val) # 直接传 UI 的情况(向后兼容)
|
||
conditions.append(or_(*[
|
||
GlobalLiterature.mesh_headings.contains([{"ui": ui}])
|
||
for ui in set(age_uis)
|
||
]))
|
||
|
||
# MEDLINE only(统一使用 journal.nlm_subsets,与筛选面板计数一致)
|
||
if medline_only:
|
||
_med_subq = select(GlobalJournal.issn).where(GlobalJournal.nlm_subsets.overlap(["M"]))
|
||
conditions.append(GlobalLiterature.journal_issn.in_(_med_subq))
|
||
|
||
# Exclude Preprints
|
||
if exclude_preprints:
|
||
conditions.append(GlobalLiterature.is_preprint != True)
|
||
|
||
# ── 按年份统计(Results by year,使用完整筛选条件) ──
|
||
_yr_before = len(conditions)
|
||
year_counts = []
|
||
|
||
# 先从 facet 缓存取(不含 page/cursor,所有页共享,TTL 更长)
|
||
yc_cached = await _cache.get(_facet_cache_key)
|
||
if yc_cached is not None:
|
||
year_counts = yc_cached if isinstance(yc_cached, list) else yc_cached.get("year_counts", [])
|
||
|
||
# 判断是否有任何筛选器/文本查询活跃
|
||
# R24: exclude query.strip() — PubMed queries are fully consumed by conditions
|
||
_has_any_filter = (year_from or year_to or date_from or date_to
|
||
or journal_tiers or pub_types or tag_ids
|
||
or retracted or negative_result or is_oa is not None
|
||
or language or languages or nlm_subsets
|
||
or has_abstract or is_free_full_text or has_full_text
|
||
or has_associated_data or species or sex or age
|
||
or medline_only or exclude_preprints)
|
||
|
||
if year_counts:
|
||
pass # facet 缓存命中
|
||
elif not _has_any_filter and not conditions:
|
||
_cached = await _cache.get("search:year_counts:all")
|
||
if _cached is not None:
|
||
year_counts = _cached
|
||
else:
|
||
try:
|
||
yr_conds = conditions[:_yr_before]
|
||
yr_subq = select(GlobalLiterature.pub_year).where(
|
||
and_(*yr_conds) if yr_conds else text("TRUE")
|
||
).subquery()
|
||
year_count_q = select(
|
||
yr_subq.c.pub_year, func.count().label("cnt")
|
||
).group_by(yr_subq.c.pub_year).order_by(yr_subq.c.pub_year.desc())
|
||
year_rows = await db.execute(year_count_q)
|
||
year_counts = [
|
||
{"year": y, "count": c} for y, c in year_rows if y is not None
|
||
]
|
||
except Exception:
|
||
logger.exception("Year counts query failed")
|
||
year_counts = []
|
||
await _cache.set("search:year_counts:all", year_counts, ttl=1800)
|
||
|
||
elif conditions:
|
||
try:
|
||
yr_conds = conditions[:_yr_before]
|
||
yr_subq = select(GlobalLiterature.pub_year).where(
|
||
and_(*yr_conds)
|
||
).subquery()
|
||
year_count_q = select(
|
||
yr_subq.c.pub_year, func.count().label("cnt")
|
||
).group_by(yr_subq.c.pub_year).order_by(yr_subq.c.pub_year.desc())
|
||
year_rows = await db.execute(year_count_q)
|
||
year_counts = [
|
||
{"year": y, "count": c} for y, c in year_rows if y is not None
|
||
]
|
||
except Exception:
|
||
logger.exception("Year counts query failed")
|
||
year_counts = []
|
||
# 第 1 页结束时统一写 facet 缓存(含 total),此处不重复写入
|
||
|
||
# 排序(PubMed 查询时跳过 ts_rank,避免语法标签噪音)
|
||
_relevance_query = query
|
||
if _pubmed_parsed and sort in ("relevance", "best_match"):
|
||
# 用纯文本词做相关性排序,去掉 [field] 标签
|
||
# P12: filter out negated terms from relevance ranking
|
||
plain_parts = [t.text for t in _pubmed_parsed.plain_terms if not t.is_not]
|
||
plain_parts += [t.text for t in _pubmed_parsed.title_terms if not t.is_not]
|
||
plain_parts += [t.text for t in _pubmed_parsed.abstract_terms if not t.is_not]
|
||
plain_parts += [t.text for t in _pubmed_parsed.tiab_terms if not t.is_not]
|
||
# P11: MeSH-only 查询(如 breast[MAJR])会产生空 plain_parts,
|
||
# 退回到原始查询字符串保证相关性排序不退化到日期排序
|
||
_relevance_query = " ".join(plain_parts).strip() or query
|
||
# ── 通用 keyset 分页(所有列式排序模式统一,代替 OFFSET) ──
|
||
_keyset_cond = AdvancedSearchEngine._keyset_condition(sort, cursor_val, cursor_id)
|
||
q = select(GlobalLiterature)
|
||
if _keyset_cond is not None:
|
||
conditions.append(_keyset_cond)
|
||
elif page > 1:
|
||
# P1-F5: keyset 条件不存在时(键集排序但游标无效/非键集排序),用 OFFSET 翻页
|
||
q = q.offset((page - 1) * page_size)
|
||
|
||
if conditions:
|
||
q = q.where(and_(*conditions))
|
||
elif not _keyset_cond:
|
||
# R23-3: empty query with no criteria should return nothing, not all literature
|
||
q = q.where(text("FALSE"))
|
||
q = q.order_by(*AdvancedSearchEngine._apply_order_by(sort, _relevance_query))
|
||
|
||
# ── LIMIT page_size+1 探测下一页 + COUNT 第 1 页缓存 ──
|
||
result = await db.execute(q.limit(page_size + 1))
|
||
items = result.scalars().all()
|
||
has_more = len(items) > page_size
|
||
items = items[:page_size]
|
||
|
||
# COUNT 只在第 1 页计算,缓存到 facet key 供后续页复用
|
||
total = 0
|
||
if is_first_page:
|
||
count_q = select(func.count()).select_from(
|
||
select(literal_column("1"))
|
||
.select_from(GlobalLiterature)
|
||
.where(and_(*conditions) if conditions else True)
|
||
.subquery()
|
||
)
|
||
total = (await db.execute(count_q)).scalar() or 0
|
||
await _cache.set(_facet_cache_key, {"year_counts": year_counts, "total": total}, ttl=1800)
|
||
else:
|
||
# 后续页从 facet 缓存读 total
|
||
facet_cached = await _cache.get(_facet_cache_key)
|
||
if facet_cached:
|
||
if isinstance(facet_cached, dict):
|
||
total = facet_cached.get("total", 0)
|
||
# P2-F11: 旧格式 list(只有 year_counts),保持 total=0
|
||
|
||
# 构建游标供翻页
|
||
next_cursor_val = None
|
||
next_cursor_id = None
|
||
if has_more and items:
|
||
last = items[-1]
|
||
next_cursor_val = AdvancedSearchEngine._cursor_from_item(last, sort)
|
||
next_cursor_id = str(last.id)
|
||
|
||
tm = await load_tags_for_literature(db, [str(lit.id) for lit in items])
|
||
|
||
# 从全局缓存加载期刊 tier/name(避免每页 IN 查询)
|
||
tier_map, name_map = await AdvancedSearchEngine._get_journal_map(db)
|
||
|
||
results = AdvancedSearchEngine._build_item_dicts(items, tm, tier_map, name_map)
|
||
|
||
result = {
|
||
"items": results,
|
||
"total": total,
|
||
"page": page,
|
||
"page_size": page_size,
|
||
"has_more": has_more,
|
||
"year_counts": year_counts,
|
||
"cursor_val": next_cursor_val,
|
||
"cursor_id": next_cursor_id,
|
||
}
|
||
|
||
if _search_cache_key is not None:
|
||
# 只缓存 lit_ids + 聚合数据,tags/journal 从 ②③ 缓存取
|
||
slim = {
|
||
"lit_ids": [str(lit.id) for lit in items],
|
||
"total": total,
|
||
"page": page,
|
||
"page_size": page_size,
|
||
"has_more": has_more,
|
||
"cursor_val": next_cursor_val,
|
||
"cursor_id": next_cursor_id,
|
||
}
|
||
await _cache.set(_search_cache_key, slim, ttl=AdvancedSearchEngine.SEARCH_CACHE_TTL)
|
||
|
||
return result
|
||
|
||
@staticmethod
|
||
async def _get_journal_map(db: AsyncSession) -> tuple[dict[str, int | None], dict[str, str | None]]:
|
||
"""返回全局 (tier_map, name_map),以 journal_issn 为 key。
|
||
|
||
缓存在 Redis 中(journals:map),TTL 3600s。
|
||
期刊 tier/name 几乎从不变化,全表查询一次即可。
|
||
"""
|
||
cached = await _cache.get("journals:map")
|
||
if cached is not None:
|
||
tier_map = cached.get("tier_map", {})
|
||
name_map = cached.get("name_map", {})
|
||
else:
|
||
rows = (await db.execute(
|
||
select(GlobalJournal.issn, GlobalJournal.tier, GlobalJournal.name)
|
||
)).all()
|
||
tier_map = {issn: tier for issn, tier, name_ in rows}
|
||
name_map = {issn: name_ for issn, tier, name_ in rows}
|
||
await _cache.set("journals:map", {"tier_map": tier_map, "name_map": name_map}, ttl=3600)
|
||
return tier_map, name_map
|
||
|
||
@staticmethod
|
||
def _build_item_dicts(
|
||
items: list,
|
||
tag_map: dict[str, list[dict]],
|
||
tier_map: dict[str, int | None],
|
||
name_map: dict[str, str | None],
|
||
) -> list[dict]:
|
||
"""将 GlobalLiterature 模型列表转为 API 响应所需的 dict 列表"""
|
||
results = []
|
||
for lit in items:
|
||
authors = lit.authors or []
|
||
results.append({
|
||
"id": str(lit.id), "pmid": lit.pmid, "title": lit.title,
|
||
"first_author": authors[0].get("family", "") if authors else "",
|
||
"journal": name_map.get(lit.journal_issn) or lit.journal,
|
||
"pub_date": cap_pub_date(lit.pub_date),
|
||
"article_date": lit.article_date.isoformat() if lit.article_date else None,
|
||
"pub_year": lit.pub_year, "tags": tag_map.get(str(lit.id), []),
|
||
"abstract": lit.abstract[:300] if lit.abstract else None,
|
||
"doi": lit.doi,
|
||
"pmc_id": lit.pmc_id,
|
||
"is_oa": lit.is_oa,
|
||
"cited_by_count": lit.cited_by_count,
|
||
"created_at": lit.created_at.isoformat() if lit.created_at else None,
|
||
"updated_at": lit.updated_at.isoformat() if lit.updated_at else None,
|
||
"journal_issn": lit.journal_issn,
|
||
"journal_tier": tier_map.get(lit.journal_issn),
|
||
"pub_types": lit.pub_types,
|
||
"affiliation": authors[0].get("affiliation", "") if authors else "",
|
||
"study_design": lit.study_design,
|
||
"trial_reg": lit.trial_reg,
|
||
"retracted": lit.retracted,
|
||
"is_negative_result": lit.is_negative_result,
|
||
"rct_detection": lit.rct_detection,
|
||
})
|
||
return results
|
||
|
||
@staticmethod
|
||
async def _hydrate_items(
|
||
db: AsyncSession,
|
||
lit_ids: list[str],
|
||
tier_map: dict[str, int | None],
|
||
name_map: dict[str, str | None],
|
||
) -> list[dict]:
|
||
"""从 lit_ids 重新构建 items(用于缓存命中时的回填)
|
||
|
||
从 DB 按 ID 查询文献、从 ②③ 缓存加载 tags/journal,避免重算复杂 WHERE。
|
||
"""
|
||
import uuid as _uuid
|
||
uids = [_uuid.UUID(s) for s in lit_ids]
|
||
rows = (await db.execute(
|
||
select(GlobalLiterature).where(GlobalLiterature.id.in_(uids))
|
||
)).scalars().all()
|
||
# 保持传入顺序(PostgreSQL WHERE id IN 不保证排序)
|
||
id_order = {str(lit.id): lit for lit in rows}
|
||
ordered = [id_order[sid] for sid in lit_ids if sid in id_order]
|
||
tag_map = await load_tags_for_literature(db, lit_ids)
|
||
return AdvancedSearchEngine._build_item_dicts(ordered, tag_map, tier_map, name_map)
|
||
|
||
@staticmethod
|
||
async def _pubmed_conditions(
|
||
db: AsyncSession,
|
||
pp,
|
||
existing_conditions: list,
|
||
) -> list:
|
||
"""将解析后的 PubMed 查询转换为 SQLAlchemy 条件列表。"""
|
||
from sqlalchemy.dialects.postgresql import JSONB
|
||
|
||
# 创建副本避免就地修改传入的列表(调用方会复用 conditions 做其他过滤)
|
||
conditions = list(existing_conditions)
|
||
# term_conditions 收集与查询词相关的条件,用于 OR 模式下统一包裹
|
||
term_conditions: list = []
|
||
|
||
# 1. 字段级搜索 [TI] [AB] [TIAB] [AU] [TA] [LA] [VI] [IP] [PG] [LID]
|
||
field_combine = or_ if pp.boolean_operator in ("or", "mixed") else and_
|
||
field_map = {
|
||
"title": pp.title_terms,
|
||
"abstract": pp.abstract_terms,
|
||
"all": pp.tiab_terms,
|
||
"author": pp.author_terms,
|
||
"journal": pp.journal_terms,
|
||
"affiliation": pp.affiliation_terms,
|
||
"language": pp.language_terms,
|
||
"volume": pp.volume_terms,
|
||
"issue": pp.issue_terms,
|
||
"pages": pp.pages_terms,
|
||
"lid": pp.lid_terms,
|
||
}
|
||
_NULL_SAFE_COL_MAP = {
|
||
"title": GlobalLiterature.title,
|
||
"abstract": GlobalLiterature.abstract,
|
||
"author": GlobalLiterature.author_names_text,
|
||
"journal": GlobalLiterature.journal,
|
||
"affiliation": GlobalLiterature.authors,
|
||
"language": GlobalLiterature.language,
|
||
"volume": GlobalLiterature.volume,
|
||
"issue": GlobalLiterature.issue,
|
||
"pages": GlobalLiterature.pages,
|
||
"lid": GlobalLiterature.doi,
|
||
"all": GlobalLiterature.title,
|
||
}
|
||
for fld, terms in field_map.items():
|
||
if not terms:
|
||
continue
|
||
pos_conds = []
|
||
neg_conds = []
|
||
for term in terms:
|
||
cond = AdvancedSearchEngine._field_condition(fld, term.text, term.exact)
|
||
if term.is_not:
|
||
_ns_col = _NULL_SAFE_COL_MAP.get(fld)
|
||
if _ns_col is not None:
|
||
neg_conds.append(or_(not_(cond), _ns_col.is_(None)))
|
||
else:
|
||
neg_conds.append(not_(cond))
|
||
else:
|
||
pos_conds.append(cond)
|
||
if pos_conds:
|
||
term_conditions.append(field_combine(*pos_conds) if len(pos_conds) > 1 else pos_conds[0])
|
||
term_conditions.extend(neg_conds)
|
||
|
||
# 2. 纯文本词(无字段标签)— P0-2: 对无标签词补充 ATM MeSH 展开
|
||
# P0-F1: ATM 只展开肯定词,否定词独立 AND,避免被 ATM OR 短路
|
||
if pp.plain_terms:
|
||
pos_conds = [AdvancedSearchEngine._field_condition("all", t.text, t.exact)
|
||
for t in pp.plain_terms if not t.is_not]
|
||
neg_conds = [not_(AdvancedSearchEngine._field_condition("all", t.text, t.exact))
|
||
for t in pp.plain_terms if t.is_not]
|
||
combined_pos_text = " ".join(t.text for t in pp.plain_terms if not t.is_not).strip()
|
||
if combined_pos_text and pos_conds and not re.search(r'[一-鿿㐀-䶿豈-]', combined_pos_text):
|
||
from app.services.query_expansion import expand_atm as _expand_atm_inline
|
||
try:
|
||
atm_cond = await _expand_atm_inline(db, combined_pos_text)
|
||
except Exception:
|
||
logger.exception("ATM expansion failed (pubmed plain_terms): %s", combined_pos_text[:100])
|
||
atm_cond = None
|
||
else:
|
||
atm_cond = None
|
||
term_cond = None
|
||
if pos_conds:
|
||
pos_combined = field_combine(*pos_conds) if len(pos_conds) > 1 else pos_conds[0]
|
||
if atm_cond is not None:
|
||
term_cond = or_(atm_cond, pos_combined)
|
||
else:
|
||
term_cond = pos_combined
|
||
elif atm_cond is not None:
|
||
term_cond = atm_cond
|
||
if term_cond is not None:
|
||
term_conditions.append(term_cond)
|
||
# 否定词独立 AND(不参与 ATM 展开)
|
||
term_conditions.extend(neg_conds)
|
||
|
||
# 3. [MH] → tree_number 展开,支持 is_not 和 _noexp
|
||
if pp.mesh_terms:
|
||
for is_neg in (False, True):
|
||
subset = [t for t in pp.mesh_terms if t.is_not == is_neg]
|
||
if not subset:
|
||
continue
|
||
noexp_names = [t.text for t in subset if t._noexp]
|
||
exp_names = [t.text for t in subset if not t._noexp]
|
||
if exp_names:
|
||
cond = await AdvancedSearchEngine._expand_mesh_tag_ids(db, exp_names, major_only=False)
|
||
if cond is not None:
|
||
term_conditions.append(not_(cond) if is_neg else cond)
|
||
elif not is_neg:
|
||
# P12: unmatched MeSH term → FALSE (no results)
|
||
term_conditions.append(text("FALSE"))
|
||
if noexp_names:
|
||
cond = await AdvancedSearchEngine._expand_mesh_tag_ids(db, noexp_names, major_only=False, noexp=True)
|
||
if cond is not None:
|
||
term_conditions.append(not_(cond) if is_neg else cond)
|
||
elif not is_neg:
|
||
term_conditions.append(text("FALSE"))
|
||
|
||
# 4. [MAJR] → tree_number 展开 + is_major=True,支持 is_not
|
||
if pp.majr_terms:
|
||
pos_names = [t.text for t in pp.majr_terms if not t.is_not]
|
||
neg_names = [t.text for t in pp.majr_terms if t.is_not]
|
||
if pos_names:
|
||
cond = await AdvancedSearchEngine._expand_mesh_tag_ids(db, pos_names, major_only=True)
|
||
if cond is not None:
|
||
term_conditions.append(cond)
|
||
else:
|
||
term_conditions.append(text("FALSE"))
|
||
if neg_names:
|
||
cond = await AdvancedSearchEngine._expand_mesh_tag_ids(db, neg_names, major_only=True)
|
||
if cond is not None:
|
||
term_conditions.append(not_(cond))
|
||
|
||
# 5. [PT] → pub_types JSONB contains,支持 is_not
|
||
if pp.pub_types:
|
||
pos = [t for t in pp.pub_types if not t.is_not]
|
||
neg = [t for t in pp.pub_types if t.is_not]
|
||
if pos:
|
||
pos_conds = [GlobalLiterature.pub_types.cast(JSONB).contains([t.text]) for t in pos]
|
||
term_conditions.append(or_(*pos_conds))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.pub_types.cast(JSONB).contains([t.text]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.pub_types.is_(None)) for c in neg_conds
|
||
)
|
||
|
||
# 5b. [GR] [SH] [RN] [NM] [SI] [PA] → JSONB contains,支持 is_not
|
||
if pp.grant_terms:
|
||
pos = [t for t in pp.grant_terms if not t.is_not]
|
||
neg = [t for t in pp.grant_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.grants.cast(JSONB).contains([{"grant_id": t.text}])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.grants.cast(JSONB).contains([{"grant_id": t.text}]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.grants.is_(None)) for c in neg_conds
|
||
)
|
||
pos = [t for t in pp.subheading_terms if not t.is_not]
|
||
neg = [t for t in pp.subheading_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.mesh_headings.cast(JSONB).contains([{"qualifiers": [t.text]}])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.mesh_headings.cast(JSONB).contains([{"qualifiers": [t.text]}]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.mesh_headings.is_(None)) for c in neg_conds
|
||
)
|
||
if pp.registry_terms:
|
||
pos = [t for t in pp.registry_terms if not t.is_not]
|
||
neg = [t for t in pp.registry_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.chemical_list.cast(JSONB).contains([{"registry_number": t.text}])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.chemical_list.cast(JSONB).contains([{"registry_number": t.text}]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.chemical_list.is_(None)) for c in neg_conds
|
||
)
|
||
if pp.substance_terms:
|
||
pos = [t for t in pp.substance_terms if not t.is_not]
|
||
neg = [t for t in pp.substance_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.chemical_list.cast(JSONB).contains([{"name": t.text}])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.chemical_list.cast(JSONB).contains([{"name": t.text}]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.chemical_list.is_(None)) for c in neg_conds
|
||
)
|
||
if pp.databank_terms:
|
||
pos = [t for t in pp.databank_terms if not t.is_not]
|
||
neg = [t for t in pp.databank_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.databank_list.cast(JSONB).contains([{"accession_numbers": [t.text]}])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.databank_list.cast(JSONB).contains([{"accession_numbers": [t.text]}]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.databank_list.is_(None)) for c in neg_conds
|
||
)
|
||
|
||
# 5c. [PA] → pharmacological_actions JSONB contains (by name or ui),支持 is_not
|
||
if pp.pharmaco_terms:
|
||
pos = [t for t in pp.pharmaco_terms if not t.is_not]
|
||
neg = [t for t in pp.pharmaco_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
or_(
|
||
GlobalLiterature.pharmacological_actions.cast(JSONB).contains([{"name": t.text}]),
|
||
GlobalLiterature.pharmacological_actions.cast(JSONB).contains([{"ui": t.text}]),
|
||
)
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [
|
||
or_(
|
||
GlobalLiterature.pharmacological_actions.cast(JSONB).contains([{"name": t.text}]),
|
||
GlobalLiterature.pharmacological_actions.cast(JSONB).contains([{"ui": t.text}]),
|
||
)
|
||
for t in neg
|
||
]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.pharmacological_actions.is_(None)) for c in neg_conds
|
||
)
|
||
|
||
# P4: [OT] → keywords JSONB contains(不再映射到 all)
|
||
if pp.ot_terms:
|
||
pos = [t for t in pp.ot_terms if not t.is_not]
|
||
neg = [t for t in pp.ot_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.keywords.cast(JSONB).contains([t.text])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.keywords.cast(JSONB).contains([t.text]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.keywords.is_(None)) for c in neg_conds
|
||
)
|
||
|
||
# P4: [GEN] → gene_symbols JSONB contains
|
||
if pp.gene_terms:
|
||
pos = [t for t in pp.gene_terms if not t.is_not]
|
||
neg = [t for t in pp.gene_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.gene_symbols.cast(JSONB).contains([t.text])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.gene_symbols.cast(JSONB).contains([t.text]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.gene_symbols.is_(None)) for c in neg_conds
|
||
)
|
||
|
||
# 5d. [ED] [IR] [PS] [PUBN] [AUID] [COIS] [TT] → 新增字段搜索,支持 is_not
|
||
if pp.ed_terms:
|
||
pos = [t for t in pp.ed_terms if not t.is_not]
|
||
neg = [t for t in pp.ed_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.authors.cast(JSONB).contains([{"type": "editor", "family": t.text}])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.authors.cast(JSONB).contains([{"type": "editor", "family": t.text}]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.authors.is_(None)) for c in neg_conds
|
||
)
|
||
if pp.investigator_terms:
|
||
pos = [t for t in pp.investigator_terms if not t.is_not]
|
||
neg = [t for t in pp.investigator_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.investigators.cast(JSONB).contains([{"family": t.text}])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.investigators.cast(JSONB).contains([{"family": t.text}]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.investigators.is_(None)) for c in neg_conds
|
||
)
|
||
if pp.personal_name_terms:
|
||
pos = [t for t in pp.personal_name_terms if not t.is_not]
|
||
neg = [t for t in pp.personal_name_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.personal_name_subjects.cast(JSONB).contains([{"family": t.text}])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.personal_name_subjects.cast(JSONB).contains([{"family": t.text}]) for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.personal_name_subjects.is_(None)) for c in neg_conds
|
||
)
|
||
if pp.pubnote_terms:
|
||
pos = [t for t in pp.pubnote_terms if not t.is_not]
|
||
neg = [t for t in pp.pubnote_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
cast(GlobalLiterature.publication_notes, String).ilike(f"%{_escape_ilike(t.text)}%")
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [cast(GlobalLiterature.publication_notes, String).ilike(f"%{_escape_ilike(t.text)}%") for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.publication_notes.is_(None)) for c in neg_conds
|
||
)
|
||
if pp.auid_terms:
|
||
pos = [t for t in pp.auid_terms if not t.is_not]
|
||
neg = [t for t in pp.auid_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.auid_data.cast(JSONB).contains([{"value": t.text}])
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.auid_data.cast(JSONB).contains([{"value": t.text}]) for t in neg]
|
||
# R23-3: NULL-safe NOT — NULL auid_data should also match NOT
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.auid_data.is_(None)) for c in neg_conds
|
||
)
|
||
if pp.cois_terms:
|
||
pos = [t for t in pp.cois_terms if not t.is_not]
|
||
neg = [t for t in pp.cois_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.cois_statement.ilike(f"%{_escape_ilike(t.text)}%")
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.cois_statement.ilike(f"%{_escape_ilike(t.text)}%") for t in neg]
|
||
# R23-3: NULL-safe NOT — NULL cois_statement should also match NOT
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.cois_statement.is_(None)) for c in neg_conds
|
||
)
|
||
if pp.tt_terms:
|
||
pos = [t for t in pp.tt_terms if not t.is_not]
|
||
neg = [t for t in pp.tt_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.vernacular_title.ilike(f"%{_escape_ilike(t.text)}%")
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.vernacular_title.ilike(f"%{_escape_ilike(t.text)}%") for t in neg]
|
||
# R23-3: NULL-safe NOT — NULL vernacular_title should also match NOT
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.vernacular_title.is_(None)) for c in neg_conds
|
||
)
|
||
|
||
# P1-2: [SB] Subset
|
||
# medline[SB] → citation_status='medline'(记录级)
|
||
# pubmed[SB] → no-op(全部记录都在 PubMed 中)
|
||
# 单字母代码(AIM/M/S/D/N/Q/T/X)→ nlm_subsets(期刊级)
|
||
if pp.sb_terms:
|
||
pos = [t for t in pp.sb_terms if not t.is_not]
|
||
neg = [t for t in pp.sb_terms if t.is_not]
|
||
for subl, is_neg in [(pos, False), (neg, True)]:
|
||
for t in subl:
|
||
val = t.text.upper()
|
||
if val == "PUBMED":
|
||
if is_neg:
|
||
term_conditions.append(text("FALSE"))
|
||
continue
|
||
elif val == "MEDLINE":
|
||
cond = GlobalLiterature.citation_status == "medline"
|
||
elif val.isalpha():
|
||
# P1-F7: 所有字母子集码(含 AIM 等多字母)统一走 nlm_subsets
|
||
subq = select(GlobalJournal.issn).where(
|
||
GlobalJournal.nlm_subsets.overlap([val])
|
||
)
|
||
cond = GlobalLiterature.journal_issn.in_(subq)
|
||
else:
|
||
cond = GlobalLiterature.citation_status == val.lower()
|
||
term_conditions.append(not_(cond) if is_neg else cond)
|
||
|
||
# P1-2: [STAT] Status → citation_status
|
||
if pp.stat_terms:
|
||
pos = [t for t in pp.stat_terms if not t.is_not]
|
||
neg = [t for t in pp.stat_terms if t.is_not]
|
||
if pos:
|
||
term_conditions.append(or_(*[
|
||
GlobalLiterature.citation_status == t.text.lower()
|
||
for t in pos
|
||
]))
|
||
if neg:
|
||
neg_conds = [GlobalLiterature.citation_status == t.text.lower() for t in neg]
|
||
term_conditions.extend(
|
||
or_(not_(c), GlobalLiterature.citation_status.is_(None)) for c in neg_conds
|
||
)
|
||
|
||
# P1-2: [UID] → PMID 优先,兜底 DOI
|
||
if pp.uid_terms:
|
||
for t in pp.uid_terms:
|
||
cond = None
|
||
try:
|
||
cond = GlobalLiterature.pmid == int(t.text)
|
||
except ValueError:
|
||
cond = GlobalLiterature.doi.ilike(f"%{_escape_ilike(t.text)}%")
|
||
if cond is not None:
|
||
term_conditions.append(not_(cond) if t.is_not else cond)
|
||
|
||
# R23-3: track date fields already handled inside negated groups (De Morgan fix)
|
||
_handled_neg_group_date_fields: set[str] = set()
|
||
|
||
# 处理括号分组的词(保留 OR/AND 嵌套结构,P2-2)
|
||
if pp.groups:
|
||
for idx, group in enumerate(pp.groups):
|
||
# P20: skip sub-groups that are children of a parent group
|
||
_is_child = any(
|
||
idx in _refs for _refs in pp.sub_group_refs
|
||
) if pp.sub_group_refs else False
|
||
if _is_child:
|
||
continue
|
||
|
||
# P16: group_negated[idx] tracks external NOT wrapper (NOT (...)),
|
||
# vs is_not per-term (internal NOT, set by _parse_not_expr).
|
||
_negated = (pp.group_negated[idx]
|
||
if idx < len(pp.group_negated)
|
||
else False)
|
||
g_pos = []
|
||
g_neg = []
|
||
# R23-3: track date fields referenced in negated groups (De Morgan fix)
|
||
_neg_group_date_fields: set[str] = set()
|
||
for t in group:
|
||
# R27: process date markers for ALL groups (follows text term pattern)
|
||
if getattr(t, '_is_range_end', False):
|
||
_marker_cond = AdvancedSearchEngine._build_marker_condition(t)
|
||
if _negated and _marker_cond is not None:
|
||
# R23-3: identify which date field this marker references
|
||
_mf = t.field or ""
|
||
if _mf.startswith("__RANGE_") and _mf.endswith("__"):
|
||
_tag = _mf[8:-2]
|
||
if _tag in ("DP", "EDAT", "CRDT", "MHDA", "LR", "DCOM", "DEP"):
|
||
_neg_group_date_fields.add(_tag)
|
||
# R27: negated group → all go to g_neg
|
||
# is_not=True means inner NOT: not_(cond), outer NOT wraps: not_(not_(cond)) = cond
|
||
g_neg.append(not_(_marker_cond) if getattr(t, 'is_not', False) else _marker_cond)
|
||
elif _marker_cond is not None:
|
||
# Non-negated group: is_not → negated range; otherwise → positive
|
||
if getattr(t, 'is_not', False):
|
||
# R28: NULL-safe NOT for group marker date ranges
|
||
_mf2 = t.field or ""
|
||
_tag2 = _mf2[8:-2] if _mf2.startswith("__RANGE_") and _mf2.endswith("__") else None
|
||
if _tag2 == "DP":
|
||
g_neg.append(or_(not_(_marker_cond), GlobalLiterature.pub_year.is_(None), GlobalLiterature.pub_date.is_(None)))
|
||
elif _tag2 in ("EDAT", "CRDT", "MHDA", "LR", "DCOM", "DEP"):
|
||
_tag2_col = {
|
||
"EDAT": GlobalLiterature.entrez_date,
|
||
"CRDT": GlobalLiterature.create_date,
|
||
"MHDA": GlobalLiterature.meshed_date,
|
||
"LR": GlobalLiterature.pubmed_revised,
|
||
"DCOM": GlobalLiterature.date_completed,
|
||
"DEP": GlobalLiterature.pub_date,
|
||
}.get(_tag2)
|
||
if _tag2_col:
|
||
g_neg.append(or_(not_(_marker_cond), _tag2_col.is_(None)))
|
||
else:
|
||
g_neg.append(not_(_marker_cond))
|
||
else:
|
||
g_neg.append(not_(_marker_cond))
|
||
else:
|
||
g_pos.append(_marker_cond)
|
||
continue
|
||
cond = await AdvancedSearchEngine._single_term_condition(db, t)
|
||
if cond is not None:
|
||
if _negated:
|
||
# R22: preserve inner is_not when group is externally negated
|
||
g_neg.append(not_(cond) if t.is_not else cond)
|
||
elif t.is_not:
|
||
g_neg.append(not_(cond))
|
||
else:
|
||
g_pos.append(cond)
|
||
|
||
# P20: process child sub-groups and fold them into this parent group
|
||
_child_gids = pp.sub_group_refs[idx] if idx < len(pp.sub_group_refs) else []
|
||
for child_gid in _child_gids:
|
||
child_group = pp.groups[child_gid]
|
||
child_op = pp.group_operators[child_gid] if child_gid < len(pp.group_operators) else "and"
|
||
child_fn = or_ if child_op == "or" else and_
|
||
child_pos = []
|
||
child_neg = []
|
||
for t in child_group:
|
||
if getattr(t, '_is_range_end', False):
|
||
# R29: process date range markers in child sub-groups
|
||
_c_marker_cond = AdvancedSearchEngine._build_marker_condition(t)
|
||
if _c_marker_cond is not None:
|
||
if getattr(t, 'is_not', False):
|
||
_cmf = t.field or ""
|
||
_ctag = _cmf[8:-2] if _cmf.startswith("__RANGE_") and _cmf.endswith("__") else None
|
||
if _ctag == "DP":
|
||
child_neg.append(or_(not_(_c_marker_cond), GlobalLiterature.pub_year.is_(None), GlobalLiterature.pub_date.is_(None)))
|
||
elif _ctag in ("EDAT", "CRDT", "MHDA", "LR", "DCOM", "DEP"):
|
||
_ctag_col = {
|
||
"EDAT": GlobalLiterature.entrez_date,
|
||
"CRDT": GlobalLiterature.create_date,
|
||
"MHDA": GlobalLiterature.meshed_date,
|
||
"LR": GlobalLiterature.pubmed_revised,
|
||
"DCOM": GlobalLiterature.date_completed,
|
||
"DEP": GlobalLiterature.pub_date,
|
||
}.get(_ctag)
|
||
if _ctag_col:
|
||
child_neg.append(or_(not_(_c_marker_cond), _ctag_col.is_(None)))
|
||
else:
|
||
child_neg.append(not_(_c_marker_cond))
|
||
else:
|
||
child_neg.append(not_(_c_marker_cond))
|
||
else:
|
||
child_pos.append(_c_marker_cond)
|
||
continue
|
||
cond = await AdvancedSearchEngine._single_term_condition(db, t)
|
||
if cond is not None:
|
||
if t.is_not:
|
||
child_neg.append(not_(cond))
|
||
else:
|
||
child_pos.append(cond)
|
||
child_combined = None
|
||
if child_pos:
|
||
child_combined = child_fn(*child_pos) if len(child_pos) > 1 else child_pos[0]
|
||
if child_neg:
|
||
_negc = child_fn(*child_neg) if len(child_neg) > 1 else child_neg[0]
|
||
child_combined = child_fn(child_combined, _negc) if child_combined is not None else _negc
|
||
if child_combined is not None:
|
||
if _negated:
|
||
g_neg.append(child_combined)
|
||
else:
|
||
g_pos.append(child_combined)
|
||
|
||
# R27: date conditions are now built directly from markers in the group loop above.
|
||
# Only track handled fields for top-level skip logic.
|
||
if _negated and _neg_group_date_fields:
|
||
_handled_neg_group_date_fields.update(_neg_group_date_fields)
|
||
|
||
gop = (pp.group_operators[idx]
|
||
if idx < len(pp.group_operators)
|
||
else "and")
|
||
combine_fn = or_ if gop == "or" else and_
|
||
|
||
if _negated:
|
||
# NOT(A OR B): single UnaryExpression → 顶层 OR/NOT 分离时被检测为 neg → 独立 AND
|
||
if g_neg:
|
||
combined = combine_fn(*g_neg) if len(g_neg) > 1 else g_neg[0]
|
||
_notted = not_(combined)
|
||
# R28: NULL-safe NOT for date fields in negated groups
|
||
if _neg_group_date_fields:
|
||
_ns_cols = []
|
||
if "DP" in _neg_group_date_fields:
|
||
_ns_cols.extend([GlobalLiterature.pub_year.is_(None), GlobalLiterature.pub_date.is_(None)])
|
||
_NS_COL_MAP = {
|
||
"EDAT": GlobalLiterature.entrez_date,
|
||
"CRDT": GlobalLiterature.create_date,
|
||
"MHDA": GlobalLiterature.meshed_date,
|
||
"LR": GlobalLiterature.pubmed_revised,
|
||
"DCOM": GlobalLiterature.date_completed,
|
||
"DEP": GlobalLiterature.pub_date,
|
||
}
|
||
for _ns_tag in _neg_group_date_fields:
|
||
_ns_col = _NS_COL_MAP.get(_ns_tag)
|
||
if _ns_col is not None:
|
||
_ns_cols.append(_ns_col.is_(None))
|
||
if _ns_cols:
|
||
_notted = or_(_notted, *_ns_cols)
|
||
term_conditions.append(_notted)
|
||
else:
|
||
# 混合/正组:将 pos 和 neg 按组操作符组合,保留组内结构
|
||
# 避免 neg 被 OR/NOT 分离拉出来破坏语义
|
||
combined = None
|
||
if g_pos:
|
||
combined = combine_fn(*g_pos) if len(g_pos) > 1 else g_pos[0]
|
||
if g_neg:
|
||
neg_combined = combine_fn(*g_neg) if len(g_neg) > 1 else g_neg[0]
|
||
combined = combine_fn(combined, neg_combined) if combined is not None else neg_combined
|
||
if combined is not None:
|
||
term_conditions.append(combined)
|
||
|
||
# 将 term_conditions 加入 conditions
|
||
if term_conditions:
|
||
if pp.boolean_operator == "or":
|
||
# OR 模式:所有词条件 OR 在一起。NOT 项也一起 OR(PubMed: A OR NOT B ≠ A AND NOT B)
|
||
conditions.append(or_(*term_conditions))
|
||
elif pp.boolean_operator == "mixed":
|
||
# mixed 模式下 NOT 项应独立 AND(PubMed: A OR B NOT C = (A OR B) AND NOT C)
|
||
from sqlalchemy.sql.elements import UnaryExpression
|
||
from sqlalchemy.sql import operators as _sa_ops
|
||
|
||
def _is_negated_cond(c):
|
||
"""Detect if a condition is negated, including NULL-safe wrapped NOT."""
|
||
if isinstance(c, UnaryExpression) and c.modifier == _sa_ops.inv:
|
||
return True
|
||
# R26: NULL-safe NOT: or_(not_(inner), col.is_(None))
|
||
# R27: verify second clause is indeed is_(None) to avoid false positives
|
||
try:
|
||
if hasattr(c, 'operator') and c.operator is _sa_ops.or_:
|
||
clauses = list(getattr(c, 'clauses', ()))
|
||
if len(clauses) >= 2 and isinstance(clauses[0], UnaryExpression) and clauses[0].modifier == _sa_ops.inv:
|
||
# Verify remaining clause is an IS NULL check
|
||
for _cl in clauses[1:]:
|
||
if hasattr(_cl, 'operator') and _cl.operator in (_sa_ops.is_, _sa_ops.isnot):
|
||
return True
|
||
except Exception:
|
||
pass
|
||
return False
|
||
|
||
pos_conds = [c for c in term_conditions if not _is_negated_cond(c)]
|
||
neg_conds = [c for c in term_conditions if _is_negated_cond(c)]
|
||
if neg_conds:
|
||
if pos_conds:
|
||
conditions.append(or_(*pos_conds))
|
||
conditions.extend(neg_conds)
|
||
else:
|
||
conditions.append(or_(*term_conditions))
|
||
else:
|
||
conditions.extend(term_conditions)
|
||
|
||
# 以下为非词条件(日期、PMID、DOI),始终 AND
|
||
# 6. [DP] → 年份/日期范围
|
||
# P25: skip DP only when ALL occurrences are inside negated groups (not at top level)
|
||
_dp_in_neg_groups = "DP" in _handled_neg_group_date_fields
|
||
_dp_at_top = "DP" in getattr(pp, '_top_level_date_fields', set())
|
||
if not (_dp_in_neg_groups and not _dp_at_top):
|
||
# R26: negated single-date/range bounds from _neg_single_dates supersede dp_negated
|
||
_dp_neg_bounds = getattr(pp, '_neg_single_dates', {}).get("DP", [])
|
||
dp_negated = "DP" in getattr(pp, 'negated_date_ranges', set())
|
||
dp_conds = []
|
||
if pp.year_from is not None:
|
||
dp_conds.append(GlobalLiterature.pub_year >= pp.year_from)
|
||
if pp.year_to is not None:
|
||
dp_conds.append(GlobalLiterature.pub_year <= pp.year_to)
|
||
if pp.date_from:
|
||
from datetime import date as _dt_date
|
||
try:
|
||
dp_conds.append(GlobalLiterature.pub_date >= _dt_date.fromisoformat(pp.date_from))
|
||
except ValueError:
|
||
pass
|
||
if pp.date_to:
|
||
from datetime import date as _dt_date
|
||
try:
|
||
dp_conds.append(GlobalLiterature.pub_date <= _dt_date.fromisoformat(pp.date_to))
|
||
except ValueError:
|
||
pass
|
||
if _dp_neg_bounds:
|
||
# _neg_single_dates present: generate positive range + separate NOTs
|
||
if dp_conds:
|
||
conditions.append(and_(*dp_conds) if len(dp_conds) > 1 else dp_conds[0])
|
||
from datetime import date as _dt_date
|
||
for _neg_from, _neg_to in _dp_neg_bounds:
|
||
# R28: detect full-year bounds → use pub_year for consistency
|
||
if _neg_from and _neg_to and _neg_from.endswith("-01-01") and _neg_to.endswith("-12-31"):
|
||
_yr_f = int(_neg_from[:4])
|
||
_yr_t = int(_neg_to[:4])
|
||
if _yr_f == _yr_t:
|
||
_neg_yr_cond = and_(
|
||
GlobalLiterature.pub_year >= _yr_f,
|
||
GlobalLiterature.pub_year <= _yr_f,
|
||
)
|
||
conditions.append(or_(not_(_neg_yr_cond), GlobalLiterature.pub_year.is_(None)))
|
||
else:
|
||
_neg_yr_cond = and_(
|
||
GlobalLiterature.pub_year >= _yr_f,
|
||
GlobalLiterature.pub_year <= _yr_t,
|
||
)
|
||
conditions.append(or_(not_(_neg_yr_cond), GlobalLiterature.pub_year.is_(None)))
|
||
continue
|
||
neg_conds = []
|
||
if _neg_from:
|
||
try:
|
||
neg_conds.append(GlobalLiterature.pub_date >= _dt_date.fromisoformat(_neg_from))
|
||
except ValueError:
|
||
pass
|
||
if _neg_to:
|
||
try:
|
||
neg_conds.append(GlobalLiterature.pub_date <= _dt_date.fromisoformat(_neg_to))
|
||
except ValueError:
|
||
pass
|
||
if neg_conds:
|
||
# R27: NULL-safe NOT — rows with NULL pub_date should not be excluded
|
||
_range_cond = and_(*neg_conds) if len(neg_conds) > 1 else neg_conds[0]
|
||
conditions.append(or_(not_(_range_cond), GlobalLiterature.pub_date.is_(None)))
|
||
elif dp_conds:
|
||
cond = and_(*dp_conds) if len(dp_conds) > 1 else dp_conds[0]
|
||
if dp_negated:
|
||
# R28: NULL-safe NOT — rows with NULL pub_year/pub_date should not be excluded
|
||
conditions.append(or_(not_(cond), GlobalLiterature.pub_year.is_(None), GlobalLiterature.pub_date.is_(None)))
|
||
else:
|
||
conditions.append(cond)
|
||
elif dp_negated:
|
||
# negated_date_ranges includes DP but no conditions built — edge case guard
|
||
pass
|
||
|
||
# 6b. [EDAT] [CRDT] [MHDA] [LR] [DCOM] [DEP] → 日期字段范围
|
||
DATE_FIELD_COLS = {
|
||
"edat": (GlobalLiterature.entrez_date, "EDAT"),
|
||
"crdt": (GlobalLiterature.create_date, "CRDT"),
|
||
"mhda": (GlobalLiterature.meshed_date, "MHDA"),
|
||
"lr": (GlobalLiterature.pubmed_revised, "LR"),
|
||
"dcom": (GlobalLiterature.date_completed, "DCOM"),
|
||
"dep": (GlobalLiterature.pub_date, "DEP"), # P1-2: [DEP] → pub_date
|
||
}
|
||
for prefix, (col, field_tag) in DATE_FIELD_COLS.items():
|
||
# P25: skip only when field is exclusively in negated groups (not at top level)
|
||
_neg_only = field_tag in _handled_neg_group_date_fields and field_tag not in getattr(pp, '_top_level_date_fields', set())
|
||
if _neg_only:
|
||
continue
|
||
_from = getattr(pp, f"{prefix}_from", None)
|
||
_to = getattr(pp, f"{prefix}_to", None)
|
||
field_conds = []
|
||
if _from:
|
||
from datetime import date as _dt_date
|
||
try:
|
||
field_conds.append(col >= _dt_date.fromisoformat(_from))
|
||
except ValueError:
|
||
pass
|
||
if _to:
|
||
from datetime import date as _dt_date
|
||
try:
|
||
field_conds.append(col <= _dt_date.fromisoformat(_to))
|
||
except ValueError:
|
||
pass
|
||
# R26: _neg_single_dates supersedes negated_date_ranges for separate NOT conditions
|
||
_neg_singles = getattr(pp, '_neg_single_dates', {}).get(field_tag, [])
|
||
if _neg_singles:
|
||
if field_conds:
|
||
conditions.append(and_(*field_conds) if len(field_conds) > 1 else field_conds[0])
|
||
from datetime import date as _dt_date
|
||
for _neg_from, _neg_to in _neg_singles:
|
||
neg_conds = []
|
||
if _neg_from:
|
||
try:
|
||
neg_conds.append(col >= _dt_date.fromisoformat(_neg_from))
|
||
except ValueError:
|
||
pass
|
||
if _neg_to:
|
||
try:
|
||
neg_conds.append(col <= _dt_date.fromisoformat(_neg_to))
|
||
except ValueError:
|
||
pass
|
||
if neg_conds:
|
||
# R27: NULL-safe NOT — rows with NULL column should not be excluded
|
||
_range_cond = and_(*neg_conds) if len(neg_conds) > 1 else neg_conds[0]
|
||
conditions.append(or_(not_(_range_cond), col.is_(None)))
|
||
elif field_conds:
|
||
cond = and_(*field_conds) if len(field_conds) > 1 else field_conds[0]
|
||
negated = field_tag in getattr(pp, 'negated_date_ranges', set())
|
||
if negated:
|
||
# R28: NULL-safe NOT — rows with NULL col should not be excluded
|
||
conditions.append(or_(not_(cond), col.is_(None)))
|
||
else:
|
||
conditions.append(cond)
|
||
|
||
# 7. [PMID] → 精确匹配,支持 is_not
|
||
for term in pp.pmid_terms:
|
||
try:
|
||
cond = GlobalLiterature.pmid == int(term.text)
|
||
except ValueError:
|
||
cond = GlobalLiterature.doi.ilike(f"%{_escape_ilike(term.text)}%")
|
||
if term.is_not:
|
||
cond = not_(cond)
|
||
conditions.append(cond)
|
||
|
||
# 8. [DOI] → ILIKE,支持 is_not
|
||
for term in pp.doi_terms:
|
||
cond = GlobalLiterature.doi.ilike(f"%{_escape_ilike(term.text)}%")
|
||
if term.is_not:
|
||
cond = not_(cond)
|
||
conditions.append(cond)
|
||
|
||
# P4: [PMC] → pmc_id 精确匹配,支持 is_not
|
||
for term in pp.pmc_terms:
|
||
cond = GlobalLiterature.pmc_id == term.text
|
||
if term.is_not:
|
||
cond = not_(cond)
|
||
conditions.append(cond)
|
||
|
||
# R21: OR 模式下日期/PMID/DOI/PMC 条件应 OR 进词条件,而非始终 AND
|
||
# A OR 2000:2020[DP] 应 = articles about A OR articles in 2000-2020
|
||
if pp.boolean_operator == "or" and conditions:
|
||
conditions = [or_(*conditions)]
|
||
|
||
return conditions
|
||
|
||
@staticmethod
|
||
async def _single_term_condition(db: AsyncSession, term) -> object | None:
|
||
"""将单个解析后的 term 转为 SQLAlchemy condition,处理所有字段类型。
|
||
|
||
P0-3: 确保括号组内的特殊字段(MH/PT/GR/RN 等)不被降级到 "all",走正确的条件生成。
|
||
"""
|
||
from sqlalchemy.dialects.postgresql import JSONB
|
||
|
||
field = term.field
|
||
# 简单 ILIKE 字段
|
||
if field in ("title", "abstract", "all", "author", "journal", "affiliation",
|
||
"language", "volume", "issue", "pages", "lid") or field is None:
|
||
return AdvancedSearchEngine._field_condition(field or "all", term.text, term.exact)
|
||
|
||
# 特殊字段 — 与 _pubmed_conditions 中 top-level dispatch 一致
|
||
if field == "MH":
|
||
cond = await AdvancedSearchEngine._expand_mesh_tag_ids(db, [term.text], major_only=False, noexp=term._noexp)
|
||
return cond if cond is not None else text("FALSE")
|
||
if field == "MAJR":
|
||
cond = await AdvancedSearchEngine._expand_mesh_tag_ids(db, [term.text], major_only=True)
|
||
return cond if cond is not None else text("FALSE")
|
||
if field == "PT":
|
||
return GlobalLiterature.pub_types.cast(JSONB).contains([term.text])
|
||
if field == "PMID":
|
||
try:
|
||
return GlobalLiterature.pmid == int(term.text)
|
||
except ValueError:
|
||
return GlobalLiterature.doi.ilike(f"%{_escape_ilike(term.text)}%")
|
||
if field == "DOI":
|
||
return GlobalLiterature.doi.ilike(f"%{_escape_ilike(term.text)}%")
|
||
if field == "GR":
|
||
return GlobalLiterature.grants.cast(JSONB).contains([{"grant_id": term.text}])
|
||
if field == "SH":
|
||
return GlobalLiterature.mesh_headings.cast(JSONB).contains([{"qualifiers": [term.text]}])
|
||
if field == "RN":
|
||
return GlobalLiterature.chemical_list.cast(JSONB).contains([{"registry_number": term.text}])
|
||
if field == "NM":
|
||
return GlobalLiterature.chemical_list.cast(JSONB).contains([{"name": term.text}])
|
||
if field == "SI":
|
||
return GlobalLiterature.databank_list.cast(JSONB).contains([{"accession_numbers": [term.text]}])
|
||
if field == "PA":
|
||
return or_(
|
||
GlobalLiterature.pharmacological_actions.cast(JSONB).contains([{"name": term.text}]),
|
||
GlobalLiterature.pharmacological_actions.cast(JSONB).contains([{"ui": term.text}]),
|
||
)
|
||
if field == "ED":
|
||
return GlobalLiterature.authors.cast(JSONB).contains([{"type": "editor", "family": term.text}])
|
||
if field == "IR":
|
||
return GlobalLiterature.investigators.cast(JSONB).contains([{"family": term.text}])
|
||
if field == "PS":
|
||
return GlobalLiterature.personal_name_subjects.cast(JSONB).contains([{"family": term.text}])
|
||
if field == "PUBN":
|
||
return cast(GlobalLiterature.publication_notes, String).ilike(f"%{_escape_ilike(term.text)}%")
|
||
if field == "AUID":
|
||
return GlobalLiterature.auid_data.cast(JSONB).contains([{"value": term.text}])
|
||
if field == "COIS":
|
||
return GlobalLiterature.cois_statement.ilike(f"%{_escape_ilike(term.text)}%")
|
||
if field == "TT":
|
||
return GlobalLiterature.vernacular_title.ilike(f"%{_escape_ilike(term.text)}%")
|
||
if field == "SB":
|
||
val = term.text.upper()
|
||
if val == "PUBMED":
|
||
return text("TRUE") # no-op: 所有记录都是 PubMed
|
||
elif val == "MEDLINE":
|
||
return GlobalLiterature.citation_status == "medline"
|
||
elif val.isalpha():
|
||
subq = select(GlobalJournal.issn).where(
|
||
GlobalJournal.nlm_subsets.overlap([val])
|
||
)
|
||
return GlobalLiterature.journal_issn.in_(subq)
|
||
else:
|
||
return GlobalLiterature.citation_status == val.lower()
|
||
if field == "STAT":
|
||
return GlobalLiterature.citation_status == term.text.lower()
|
||
if field == "UID":
|
||
try:
|
||
return GlobalLiterature.pmid == int(term.text)
|
||
except ValueError:
|
||
return GlobalLiterature.doi.ilike(f"%{_escape_ilike(term.text)}%")
|
||
# P4: [OT] → keywords JSONB contains
|
||
if field == "OT":
|
||
return GlobalLiterature.keywords.cast(JSONB).contains([term.text])
|
||
# P4: [GEN] → gene_symbols JSONB contains
|
||
if field == "GEN":
|
||
return GlobalLiterature.gene_symbols.cast(JSONB).contains([term.text])
|
||
# P4: [PMC] → pmc_id 精确匹配
|
||
if field == "PMC":
|
||
return GlobalLiterature.pmc_id == term.text
|
||
|
||
# P14: 日期字段在括号组内
|
||
if field in ("DP", "EDAT", "CRDT", "MHDA", "LR", "DCOM", "DEP"):
|
||
from app.services.pubmed_query_parser import _validate_date_str as _vds
|
||
_term_text = term.text
|
||
if _term_text.isdigit() and len(_term_text) == 4:
|
||
year = int(_term_text)
|
||
if field == "DP":
|
||
return and_(GlobalLiterature.pub_year >= year, GlobalLiterature.pub_year <= year)
|
||
col_map = {
|
||
"EDAT": GlobalLiterature.entrez_date,
|
||
"CRDT": GlobalLiterature.create_date,
|
||
"MHDA": GlobalLiterature.meshed_date,
|
||
"LR": GlobalLiterature.pubmed_revised,
|
||
"DCOM": GlobalLiterature.date_completed,
|
||
"DEP": GlobalLiterature.pub_date,
|
||
}
|
||
col = col_map[field]
|
||
from datetime import date as _d
|
||
return and_(col >= _d(year, 1, 1), col <= _d(year, 12, 31))
|
||
# R28: partial date YYYY-MM inside group
|
||
from app.services.pubmed_query_parser import _PARTIAL_DATE_RE as _PDR, _expand_partial_date
|
||
if _PDR.match(_term_text):
|
||
_df, _dt = _expand_partial_date(_term_text)
|
||
from datetime import date as _d
|
||
try:
|
||
d_from, d_to = _d.fromisoformat(_df), _d.fromisoformat(_dt)
|
||
if field == "DP":
|
||
return and_(GlobalLiterature.pub_date >= d_from, GlobalLiterature.pub_date <= d_to)
|
||
col_map = {
|
||
"EDAT": GlobalLiterature.entrez_date,
|
||
"CRDT": GlobalLiterature.create_date,
|
||
"MHDA": GlobalLiterature.meshed_date,
|
||
"LR": GlobalLiterature.pubmed_revised,
|
||
"DCOM": GlobalLiterature.date_completed,
|
||
"DEP": GlobalLiterature.pub_date,
|
||
}
|
||
return and_(col_map[field] >= d_from, col_map[field] <= d_to)
|
||
except ValueError:
|
||
pass
|
||
if _vds(_term_text):
|
||
from datetime import date as _d
|
||
try:
|
||
d = _d.fromisoformat(_term_text)
|
||
if field == "DP":
|
||
return GlobalLiterature.pub_date == d
|
||
col_map = {
|
||
"EDAT": GlobalLiterature.entrez_date,
|
||
"CRDT": GlobalLiterature.create_date,
|
||
"MHDA": GlobalLiterature.meshed_date,
|
||
"LR": GlobalLiterature.pubmed_revised,
|
||
"DCOM": GlobalLiterature.date_completed,
|
||
"DEP": GlobalLiterature.pub_date,
|
||
}
|
||
return col_map[field] == d
|
||
except ValueError:
|
||
pass
|
||
|
||
# 回退
|
||
return AdvancedSearchEngine._field_condition("all", term.text, term.exact)
|
||
|
||
@staticmethod
|
||
def _field_condition(field: str, term: str, exact: bool) -> callable:
|
||
_wildcard = term.endswith('*') and not exact
|
||
_stem = term[:-1] if _wildcard else term
|
||
# R24: pure wildcard '*' matches nothing
|
||
if not _stem:
|
||
return text("FALSE")
|
||
_escaped = _escape_ilike(_stem)
|
||
|
||
def _pt() -> str:
|
||
"""P2-1: wildcard → 右截断 ILIKE;其他 → 双百搭"""
|
||
return f"{_escaped}%" if _wildcard else f"%{_escaped}%"
|
||
|
||
if field == "title":
|
||
return GlobalLiterature.title.ilike(_pt())
|
||
elif field == "abstract":
|
||
return GlobalLiterature.abstract.ilike(_pt())
|
||
elif field == "author":
|
||
pat = _pt()
|
||
return or_(
|
||
text(
|
||
"EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _a "
|
||
"WHERE _a->>'family' ILIKE :author_pat)"
|
||
).bindparams(author_pat=pat),
|
||
GlobalLiterature.author_names_text.ilike(pat),
|
||
)
|
||
elif field == "journal":
|
||
pat = _pt()
|
||
return or_(
|
||
GlobalLiterature.journal.ilike(pat),
|
||
GlobalLiterature.journal_iso.ilike(pat),
|
||
)
|
||
elif field == "affiliation":
|
||
# P0-F2: 用 jsonb_array_elements 提取 affiliation 值,避免 JSON 键名假阳性
|
||
_pat = _pt()
|
||
return text(
|
||
"EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
|
||
"WHERE _e->>'affiliation' ILIKE :aff_pat)"
|
||
).bindparams(aff_pat=_pat)
|
||
elif field == "language":
|
||
return GlobalLiterature.language.ilike(_pt())
|
||
elif field == "volume":
|
||
return GlobalLiterature.volume.ilike(_pt())
|
||
elif field == "issue":
|
||
return GlobalLiterature.issue.ilike(_pt())
|
||
elif field == "pages":
|
||
return GlobalLiterature.pages.ilike(_pt())
|
||
elif field == "lid":
|
||
pat = _pt()
|
||
return or_(
|
||
GlobalLiterature.doi.ilike(pat),
|
||
GlobalLiterature.pmc_id.ilike(pat),
|
||
)
|
||
else: # "all" default
|
||
pat = _pt()
|
||
if exact and not _wildcard:
|
||
# P4: 精确短语 → phraseto_tsquery(利用 GIN 索引,保留词序)
|
||
# P13: 加入 journal/journal_iso ILIKE(不在 tsvector 中)
|
||
# R29: 加入 title/abstract ILIKE(非英文如中文无法被 tsquery 匹配)
|
||
return or_(
|
||
GlobalLiterature.search_tsv.op("@@")(func.phraseto_tsquery("english", term)),
|
||
GlobalLiterature.title.ilike(pat),
|
||
GlobalLiterature.abstract.ilike(pat),
|
||
GlobalLiterature.journal.ilike(pat),
|
||
GlobalLiterature.journal_iso.ilike(pat),
|
||
text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
|
||
"WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat),
|
||
)
|
||
if _wildcard:
|
||
# wildcard → ILIKE 右截断(tsvector 不支持 *),多字段覆盖
|
||
return or_(
|
||
GlobalLiterature.title.ilike(pat),
|
||
GlobalLiterature.abstract.ilike(pat),
|
||
GlobalLiterature.author_names_text.ilike(pat),
|
||
GlobalLiterature.journal.ilike(pat),
|
||
GlobalLiterature.journal_iso.ilike(pat),
|
||
cast(GlobalLiterature.pmid, String).ilike(pat),
|
||
GlobalLiterature.doi.ilike(pat),
|
||
text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
|
||
"WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat),
|
||
)
|
||
like_val = f"%{_escaped}%"
|
||
if "/" in term:
|
||
if term.startswith("10."):
|
||
return or_(
|
||
GlobalLiterature.doi.ilike(_escape_ilike(term)),
|
||
GlobalLiterature.doi.ilike(like_val),
|
||
text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
|
||
"WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat),
|
||
)
|
||
return or_(
|
||
GlobalLiterature.title.ilike(like_val),
|
||
cast(GlobalLiterature.pmid, String).ilike(like_val),
|
||
GlobalLiterature.doi.ilike(like_val),
|
||
GlobalLiterature.abstract.ilike(like_val),
|
||
GlobalLiterature.author_names_text.ilike(like_val),
|
||
GlobalLiterature.journal.ilike(like_val),
|
||
GlobalLiterature.journal_iso.ilike(like_val),
|
||
text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
|
||
"WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat),
|
||
)
|
||
# P7-D2: Chinese → ILIKE fallback (tsvector is English-only)
|
||
if re.search(r'[一-鿿㐀-䶿豈-]', term):
|
||
return or_(
|
||
GlobalLiterature.title.ilike(like_val),
|
||
GlobalLiterature.abstract.ilike(like_val),
|
||
GlobalLiterature.author_names_text.ilike(like_val),
|
||
GlobalLiterature.journal.ilike(like_val),
|
||
GlobalLiterature.journal_iso.ilike(like_val),
|
||
cast(GlobalLiterature.pmid, String).ilike(like_val),
|
||
GlobalLiterature.doi.ilike(like_val),
|
||
text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
|
||
"WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat),
|
||
)
|
||
# tsvector 索引主覆盖 title/abstract/author_names/chemicals/genes/mesh/keywords
|
||
# journal/journal_iso/affiliation 不在 tsvector 中,以 ILIKE 兜底
|
||
return or_(
|
||
GlobalLiterature.search_tsv.op("@@")(func.plainto_tsquery("english", term)),
|
||
GlobalLiterature.journal.ilike(like_val),
|
||
GlobalLiterature.journal_iso.ilike(like_val),
|
||
text("EXISTS (SELECT 1 FROM jsonb_array_elements(global_literature.authors) AS _e "
|
||
"WHERE _e->>'affiliation' ILIKE :aff_pat)").bindparams(aff_pat=pat),
|
||
)
|
||
|
||
@staticmethod
|
||
async def _expand_mesh_tag_ids(
|
||
db: AsyncSession,
|
||
mesh_names: list[str],
|
||
major_only: bool = False,
|
||
noexp: bool = False, # P1-4: [MH:noexp] 抑制树展开
|
||
) -> object | None:
|
||
"""用 GlobalTagTreeNumber 展开 [MH]/[MAJR]
|
||
|
||
1. 先精确入口词匹配(entry_terms JSONB @>),再退到 name_en ILIKE
|
||
2. 用 tree_number 前缀展开子节点(C04.588 → 所有子树号)
|
||
3. [MAJR] 额外 AND is_major=True
|
||
返回 SQLAlchemy condition 或 None(无匹配时)。
|
||
|
||
noexp=True 时跳过第 2 步(树展开),只搜精确词。
|
||
|
||
缓存策略:atm:{md5(归一化查询+参数)} → expanded_tag_ids list。
|
||
GlobalTag/TreeNumber 几乎不变,TTL 1 小时。
|
||
"""
|
||
import hashlib as _hashlib
|
||
import uuid as _uuid
|
||
|
||
# 计算缓存键
|
||
items = sorted(m.strip().lower() for m in mesh_names if m.strip())
|
||
if not items:
|
||
return None
|
||
raw_key = f"{'|'.join(items)}:major={major_only}:noexp={noexp}"
|
||
cache_key = f"atm:{_hashlib.md5(raw_key.encode()).hexdigest()}"
|
||
|
||
# 查缓存
|
||
cached = await _cache.get(cache_key)
|
||
if cached is not None:
|
||
mesh_tag_ids = {_uuid.UUID(uid) for uid in cached["tag_ids"]}
|
||
else:
|
||
mesh_tag_ids: set[_uuid.UUID] = set()
|
||
|
||
# Batch all mesh name lookups — 2 queries instead of 2N
|
||
entry_conds = []
|
||
name_conds = []
|
||
for m in mesh_names:
|
||
q = m.strip().lower()
|
||
if not q:
|
||
continue
|
||
# entry_terms 在 import_mesh_full.py 时已统一小写,可安全用 @> 精确匹配
|
||
entry_conds.append(GlobalTag.entry_terms.contains([q]))
|
||
name_conds.append(GlobalTag.name_en.ilike(_escape_ilike(m)))
|
||
|
||
try:
|
||
if entry_conds:
|
||
rows = (await db.execute(
|
||
select(GlobalTag.id).where(
|
||
GlobalTag.source.in_(["mesh", "manual"]),
|
||
GlobalTag.mesh_ui.isnot(None),
|
||
GlobalTag.entry_terms.isnot(None),
|
||
or_(*entry_conds),
|
||
)
|
||
)).all()
|
||
for (tid,) in rows:
|
||
mesh_tag_ids.add(tid)
|
||
|
||
if name_conds:
|
||
rows = (await db.execute(
|
||
select(GlobalTag.id).where(
|
||
GlobalTag.source.in_(["mesh", "manual"]),
|
||
GlobalTag.mesh_ui.isnot(None),
|
||
or_(*name_conds),
|
||
)
|
||
)).all()
|
||
for (tid,) in rows:
|
||
mesh_tag_ids.add(tid)
|
||
except Exception:
|
||
logger.exception("MeSH tag lookup failed for mesh_names=%s", mesh_names[:5])
|
||
|
||
if mesh_tag_ids:
|
||
# tree_number 前缀展开
|
||
if not noexp:
|
||
try:
|
||
tns = (await db.execute(
|
||
select(GlobalTagTreeNumber.tree_number).where(
|
||
GlobalTagTreeNumber.tag_id.in_(list(mesh_tag_ids))
|
||
).distinct()
|
||
)).scalars().all()
|
||
|
||
if tns:
|
||
child_conds = [or_(
|
||
GlobalTagTreeNumber.tree_number == tn,
|
||
GlobalTagTreeNumber.tree_number.like(f"{tn}.%"),
|
||
) for tn in tns]
|
||
children = (await db.execute(
|
||
select(GlobalTagTreeNumber.tag_id).where(or_(*child_conds))
|
||
)).scalars().all()
|
||
mesh_tag_ids.update(children)
|
||
except Exception:
|
||
logger.exception("Tree number expansion failed for mesh_names=%s", mesh_names[:5])
|
||
|
||
# 写缓存(即使为空也缓存,避免重复查空)
|
||
await _cache.set(cache_key, {"tag_ids": [str(tid) for tid in mesh_tag_ids]}, ttl=3600)
|
||
|
||
if not mesh_tag_ids:
|
||
return None
|
||
|
||
uids = list(mesh_tag_ids)
|
||
if major_only:
|
||
subq = select(func.distinct(GlobalLiteratureTag.literature_id)).where(
|
||
GlobalLiteratureTag.tag_id.in_(uids),
|
||
GlobalLiteratureTag.is_major == True,
|
||
)
|
||
else:
|
||
subq = select(func.distinct(GlobalLiteratureTag.literature_id)).where(
|
||
GlobalLiteratureTag.tag_id.in_(uids),
|
||
)
|
||
return GlobalLiterature.id.in_(subq)
|
||
|
||
@staticmethod
|
||
def _best_match_order(tsq):
|
||
"""构建 best_match 排序表达式:ts_rank + 引用数对数 + 近期度梯度加分
|
||
|
||
近期度加分通过 EXTRACT(YEAR FROM NOW()) 动态计算,分三档:
|
||
- 当年: +10
|
||
- 去年: +7
|
||
- 前年: +3
|
||
"""
|
||
_cy = func.extract("year", func.now())
|
||
score = (
|
||
func.ts_rank(GlobalLiterature.search_tsv, tsq) * 0.3
|
||
+ func.ln(func.coalesce(GlobalLiterature.cited_by_count, 0) + 1) * 2
|
||
+ case(
|
||
(GlobalLiterature.pub_year >= _cy, 10),
|
||
(GlobalLiterature.pub_year >= _cy - 1, 7),
|
||
(GlobalLiterature.pub_year >= _cy - 2, 3),
|
||
else_=0,
|
||
)
|
||
)
|
||
return score.desc()
|
||
|
||
KEYSET_COLUMN_SORTS = {"date", "cited", "title", "journal", "first_author"}
|
||
|
||
@staticmethod
|
||
def _apply_order_by(sort: str, relevance_query: str):
|
||
"""Return a list of order_by expressions for the given sort mode."""
|
||
if sort == "date":
|
||
return [GlobalLiterature.pub_date.desc().nullslast(), GlobalLiterature.id.desc()]
|
||
elif sort == "cited":
|
||
return [GlobalLiterature.cited_by_count.desc().nullslast(), GlobalLiterature.id.desc()]
|
||
elif sort == "best_match" and relevance_query.strip():
|
||
tsq = func.plainto_tsquery("english", relevance_query)
|
||
return [AdvancedSearchEngine._best_match_order(tsq), GlobalLiterature.id.desc()]
|
||
elif sort == "relevance" and relevance_query.strip():
|
||
tsq = func.plainto_tsquery("english", relevance_query)
|
||
rank = func.ts_rank(GlobalLiterature.search_tsv, tsq)
|
||
return [rank.desc(), GlobalLiterature.id.desc()]
|
||
elif sort == "first_author":
|
||
return [GlobalLiterature.authors[0]['family'].astext.asc().nullslast(),
|
||
GlobalLiterature.id.asc()]
|
||
elif sort == "journal":
|
||
return [GlobalLiterature.journal.asc().nullslast(),
|
||
GlobalLiterature.id.asc()]
|
||
elif sort == "title":
|
||
return [GlobalLiterature.title.asc().nullslast(),
|
||
GlobalLiterature.id.asc()]
|
||
else:
|
||
return [GlobalLiterature.pub_date.desc().nullslast(), GlobalLiterature.id.desc()]
|
||
|
||
@staticmethod
|
||
def _keyset_condition(sort: str, cursor_val: str | None, cursor_id: str | None) -> object | None:
|
||
"""构建 keyset WHERE 条件(适用所有列式排序模式)。
|
||
|
||
date/cited → DESC(col < val)
|
||
title/journal/first_author → ASC(col > val)
|
||
best_match/relevance 为计算表达式,降级到 OFFSET(返回 None)。
|
||
统一用 id 做 tiebreaker。
|
||
"""
|
||
if sort not in AdvancedSearchEngine.KEYSET_COLUMN_SORTS:
|
||
return None
|
||
if not cursor_val or cursor_id is None:
|
||
return None
|
||
import uuid as _uuid
|
||
try:
|
||
uid = _uuid.UUID(cursor_id)
|
||
except (ValueError, AttributeError):
|
||
return None
|
||
|
||
try:
|
||
if cursor_val == "__NULL__":
|
||
if sort == "date":
|
||
return and_(GlobalLiterature.pub_date.is_(None), GlobalLiterature.id < uid)
|
||
elif sort == "cited":
|
||
return and_(GlobalLiterature.cited_by_count.is_(None), GlobalLiterature.id < uid)
|
||
elif sort == "title":
|
||
return and_(GlobalLiterature.title.is_(None), GlobalLiterature.id > uid)
|
||
elif sort == "journal":
|
||
return and_(GlobalLiterature.journal.is_(None), GlobalLiterature.id > uid)
|
||
elif sort == "first_author":
|
||
return and_(GlobalLiterature.authors[0]['family'].astext.is_(None), GlobalLiterature.id > uid)
|
||
if sort == "date":
|
||
from datetime import date as dt_date
|
||
val = dt_date.fromisoformat(cursor_val)
|
||
return or_(
|
||
GlobalLiterature.pub_date < val,
|
||
and_(GlobalLiterature.pub_date == val, GlobalLiterature.id < uid),
|
||
GlobalLiterature.pub_date.is_(None), # P12: nullslat transition, no id constraint
|
||
)
|
||
elif sort == "cited":
|
||
val = int(cursor_val)
|
||
return or_(
|
||
GlobalLiterature.cited_by_count < val,
|
||
and_(GlobalLiterature.cited_by_count == val, GlobalLiterature.id < uid),
|
||
GlobalLiterature.cited_by_count.is_(None), # P12
|
||
)
|
||
elif sort == "title":
|
||
return or_(
|
||
GlobalLiterature.title > cursor_val,
|
||
and_(GlobalLiterature.title == cursor_val, GlobalLiterature.id > uid),
|
||
GlobalLiterature.title.is_(None), # P12
|
||
)
|
||
elif sort == "journal":
|
||
return or_(
|
||
GlobalLiterature.journal > cursor_val,
|
||
and_(GlobalLiterature.journal == cursor_val, GlobalLiterature.id > uid),
|
||
GlobalLiterature.journal.is_(None), # P12
|
||
)
|
||
elif sort == "first_author":
|
||
family_col = GlobalLiterature.authors[0]['family'].astext
|
||
return or_(
|
||
family_col > cursor_val,
|
||
and_(family_col == cursor_val, GlobalLiterature.id > uid),
|
||
family_col.is_(None), # P12
|
||
)
|
||
except (ValueError, TypeError):
|
||
return None
|
||
return None
|
||
|
||
@staticmethod
|
||
def _cursor_from_item(lit, sort: str) -> str | None:
|
||
"""从末尾条目标提取 keyset 游标值。"""
|
||
if sort == "date":
|
||
# R24: ignore article_date for cursor — ORDER BY is only pub_date DESC NULLS LAST
|
||
if lit.pub_date is not None:
|
||
return str(lit.pub_date)
|
||
return "__NULL__"
|
||
elif sort == "cited":
|
||
if lit.cited_by_count is not None:
|
||
return str(lit.cited_by_count)
|
||
return "__NULL__"
|
||
elif sort == "title":
|
||
return lit.title # title is NOT NULL per model
|
||
elif sort == "journal":
|
||
return lit.journal if lit.journal is not None else "__NULL__"
|
||
elif sort == "first_author":
|
||
authors = lit.authors or []
|
||
if authors and isinstance(authors[0], dict):
|
||
return authors[0].get("family") or "__NULL__"
|
||
return "__NULL__"
|
||
return None
|
||
|
||
@staticmethod
|
||
def _build_date_cond_from_pp(pp, field_tag):
|
||
"""Build date range SQLAlchemy condition from ParsedPubmedQuery for a field tag.
|
||
Used by R23-3 to include date ranges inside negated group scope for correct De Morgan.
|
||
"""
|
||
from datetime import date as _dt_date
|
||
|
||
if field_tag == "DP":
|
||
conds = []
|
||
if pp.year_from is not None:
|
||
conds.append(GlobalLiterature.pub_year >= pp.year_from)
|
||
if pp.year_to is not None:
|
||
conds.append(GlobalLiterature.pub_year <= pp.year_to)
|
||
if getattr(pp, 'date_from', None):
|
||
try:
|
||
conds.append(GlobalLiterature.pub_date >= _dt_date.fromisoformat(pp.date_from))
|
||
except ValueError:
|
||
pass
|
||
if getattr(pp, 'date_to', None):
|
||
try:
|
||
conds.append(GlobalLiterature.pub_date <= _dt_date.fromisoformat(pp.date_to))
|
||
except ValueError:
|
||
pass
|
||
return and_(*conds) if conds else None
|
||
|
||
_DATE_COL_MAP = {
|
||
"EDAT": GlobalLiterature.entrez_date,
|
||
"CRDT": GlobalLiterature.create_date,
|
||
"MHDA": GlobalLiterature.meshed_date,
|
||
"LR": GlobalLiterature.pubmed_revised,
|
||
"DCOM": GlobalLiterature.date_completed,
|
||
"DEP": GlobalLiterature.pub_date,
|
||
}
|
||
col = _DATE_COL_MAP.get(field_tag)
|
||
if col is None:
|
||
return None
|
||
_from = getattr(pp, f"{field_tag.lower()}_from", None)
|
||
_to = getattr(pp, f"{field_tag.lower()}_to", None)
|
||
conds = []
|
||
if _from:
|
||
try:
|
||
conds.append(col >= _dt_date.fromisoformat(_from))
|
||
except ValueError:
|
||
pass
|
||
if _to:
|
||
try:
|
||
conds.append(col <= _dt_date.fromisoformat(_to))
|
||
except ValueError:
|
||
pass
|
||
return and_(*conds) if conds else None
|
||
|
||
@staticmethod
|
||
def _build_marker_condition(marker_term) -> object | None:
|
||
"""Build SQL condition from a __RANGE_* marker term for group-internal date ranges.
|
||
|
||
R27: used by the group processing path to build date conditions from markers,
|
||
replacing the old approach that read conflated global attributes.
|
||
"""
|
||
_mf = marker_term.field or ""
|
||
if not _mf.startswith("__RANGE_") or not _mf.endswith("__"):
|
||
return None
|
||
_tag = _mf[8:-2]
|
||
if _tag not in ("DP", "EDAT", "CRDT", "MHDA", "LR", "DCOM", "DEP"):
|
||
return None
|
||
|
||
parts = marker_term.text.split(":", 1)
|
||
if len(parts) != 2:
|
||
return None
|
||
_from_str, _to_str = parts
|
||
|
||
from datetime import date as _dt_date
|
||
|
||
if _tag == "DP":
|
||
conds = []
|
||
if _from_str.isdigit() and len(_from_str) == 4:
|
||
conds.append(GlobalLiterature.pub_year >= int(_from_str))
|
||
else:
|
||
try:
|
||
conds.append(GlobalLiterature.pub_date >= _dt_date.fromisoformat(_from_str))
|
||
except ValueError:
|
||
pass
|
||
if _to_str.isdigit() and len(_to_str) == 4:
|
||
conds.append(GlobalLiterature.pub_year <= int(_to_str))
|
||
else:
|
||
try:
|
||
conds.append(GlobalLiterature.pub_date <= _dt_date.fromisoformat(_to_str))
|
||
except ValueError:
|
||
pass
|
||
return and_(*conds) if conds else None
|
||
|
||
_DATE_COL_MAP = {
|
||
"EDAT": GlobalLiterature.entrez_date,
|
||
"CRDT": GlobalLiterature.create_date,
|
||
"MHDA": GlobalLiterature.meshed_date,
|
||
"LR": GlobalLiterature.pubmed_revised,
|
||
"DCOM": GlobalLiterature.date_completed,
|
||
"DEP": GlobalLiterature.pub_date,
|
||
}
|
||
col = _DATE_COL_MAP.get(_tag)
|
||
if col is None:
|
||
return None
|
||
conds = []
|
||
if _from_str.isdigit() and len(_from_str) == 4:
|
||
_from_str = f"{_from_str}-01-01"
|
||
try:
|
||
conds.append(col >= _dt_date.fromisoformat(_from_str))
|
||
except ValueError:
|
||
pass
|
||
if _to_str.isdigit() and len(_to_str) == 4:
|
||
_to_str = f"{_to_str}-12-31"
|
||
try:
|
||
conds.append(col <= _dt_date.fromisoformat(_to_str))
|
||
except ValueError:
|
||
pass
|
||
return and_(*conds) if conds else None
|