feat: initial commit - oncology literature search platform
CI / backend (push) Canceled after 0s
CI / frontend (push) Canceled after 0s

OncoLit: a multi-tenant oncology literature search, feed, and
collaboration platform. Built with FastAPI + Vue 3 + PostgreSQL.
Includes PubMed pipeline, drug approvals, AI summaries, and
systematic review tools.
This commit is contained in:
34047007@qq.com
2026-07-27 07:59:18 +08:00
commit a6cd99a4ca
473 changed files with 151472 additions and 0 deletions
+73
View File
@@ -0,0 +1,73 @@
"""Check if article_date also has YYYY-01-01 issue"""
import asyncio, httpx, lxml.etree as ET
from datetime import date
from sqlalchemy import text
from sqlalchemy.ext.asyncio import create_async_engine
DATABASE_URL = 'postgresql+asyncpg://scilit:scilit_prod_2026@postgres:5432/scilit'
async def main():
engine = create_async_engine(DATABASE_URL)
async with engine.connect() as conn:
# 1. Count article_date YYYY-01-01
r = await conn.execute(text("""
SELECT COUNT(*) FROM global_literature WHERE article_date IS NOT NULL
AND EXTRACT(MONTH FROM article_date)=1 AND EXTRACT(DAY FROM article_date)=1
"""))
ad_jan01 = r.scalar()
r = await conn.execute(text("SELECT COUNT(*) FROM global_literature WHERE article_date IS NOT NULL"))
ad_total = r.scalar()
print(f'article_date Jan-01: {ad_jan01} / {ad_total} total ({(ad_jan01/ad_total*100) if ad_total else 0:.1f}%)')
# 2. Count records where pub_date is YYYY-01-01 but article_date is not
r = await conn.execute(text("""
SELECT COUNT(*) FROM global_literature
WHERE EXTRACT(MONTH FROM pub_date)=1 AND EXTRACT(DAY FROM pub_date)=1
AND article_date IS NOT NULL
AND NOT (EXTRACT(MONTH FROM article_date)=1 AND EXTRACT(DAY FROM article_date)=1)
"""))
fixed_but_ad_not = r.scalar()
print(f'pub_date Jan-01 but article_date has real date: {fixed_but_ad_not}')
# 3. Count by year for article_date Jan-01
r = await conn.execute(text("""
SELECT EXTRACT(YEAR FROM article_date) as y, COUNT(*) as c FROM global_literature
WHERE article_date IS NOT NULL AND EXTRACT(MONTH FROM article_date)=1 AND EXTRACT(DAY FROM article_date)=1
GROUP BY y ORDER BY y DESC LIMIT 10
"""))
print('\narticle_date Jan-01 by year:')
for row in r:
print(f' {int(row.y)}: {row.c}')
# 4. Check for any pub_date that's not YYYY-01-01 but pub_year differs from EXTRACT(YEAR FROM pub_date)
r = await conn.execute(text("""
SELECT COUNT(*) FROM global_literature
WHERE pub_date IS NOT NULL AND pub_year IS NOT NULL
AND pub_year != EXTRACT(YEAR FROM pub_date)
"""))
year_mismatch = r.scalar()
print(f'\npub_year != EXTRACT(YEAR FROM pub_date): {year_mismatch}')
# 5. Check for orphan feeds (articles that don't exist)
r = await conn.execute(text("""
SELECT COUNT(*) FROM user_feed uf
LEFT JOIN global_literature gl ON uf.literature_id = gl.id
WHERE gl.id IS NULL
"""))
orphan_feeds = r.scalar()
print(f'Orphan feeds (no matching literature): {orphan_feeds}')
# 6. Check the articles that were backfilled (2018-2023) - do they have proper dates now?
r = await conn.execute(text("""
SELECT EXTRACT(YEAR FROM pub_date) as y, COUNT(*) as c FROM global_literature
WHERE EXTRACT(YEAR FROM pub_date) BETWEEN 2018 AND 2023
AND EXTRACT(MONTH FROM pub_date)=1 AND EXTRACT(DAY FROM pub_date)=1
GROUP BY y ORDER BY y DESC
"""))
print('\nBackfill years (2018-2023) still Jan-01:')
for row in r:
print(f' {int(row.y)}: {row.c}')
await engine.dispose()
asyncio.run(main())