Repository navigation
Expand file tree
/
Copy pathtypes.py
More file actions
385 lines (329 loc) · 18.4 KB
/
Copy pathtypes.py
File metadata and controls
385 lines (329 loc) · 18.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
"""
paper-search-pro Skill - Shared types across helper scripts.
In v2.0 architecture, this file holds ONLY data types for deterministic helpers.
LLM orchestration (state machine, budget controller, tools) belongs to the main
Claude Code agent driven by SKILL.md, not Python — those types have been removed.
Helper scripts that consume these types:
- openalex_helper.py, ss_helper.py, crossref_helper.py, pubmed_helper.py, arxiv_helper.py
- federated_kg_resolver.py
- discovery_curve.py, rcs_parser.py
- data_materialization.py, html_renderer_webartifacts.py, generate_exports.py, md_report.py
- prisma_s_logger.py, semantic_cache.py
"""
import hashlib
from dataclasses import dataclass, field
from typing import Dict, List, Literal, Optional
# =============================================================================
# Tier
# =============================================================================
Tier = Literal["quick", "standard", "deep", "audit"]
# =============================================================================
# Paper & Authors (unified cross-source representation)
# =============================================================================
@dataclass
class Author:
"""A single author. ORCID and country are best-effort (OpenAlex provides; SS often null)."""
name: str
orcid: Optional[str] = None
affiliation: Optional[str] = None
country: Optional[str] = None
is_first: bool = False
is_corresponding: bool = False
@dataclass
class JournalMetric:
"""Journal influence / quartile for one paper's venue (v2.2 Feature A).
Additive, opt-in: ``UnifiedPaperEntity.journal_metric`` defaults to None so
existing serialization / behavior is unchanged. Populated only when the SJR
lookup and/or OpenAlex source stats are available.
Naming is deliberate (R-04 / R-09): this is **SJR分区 / 期刊影响力**, never a
JCR Impact Factor.
- sjr_quartile : "Q1".."Q4" or None — SCImago Best Quartile
(or the chosen category's quartile). Data © SCImago (scimagojr.com),
non-commercial cited use — SCImago custom terms, NOT a Creative Commons /
CC BY-NC licence; attribution required.
- sjr_category_quartiles : {category: "Qn"} — per-category SJR quartiles.
- openalex_2yr_mean_citedness: OpenAlex source summary_stats — an OPEN
journal-impact figure. R-09: this is NOT the official JIF (e.g. JPSP reads
~2.7 here vs ~7-8 official); use for relative ranking/filtering only.
- h_index : OpenAlex source h-index (CC0).
- cwts_snip : optional cross-field-normalised SNIP (reserved).
- sjr_attribution : the mandatory SJR citation string (R-03) when
any SJR field is present; None when no SJR data was joined.
- issn_backfill_needed : True when the paper has no ISSN to join on
(SS-primary records frequently lack publicationVenue.issn — R-08). Marks the
gap so an OpenAlex DOI-lookup backfill can be wired in as an integration
point.
"""
sjr_quartile: Optional[str] = None
sjr_category_quartiles: Dict[str, str] = field(default_factory=dict)
openalex_2yr_mean_citedness: Optional[float] = None
h_index: Optional[int] = None
cwts_snip: Optional[float] = None
sjr_attribution: Optional[str] = None
issn_backfill_needed: bool = False
@dataclass
class CASRank:
"""CAS (中科院文献情报中心) journal partition for one venue (v2.2 Feature A).
Additive / optional — defaults keep serialization byte-compatible. 区 1-4 is a
PARTITION (大类分区), NOT an Impact Factor (R-04). 勿公开传播 (personal use).
- tier : 大类分区 区号 1-4 (1 = top).
- rank : within-大类 rank string, e.g. "168/495".
- top : CAS Top 期刊 flag.
- minor : up to six 小类 (sub-category) partitions
[{category, tier, rank}].
- source_year : the CAS table year (e.g. 2025).
"""
tier: Optional[int] = None
rank: Optional[str] = None
top: bool = False
minor: List[Dict] = field(default_factory=list)
source_year: Optional[int] = None
@dataclass
class JCRRank:
"""JCR (Clarivate) journal record for one venue (v2.2 Feature A).
Additive / optional. This is the ONLY source of a real **Impact Factor**
(R-04 / R-09): ``impact_factor`` here is the genuine JCR IF(2024); OpenAlex
2yr-mean-citedness elsewhere is NOT this.
- quartile : "Q1".."Q4" (best across categories).
- impact_factor: the real JCR IF(2024). © Clarivate.
- rank : category rank, e.g. "1/326".
- category : raw JCR Category string (may carry multiple, ``;``-joined).
- source_year : the JCR table year (e.g. 2024).
"""
quartile: Optional[str] = None
impact_factor: Optional[float] = None
rank: Optional[str] = None
category: Optional[str] = None
source_year: Optional[int] = None
@dataclass
class SJRRank:
"""SJR (SCImago) journal record for one venue (v2.2 Feature A).
Additive / optional. SJR quartile is a PARTITION/quartile, NOT an IF (R-04).
Data © SCImago (scimagojr.com), non-commercial cited use — NOT CC BY-NC.
- best_quartile: "Q1".."Q4" (best across categories).
- sjr : the SJR indicator value.
- per_category : per-category quartiles [{category, quartile}].
- source_year : the SJR table year (e.g. 2024).
"""
best_quartile: Optional[str] = None
sjr: Optional[float] = None
per_category: List[Dict] = field(default_factory=list)
source_year: Optional[int] = None
@dataclass
class JournalRank:
"""Unified multi-platform journal rank for one venue (v2.2 Feature A).
Additive / optional: ``UnifiedPaperEntity.journal_rank`` defaults to None so
existing serialization / behavior is byte-compatible. Each platform slot is
independently optional (a journal found on only one platform still yields a
valid record). Populated by journal_rank.py when ranking CSVs are cached.
NOTE this is the v2.2 A-line *multi-platform* schema (CAS + JCR + SJR + an
OpenAlex open-impact slot). As of the v2.2 single-layer collapse it is the ONE
journal-rank record on a paper: the older SJR-only ``JournalMetric`` above is no
longer populated by the search pipeline (kept only for back-compat decoding of
pre-collapse kg.json).
- title : journal title (from whichever platform supplied it).
- issns : normalised "XXXX-XXXX" keys (print + electronic).
- cas / jcr / sjr : per-platform sub-records (None when not on that platform).
- openalex : OPEN journal-impact slot {mean_citedness_2yr, h_index} from
OpenAlex summary_stats (CC0). R-04/R-09: this is an OPEN
impact figure, NEVER an Impact Factor — only JCR's IF(2024)
is a real IF. Filled by the search pipeline, not the CSV
parsers; None when no impact was looked up / reachable.
- matched_issn : the ISSN a lookup hit on.
- matched_platforms : which platforms contributed (["cas","jcr","sjr"] subset).
"""
title: Optional[str] = None
issns: List[str] = field(default_factory=list)
cas: Optional["CASRank"] = None
jcr: Optional["JCRRank"] = None
sjr: Optional["SJRRank"] = None
openalex: Optional[Dict] = None # {mean_citedness_2yr: float|None, h_index: int|None}
matched_issn: Optional[str] = None
matched_platforms: List[str] = field(default_factory=list)
@dataclass
class UnifiedPaperEntity:
"""Cross-source unified paper representation. DOI is Primary Key.
Source merging priority (see federated_kg_resolver.py):
- title / authors / year: OpenAlex
- abstract: OpenAlex (reconstructed) -> SS fallback
- citation_count: OpenAlex
- influential_citation_count: SS ONLY (unique signal)
- references: OpenAlex -> CrossRef supplement (skip arXiv DOIs)
- funder / license / clinical_trial_number: CrossRef
- mesh_terms / pmcid: PubMed ONLY
- arxiv_id: arXiv ONLY (also in OpenAlex.locations[])
"""
# Identifiers
doi: Optional[str] = None # lowercase, no URL prefix
arxiv_id: Optional[str] = None # without version suffix, e.g. "1706.03762"
openalex_id: Optional[str] = None # "W..." prefix
ss_paper_id: Optional[str] = None # SS hex paperId
pmid: Optional[str] = None
pmcid: Optional[str] = None
# Source-native stable ID for records that carry no DOI / arXiv / PMID /
# OpenAlex / SS identifier (e.g. NSSD ``nssd:JJYJ2024005009`` or yiigle
# ``yiigle:...``). Additive / optional (v2.2.1 Phase 0, 0.2): defaults to None
# so canonical_key / paper_id / serialization stay byte-identical (R-19). The
# ('native', id) key branch and the paper_id fallback are reached ONLY when
# this is set — which no OA / SS / CrossRef / PubMed / arXiv record does — so
# every existing English record behaves exactly as before.
source_native_id: Optional[str] = None
# Core metadata
title: str = ""
abstract: Optional[str] = None
authors: List[Author] = field(default_factory=list)
year: Optional[int] = None
venue: Optional[str] = None
issn: Optional[str] = None # journal ISSN (e.g. SS publicationVenue.issn / OA source.issn) — used downstream for SJR join
# Every ISSN the source lists for the journal (OpenAlex ``source.issn[]``, SS
# ``publicationVenue.issn`` + ``alternate_issns``). ``issn`` above stays the
# single preferred key; the rank join falls back to these when it misses,
# because a journal's linking ISSN is often not the one the rank tables
# list (The Lancet, renamed journals). Additive: empty for sources that
# list no ISSN, and the report does not display it.
issns: List[str] = field(default_factory=list)
type: Optional[str] = None # article / preprint / review / book / dataset
# Citations
citation_count: int = 0
referenced_works_count: Optional[int] = None
# OpenAlex-specific
fwci: Optional[float] = None
cited_by_percentile_year: Optional[float] = None
topics: List[Dict] = field(default_factory=list)
keywords: List[str] = field(default_factory=list)
sdgs: List[Dict] = field(default_factory=list)
is_oa: Optional[bool] = None
# Semantic Scholar-specific
tldr: Optional[str] = None
influential_citation_count: Optional[int] = None
# CrossRef-specific
funders: List[Dict] = field(default_factory=list) # [{name, doi}]
license: List[Dict] = field(default_factory=list) # [{URL, content-version, delay-in-days}]
clinical_trial_number: Optional[str] = None
# PubMed-specific
mesh_terms: List[str] = field(default_factory=list)
publication_types: List[str] = field(default_factory=list) # ["Clinical Trial", "Review", ...]
# arXiv-specific
arxiv_categories: List[str] = field(default_factory=list) # ["cs.CL", "cs.AI"]
arxiv_comment: Optional[str] = None # "Accepted at NeurIPS 2024"
# URLs
doi_url: Optional[str] = None
openalex_url: Optional[str] = None
pdf_url: Optional[str] = None
pmc_url: Optional[str] = None
# Additional open-access copy URLs parsed from OpenAlex ``locations[]``
# (v2.2.1 Phase 0, 0.3 — additive). ``pdf_url`` above still holds the single
# best OA URL (open_access.oa_url); this is the fuller multi-copy list (e.g.
# publisher OA + PMC + repository). Empty list for papers with no OA copies,
# so a non-OA record is byte-identical to pre-0.3 (R-19); the display path
# (_render_paper) does not emit it, so the deliverable is unchanged.
oa_locations: List[str] = field(default_factory=list)
# Journal influence / quartile (v2.2 Feature A, additive — default None so
# existing serialization is byte-compatible). Populated by sjr_helper +
# OpenAlex source stats when available; None means "not looked up".
journal_metric: Optional["JournalMetric"] = None
# Multi-platform journal rank (v2.2 Feature A-line, additive — default None so
# existing serialization is byte-compatible). Populated by journal_rank.py
# (CAS + JCR + SJR) when ranking CSVs are cached; None means "not looked up".
# Wave A-2 wires this into agent_search; until then it is reserved.
journal_rank: Optional["JournalRank"] = None
# Skill-internal (set by main Claude Code agent or scripts)
rcs: Optional[int] = None # 0-10, set during classification
rcs_reasoning: Optional[str] = None
rcs_flag: Optional[str] = None # parse_failed_uncertain / off_topic_despite_keywords / abstract_unavailable / no_abstract_uncertain
sources: List[str] = field(default_factory=list) # ["openalex", "semantic_scholar", "crossref", "pubmed", "arxiv"]
discovery_path: Optional[str] = None # "query: prospect theory" / "ref of W12345" / "cites W12345" / "arxiv:T-0~T-4"
@property
def paper_id(self) -> str:
"""Stable identifier across sources. Priority: DOI > arxiv > openalex_id >
pmid > ss_paper_id > source_native_id > title-hash.
``source_native_id`` sits just above the title-hash fallback (0.2): it only
applies to records lacking every standard ID (e.g. NSSD / yiigle), so any
record with a DOI / arXiv / OpenAlex / PMID / SS id keeps its prior
paper_id byte-for-byte (R-19)."""
return (
self.doi
or (f"arxiv:{self.arxiv_id}" if self.arxiv_id else None)
or self.openalex_id
or (f"pmid:{self.pmid}" if self.pmid else None)
or self.ss_paper_id
or self.source_native_id
# md5, not hash(): hash() is salted per process, so the same paper got
# a different id in the export than in the report.
or f"untitled_{hashlib.md5((self.title or '').encode('utf-8')).hexdigest()[:16]}"
)
@dataclass
class ParsedRCS:
"""RCS parser output for a single paper. See rcs_parser.py."""
paper_id: str
rcs: int
reasoning: str
flag: Optional[str] = None
# =============================================================================
# Configuration (loaded from ~/.paper-search-pro/config.yaml)
# =============================================================================
@dataclass
class Config:
"""User config + defaults. Loaded by config.py.
No state-machine / budget fields — those are managed by the main Claude Code
agent following SKILL.md, not Python.
"""
# ---- Data source credentials ----
openalex_email: str = "" # Required for OpenAlex polite pool (or use API key)
openalex_api_key: str = "" # Optional: OpenAlex Premium / Bearer token
semantic_scholar_api_key: str = "" # Optional: SS API key (15x rate limit boost)
ncbi_email: str = "" # Required for PubMed (always)
ncbi_api_key: str = "" # Optional: NCBI key (3 req/s -> 10 req/s)
crossref_email: str = "" # Required for CrossRef polite pool
# ---- Output ----
output_dir: str = "./paper-search-results"
default_tier: Tier = "standard"
language: str = "en" # REPORT UI chrome language (en | zh); auto-detected by detect_language.py. ORTHOGONAL to search_language below — this is "what language the report speaks", not "which literature to fish in".
# ---- Language routing (v2.3, additive — default "auto" preserves v2.2 behavior) ----
# The search "language space" (axis 2 of the three-axis model): which ocean of
# literature to search — English, Chinese, or both. Additive / opt-in: a config
# missing this key defaults to "auto", so every existing (English) run is
# byte-for-byte unchanged (same additive precedent as primary_source — R-19).
# auto (factory) — English query -> en space (byte-identical to v2.2); Chinese
# query -> follow explicit in-query signals, else the human
# path asks once and the agent path passes the query through.
# en — never enter the Chinese space; a Chinese query is planned as an
# English search (with a one-line notice, never a silent translation).
# zh — Chinese query runs in the Chinese space (OpenAlex Chinese base +
# discipline-routed NSSD/yiigle boosters), search terms kept in Chinese.
# both — dual-space federated search (English + Chinese strategy sets).
# ORTHOGONAL to `language` above (UI chrome) — neither reads nor overrides the
# other. SSOT for parsing priority / markers: references/source_routing.md
# §"Language scope".
search_language: str = "auto" # auto | en | zh | both
# What "最新 / recent" means when the user gives no number: N years back
# from the current year. None = ask once per run (SKILL.md STEP 1 "Time
# scope"). Written only when the user says "以后都这样".
recent_years: Optional[int] = None
# ---- Source routing (v2.2, additive — defaults preserve v2.0/2.1 behavior) ----
# Which source the search entry-point treats as primary. "openalex" = current
# behavior unchanged. "semantic_scholar" = use SS bulk search as primary.
# "auto" = start on OpenAlex but let quota_guard (run mode) stickily fall
# back to SS for the rest of a run when OpenAlex USD budget runs low.
primary_source: str = "openalex" # openalex | semantic_scholar | auto
# Serve calls OpenAlex cannot serve (budget spent / throttled / down) from
# Semantic Scholar; also gates the "auto" pre-flight switch.
quota_fallback: bool = True
# USD budget remaining (per OpenAlex X-RateLimit-Remaining-USD) at or below
# which "auto" mode switches to SS for the remainder of the run.
quota_fallback_threshold_usd: float = 0.05
# ---- Journal rank (v2.2 Feature A — additive; default None = built-ins) ----
# Nested dict from config ``rank:`` (default_platform / cache_dir / sources).
# None means "use journal_rank.py built-in defaults". Multi-platform partition
# / quartile (CAS / JCR / SJR); data fetched at runtime, never bundled.
rank: Optional[Dict] = None
# ---- HTML rendering ----
# (No size cap or alternative renderer as of 2026-05-23. The Skill
# always uses html_renderer_webartifacts; the size-driven jinja2
# fallback was removed for UX consistency.)
# ---- Cache ----
cache_enabled: bool = True
cache_ttl_days: int = 7
cache_max_size_mb: int = 500
# ---- Logging ----
log_level: str = "INFO"