From 953da65081670febd3a3bd4f64b35c52b1047734 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ahmet=20O=C4=9Fuzhan=20K=C3=B6k=C3=BCl=C3=BC?= Date: Fri, 14 Aug 2026 20:42:31 +0300 Subject: [PATCH 1/8] feat: add PubTator3 and Paperclip literature-evidence agents Two independent literature-evidence agents, each in its own package alongside agent_tools/. Purely additive: no existing file is modified. crossbar_llm/pubtator3_tools/ NCBI PubTator3 (autocomplete, relations, search, BioC export) crossbar_llm/paperclip_tools/ Paperclip full-text corpora, REST-primary with an automatic MCP fallback Each package holds its own client/adapter, LangGraph agent, prompts, schemas and tests, and neither imports the other - verified by asserting that importing one leaves the other absent from sys.modules. The two small provider-agnostic helpers they both need (structured-output-with-JSON-fallback, and token usage capture) are duplicated rather than shared, so a change to one agent cannot break the other. Both expose the same contract: question (+ state) in -> {final_answer, citations, warnings, usage} out where question_type == "out_of_scope" with final_answer None means the agent declines, letting a caller distinguish that from an answer or a failure. `build_graph` takes an already-built `chat_model`, which keeps the graph independent of model configuration and is the seam the tests inject fakes at. Each package's `llm.py` is the convenience path for callers, building that model through agent_tools.llm_factory. Nothing calls the agents yet. Wiring them into the API is a separate change. New dependencies: aiolimiter>=1.2.1 and langchain-mcp-adapters>=0.3.2 at runtime, pytest-asyncio>=1.4.0 and pytest-httpx>=0.36.2 for tests. Resolved against this branch's lockfile on Python 3.12.11 with no change to existing pins (langchain-core 1.5.0, langgraph 1.2.6, httpx 0.28.1, pydantic 2.13.4). pytest -c crossbar_llm/pubtator3_tools/tests/pytest.ini crossbar_llm/pubtator3_tools/tests pytest -c crossbar_llm/paperclip_tools/tests/pytest.ini crossbar_llm/paperclip_tools/tests Note: Paperclip caps its full-text `map` step at 100 operations/day per API key, shared across all users of a deployment. --- crossbar_llm/paperclip_tools/__init__.py | 12 + crossbar_llm/paperclip_tools/adapter.py | 1302 +++++++++++++++++ crossbar_llm/paperclip_tools/agent.py | 440 ++++++ crossbar_llm/paperclip_tools/llm.py | 41 + crossbar_llm/paperclip_tools/nodes.py | 464 ++++++ crossbar_llm/paperclip_tools/prompts.py | 169 +++ crossbar_llm/paperclip_tools/schemas.py | 360 +++++ .../paperclip_tools/structured_output.py | 108 ++ .../paperclip_tools/tests/__init__.py | 0 .../paperclip_tools/tests/conftest.py | 18 + crossbar_llm/paperclip_tools/tests/pytest.ini | 6 + .../paperclip_tools/tests/test_adapter.py | 1052 +++++++++++++ .../paperclip_tools/tests/test_agent.py | 846 +++++++++++ .../paperclip_tools/tests/test_live.py | 275 ++++ crossbar_llm/paperclip_tools/tools.py | 222 +++ crossbar_llm/paperclip_tools/usage.py | 90 ++ crossbar_llm/pubtator3_tools/__init__.py | 12 + crossbar_llm/pubtator3_tools/agent.py | 390 +++++ crossbar_llm/pubtator3_tools/client.py | 427 ++++++ crossbar_llm/pubtator3_tools/llm.py | 41 + crossbar_llm/pubtator3_tools/nodes.py | 352 +++++ crossbar_llm/pubtator3_tools/prompts.py | 486 ++++++ crossbar_llm/pubtator3_tools/schemas.py | 286 ++++ .../pubtator3_tools/structured_output.py | 108 ++ .../pubtator3_tools/tests/__init__.py | 0 .../pubtator3_tools/tests/conftest.py | 36 + .../pubtator3_autocomplete_example.json | 47 + .../fixtures/pubtator3_export_example.json | 1 + .../fixtures/pubtator3_relations_example.json | 1 + .../fixtures/pubtator3_search_example.json | 424 ++++++ crossbar_llm/pubtator3_tools/tests/pytest.ini | 6 + .../pubtator3_tools/tests/test_graph.py | 827 +++++++++++ .../pubtator3_tools/tests/test_models.py | 234 +++ .../tests/test_new_features.py | 412 ++++++ .../pubtator3_tools/tests/test_tools_split.py | 189 +++ crossbar_llm/pubtator3_tools/tools.py | 293 ++++ crossbar_llm/pubtator3_tools/usage.py | 90 ++ 37 files changed, 10067 insertions(+) create mode 100644 crossbar_llm/paperclip_tools/__init__.py create mode 100644 crossbar_llm/paperclip_tools/adapter.py create mode 100644 crossbar_llm/paperclip_tools/agent.py create mode 100644 crossbar_llm/paperclip_tools/llm.py create mode 100644 crossbar_llm/paperclip_tools/nodes.py create mode 100644 crossbar_llm/paperclip_tools/prompts.py create mode 100644 crossbar_llm/paperclip_tools/schemas.py create mode 100644 crossbar_llm/paperclip_tools/structured_output.py create mode 100644 crossbar_llm/paperclip_tools/tests/__init__.py create mode 100644 crossbar_llm/paperclip_tools/tests/conftest.py create mode 100644 crossbar_llm/paperclip_tools/tests/pytest.ini create mode 100644 crossbar_llm/paperclip_tools/tests/test_adapter.py create mode 100644 crossbar_llm/paperclip_tools/tests/test_agent.py create mode 100644 crossbar_llm/paperclip_tools/tests/test_live.py create mode 100644 crossbar_llm/paperclip_tools/tools.py create mode 100644 crossbar_llm/paperclip_tools/usage.py create mode 100644 crossbar_llm/pubtator3_tools/__init__.py create mode 100644 crossbar_llm/pubtator3_tools/agent.py create mode 100644 crossbar_llm/pubtator3_tools/client.py create mode 100644 crossbar_llm/pubtator3_tools/llm.py create mode 100644 crossbar_llm/pubtator3_tools/nodes.py create mode 100644 crossbar_llm/pubtator3_tools/prompts.py create mode 100644 crossbar_llm/pubtator3_tools/schemas.py create mode 100644 crossbar_llm/pubtator3_tools/structured_output.py create mode 100644 crossbar_llm/pubtator3_tools/tests/__init__.py create mode 100644 crossbar_llm/pubtator3_tools/tests/conftest.py create mode 100644 crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_autocomplete_example.json create mode 100644 crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_export_example.json create mode 100644 crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_relations_example.json create mode 100644 crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_search_example.json create mode 100644 crossbar_llm/pubtator3_tools/tests/pytest.ini create mode 100644 crossbar_llm/pubtator3_tools/tests/test_graph.py create mode 100644 crossbar_llm/pubtator3_tools/tests/test_models.py create mode 100644 crossbar_llm/pubtator3_tools/tests/test_new_features.py create mode 100644 crossbar_llm/pubtator3_tools/tests/test_tools_split.py create mode 100644 crossbar_llm/pubtator3_tools/tools.py create mode 100644 crossbar_llm/pubtator3_tools/usage.py diff --git a/crossbar_llm/paperclip_tools/__init__.py b/crossbar_llm/paperclip_tools/__init__.py new file mode 100644 index 0000000..f0d09f7 --- /dev/null +++ b/crossbar_llm/paperclip_tools/__init__.py @@ -0,0 +1,12 @@ +"""Paperclip literature-evidence agent. + +Answers a biomedical question from Paperclip's full-text corpora with a cited +answer. Self-contained: nothing here imports the PubTator3 agent, so the two +evolve independently. + + from crossbar_llm.paperclip_tools.agent import build_graph + from crossbar_llm.paperclip_tools.llm import build_chat_model + + graph = build_graph(chat_model=build_chat_model(model="gpt-4o-mini")) + state = await graph.ainvoke({"question": "...", "warnings": []}) +""" diff --git a/crossbar_llm/paperclip_tools/adapter.py b/crossbar_llm/paperclip_tools/adapter.py new file mode 100644 index 0000000..9fec3e2 --- /dev/null +++ b/crossbar_llm/paperclip_tools/adapter.py @@ -0,0 +1,1302 @@ +"""Thin typed async adapter over Paperclip, with two transports. + +Paperclip (https://paperclip.gxl.ai) exposes the same command set two ways: a +first-class MCP server, and a REST endpoint (`/api/cli/execute`) that turns +out to also accept our API key even though it's undocumented for that auth +mode. **REST is primary** (it supports a genuine +unscoped/all-sources `search` and returns structured JSON, neither of which +MCP's `search` allows); **MCP is the fallback** (documented, guaranteed to +keep working, used automatically whenever REST fails or is disabled via +`PAPERCLIP_DISABLE_REST=1`). + +This module wraps both transports behind typed helpers (`search`, `get_meta`, +`get_content`) so the rest of the agent never sees raw command strings, REST +JSON, or the MCP protocol. `PaperclipAdapterProtocol` is the single injectable +seam: tests pass a fake with the same signatures into `build_graph(adapter=...)`; +production uses `PaperclipAdapter`. (Renamed from `mcp_adapter.py`/ +`PaperclipMCPAdapter` once REST became primary; not `PaperclipClient` since +that collides with the real `gxl_paperclip` SDK's own class name.) + +Design notes: +- **No agent logic here.** Typed models in, typed models out. Retrieval choices + (source, depth, fallback) live in the node layer. +- **Per-event-loop singletons** for the MCP client + loaded tool, keyed in a + `WeakKeyDictionary`, so `asyncio.run` in tests/scripts doesn't hit "event loop + is closed" (mirrors `pubtator3/client.py`). +- Command errors surface as `PaperclipError`; the tool layer converts those to + never-raise error envelopes. REST failures surface as + `PaperclipRestUnavailable` internally, caught by `_execute`/`search` to + trigger the MCP fallback — they never escape this module as that type. + +No client-side rate limiter is applied: Paperclip publishes no request-rate +policy, so throttling would be guesswork. Add one here if the server later +documents limits. + +CLI facts pinned against the live server: +- `search -s "" -n ` — over MCP, the `-s` source flag is + REQUIRED (confirmed live, contradicts Paperclip's own docs/help text). Over + REST, `-s` is optional and omitting it searches broadly across sources — + `source=None` uses this; the MCP fallback substitutes a paper-corpora list + since MCP has no unscoped mode. +- `cat /papers//meta.json` — clean JSON: pmid, doi, pmc_id, title, + authors, abstract, journal, pub_year, ... (the citable-ID source). +- `cat`/`head /papers//content.lines` — line-numbered `L: ...` body. +""" +from __future__ import annotations + +import asyncio +import json +import logging +import os +import re +import weakref +from typing import Literal, Protocol, runtime_checkable + +import httpx +from pydantic import BaseModel, Field, field_validator + +# --- Module constants --------------------------------------------------------- +MCP_URL = "https://paperclip.gxl.ai/mcp" +# Undocumented for API-key auth, but confirmed working and used as the primary +# transport anyway. `PAPERCLIP_DISABLE_REST=1` forces MCP-only — a safety valve +# if this endpoint ever changes or locks down. +REST_URL = "https://paperclip.gxl.ai/api/cli/execute" +DISABLE_REST_ENV = "PAPERCLIP_DISABLE_REST" +API_KEY_ENV = "PAPERCLIP_API_KEY" +DEFAULT_TIMEOUT_S = 60.0 # search/cat/head/ls — typically fast +# `map`/`ask-image` read full text server-side and can take minutes. 480s, not +# 300s: 300 is langchain_mcp_adapters' default `sse_read_timeout`, and sitting on +# that edge lets a merely-slow job look like a dead stream and trigger a +# reconnect — a path confirmed to fail auth server-side. +SLOW_TIMEOUT_S = 480.0 +_SLOW_COMMANDS = frozenset({"map", "ask-image"}) +DEFAULT_SOURCE = "pmc" +DEFAULT_CONTENT_MAX_LINES = 400 # cap full-text body pulled per paper + +_log = logging.getLogger(__name__) + +# Paperclip source scopes accepted by `search -s`. Kept as a Literal so the tool +# and router schemas can reference one canonical list. +PaperclipSource = Literal[ + "pmc", + "biorxiv", + "medrxiv", + "arxiv", + "fda", + "fda/us", + "fda/jp", + "fda/eu", + "trials", + "trials/us", + "trials/eu", + "trials/jp", + "trials/cn", + "proteins", + "pdb", + "chembl", +] + +# Document-id prefixes Paperclip uses across corpora, for parsing search output +# and for choosing the citation-URL namespace. arXiv ids are `YYMM.NNNNN` +# (contain a literal dot) -- NOT a hex-digit class, unlike bio_/med_'s hashes. +_DOC_ID_RE = re.compile(r"(PMC\d+|bio_[0-9a-fA-F]+|med_[0-9a-fA-F]+|arx_[\w.]+|fda_\w+|tri_\w+|NCT\w+)") +# Trailing timing/footer lines the CLI appends, e.g. "[75ms]" or +# "[357ms, saved to s_c7326471]". Stripped before JSON parsing. +_TIMING_FOOTER_RE = re.compile(r"^\s*\[\d+ms.*\]\s*$", re.MULTILINE) + +# Canonical body-section names the router/nodes can request, mapped to the +# lowercase keywords we match against Paperclip's per-paper section filenames +# (which vary: "2. Materials and Methods", "3. Results", "5. Conclusions", ...). +# Title + abstract are NOT here — abstracts come from meta.json and are the +# default depth; these are the *body* sections fetched only for full-text depth. +PaperclipSectionName = Literal[ + "introduction", + "methods", + "results", + "discussion", + "conclusion", +] + +SECTION_KEYWORDS: dict[str, tuple[str, ...]] = { + "introduction": ("introduction", "background"), + "methods": ("method", "materials and methods", "methodology", "experimental"), + "results": ("result", "findings"), + "discussion": ("discussion",), + "conclusion": ("conclusion",), +} + +# One directory listing entry ends in ".lines"; names may contain single spaces +# ("2. Materials and Methods"), while entries are separated by 2+ spaces. +_SECTION_FILE_RE = re.compile(r"(?P.+?)\.lines(?=\s{2,}|\s*$)") + +# Maps a search `-s ` scope to the read-only VFS root the corpus lives +# under. The paper corpora all live under /papers/; regulatory, trials, and +# proteins have their own roots. Used to address `meta.json` / `content.lines` +# for a hit (proteins are NOT under /papers/ — the earlier bug). +_SOURCE_ROOT: dict[str, str] = { + "pmc": "papers", + "papers": "papers", # the _BROAD_MCP_SOURCES alias, made explicit rather + # than relying on the dict's own fallback default to resolve it the same way + "biorxiv": "papers", + "medrxiv": "papers", + "arxiv": "papers", + "fda": "fda", + "fda/us": "fda", + "fda/jp": "fda", + "fda/eu": "fda", + "trials": "trials", + "trials/us": "trials", + "trials/eu": "trials", + "trials/jp": "trials", + "trials/cn": "trials", + "proteins": "proteins", + "uniprot": "proteins", + "pdb": "proteins", + "chembl": "proteins", +} + +# MCP has no unscoped search mode, so a broad/unscoped request falls back to +# this scope. `-s papers` is Paperclip's own documented alias for +# pmc+biorxiv+medrxiv+arxiv (confirmed live on both transports) — used +# instead of hand-joining those four so we track their definition if it +# changes. Deliberately excludes fda/trials/proteins: mixing those into one +# multi-source call was confirmed live to be slow and to silently drop them +# from the results. +_BROAD_MCP_SOURCES = "papers" + +# The `s_xxxx` result-set id in a search footer, and the `m_xxxx` results id a +# map run reports. Both are needed to chain search -> map -> cat full results. +_SEARCH_ID_RE = re.compile(r"\[(s_[0-9a-fA-F]+)\]") +_MAP_ID_RE = re.compile(r"\b(m_[0-9a-fA-F]+)\b") +# Per-paper block header in the full map results file: +# --- [1/3] [success] --- +_MAP_BLOCK_RE = re.compile(r"^---\s*\[\d+/\d+\]\s*\[(?P<status>\w+)\].*?---\s*$", re.MULTILINE) + +# Paperclip intermittently fails to resolve a search id it issued moments +# earlier — `ls /.gxl/` lists the result file while `map --from <id>` reports +# it missing, and consecutive identical calls disagree. Measured at ~2/10 +# calls; a plain retry recovered about half. Server-side state we can't +# address from here (there is no session cookie to pin to an instance), so +# retry is the mitigation, not a fix. +# +# It arrives as HTTP 200 with `exit_code: 0` and the error only in the body +# text, so it MUST be detected by string match — no status code reveals it. +_MAP_NOT_FOUND = "Results not found" +_MAP_NOT_FOUND_RETRIES = 2 +_MAP_RETRY_BACKOFF_S = 0.5 + +# The per-paper answer contract `run_map` asks for. Deliberately minimal — +# `found` lets us gate on "this paper doesn't address the question" explicitly +# instead of the old heuristic of treating any non-empty free-text answer as +# usable evidence. Kept as a dict because it documents the shape +# `_map_answer_and_found` parses; the wire format is the prompt text below. +MAP_OUTPUT_SCHEMA: dict = { + "type": "object", + "properties": { + "answer": { + "type": "string", + "description": "Direct answer to the question from this paper, or empty if not addressed.", + }, + "found": { + "type": "boolean", + "description": "Whether this paper actually addresses the question.", + }, + }, + "required": ["answer", "found"], +} + +# We ask for that contract in the QUESTION TEXT rather than via Paperclip's +# `--output_schema` flag, because that flag makes their REST endpoint return +# HTTP 500 for any schema value whatsoever — confirmed live down to an 18-byte +# `{"type": "object"}`, while the identical call succeeds over MCP. REST is our +# primary transport, and forcing every map onto MCP is what exposed us to their +# stream-resume auth failure, so the flag costs far more than it buys. The map +# model honors this instruction reliably; `_map_extract_from_text` parses the +# result identically either way. +# +# The one thing lost is Paperclip's automatic `_citations` field (server-computed +# line provenance), which only ships alongside `--output_schema`. The model still +# writes line refs inline in its answer, so `_map_citation_lines_from_text` +# recovers them — LLM-emitted rather than server-verified, but present. +_MAP_JSON_CONTRACT = ( + "Respond with ONLY a JSON object, no prose before or after, in exactly this form: " + '{"answer": "<your answer, citing supporting line numbers inline like (L12, L20-L25)>", ' + '"found": true or false}. ' + 'Set "found" to false if this paper does not address the question.' +) + + +class PaperclipError(RuntimeError): + """A Paperclip command failed (transport, auth, or CLI-level error).""" + + +class PaperclipConfigError(PaperclipError): + """The adapter is misconfigured (e.g. missing API key).""" + + +class PaperclipRestUnavailable(PaperclipError): + """The REST endpoint failed (network, timeout, non-200, bad JSON) — signals + the caller to fall back to the MCP path. Never raised past `_execute`.""" + + +class PaperHit(BaseModel): + """One ranked search result. `doc_id` is the stable handle for follow-up + `get_meta` / `get_content` calls and for building citation URLs. + + `score`/`corpus`/`backend`/`doi`/`pub_year` are only populated when the + hit came from the REST path's structured `result_data.papers` — they stay + `None` on the MCP text-parsing fallback path (nothing was lost; those + fields were never available there either).""" + doc_id: str + title: str = "" + authors: str = "" + source: str = "" + date: str = "" + url: str = "" + snippet: str = "" + score: float | None = None + corpus: str | None = None + backend: str | None = None + doi: str | None = None + pub_year: int | None = None + + +class PaperMeta(BaseModel): + """Structured metadata from a corpus record's `meta.json` — the reliable + provenance record the synthesis contract cites. + + Covers both the paper corpora (`doi`/`pmid`/`abstract`/`journal`) and the + `proteins` corpus (`accession`/`uniprot_id`/`protein_name`/`gene_name`/ + `organism`), which has no DOI/abstract. `extra="ignore"` tolerates the + schema differences between corpora.""" + document_id: str = Field(alias="document_id") + pmc_id: str | None = None + pmid: str | None = None + doi: str | None = None + title: str = "" + authors: str = "" + abstract: str = "" + journal: str | None = None + pub_year: int | None = None + pub_date: str | None = None + article_type: str | None = None + source: str | None = None + # proteins corpus fields + accession: str | None = None + uniprot_id: str | None = None + protein_name: str | None = None + gene_name: str | None = None + organism: str | None = None + + model_config = {"populate_by_name": True, "extra": "ignore"} + + # meta.json carries an explicit `null` for these on records that genuinely + # lack them (conference posters, meeting abstracts, editorials). A field + # default only applies when the key is ABSENT, so without this a valid + # `"abstract": null` response fails validation and the whole record is + # discarded — which cost us the journal/article_type of ~30% of hits. + @field_validator("title", "authors", "abstract", mode="before") + @classmethod + def _null_to_empty(cls, v): + return "" if v is None else v + + +class SearchResult(BaseModel): + """A search response: ranked hits plus the server-side result-set id + (`s_xxxx`) that `map` operates on. The id survives across MCP calls (it is + keyed by the API key/workspace, not the connection), so a later `run_map` + can reference it.""" + hits: list[PaperHit] = [] + search_id: str | None = None + + +class SqlResult(BaseModel): + """Rows from a Paperclip `sql` query, parsed from its ASCII table output. + + Unlike `search`/`map`, `sql` gives no structured JSON on either transport + (confirmed live: REST's `result_data` is `null` for this command) — every + row's values come back as strings, exactly as Paperclip renders them in + its table (no numeric/type coercion). `documents` is not one unified + table; see `PaperclipAdapter.sql`'s docstring for shard-scoping caveats + confirmed live.""" + columns: list[str] = [] + rows: list[dict] = [] + + +class MapExtract(BaseModel): + """One paper's answer from a `map` run — a full-text-derived extraction of + the question against a single paper (produced server-side, so it does not + cost us full-body tokens). + + `data` carries the full parsed `{answer, found, _citations}` object when + the server honored `MAP_OUTPUT_SCHEMA`, `None` if it fell back to raw + prose (never assume it's populated). `found`/`citation_lines` are + normalized projections of `data` — prefer these over reading `data` + directly, since the server has been observed live to return the + structured answer in two shapes for the same schema (flat, or a + malformed nested echo of the schema's own `properties`); + `_map_extract_from_text` normalizes both.""" + doc_id: str + text: str = "" + success: bool = True + data: dict | None = None + found: bool | None = None + citation_lines: list[int] = [] + + +@runtime_checkable +class PaperclipAdapterProtocol(Protocol): + """The single injectable seam. Production is `PaperclipAdapter`; tests + pass a fake implementing these coroutines.""" + + async def search( + self, + query: str, + *, + source: str | None = None, + limit: int = 10, + sort: str | None = None, + year: str | None = None, + ranking: str | None = None, + ) -> SearchResult: ... + + async def get_meta(self, doc_id: str, *, source: str = DEFAULT_SOURCE) -> PaperMeta: ... + + async def get_content( + self, + doc_id: str, + *, + source: str = DEFAULT_SOURCE, + sections: list[str] | None = None, + max_lines: int | None = None, + ) -> str: ... + + async def run_map( + self, search_id: str, question: str, *, limit: int | None = None + ) -> list[MapExtract]: ... + + async def sql(self, query: str, *, source: str | None = None) -> SqlResult: ... + + async def filter(self, search_id: str, query: str) -> SearchResult | None: ... + + +# --- Per-loop singletons ------------------------------------------------------ +# The MCP client + loaded tool are bound to the event loop that created them; +# key them by the running loop so repeated `asyncio.run` calls (tests, scripts) +# each get their own rather than reusing a closed-loop instance. +_tools_by_loop: "weakref.WeakKeyDictionary[asyncio.AbstractEventLoop, object]" = weakref.WeakKeyDictionary() + + +def _api_key() -> str: + key = os.environ.get(API_KEY_ENV) + if not key: + raise PaperclipConfigError( + f"{API_KEY_ENV} is not set; Paperclip requires an API key " + f"(create one at https://paperclip.gxl.ai). " + ) + return key + + +async def _paperclip_tool(*, timeout_s: float = DEFAULT_TIMEOUT_S): + """Lazily load and cache the single `paperclip` MCP tool for this loop.""" + loop = asyncio.get_running_loop() + tool = _tools_by_loop.get(loop) + if tool is not None: + return tool + + # Imported lazily so importing this module doesn't require the MCP client + # (keeps `from ... import PaperHit` cheap for schema-only consumers). + from langchain_mcp_adapters.client import MultiServerMCPClient + + client = MultiServerMCPClient( + { + "paperclip": { + "transport": "streamable_http", + "url": MCP_URL, + "headers": {"X-API-Key": _api_key()}, + "timeout": timeout_s, + # Explicit, not left to the library default — see SLOW_TIMEOUT_S. + "sse_read_timeout": timeout_s, + } + } + ) + tools = await client.get_tools() + if not tools: + raise PaperclipError("Paperclip MCP server exposed no tools.") + # The server exposes exactly one tool ("paperclip"); pick it by name if + # present, else the first (defensive against future renames). + tool = next((t for t in tools if t.name == "paperclip"), tools[0]) + _tools_by_loop[loop] = tool + return tool + + +# REST returns the CLI's terminal output verbatim, ANSI colour codes included; +# MCP does not. Left in, they break parsers in non-obvious ways: the dim code +# `\x1b[2m` ends in the letter `m`, so `Results ID: \x1b[2mm_7049c74c` has no +# word boundary before the id and `_MAP_ID_RE`'s `\b` never matches. Strip at +# both transport boundaries so every parser downstream sees the same clean text. +_ANSI_RE = re.compile(r"\x1b\[[0-9;]*[a-zA-Z]") + + +def _strip_ansi(text: str) -> str: + return _ANSI_RE.sub("", text) + + +def _strip_timing(text: str) -> str: + return _TIMING_FOOTER_RE.sub("", text).strip() + + +def _parse_section_listing(text: str) -> list[str]: + """Parse `ls /papers/<id>/sections/` output into section basenames. + + The listing is space-separated `<Name>.lines` entries (names may contain + single spaces; entries are separated by 2+ spaces), followed by a + "(read-only ...)" note and a `[..ms]` footer. Returns names without the + `.lines` suffix, e.g. ["Abstract", "1. Introduction", "3. Results", ...]. + """ + body = _strip_timing(text) + # Drop the parenthetical read-only note line(s). + body = "\n".join( + ln for ln in body.splitlines() if not ln.strip().startswith("(") + ) + return [m.group("name").strip() for m in _SECTION_FILE_RE.finditer(body)] + + +def _select_section_files(available: list[str], wanted: list[str]) -> list[str]: + """Pick the section files that satisfy the requested canonical sections. + + Matching is keyword-based (see `SECTION_KEYWORDS`) against the paper's own + section names. Because Paperclip splits a section into fine-grained + subsection files ("2. Materials and Methods" + "2.1. Study Design" + ...), + when a numbered top-level section matches a keyword we also pull all of its + subsections (files sharing the same leading integer). Non-numbered sections + (e.g. a bare "Conclusions") are matched by keyword directly. + """ + keywords: list[str] = [] + for w in wanted: + keywords.extend(SECTION_KEYWORDS.get(w.lower(), (w.lower(),))) + + matched_numbers: set[str] = set() + for name in available: + m = re.match(r"^(\d+)[.\s]", name) + if m and any(kw in name.lower() for kw in keywords): + matched_numbers.add(m.group(1)) + + selected: list[str] = [] + for name in available: + num = re.match(r"^(\d+)", name) + if num and num.group(1) in matched_numbers: + selected.append(name) + elif not num and any(kw in name.lower() for kw in keywords): + selected.append(name) + return selected + + +def _extract_json_object(text: str) -> dict: + """Pull the first top-level JSON object out of a CLI response (which may + carry a trailing `[75ms]` timing footer).""" + stripped = _strip_timing(text) + decoder = json.JSONDecoder() + for idx, ch in enumerate(stripped): + if ch != "{": + continue + try: + obj, _ = decoder.raw_decode(stripped[idx:]) + except json.JSONDecodeError: + continue + if isinstance(obj, dict): + return obj + raise PaperclipError("Paperclip response did not contain a JSON object.") + + +def _parse_search(text: str) -> list[PaperHit]: + """Parse the human-readable `search` listing into PaperHit rows. + + The format per hit is: + + 1. <title> + <authors> + <doc_id> · <source> · <date> + <url> + "<snippet>" + + Only `doc_id` is load-bearing (it drives `get_meta`/`get_content` and the + citation URL); title/snippet are captured as a cheap preview. Reliable + fields (doi, pmid, authors, journal) come from `get_meta`, not from here. + """ + hits: list[PaperHit] = [] + # Split into per-result blocks starting at "<n>. ". + blocks = re.split(r"\n\s*\d+\.\s", "\n" + _strip_timing(text)) + for block in blocks[1:]: + lines = [ln.rstrip() for ln in block.splitlines()] + if not lines: + continue + title = lines[0].strip() + m = _DOC_ID_RE.search(block) + if not m: + continue # header line ("Found N papers") or a malformed block + doc_id = m.group(1) + + authors = source = date = url = snippet = "" + for ln in lines[1:]: + s = ln.strip() + if not s: + continue + if " · " in s and _DOC_ID_RE.search(s): + parts = [p.strip() for p in s.split(" · ")] + # parts: [<doc_id>, <source>, <date>] + if len(parts) >= 2: + source = parts[1] + if len(parts) >= 3: + date = parts[2] + elif s.startswith("http"): + url = s + elif s.startswith('"') and s.endswith('"') and len(s) > 1: + snippet = s[1:-1] + elif not authors and not source and not url: + # First non-empty line after the title, before the id line. + authors = s + hits.append( + PaperHit( + doc_id=doc_id, title=title, authors=authors, + source=source, date=date, url=url, snippet=snippet, + ) + ) + return hits + + +def _doc_root(source: str | None) -> str: + """VFS root directory for a hit from the given search source.""" + return _SOURCE_ROOT.get((source or DEFAULT_SOURCE).lower(), "papers") + + +# UniProt accession format (proteins corpus doc ids), e.g. O95251, Q09472, P04637. +_UNIPROT_ACC_RE = re.compile(r"^[OPQ][0-9][A-Z0-9]{3}[0-9]|[A-NR-Z][0-9]([A-Z][A-Z0-9]{2}[0-9]){1,2}$") + + +def infer_source_from_doc_id(doc_id: str) -> str: + """Best-effort corpus guess from a doc_id's shape alone. + + Needed once a search can be broad/unscoped (multiple corpora in one + result set, e.g. the REST default-all path, or the MCP fallback's + comma-separated paper-corpora substitute): a single blanket `source` + string is no longer necessarily correct for every hit, so per-hit + follow-ups (`get_meta`/`get_content`/citation URLs) need to resolve each + hit's own VFS root from its doc_id rather than the search-level source. + Mirrors the prefix logic already used by `citation_url` in + `paperclip_nodes.py`. + """ + if doc_id.startswith(("PMC", "bio_", "med_", "arx_")): + return "pmc" + if doc_id.startswith("fda_"): + return "fda" + if doc_id.startswith(("tri_", "NCT")): + return "trials" + if _UNIPROT_ACC_RE.match(doc_id): + return "proteins" + return "pmc" + + +def _parse_protein_search(text: str) -> list[PaperHit]: + """Parse the `-s proteins` listing, whose per-hit format differs from papers: + + 1. KAT7 - Histone acetyltransferase KAT7 + O95251 + Homo sapiens · 611 aa + + The accession (line 2) is the `doc_id`; the title is line 1; the organism + + length line becomes the snippet. No DOI/URL here (those come from get_meta). + """ + hits: list[PaperHit] = [] + blocks = re.split(r"\n\s*\d+\.\s", "\n" + _strip_timing(text)) + for block in blocks[1:]: + lines = [ln.strip() for ln in block.splitlines() if ln.strip()] + if not lines: + continue + title = lines[0] + acc = next((ln for ln in lines[1:] if _UNIPROT_ACC_RE.match(ln)), "") + if not acc: + continue + snippet = next((ln for ln in lines[1:] if " · " in ln), "") + hits.append( + PaperHit(doc_id=acc, title=title, source="proteins", snippet=snippet) + ) + return hits + + +def _map_answer_and_found(parsed: dict) -> tuple[str, bool | None]: + """Normalize the two structured-answer shapes confirmed live for + `MAP_OUTPUT_SCHEMA`: the flat `{"answer": str, "found": bool}` we + requested, and a malformed nested echo of the schema's own `properties`, + `{"answer": {"answer": str, "found": bool}}`. Returns `("", None)` if + `answer` is neither a string nor this specific nested shape — never + raises, so one paper's malformed response can't take down the batch.""" + answer = parsed.get("answer") + found = parsed.get("found") + if isinstance(answer, dict): + found = answer.get("found", found) + answer = answer.get("answer", "") + if not isinstance(answer, str): + answer = "" + return answer, found if isinstance(found, bool) else None + + +def _map_citation_lines(parsed: dict) -> list[int]: + """Extract Paperclip's own line-level provenance for the `answer` field — + `_citations`, an automatic bonus field added regardless of the requested + schema (confirmed live: `[{"field": "answer", "line": <int>, "content": + "<supporting text>"}, ...]`, a top-level key in both known `answer` + shapes). Sorted + deduped; empty if absent/malformed. This is the + deterministic, server-provided data line-level citation URLs are built + from — not something the synthesizing LLM has to identify or copy.""" + raw = parsed.get("_citations") + if not isinstance(raw, list): + return [] + lines = { + c["line"] for c in raw + if isinstance(c, dict) and isinstance(c.get("line"), int) + } + return sorted(lines) + + +# Line refs the map model writes inline in its answer: "(L8, L14-L16, L72)". +# Matches a single `L12` or a range `L20-L25` / `L20-25`. +_INLINE_LINE_REF_RE = re.compile(r"\bL(\d+)(?:\s*-\s*L?(\d+))?") + +# Only refs inside a parenthesised group that contains NOTHING BUT refs are +# accepted. Biomedical prose is full of `L<n>`-shaped tokens that are not line +# numbers at all — L1CAM, the L1/L2 vertebrae, the L5 nerve root, L3-stage +# larvae — and a bare scan turns each into a citation anchor pointing at a line +# that does not support the claim. `_MAP_JSON_CONTRACT` asks for the +# parenthesised form, and every ref observed live uses it, so requiring it +# costs nothing real and removes the whole class of false positive. +_PAREN_GROUP_RE = re.compile(r"\(([^)]{1,200})\)") +_REF_SEPARATORS_ONLY_RE = re.compile(r"^[\s,;.&+]*(?:and[\s,;.&+]*)*$", re.IGNORECASE) + +# A range wider than this is almost certainly the model gesturing at a whole +# section rather than citing specific support; keep its endpoints instead of +# expanding it into a citation anchor with hundreds of line numbers. +_MAX_LINE_RANGE_SPAN = 25 + + +def _map_citation_lines_from_text(answer: str) -> list[int]: + """Recover supporting line numbers from line refs the model wrote inline + in its answer — the fallback for `_citations`, which only arrives with + `--output_schema` (see `_MAP_JSON_CONTRACT` for why we can't use that). + + Ranges are expanded so `format_line_anchor` can re-collapse them into a + `#L20-L25` anchor; over-wide ranges keep only their endpoints. Only + citation-shaped parenthesised groups are read — see `_PAREN_GROUP_RE`.""" + lines: set[int] = set() + for group in _PAREN_GROUP_RE.findall(answer): + matches = list(_INLINE_LINE_REF_RE.finditer(group)) + if not matches: + continue + # Anything left after removing the refs must be pure separators, + # otherwise this is prose that merely contains an `L<n>`-shaped token. + if not _REF_SEPARATORS_ONLY_RE.match(_INLINE_LINE_REF_RE.sub("", group)): + continue + for m in matches: + start = int(m.group(1)) + end = int(m.group(2)) if m.group(2) else start + if end < start: + start, end = end, start + if end - start > _MAX_LINE_RANGE_SPAN: + lines.update((start, end)) + else: + lines.update(range(start, end + 1)) + return sorted(lines) + + +def _map_extract_from_text(doc_id: str, raw_text: str, success: bool) -> MapExtract: + """Build a `MapExtract` from one paper's raw map answer. + + `raw_text` is a JSON string matching `MAP_OUTPUT_SCHEMA` when the server + honored the requested schema, or plain prose otherwise (schema not + applied for that paper, or an older/non-schema call). Best-effort: try + JSON first, fall back to treating it as prose — never raises, since a + single paper's shape shouldn't take down the whole map result set. + """ + stripped = raw_text.strip() + if stripped.startswith("{"): + try: + parsed = json.loads(stripped) + except json.JSONDecodeError: + parsed = None + if isinstance(parsed, dict) and "answer" in parsed: + answer, found = _map_answer_and_found(parsed) + # `_citations` when the server supplied it, else the model's own + # inline refs — see `_map_citation_lines_from_text`. + citation_lines = _map_citation_lines(parsed) or _map_citation_lines_from_text(answer) + return MapExtract( + doc_id=doc_id, text=answer, success=success, data=parsed, + found=found, citation_lines=citation_lines, + ) + return MapExtract( + doc_id=doc_id, text=raw_text, success=success, + citation_lines=_map_citation_lines_from_text(raw_text), + ) + + +def _parse_map_results(text: str) -> list[MapExtract]: + """Parse the full `cat /.gxl/map_<id>.txt` results into per-paper extracts. + + Blocks look like: + --- [1/3] [success] <title> --- + doc_id: PMC12511219 + <multi-line answer, JSON or prose — see _map_extract_from_text> + """ + body = _strip_timing(text) + extracts: list[MapExtract] = [] + matches = list(_MAP_BLOCK_RE.finditer(body)) + for i, m in enumerate(matches): + start = m.end() + end = matches[i + 1].start() if i + 1 < len(matches) else len(body) + block = body[start:end].strip() + did_m = re.search(r"doc_id:\s*(\S+)", block) + if not did_m: + continue + doc_id = did_m.group(1) + # Answer text is everything after the doc_id line. + answer = block[did_m.end():].strip() + extracts.append( + _map_extract_from_text(doc_id, answer, m.group("status").lower() == "success") + ) + return extracts + + +# Trailing "(N rows, Xms) [shard breakdown]" or "(1 row, Xms)" line `sql` +# appends after its ASCII table. Distinct from `_TIMING_FOOTER_RE` (which +# only matches `[NNms]`-only lines) since this one starts with `(`, not `[`. +_SQL_FOOTER_RE = re.compile(r"^\(\d+ rows?,\s*[\d.]+ms\).*$", re.MULTILINE) + + +def _parse_sql_output(text: str) -> SqlResult: + """Parse `sql`'s ASCII table output — confirmed identical on REST and MCP + (no structured JSON available for this command on either transport, + unlike search/map). Shape: + + title | doi | source + ---------------------+---------------+------- + Some Paper Title... | 10.1/xyz | pmc + (1 row, 14ms) + + Raises `PaperclipError` on a server-reported query error (e.g. a + non-SELECT statement, unknown column, or the 15s statement timeout) — + these come back as `ERR: sql: <message>` in `output` with HTTP 200 / + MCP success, not as a transport-level failure, so we must check for the + prefix ourselves rather than relying on `_run`/`_run_rest` to raise. + """ + stripped = _strip_timing(text).strip() + if stripped.startswith("ERR:"): + raise PaperclipError(stripped.splitlines()[0][len("ERR:"):].strip()) + body = _SQL_FOOTER_RE.sub("", stripped).rstrip() + lines = [ln for ln in body.splitlines() if ln.strip()] + if len(lines) < 2: + return SqlResult(columns=[], rows=[]) + columns = [c.strip() for c in lines[0].split("|")] + rows: list[dict] = [] + for ln in lines[2:]: # lines[1] is the "---+---+---" divider + cells = [c.strip() for c in ln.split("|")] + if len(cells) != len(columns): + continue # malformed row (shouldn't happen); skip rather than misalign + rows.append(dict(zip(columns, cells))) + return SqlResult(columns=columns, rows=rows) + + +def _shell_quote(s: str) -> str: + """Quote an argument for Paperclip's server-side `vsh` parser. + + Single-quote style, not the double quotes used in Paperclip's own CLI + examples. Their double-quote handling breaks when an argument contains + BOTH an embedded double quote and a newline — confirmed live as + `ERR: vsh: parse error: No closing quotation`, with either feature alone + parsing fine. That combination is exactly the shape of + `_MAP_JSON_CONTRACT`, and an LLM-written query could reproduce it + anywhere else too. Single quotes with `'\\''` escaping parsed correctly + for every combination tested: embedded double quotes, newlines, + apostrophes ("Alzheimer's"), and all of them together. + """ + return "'" + s.replace("'", "'\\''") + "'" + + +def _paper_hit_from_rest(p: dict) -> PaperHit: + """Map one entry of the REST path's structured `result_data.papers` into + a `PaperHit` — no regex needed (unlike the MCP text-parsing fallback). + + Field shape confirmed live for pmc/biorxiv hits (see + fda/trials/proteins shape + is unverified (open question in the migration plan) — fall back + defensively rather than assume every field is present. + """ + doc_id = str(p.get("document_id") or p.get("id") or p.get("accession") or "") + pub_year = p.get("pub_year") + date = p.get("pub_date") or (str(pub_year) if pub_year else "") + return PaperHit( + doc_id=doc_id, + title=p.get("title") or "", + authors=p.get("authors") or "", + source=p.get("source") or p.get("corpus") or "", + date=date, + url=p.get("url") or "", + snippet=p.get("tldr") or p.get("abstract_snippet") or "", + score=p.get("score"), + corpus=p.get("corpus"), + backend=p.get("backend"), + doi=p.get("doi"), + pub_year=pub_year, + ) + + +def _hits_from_rest_payload( + data: dict, output_text: str, source: str | None +) -> list[PaperHit]: + """Extract hits from a REST search/filter response. + + Prefers the structured `result_data.papers`, but falls back to parsing the + text listing when that key is absent. The server intermittently returns a + complete listing in `output` — "Found 5 papers", every record present, a + valid result id — while omitting `result_data` entirely. Reading only the + structured field reported those searches as zero-hit (measured at ~14% of + calls), which then tripped the node's zero-result fallback as though + nothing had matched. + + The text parsers are the same ones the MCP path uses, which never has + `result_data` at all — so this is a fallback we already trust. + """ + result_data = data.get("result_data") + papers = None + if isinstance(result_data, dict): + papers = result_data.get("papers") + if papers is None: + papers = result_data.get("results") + if papers: + return [_paper_hit_from_rest(p) for p in papers if isinstance(p, dict)] + parse = _parse_protein_search if _doc_root(source) == "proteins" else _parse_search + return parse(output_text) + + +# Set once per process the first time a REST call fails and we fall back to +# MCP — logged once, not per-call, so a down/locked-down REST endpoint is +# noticeable without spamming logs for the rest of the session. +_rest_fallback_warned = False + + +def _warn_rest_fallback(err: Exception) -> None: + global _rest_fallback_warned + if not _rest_fallback_warned: + _rest_fallback_warned = True + _log.warning( + "Paperclip REST endpoint (%s) unavailable, falling back to MCP: %s", + REST_URL, err, + ) + + +class PaperclipAdapter: + """Production adapter: issues CLI command strings to the single `paperclip` + MCP tool and parses the responses into typed models.""" + + def __init__( + self, + *, + timeout_s: float = DEFAULT_TIMEOUT_S, + slow_timeout_s: float = SLOW_TIMEOUT_S, + ): + self._timeout_s = timeout_s + self._slow_timeout_s = slow_timeout_s + # One pooled HTTP client per event loop, per adapter. Per-loop because + # an AsyncClient binds to the loop it was created on; per-adapter (not + # module-global) because `build_graph` constructs one adapter per run, + # so pooling stays scoped to a single user's request rather than shared + # across everyone. + self._clients: "weakref.WeakKeyDictionary[asyncio.AbstractEventLoop, httpx.AsyncClient]" = ( + weakref.WeakKeyDictionary() + ) + + def _rest_client(self) -> "httpx.AsyncClient": + """The pooled client for this loop, created on first use. + + Reusing one client keeps connections alive across calls, so the + ~14-wide `asyncio.gather` fan-out in `assemble_context_node` stops + paying a fresh TCP + TLS handshake per document. + """ + loop = asyncio.get_running_loop() + client = self._clients.get(loop) + if client is None or client.is_closed: + client = httpx.AsyncClient( + # Keep-alive headroom for the fan-out; `max_connections` caps + # how hard one request can hit Paperclip concurrently. + limits=httpx.Limits(max_connections=32, max_keepalive_connections=16), + follow_redirects=True, + ) + self._clients[loop] = client + return client + + async def aclose(self) -> None: + """Release pooled connections. Optional — clients are garbage-collected + with their loop — but lets long-lived callers clean up deterministically.""" + for client in list(self._clients.values()): + if not client.is_closed: + await client.aclose() + self._clients.clear() + + async def _run(self, command: str) -> str: + """Invoke the `paperclip` tool with one command; return its text output. + + Raises `PaperclipError` on transport / auth / CLI-level failures so the + tool layer can wrap it in a never-raise envelope. This is the MCP + path — kept exactly as before, since it's now the fallback path + `_execute` uses when REST is unavailable, and must stay correct on + its own (not just until the REST migration lands). + + The underlying MCP client is a per-event-loop singleton + (`_paperclip_tool`) with ONE fixed transport-level timeout baked in + at creation — `ainvoke()` calls on it can't override it per-call. So + this always requests the client with `_slow_timeout_s`: MCP is now + the fallback path only (REST is primary), so a generous ceiling here + costs nothing on the common path, and avoids under-timing out a + `map` call that happens to fall back to MCP. + """ + tool = await _paperclip_tool(timeout_s=self._slow_timeout_s) + try: + out = await tool.ainvoke({"command": command}) + except PaperclipError: + raise + except Exception as e: # ToolException, transport errors, timeouts + raise PaperclipError(f"{type(e).__name__}: {e}") from e + return _strip_ansi(out if isinstance(out, str) else str(out)) + + async def _run_rest(self, verb: str, raw: str) -> dict: + """POST to the REST execute endpoint — see + Undocumented for API-key auth but confirmed working. Raises + `PaperclipRestUnavailable` on any failure (never `PaperclipError` + directly) so `_execute`/`search` can catch specifically that and + fall back to MCP, without swallowing genuine config errors (a + missing API key fails identically on both transports, so there's no + point falling back for that case — `_api_key()` is called outside + the try so `PaperclipConfigError` propagates immediately). + + REST timeouts are set per-call (unlike MCP's fixed client-level + one), so `map`/`ask-image` get `_slow_timeout_s` precisely — no need + for MCP's blanket-generous workaround here. + + The API key is sent per-request rather than baked into the pooled + client's headers, so rotating `PAPERCLIP_API_KEY` takes effect without + rebuilding the client. + """ + if os.environ.get(DISABLE_REST_ENV): + raise PaperclipRestUnavailable(f"disabled via {DISABLE_REST_ENV}") + key = _api_key() + + timeout = self._slow_timeout_s if verb in _SLOW_COMMANDS else self._timeout_s + try: + resp = await self._rest_client().post( + REST_URL, + json={"command": verb, "raw": raw}, + headers={"X-API-Key": key}, + timeout=timeout, + ) + except Exception as e: # network errors, timeouts + raise PaperclipRestUnavailable(f"{type(e).__name__}: {e}") from e + # 401/429 are account-level verdicts, not transport failures: MCP + # carries the same key against the same quota, so falling back to it + # can only burn time and produce a confusing retry storm. Raise the + # base error so `_execute` doesn't catch it and callers see the real + # reason (`map` is capped at 100/day, resetting midnight UTC). + if resp.status_code in (401, 429): + raise PaperclipError(f"HTTP {resp.status_code}: {resp.text[:200]}") + if resp.status_code != 200: + raise PaperclipRestUnavailable(f"HTTP {resp.status_code}: {resp.text[:200]}") + try: + data = resp.json() + except ValueError as e: + raise PaperclipRestUnavailable(f"invalid JSON response: {e}") from e + if not isinstance(data, dict): + raise PaperclipRestUnavailable("REST response was not a JSON object.") + if isinstance(data.get("output"), str): + data["output"] = _strip_ansi(data["output"]) + return data + + async def _execute(self, verb: str, raw: str) -> tuple[str, str | None]: + """Try REST first, fall back to MCP on any REST failure. Returns + `(output_text, result_id)`. + + Used for `cat`/`ls`/`head`/`map` — commands where the raw argument + string is identical on both transports. `search` does NOT use this; + it needs a different raw string per transport when `source` is + omitted (REST: no `-s` at all, for the real all-sources default; + MCP: substitute a source list, since MCP requires `-s`) — see + `search()`. + """ + try: + data = await self._run_rest(verb, raw) + return data.get("output", ""), data.get("result_id") + except PaperclipRestUnavailable as e: + _warn_rest_fallback(e) + full_command = f"{verb} {raw}".strip() if raw else verb + text = await self._run(full_command) + return text, None + + async def search( + self, + query: str, + *, + source: str | None = None, + limit: int = 10, + sort: str | None = None, + year: str | None = None, + ranking: str | None = None, + ) -> SearchResult: + """Search Paperclip. `source=None` searches broadly (the default) — + REST supports this natively (Paperclip's own documented default); + MCP requires an explicit `-s`, so the fallback path substitutes a + fast, same-shaped multi-source list (`_BROAD_MCP_SOURCES`) instead. + The two transports genuinely differ here. + + `ranking` (e.g. `"analogical"` — confirmed live to work over REST, + same `result_data.papers` shape as the default `hybrid` ranking) is + REST-only, deliberately not threaded into the MCP fallback command: + it's a quality/relevance-mode knob, not a correctness requirement, so + it degrades to the default ranking rather than failing when REST is + unavailable — same philosophy as `filter`. + """ + base_args = f"{_shell_quote(query)} -n {int(limit)}" + # `--all` means "search all papers, not just recent" — without it the + # server silently applies a recency restriction and returns far fewer + # results than `-n` asks for. Measured over 8 benchmark queries at + # `-n 14`: 5.0 hits without it, 13.6 with. It also surfaces the older + # canonical papers a factual question usually needs — the Denosumab + # question returns the 2007-2014 RANKL literature with it, and only + # 2025-2026 papers without. `--year` is a deliberate recency filter, so + # the two are mutually exclusive. + if not year: + base_args += " --all" + if sort: + base_args += f" --sort {sort}" + if year: + base_args += f" --year {year}" + + rest_raw = f"-s {source} {base_args}" if source else base_args + if ranking: + rest_raw += f" --ranking {ranking}" + try: + data = await self._run_rest("search", rest_raw) + except PaperclipRestUnavailable as e: + _warn_rest_fallback(e) + else: + output_text = data.get("output", "") + hits = _hits_from_rest_payload(data, output_text, source) + id_m = _SEARCH_ID_RE.search(output_text) + result_id = data.get("result_id") or (id_m.group(1) if id_m else None) + return SearchResult(hits=hits, search_id=result_id) + + # MCP fallback: `-s` is mandatory. Substitute the broad paper-corpora + # list when the caller wanted a default/unscoped search. + mcp_source = source or _BROAD_MCP_SOURCES + cmd = f"search -s {mcp_source} {base_args}" + raw = await self._run(cmd) + id_m = _SEARCH_ID_RE.search(raw) + parse = _parse_protein_search if _doc_root(mcp_source) == "proteins" else _parse_search + return SearchResult( + hits=parse(raw), + search_id=id_m.group(1) if id_m else None, + ) + + async def get_meta(self, doc_id: str, *, source: str = DEFAULT_SOURCE) -> PaperMeta: + root = _doc_root(source) + raw, _ = await self._execute("cat", f"/{root}/{doc_id}/meta.json") + data = _extract_json_object(raw) + return PaperMeta.model_validate(data) + + async def list_sections(self, doc_id: str, *, source: str = DEFAULT_SOURCE) -> list[str]: + """List a document's available body-section names (without `.lines`).""" + root = _doc_root(source) + text, _ = await self._execute("ls", f"/{root}/{doc_id}/sections/") + return _parse_section_listing(text) + + async def get_content( + self, + doc_id: str, + *, + source: str = DEFAULT_SOURCE, + sections: list[str] | None = None, + max_lines: int | None = None, + ) -> str: + """Fetch full-text body for a document. + + `sections=None` returns the whole line-numbered body (`content.lines`), + capped at `max_lines`. When `sections` is given (canonical names like + "methods", "discussion"), only the matching per-section files are + fetched and concatenated — cheaper and more focused than the whole body. + Falls back to the whole body if the requested sections can't be matched. + """ + root = _doc_root(source) + n = DEFAULT_CONTENT_MAX_LINES if max_lines is None else int(max_lines) + + if not sections: + text, _ = await self._execute("head", f"-n {n} /{root}/{doc_id}/content.lines") + return _strip_timing(text) + + available = await self.list_sections(doc_id, source=source) + selected = _select_section_files(available, sections) + if not selected: + # Requested sections not present under their expected names — return + # the whole body rather than nothing, so depth retrieval still works. + text, _ = await self._execute("head", f"-n {n} /{root}/{doc_id}/content.lines") + return _strip_timing(text) + + # Budget the per-section line cap so the concatenation stays near `n`. + per_section = max(20, n // len(selected)) + blocks: list[str] = [] + for name in selected: + path = _shell_quote(f"/{root}/{doc_id}/sections/{name}.lines") + text, _ = await self._execute("head", f"-n {per_section} {path}") + block = _strip_timing(text) + if block: + blocks.append(f"## {name}\n{block}") + return "\n\n".join(blocks) + + async def run_map( + self, search_id: str, question: str, *, limit: int | None = None + ) -> list[MapExtract]: + """Run Paperclip's `map` over a saved search result set. + + `map` reads each paper's FULL TEXT server-side and answers `question` + per paper — a high-recall extraction that costs Paperclip's tokens, not + ours. Returns one `MapExtract` per paper. We parse the full results file + (`/.gxl/map_<id>.txt`) for the complete answers (the inline listing is + truncated). + + Asks for `MAP_OUTPUT_SCHEMA`'s shape via `_MAP_JSON_CONTRACT` appended + to the question, NOT via `--output_schema` (which 500s on REST — see + that constant). `_map_extract_from_text` falls back to raw prose per + paper if the model doesn't honor it, so it's never a hard requirement. + + Retries `_MAP_NOT_FOUND` in place: Paperclip intermittently cannot + resolve a search id it just issued, and a plain retry recovers about + half of those (see `_MAP_NOT_FOUND` for the mechanism). + """ + full_question = f"{question}\n\n{_MAP_JSON_CONTRACT}" + raw_args = f"--from {search_id} {_shell_quote(full_question)}" + if limit: + raw_args += f" -n {int(limit)}" + + for attempt in range(_MAP_NOT_FOUND_RETRIES + 1): + out, _ = await self._execute("map", raw_args) + if _MAP_NOT_FOUND not in out: + break + if attempt < _MAP_NOT_FOUND_RETRIES: + await asyncio.sleep(_MAP_RETRY_BACKOFF_S * (attempt + 1)) + else: + raise PaperclipError( + f"map could not resolve search id {search_id} after " + f"{_MAP_NOT_FOUND_RETRIES + 1} attempts." + ) + + # A server-side failure arrives as HTTP 200 with `ERR: <message>` in the + # body, so it must be read out of the text. Surface that message + # verbatim — reporting only "no results id" hides the cause and makes a + # server outage look like a parsing bug on our side. + stripped = out.strip() + if stripped.startswith("ERR:"): + raise PaperclipError(stripped.splitlines()[0][len("ERR:"):].strip()) + map_m = _MAP_ID_RE.search(out) + if not map_m: + raise PaperclipError( + f"map returned no results id; output was: {stripped[:200]!r}" + ) + full, _ = await self._execute("cat", f"/.gxl/map_{map_m.group(1)}.txt") + return _parse_map_results(full) + + async def sql(self, query: str, *, source: str | None = None) -> SqlResult: + """Run a read-only SQL `SELECT` against Paperclip's `documents` table. + + Server-enforced: `SELECT`-only, 15s statement timeout, 200-row cap + regardless of the query's own `LIMIT`. No structured JSON on either + transport — `_parse_sql_output` parses the ASCII table and raises + `PaperclipError` on a server-reported query error. + + Confirmed live, caveats not obvious from the docs: + - `documents` is sharded, not unified: omitting `source` queries only + `arxiv`/`biorxiv`/`medrxiv`/`pmc` (same as `_BROAD_MCP_SOURCES`), + not "all". A `WHERE source = 'x'` clause only filters *within* + whatever shard(s) `source=` already selected. + - `source="trials"`/`"proteins"` error with `relation "documents" + does not exist` — not SQL-queryable via this table at all. + - `abstract_text ILIKE '%...%'` over the full pmc/arxiv shards + (millions of rows, unindexed for this) reliably hits the timeout — + use SQL for structured-column aggregates, not free-text search. + """ + raw = _shell_quote(query) + if source: + raw = f"-s {source} {raw}" + out, _ = await self._execute("sql", raw) + return _parse_sql_output(out) + + async def filter(self, search_id: str, query: str) -> SearchResult | None: + """Trim a saved search result set to relevant papers via Paperclip's + server-side LLM relevance judgment. MUTATES `search_id` in place + (confirmed live) — a later `run_map` against the same id sees the + trimmed set too. + + REST-only: `filter`'s text output only reports before/after counts, + never the surviving hit list, so there's no way to reconstruct + results from MCP's text output the way `search`/`map`/`sql` can. + Returns `None` (not an error) when REST is unavailable — a quality + improvement, not a correctness requirement, so it degrades by doing + nothing rather than falling back to MCP; callers should treat `None` + as "use the original unfiltered hits." + + Doesn't use `--require N`: confirmed live it doesn't block the trim, + it just adds an `ERR:`-prefixed warning on top of the same (possibly + empty) result — no simpler than handling an empty result ourselves. + """ + if os.environ.get(DISABLE_REST_ENV): + return None + raw = f"--from {search_id} {_shell_quote(query)}" + try: + data = await self._run_rest("filter", raw) + except PaperclipRestUnavailable as e: + _warn_rest_fallback(e) + return None + output_text = data.get("output", "") + if output_text.strip().startswith("ERR:"): + raise PaperclipError(output_text.strip().splitlines()[0][len("ERR:"):].strip()) + result_data = data.get("result_data") + papers = None + if isinstance(result_data, dict): + papers = result_data.get("papers") + if papers is None: + papers = result_data.get("results") + if papers is None: + # Same intermittent `result_data` omission handled in + # `_hits_from_rest_payload` — but filter's text output is only a + # summary ("Filtered: 5 -> 2 papers"), with no listing to parse. + # Report it as "couldn't filter" (caller keeps the unfiltered + # hits) rather than as "filter removed everything". + return None + hits = [_paper_hit_from_rest(p) for p in papers if isinstance(p, dict)] + return SearchResult(hits=hits, search_id=search_id) + + +__all__ = [ + "MCP_URL", + "REST_URL", + "DISABLE_REST_ENV", + "API_KEY_ENV", + "MAP_OUTPUT_SCHEMA", + "PaperclipSource", + "PaperclipSectionName", + "SECTION_KEYWORDS", + "PaperHit", + "PaperMeta", + "SearchResult", + "SqlResult", + "MapExtract", + "PaperclipError", + "PaperclipConfigError", + "PaperclipRestUnavailable", + "infer_source_from_doc_id", + "PaperclipAdapterProtocol", + "PaperclipAdapter", +] diff --git a/crossbar_llm/paperclip_tools/agent.py b/crossbar_llm/paperclip_tools/agent.py new file mode 100644 index 0000000..cff73a6 --- /dev/null +++ b/crossbar_llm/paperclip_tools/agent.py @@ -0,0 +1,440 @@ +"""LangGraph orchestrator for Paperclip literature evidence. + +`build_graph` wires the standalone data-plumbing nodes from `paperclip_nodes` +(which call the Paperclip MCP adapter) together with three inline LLM-bound nodes +(router, synthesizer, depth evaluator) that close over the chat model and +prompts. It mirrors PubTator3's `build_graph` shape and its injectable seams so +the two tools share one external contract: + + question (+ shared state) in -> {final_answer, citations, warnings, usage} out + +The Paperclip module is architecturally asymmetric with PubTator3 (a thin +MCP-adapter subgraph vs. a hand-built REST subgraph); that is deliberate — the +shared contract is what the top-level graph depends on, not the internals. +""" +from __future__ import annotations + +import re + +from langchain_core.language_models import BaseChatModel +from langchain_core.prompts import ( + ChatPromptTemplate, + HumanMessagePromptTemplate, + MessagesPlaceholder, + SystemMessagePromptTemplate, +) +from langgraph.graph import END, StateGraph + +# Reuse PubTator3's provider-agnostic structured-output helper (function-calling +# first, plain-JSON fallback) rather than re-deriving it. +from crossbar_llm.paperclip_tools.structured_output import ( + _ainvoke_structured_with_json_fallback, +) +from crossbar_llm.paperclip_tools.nodes import ( + _add_warning, + assemble_context_node, + filter_node, + search_node, + sql_node, +) +from crossbar_llm.paperclip_tools.prompts import ( + PAPERCLIP_DEPTH_EVAL_SYSTEM_PROMPT, + PAPERCLIP_ROUTER_SYSTEM_PROMPT, + PAPERCLIP_SQL_SYNTHESIZE_SYSTEM_PROMPT, + PAPERCLIP_SYNTHESIZE_SYSTEM_PROMPT, +) +from crossbar_llm.paperclip_tools.schemas import ( + PaperclipDepthEvaluation, + PaperclipEvaluatorFn, + PaperclipRouterDecision, + PaperclipRouterFn, + PaperclipState, + PaperclipSynthesizerFn, +) +from crossbar_llm.paperclip_tools.adapter import ( + PaperclipAdapterProtocol, + PaperclipAdapter, +) + + +async def _always_sufficient_evaluator(_state: PaperclipState) -> PaperclipDepthEvaluation: + """Default no-op evaluator: declares every answer sufficient and skips the + refinement loop. Used when no chat_model and no explicit evaluator are given + (test seam) so the default path stays single-pass.""" + return PaperclipDepthEvaluation(sufficient=True, missing=None, rationale="no-op evaluator") + + +def _format_documents(state: PaperclipState) -> str: + """Render the assembled papers into the numbered evidence block the + synthesizer reads. Reference numbers align with `state['citations']`. + + Includes the citation URL explicitly: the synthesis prompt's mandatory + References format ends every entry with `— <url>`, so the model needs + the real value here rather than being left to guess or omit it — a real + gap this used to have (URL was in `state['citations']` but never + actually surfaced in the text the model reads).""" + docs = state.get("documents") or [] + citations = {c.doc_id: c for c in (state.get("citations") or [])} + if not docs: + return "(no papers found)" + blocks: list[str] = [] + for doc in docs: + cit = citations.get(doc.doc_id) + ref = cit.ref_num if cit else "?" + url = cit.url if cit else "" + meta = doc.meta + header = ( + f"[{ref}] {meta.title}\n" + f" Authors: {meta.authors or 'n/a'}\n" + f" Journal: {meta.journal or 'n/a'} ({meta.pub_year or 'n/a'}) " + f"doi:{meta.doi or 'n/a'} pmid:{meta.pmid or 'n/a'}\n" + f" URL: {url or 'n/a'}\n" + f" Abstract: {meta.abstract or '(no abstract)'}" + ) + if doc.body: + header += f"\n Evidence (full text / map extraction):\n{doc.body}" + blocks.append(header) + return "\n\n".join(blocks) + + +def _format_sql_result(state: PaperclipState) -> str: + """Render the SQL query + result table for `PAPERCLIP_SQL_SYNTHESIZE_SYSTEM_PROMPT`. + + Caps what's shown to the LLM at 50 rows — `sql`'s own 200-row server cap + is already generous for an aggregate query; this is just a token-budget + guard for the rare case a query legitimately returns close to that cap. + """ + query = state.get("sql_query") or "" + columns = state.get("sql_columns") or [] + rows = state.get("sql_rows") or [] + header = f"Query:\n{query}\n\n" + if not rows: + return header + "(No rows returned.)" + col_line = " | ".join(columns) + lines = [col_line, "-" * len(col_line)] + for row in rows[:50]: + lines.append(" | ".join(str(row.get(c, "")) for c in columns)) + table = "\n".join(lines) + more = f"\n... and {len(rows) - 50} more rows" if len(rows) > 50 else "" + plural = "s" if len(rows) != 1 else "" + return f"{header}Results ({len(rows)} row{plural}):\n{table}{more}" + + +def _is_sql_success(state: PaperclipState) -> bool: + return state.get("question_type") == "sql_aggregate" and not state.get("sql_error") + + +_RULE_CHARS = re.escape("-=_*#~") +_DEGENERATE_RUN_RES = ( + # Rule characters repeat legitimately: a markdown table's alignment row is + # routinely 30+ dashes and trimming one stops the table rendering, so these + # only count as degenerate far past any real formatting. + re.compile(rf"([{_RULE_CHARS}])\1{{199,}}"), + re.compile(rf"([^\w\s{_RULE_CHARS}])\1{{29,}}"), +) + + +def _collapse_degenerate_runs(text: str) -> tuple[str, int]: + """Cut back pathological punctuation repetition in a model's answer. + + Models occasionally lock into emitting one character for thousands of + tokens, typically at the tail of a numbered reference list. The prose + before the run is intact, so the run is trimmed rather than the answer + discarded, and a recovery after the run is preserved. Thirty clears any + real use of repeated punctuation (`...`, `???`) while still catching the + shortest run we have observed, which was 41. + """ + total = 0 + for pattern in _DEGENERATE_RUN_RES: + text, n = pattern.subn(lambda m: m.group(1) * 3, text) + total += n + return text, total + + +def build_graph( + *, + chat_model: BaseChatModel | None = None, + router: PaperclipRouterFn | None = None, + synthesizer: PaperclipSynthesizerFn | None = None, + evaluator: PaperclipEvaluatorFn | None = None, + adapter: PaperclipAdapterProtocol | None = None, + max_documents: int = 7, + content_max_lines: int | None = None, + abstracts_only: bool = True, + use_map: bool = True, + use_filter: bool = False, +): + """Compile the Paperclip LangGraph. + + Pass `chat_model` for production. Pass `router`/`synthesizer`/`evaluator` + directly to bypass the LLM (test seam). Pass `adapter` to inject a fake + Paperclip MCP adapter (the single MCP seam); defaults to a live + `PaperclipAdapter`. `evaluator` is optional; when neither it nor + `chat_model` is given, the no-op default treats every answer as sufficient + and the refinement loop never fires. + + `abstracts_only` forces title+abstract retrieval regardless of the router's + `full_text` choice and short-circuits the depth-refinement pass — use it for + predictable token cost. + + `use_map` (ON by default) runs Paperclip's `map` over the search results: it + reads each paper's FULL TEXT server-side and extracts an answer to the + question per paper, used as the evidence body. This boosts recall on detail + questions without pulling whole bodies through our tokens (it costs Paperclip + tokens + latency instead). Independent of `abstracts_only`; set it False to + synthesize from abstracts only. + + `use_filter` (OFF by default) runs Paperclip's `filter` on the search hits + before `assemble`, trimming to server-judged relevant papers. REST-only and + best-effort (see `PaperclipAdapter.filter`/`filter_node`) — any failure or + unavailability reverts to the unfiltered hits rather than failing the run. + """ + if (router is None or synthesizer is None) and chat_model is None: + raise ValueError( + "build_graph requires either chat_model, or both router and synthesizer." + ) + if evaluator is None and chat_model is None: + evaluator = _always_sufficient_evaluator + if adapter is None: + adapter = PaperclipAdapter() + + async def router_node(state: PaperclipState) -> dict: + warnings = list(state.get("warnings", [])) + try: + if router is not None: + decision = await router(state["question"]) + else: + prompt = ChatPromptTemplate.from_messages([ + SystemMessagePromptTemplate.from_template(PAPERCLIP_ROUTER_SYSTEM_PROMPT), + MessagesPlaceholder("chat_history", optional=True), + HumanMessagePromptTemplate.from_template("User question: {question}"), + ]) + decision, used_json_fallback = await _ainvoke_structured_with_json_fallback( + chat_model=chat_model, + prompt=prompt, + schema=PaperclipRouterDecision, + values={ + "question": state["question"], + "chat_history": state.get("chat_history", []), + }, + json_instruction=( + "The previous instruction defines the exact routing schema. " + "Return ONLY a valid JSON object for that schema. Do not use " + "Markdown, prose, tool calls, or extra keys." + ), + ) + if used_json_fallback: + warnings.append( + "router structured-output unavailable; used JSON fallback." + ) + except Exception as e: + # Degrade to keyword_search rather than crash the run. + warnings.append( + f"router failed ({type(e).__name__}); fell back to keyword_search." + ) + decision = PaperclipRouterDecision( + question_type="keyword_search", + source="pmc", + search_query=state["question"], + map_question=state["question"], + rationale=f"router error fallback: {e}", + ) + return { + "question_type": decision.question_type, + "source": decision.source, + "search_query": decision.search_query, + "analogical_query": decision.analogical_query, + "map_question": decision.map_question, + "sql_query": decision.sql_query, + "full_text": False if abstracts_only else decision.full_text, + "sections": None if abstracts_only else decision.sections, + "year": decision.year, + "rationale": decision.rationale, + "warnings": warnings, + } + + async def _search(state): + return await search_node( + state, adapter=adapter, max_documents=max_documents, use_filter=use_filter + ) + + async def _sql(state): + return await sql_node(state, adapter=adapter) + + async def _filter(state): + return await filter_node(state, adapter=adapter, use_filter=use_filter) + + async def _assemble(state): + return await assemble_context_node( + state, + adapter=adapter, + max_documents=max_documents, + content_max_lines=content_max_lines, + use_map=use_map, + ) + + async def synthesize_node(state: PaperclipState) -> dict: + if synthesizer is not None: + answer = await synthesizer(state) + else: + if _is_sql_success(state): + evidence = _format_sql_result(state) + system_prompt = PAPERCLIP_SQL_SYNTHESIZE_SYSTEM_PROMPT + else: + evidence = _format_documents(state) + system_prompt = PAPERCLIP_SYNTHESIZE_SYSTEM_PROMPT + prompt = ChatPromptTemplate.from_messages([ + SystemMessagePromptTemplate.from_template(system_prompt), + MessagesPlaceholder("chat_history", optional=True), + HumanMessagePromptTemplate.from_template( + "User question:\n{question}\n\nEvidence:\n{evidence}\n\n" + "Write the final answer." + ), + ]) + chain = prompt | chat_model + msg = await chain.ainvoke({ + "question": state["question"], + "evidence": evidence, + "chat_history": state.get("chat_history", []), + }) + answer = msg.content if isinstance(msg.content, str) else str(msg.content) + answer, degenerate_runs = _collapse_degenerate_runs(answer) + if degenerate_runs: + return { + "final_answer": answer, + "warnings": _add_warning( + state, + f"synthesize: trimmed {degenerate_runs} degenerate character " + "run(s) from the model's answer", + ), + } + return {"final_answer": answer} + + async def evaluate_depth_node(state: PaperclipState) -> dict: + # Short-circuit cases — no escalation lever to pull. + if _is_sql_success(state): + return { + "depth_sufficient": True, + "depth_skip_reason": "sql_aggregate has no full-text escalation lever", + } + if abstracts_only: + return {"depth_sufficient": True, "depth_skip_reason": "abstracts_only enabled"} + if not state.get("final_answer") or not state.get("documents"): + return {"depth_sufficient": True, "depth_skip_reason": "no answer or no documents"} + if state.get("refinement_attempted"): + return {"depth_sufficient": True, "depth_skip_reason": "refinement already attempted"} + if state.get("full_text"): + return {"depth_sufficient": True, "depth_skip_reason": "already at full-text depth"} + + warnings = list(state.get("warnings", [])) + try: + if evaluator is not None: + verdict = await evaluator(state) + else: + prompt = ChatPromptTemplate.from_messages([ + SystemMessagePromptTemplate.from_template(PAPERCLIP_DEPTH_EVAL_SYSTEM_PROMPT), + HumanMessagePromptTemplate.from_template( + "User question:\n{question}\n\n" + "Generated answer (abstracts-only):\n{answer}" + ), + ]) + verdict, used_json_fallback = await _ainvoke_structured_with_json_fallback( + chat_model=chat_model, + prompt=prompt, + schema=PaperclipDepthEvaluation, + values={ + "question": state["question"], + "answer": state["final_answer"], + }, + json_instruction=( + "Return ONLY a valid JSON object for the depth-evaluation " + "schema. Do not use Markdown, prose, tool calls, or extra keys." + ), + ) + if used_json_fallback: + warnings.append( + "depth evaluator structured-output unavailable; used JSON fallback." + ) + except Exception as e: + warnings.append( + f"depth evaluator failed ({type(e).__name__}); accepting answer as-is." + ) + return { + "depth_sufficient": True, + "depth_skip_reason": f"evaluator error ({type(e).__name__}: {e})", + "warnings": warnings, + } + + if verdict.sufficient: + return {"depth_sufficient": True, "depth_missing": None, "warnings": warnings} + + warnings.append( + f"depth check flagged shallow answer: " + f"{verdict.missing or 'no specific gap reported'}; " + f"re-fetching with full text." + ) + return { + "depth_sufficient": False, + "depth_missing": verdict.missing, + "full_text": True, + "refinement_attempted": True, + "warnings": warnings, + } + + def _post_evaluate_route(state: PaperclipState) -> str: + if state.get("depth_sufficient", True): + return "end" + if not state.get("refinement_attempted"): + return "end" # safety net — never loop without the cap + return "refine" + + g = StateGraph(PaperclipState) + g.add_node("router", router_node) + g.add_node("search", _search) + g.add_node("sql", _sql) + g.add_node("filter", _filter) + g.add_node("assemble", _assemble) + g.add_node("synthesize", synthesize_node) + g.add_node("evaluate_depth", evaluate_depth_node) + + g.set_entry_point("router") + g.add_conditional_edges( + "router", + lambda s: s["question_type"], + { + "out_of_scope": END, + "keyword_search": "search", + "list_breadth": "search", + "full_text_depth": "search", + "analogical_search": "search", + "sql_aggregate": "sql", + }, + ) + # sql_node never fails the run: on a bad/timed-out query it sets + # sql_error and leaves hits/search_id unset, so this falls through to + # the normal search path using the router's search_query fallback — + # same shape as search_node's own zero-result fallback. + g.add_conditional_edges( + "sql", + lambda s: "search" if s.get("sql_error") else "synthesize", + {"search": "search", "synthesize": "synthesize"}, + ) + g.add_edge("search", "filter") + g.add_edge("filter", "assemble") + g.add_edge("assemble", "synthesize") + g.add_edge("synthesize", "evaluate_depth") + g.add_conditional_edges( + "evaluate_depth", + _post_evaluate_route, + {"end": END, "refine": "assemble"}, + ) + + return g.compile() + + +__all__ = [ + "build_graph", + "_always_sufficient_evaluator", + "_format_documents", + "_format_sql_result", +] diff --git a/crossbar_llm/paperclip_tools/llm.py b/crossbar_llm/paperclip_tools/llm.py new file mode 100644 index 0000000..d8d7ea4 --- /dev/null +++ b/crossbar_llm/paperclip_tools/llm.py @@ -0,0 +1,41 @@ +"""Chat model construction, via the project's shared LLM factory. + +The agent graph takes an already-built `chat_model` rather than building one +itself: that keeps the graph independent of how the model is configured, and +is the seam the tests inject fakes at. This helper is the convenience path for +callers who just want the project's configured model. +""" +from __future__ import annotations + +from langchain_core.callbacks import BaseCallbackHandler +from langchain_core.language_models import BaseChatModel + +from crossbar_llm.agent_tools.config import LLMConfig, ReasoningConfig +from crossbar_llm.agent_tools.llm_factory import LLMFactory + + +def build_chat_model( + *, + model: str, + provider: str | None = None, + temperature: float = 0.0, + callbacks: list[BaseCallbackHandler] | None = None, + reasoning: ReasoningConfig | None = None, +) -> BaseChatModel: + """Build a chat model for this agent. + + `provider` may be omitted — the factory infers it from the model name. + Temperature defaults to 0.0 because routing and synthesis both want + reproducible output, where the factory's own default is 1.0. + """ + config = LLMConfig( + model=model, + provider=provider, + temperature=temperature, + callbacks=callbacks or [], + reasoning=reasoning or ReasoningConfig(), + ) + return LLMFactory(config).get_base_model() + + +__all__ = ["build_chat_model"] diff --git a/crossbar_llm/paperclip_tools/nodes.py b/crossbar_llm/paperclip_tools/nodes.py new file mode 100644 index 0000000..e266f84 --- /dev/null +++ b/crossbar_llm/paperclip_tools/nodes.py @@ -0,0 +1,464 @@ +"""Standalone LangGraph node functions for the Paperclip pipeline. + +Each node is a plain coroutine `async def node(state, *, adapter, ...) -> dict` +returning a partial `PaperclipState` to merge. **No LLM lives here** — the +LLM-bound nodes (router, synthesize, evaluate_depth) live inside `build_graph` +because they close over the chat model and prompts. Everything here is pure data +plumbing around the never-raise Paperclip tool wrappers, so it is deterministic +and unit-testable with a fake adapter injected at the seam. +""" +from __future__ import annotations + +import asyncio + +from crossbar_llm.paperclip_tools.schemas import ( + Citation, + PaperContext, + PaperclipState, +) +from crossbar_llm.paperclip_tools.tools import ( + paperclip_filter, + paperclip_get_content, + paperclip_get_meta, + paperclip_map, + paperclip_search, + paperclip_sql, +) +from crossbar_llm.paperclip_tools.adapter import ( + PaperclipAdapterProtocol, + PaperMeta, + infer_source_from_doc_id, +) + +# Base search fan-out and the breadth-question fan-out. +_DEFAULT_LIMIT = 10 +_BREADTH_LIMIT = 25 +# Extra hits fetched beyond `max_documents` so uncitable hits can be backfilled +# from the buffer rather than shrinking the evidence set. +_BACKFILL_BUFFER = 4 +# Wider fan-out when `filter` will run afterward. Confirmed live: filter can cut +# a ~14-hit fetch to single digits or zero, which `_BACKFILL_BUFFER`'s +4 cannot +# absorb — without this, a filtered search starves assembly below max_documents +# even when filter behaves correctly. +_FILTER_FETCH_LIMIT = 25 + +# Sources whose meta.json doesn't populate title/authors/abstract the way +# paper corpora do (confirmed live). Evidence for these falls back to the +# search hit's own title/snippet, which Paperclip does populate. +_REGULATORY_TRIAL_SOURCES = frozenset({ + "fda", "fda/us", "fda/jp", "fda/eu", + "trials", "trials/us", "trials/eu", "trials/jp", "trials/cn", +}) + + +def _add_warning(state: PaperclipState, msg: str) -> list[str]: + return [*state.get("warnings", []), msg] + + +def citation_url(doc_id: str, source: str | None = None, *, line_anchor: str | None = None) -> str: + """Build the citation URL for a doc id. + + Paper/FDA/trial corpora use Paperclip's citation host; the proteins corpus + is a UniProt accession, so we link to UniProt directly (it has no gxl + citation namespace, and no line concept — `line_anchor` is ignored there). + + `line_anchor` (e.g. `"L45"`, `"L45-L52"`, `"L45,120,210"` — see + `format_line_anchor`) appends a `#<anchor>` fragment per Paperclip's own + citation-format spec, pointing at the specific supporting line(s) instead + of just the paper root.""" + src = (source or "").lower() + if src in ("proteins", "uniprot") and not doc_id.startswith(("PMC", "bio_", "med_", "arx_")): + return f"https://www.uniprot.org/uniprotkb/{doc_id}/entry" + if doc_id.startswith("fda_"): + ns = "fda" + elif doc_id.startswith("tri_") or doc_id.startswith("NCT"): + ns = "trials" + else: + ns = "papers" + url = f"https://citations.gxl.ai/{ns}/{doc_id}" + return f"{url}#{line_anchor}" if line_anchor else url + + +def format_line_anchor(lines: list[int]) -> str: + """Render supporting line numbers as a citation fragment, per Paperclip's + own format: single `L45`, contiguous range `L45-L52`, or a comma list + `L45,120,210` for non-contiguous lines. Empty string for no lines.""" + uniq = sorted(set(lines)) + if not uniq: + return "" + if len(uniq) == 1: + return f"L{uniq[0]}" + if uniq == list(range(uniq[0], uniq[-1] + 1)): + return f"L{uniq[0]}-L{uniq[-1]}" + return "L" + ",".join(str(n) for n in uniq) + + +def _proteins_summary(meta) -> str: + """A compact evidence string for a proteins-corpus hit (which has no + abstract or full text — deeper feature/domain data lives in the UniProt SQL + views, not fetched here).""" + parts = [ + meta.protein_name or meta.title, + f"gene {meta.gene_name}" if meta.gene_name else "", + meta.organism or "", + f"UniProt {meta.uniprot_id or meta.accession or meta.document_id}", + ] + return "; ".join(p for p in parts if p) + + +async def search_node( + state: PaperclipState, + *, + adapter: PaperclipAdapterProtocol, + max_documents: int = 10, + use_filter: bool = False, +) -> dict: + """Run the primary search, with a zero-result free-text fallback. + + Fetches a few more hits than `max_documents` (the backfill buffer) so + `assemble` can replace uncitable hits instead of shrinking the evidence set. + `source=None` (the router's default) searches broadly across the general + literature corpora in one call rather than guessing a single corpus — see + `PaperclipAdapter.search`. + The router always fills `search_query`, so any route can degrade to a + broader retry rather than returning "no information": if the primary search + yields nothing, we retry once against the broad `abstracts` corpus. + + `analogical_search` (§5.13) uses `analogical_query` — a method/problem- + description sentence, not keywords — for the PRIMARY call only, with + `ranking="analogical"`. The zero-result fallback always uses + `search_query` (keywords) with the default ranking, same as every other + route: a keyword-shaped retry is a more useful safety net than repeating + an unranked analogical query, and matches `search_query`'s own documented + role as the universal fallback. + + `use_filter=True` widens the fetch to `_FILTER_FETCH_LIMIT` (unless + `list_breadth`, which is already wide and which `filter_node` skips + outright) — confirmed live that `filter` can cut a fetch down to single + digits or zero, and the normal backfill buffer alone leaves no room for + that cut on top of the usual uncitable-hit backfill. + """ + question_type = state.get("question_type") + search_query = (state.get("search_query") or state.get("question") or "").strip() + analogical_query = (state.get("analogical_query") or "").strip() + use_analogical = question_type == "analogical_search" and bool(analogical_query) + query = analogical_query if use_analogical else search_query + ranking = "analogical" if use_analogical else None + source = state.get("source") + year = state.get("year") + if question_type == "list_breadth": + limit = _BREADTH_LIMIT + elif use_filter: + limit = max(_FILTER_FETCH_LIMIT, max_documents + _BACKFILL_BUFFER) + else: + limit = max(_DEFAULT_LIMIT, max_documents + _BACKFILL_BUFFER) + + warnings = list(state.get("warnings", [])) + queries_used: list[str] = [] + + if not query: + warnings.append("no search query available; nothing to retrieve.") + return {"hits": [], "search_id": None, "queries_used": [], "warnings": warnings} + + out = await paperclip_search(adapter, query, source=source, limit=limit, year=year, ranking=ranking) + queries_used.append(f"[{source or 'broad'}{' analogical' if use_analogical else ''}] {query}") + hits = list(out.hits) if not out.error else [] + search_id = out.search_id + if out.error: + warnings.append(f"search failed for '{query}' (-s {source}): {out.error}") + + # Zero-result fallback: drop the corpus scope, the year filter and any + # analogical ranking, and retry unscoped with the plain keyword + # `search_query` — a keyword retry is the useful safety net here. Only run + # it when that actually differs from what was just tried; an identical + # repeat would return the same nothing. + # + # This used to retry against `-s abstracts`, which is listed in Paperclip's + # own `help search` but is not a real corpus: it is absent from `ls /` and + # returns "No papers found" for every query tried, so the fallback could + # never recover anything. + retry_differs = bool(source or year or ranking or query != search_query) + if not hits and retry_differs: + fb_out = await paperclip_search(adapter, search_query, source=None, limit=limit) + queries_used.append(f"[broad] {search_query}") + if fb_out.error: + warnings.append(f"fallback broad search failed: {fb_out.error}") + elif fb_out.hits: + warnings.append( + f"primary search returned 0 results; fell back to an unscoped " + f"search ({len(fb_out.hits)} hits)." + ) + hits = list(fb_out.hits) + search_id = fb_out.search_id + source = None + + if not hits: + warnings.append("No papers found for the query.") + + return { + "hits": hits, + "search_id": search_id, + "source": source, + "queries_used": queries_used, + "warnings": warnings, + } + + +async def sql_node(state: PaperclipState, *, adapter: PaperclipAdapterProtocol) -> dict: + """Run the router's SQL query for `question_type == "sql_aggregate"`. + + SQL is a precision tool for structured aggregates (counts/rankings by + source, year, journal, ...) and can legitimately fail in ways full-text + search wouldn't — a malformed query, an unsupported source (trials/ + proteins aren't SQL-queryable at all — confirmed live, see + `PaperclipAdapter.sql`), or the server's 15s statement timeout on an + unindexed pattern. On failure this does NOT fail the run: it records + `sql_error` and leaves `hits`/`search_id` unset, so the graph's + conditional edge can route to the normal `search` node instead, using + the router's `search_query` fallback exactly like the zero-result path. + """ + query = (state.get("sql_query") or "").strip() + source = state.get("source") + warnings = list(state.get("warnings", [])) + + if not query: + warnings.append("sql_aggregate route had no sql_query; falling back to search.") + return {"sql_error": "no query", "warnings": warnings} + + out = await paperclip_sql(adapter, query, source=source) + if out.error: + warnings.append(f"SQL query failed ({out.error}); falling back to search.") + return {"sql_error": out.error, "warnings": warnings} + + return { + "sql_columns": out.columns, + "sql_rows": out.rows, + "sql_error": None, + "warnings": warnings, + } + + +async def assemble_context_node( + state: PaperclipState, + *, + adapter: PaperclipAdapterProtocol, + max_documents: int = 7, + content_max_lines: int | None = None, + use_map: bool = False, +) -> dict: + """Enrich hits into citable context for synthesis. + + For each hit we fetch `meta.json` (the citable DOI/PMID record). Evidence + body per paper is, in priority order: + 1. a `map` extraction (full-text-derived, server-side) when `use_map`; + 2. full body / target sections when `full_text` depth is on; + 3. the abstract only (from meta), otherwise. + Uncitable hits (metadata fetch fails) are dropped and **backfilled** from + the remaining hits so we still reach `max_documents` citable papers. + + A paper cited from map evidence gets a **line-anchored** citation URL + (`#L<n>`) built from Paperclip's own per-answer line provenance — see + `adapter.py`'s `_map_citation_lines` and this module's `format_line_anchor` + — instead of a blanket paper root link; abstract/full-text evidence has + no such deterministic anchor (the synthesis prompt asks the model to + self-cite a specific full-text line instead, when full-text evidence is + used). + + `source` may now be `None` (broad/unscoped search) or a comma-separated + list (the MCP fallback's paper-corpora substitute) — either way a single + result set can mix corpora, so `get_meta`/`get_content`/citation URLs + resolve each hit's VFS root from its own `doc_id` shape + (`infer_source_from_doc_id`) rather than one blanket source string. + """ + # Deduplicate by doc_id, keeping rank order: a broad search mixes corpora + # and `filter` returns a server-rebuilt list, so one document can arrive + # twice and would otherwise be cited as two independent references. + seen: set[str] = set() + hits = [ + h for h in (state.get("hits") or []) + if h.doc_id and not (h.doc_id in seen or seen.add(h.doc_id)) + ] + warnings = list(state.get("warnings", [])) + full_text = bool(state.get("full_text", False)) + sections = state.get("sections") or None + source = state.get("source") + # "papers" (the broad-search alias) and a comma list both mean "multiple + # corpora in one result set" — resolve source per-hit in either case. + is_broad = not source or source == "papers" or "," in source + + if not hits: + return {"documents": [], "citations": []} + + # Resolve once per hit so `_one()` and the final assembly loop agree. + hit_sources: dict[str, str] = { + h.doc_id: (infer_source_from_doc_id(h.doc_id) if is_broad else source) + for h in hits + } + + # Optional map pass: one server-side call reads full text across the result + # set and answers the question per paper. High-recall evidence at no full- + # body token cost to us. + map_extracts: dict[str, str] = {} + map_citation_lines: dict[str, list[int]] = {} + if use_map and state.get("search_id"): + # map_question is a full extraction question, unlike search_query's + # keywords — a vague one here yields vague per-paper answers and a + # weaker found/not-found signal. + map_question = state.get("map_question") or state.get("question", "") + map_out = await paperclip_map( + adapter, state["search_id"], map_question, limit=len(hits) + ) + if map_out.error: + warnings.append(f"map extraction failed: {map_out.error}; using abstracts.") + else: + # `run_map` asks for an {answer, found} contract. When the model + # honored it, `.found is False` means this paper explicitly + # doesn't address the question — exclude it rather than using + # "not found"-shaped text as if it were evidence. `.found is None` + # (contract not honored for that paper) is treated as usable. + usable = [e for e in map_out.extracts if e.success and e.text and e.found is not False] + map_extracts = {e.doc_id: e.text for e in usable} + # Line-level provenance, so a citation points at the supporting + # lines rather than the paper root. Server-computed when available, + # else recovered from the model's inline refs. + map_citation_lines = {e.doc_id: e.citation_lines for e in usable if e.citation_lines} + + async def _one(hit): + hit_source = hit_sources[hit.doc_id] + is_proteins_hit = hit_source in ("proteins", "uniprot", "pdb", "chembl") + meta_out = await paperclip_get_meta(adapter, hit.doc_id, source=hit_source) + # map ran once over the whole result set, so its evidence survives a + # per-hit get_meta failure — look it up regardless. + body = map_extracts.get(hit.doc_id) + # A failed get_meta doesn't make the document unreachable — assembly + # below recovers a citable record from the hit's own fields when it has + # a title. So fetch the body for anything that will survive assembly, + # independent of the get_meta outcome. + will_be_citable = not (meta_out.error or meta_out.meta is None) or bool(hit.title) + content_err = None + if body is None and full_text and not is_proteins_hit and will_be_citable: + content_out = await paperclip_get_content( + adapter, hit.doc_id, source=hit_source, + sections=sections, max_lines=content_max_lines, + ) + if content_out.error: + content_err = content_out.error + else: + body = content_out.content + return hit, meta_out, body, content_err + + results = await asyncio.gather(*(_one(h) for h in hits)) + + documents: list[PaperContext] = [] + citations: list[Citation] = [] + ref_num = 0 + for hit, meta_out, body, content_err in results: + if len(documents) >= max_documents: + break + meta = meta_out.meta + if meta_out.error or meta is None: + if hit.title: + # get_meta can fail intermittently even when the doc_id is + # valid (confirmed live for some `abstracts`/OpenAlex hits — + # a server-side inconsistency, not a parsing bug). Recover a + # citable PaperMeta from the search hit's own fields rather + # than dropping otherwise-good evidence — same "trust the + # hit's own fields" pattern as the fda/trials fallback below. + warnings.append( + f"could not fetch metadata for {hit.doc_id} " + f"({meta_out.error or 'no meta'}); recovered from search hit." + ) + meta = PaperMeta( + document_id=hit.doc_id, title=hit.title, authors=hit.authors, + doi=hit.doi, pub_year=hit.pub_year, abstract=hit.snippet, + ) + else: + warnings.append( + f"could not fetch metadata for {hit.doc_id}; dropped " + f"(uncitable, backfilling): {meta_out.error or 'no meta'}." + ) + continue + if content_err: + warnings.append( + f"full text unavailable for {hit.doc_id}; using abstract only: {content_err}." + ) + hit_source = hit_sources[hit.doc_id] + is_proteins_hit = hit_source in ("proteins", "uniprot", "pdb", "chembl") + # Proteins have no abstract/body — synthesise a compact evidence line. + if is_proteins_hit and not body: + body = _proteins_summary(meta) + # fda/trials meta.json doesn't populate title/abstract like paper + # corpora do (confirmed empty live) — backfill from the search hit's + # own title/snippet, which Paperclip DOES populate, so both the + # context block fed to the synthesizer (which reads meta.title/ + # meta.abstract directly) and the citation below see real content. + if hit_source in _REGULATORY_TRIAL_SOURCES and not meta.title and not meta.abstract: + meta.title = meta.title or hit.title + meta.abstract = meta.abstract or hit.snippet + ref_num += 1 + documents.append(PaperContext(doc_id=hit.doc_id, meta=meta, body=body)) + line_anchor = format_line_anchor(map_citation_lines.get(hit.doc_id, [])) + citations.append( + Citation( + ref_num=ref_num, + doc_id=hit.doc_id, + title=meta.title or meta.protein_name or hit.title, + authors=meta.authors, + journal=meta.journal or (meta.organism if is_proteins_hit else None), + year=meta.pub_year, + doi=meta.doi, + pmid=meta.pmid, + url=citation_url(hit.doc_id, hit_source, line_anchor=line_anchor), + ) + ) + + if not documents: + warnings.append("No citable documents assembled from the search hits.") + + return {"documents": documents, "citations": citations, "warnings": warnings} + + +async def filter_node( + state: PaperclipState, *, adapter: PaperclipAdapterProtocol, use_filter: bool = False +) -> dict: + """Trim search hits to relevant ones via Paperclip's `filter`, before the + more expensive per-hit get_meta/get_content/map work in + `assemble_context_node`. + + Opt-in (`use_filter`, off by default) and REST-only (see + `PaperclipAdapter.filter`) — a quality improvement, never a correctness + requirement, so any failure/unavailability/empty result reverts to the + original unfiltered hits rather than failing the run. Skipped for + `list_breadth`: that route wants broad coverage, which relevance-filtering + would fight against. + """ + if not use_filter: + return {} + hits = state.get("hits") or [] + search_id = state.get("search_id") + warnings = list(state.get("warnings", [])) + if not hits or not search_id or state.get("question_type") == "list_breadth": + return {} + + query = state.get("question") or state.get("search_query") or "" + out = await paperclip_filter(adapter, search_id, query) + if out.error: + warnings.append(f"filter failed ({out.error}); using unfiltered results.") + return {"warnings": warnings} + if out.skipped: + return {} + if not out.hits: + warnings.append("filter removed all hits as irrelevant; using unfiltered results.") + return {"warnings": warnings} + return {"hits": out.hits, "warnings": warnings} + + +__all__ = [ + "_add_warning", + "citation_url", + "format_line_anchor", + "search_node", + "sql_node", + "filter_node", + "assemble_context_node", +] diff --git a/crossbar_llm/paperclip_tools/prompts.py b/crossbar_llm/paperclip_tools/prompts.py new file mode 100644 index 0000000..e561192 --- /dev/null +++ b/crossbar_llm/paperclip_tools/prompts.py @@ -0,0 +1,169 @@ +"""System prompts for the Paperclip LLM-bound nodes: router, synthesize, +SQL-synthesize, depth. + +Paperclip is retrieval/full-text oriented, so the router classifies by retrieval +shape (keyword vs breadth vs full-text depth vs SQL aggregate) and picks a corpus +`source`, rather than PubTator3's entity/relation taxonomy. The literature- +synthesis path keeps a strict numbered citation contract so answers are always +provenance-backed; the SQL-synthesis path has its own prompt with no citation +contract at all (there's nothing to cite for a direct database query) — see +`PAPERCLIP_SQL_SYNTHESIZE_SYSTEM_PROMPT`. + +Field descriptions on `PaperclipRouterDecision` (paperclip_schemas.py) are the +single source of truth for each field's specific rules (source's corpus list, +sql_query's table schema/constraints, etc.) — they're shown to the model via +structured output regardless of what's in these system prompts, so keep the +system prompts focused on cross-field judgment calls and don't re-duplicate +per-field mechanics here (drifting the two out of sync is a real risk, not a +hypothetical one — it's already happened once with other docs in this project). +""" +from __future__ import annotations + + +PAPERCLIP_ROUTER_SYSTEM_PROMPT = """\ +You are the router of a Paperclip literature-retrieval agent. Paperclip searches a large +full-text corpus (PubMed Central, bioRxiv/medRxiv, arXiv, FDA regulatory documents, clinical +trial registries, protein databases) and returns papers with citable identifiers (DOI/PMID). + +Classify the user's question into exactly one `question_type`, then fill the other fields. +Each field's own description has the specific rules for filling it correctly (source's corpus +list, sql_query's table schema and constraints, etc.) — read those before deciding a value. The +principles below are the cross-cutting judgment calls that decide BETWEEN fields, not the +mechanics of any one of them: + +- Paperclip is a RETRIEVAL tool: it is strongest at "find and summarise the literature on X", + breadth/list questions, and full-text depth. It is NOT a knowledge-graph engine. +- BIAS TOWARD ANSWERING. The DEFAULT is `keyword_search`. This INCLUDES specific factual and + mechanistic questions — "which pathway/gene/drug/receptor does X involve?", "what is the + mechanism of Y?", "what triggers Z?". These are answered by retrieving the papers that state + the fact; they are NOT knowledge-graph traversal just because the answer is a single entity. + If a plausible paper would contain the answer, it is in scope. +- Reserve `out_of_scope` for questions Paperclip genuinely cannot serve: non-biomedical topics + (weather, math, opinion, news), EXPLICIT graph-algorithm requests ("shortest path between + ...", "k-hop neighbours of ..."), or requests to traverse a specific knowledge graph. When in + doubt between out_of_scope and keyword_search, choose keyword_search — the retrieval will + simply return little if the topic is truly unsupported. +- `sql_aggregate` is a NARROW, separate route for counts/rankings over STRUCTURED metadata + (source, year, journal, article type, author) — it is not a literature search. "How many + papers discuss/mention/are about X" is NOT sql_aggregate (that needs semantic retrieval over + abstracts, which SQL cannot safely do — see `sql_query`'s description for why). When in + doubt, don't use it — it's a precision tool for a narrow class of questions, not a default. +- `analogical_search` is rarer and narrower than `sql_aggregate` (full criteria in its own + description) — crosses RESEARCH FIELDS, not diseases within biomedicine, and not SPECIES. + Orthologs, conserved processes and cross-species comparisons are ordinary literature + (`keyword_search`), however much the word "analogous" seems to fit. When in doubt, don't + use it. +- `search_query` must be built around the ANSWER TYPE — the kind of thing the user wants back, + the noun right after "which"/"what"/"name all". Keep the subject the question is anchored on, + and drop every intermediate entity the question only routes THROUGH. "Which drugs target + proteins associated with Alzheimer disease?" wants drugs, about Alzheimer disease, linked via + proteins: query "drugs approved for Alzheimer disease". Querying the intermediate retrieves + papers about protein targets, which name no drugs; appending the answer type to it is worse + still. Longer chains collapse the same way: "diseases related to a gene associated with drug + X" -> "X associated diseases". Topic questions with no chain need no rewriting. +- `search_query` and `map_question` answer DIFFERENT needs and should usually read differently: + `search_query` is a handful of keywords for retrieval; `map_question` is a full, specific + question — every field you want extracted enumerated — asked to each retrieved paper + individually. Conflating them (e.g. putting keywords in `map_question`) weakens per-paper + extraction quality. Fill both, for every route. +- `source` defaults to null (a broad search across the general literature in one call) — only + narrow it when the question specifically targets one corpus's domain. Do not guess a single + corpus for a general question; when unsure, leave it null. A question about what a drug IS, + CONTAINS, TARGETS, or is USED FOR is a literature fact, not a narrow-corpus question — e.g. + "which drugs are in LONSURF?" stays null, it is NOT `fda`. The same trap exists for + `proteins`: anything a PAPER reports about a protein — which domain binds what, how a complex + assembles, what regulates it — is literature, e.g. "which domain of the MOZ/MYST3 complex + associates with histone H3?" stays null, it is NOT `proteins`, because a binding partner is a + paper finding. But the reverse case is real and `proteins` IS right for it: "what is the + sequence length of UniProt P04637?" or "what is the PDB accession for human lysozyme?" ask for + a database field, and those DO go to `proteins`/`pdb`. +- `full_text`/`sections` default to abstracts-only — only escalate when the user explicitly + wants mechanism/method/results/paragraph-level detail beyond what an abstract gives. + +Return the routing schema only.""" + + +PAPERCLIP_SYNTHESIZE_SYSTEM_PROMPT = """\ +You are the synthesis layer of a Paperclip literature-evidence agent. Given the user's +question and a set of retrieved papers — each with a reference number [N], title, abstract, +a citable URL, and (when available) full-text body passages or a per-paper extracted answer +(labeled "map extraction") — integrate the findings into one evidence-grounded answer. + +Rules: +- FORMAT: consolidate findings from the retrieved papers into ONE coherent paragraph that + integrates the evidence into a unified narrative. Do NOT use bullet points, numbered lists, + sub-headings, or multiple paragraphs in the body. The mandatory `References:` section is the + only structured part of the output. (Exception: an explicit "list all / enumerate" question + may present the items as a single short list, still followed by the References section.) +- CITE every claim inline with its reference number in square brackets, e.g. [1]. When several + papers support one claim, group them: [1, 3]. Use ONLY the reference numbers you were given; + never invent citations or IDs. +- ALWAYS end with a `References:` section listing EVERY reference number you cited, one per + line, in this format: + [1] Authors. Title. Journal (Year). doi:<doi> — <url> + Use the other metadata provided for each reference; omit a field only if it wasn't given. + This section is mandatory whenever you cite at least one reference. +- CITATION URL / LINE ANCHOR — three cases, by what that reference's URL and evidence look like: + 1. URL already ends in `#L...` (a map-extraction citation Paperclip itself pinned to specific + supporting lines): use that URL EXACTLY as given, character for character. Do not edit, + recompute, or drop the anchor. + 2. URL has NO `#L...` anchor AND that reference's evidence is full-text body content shown + with `L<n>: ` line-number prefixes: after writing the reference line, append `#L<n>` to the + URL naming the SPECIFIC line(s) that support the claim(s) you cited for that reference — + `#L45` for one line, `#L45-L52` for a contiguous range, `#L45,120,210` for several. Use + ONLY line numbers that actually appear in the evidence shown for that paper — never infer, + estimate, or invent one. + 3. URL has no anchor and there are no `L<n>: ` lines in that reference's evidence (abstract-only): + use the URL exactly as given, with no anchor added. +- If the retrieved papers do not support the user's question, say so plainly in one or two + sentences instead of speculating, and omit the References section. +- Ground claims in the provided abstracts/passages/extractions only. Do not add facts from + prior knowledge that aren't supported by the evidence.""" + + +PAPERCLIP_SQL_SYNTHESIZE_SYSTEM_PROMPT = """\ +You are the synthesis layer for a Paperclip SQL aggregate query. You are given the user's +question, the exact SQL query that was run against Paperclip's literature database, and its +result table (or a note that it returned no rows). + +Rules: +- FORMAT: one short direct answer, numerically grounded, using ONLY the values in the result + table — no bullet points, no headers. Do not estimate, round misleadingly, or add facts not + present in the results. +- Results may come back split across multiple rows (e.g. one row per underlying source shard) + rather than one pre-combined total. If the user asked for a single total, combine the + relevant rows yourself (e.g. sum a count column) and say so plainly, e.g. "7,726,938 PMC + papers total". If the split itself is informative (a per-year or per-journal breakdown), keep + it broken out. +- This is a direct database query, not a literature synthesis — do NOT invent per-paper + citations and do NOT write a `References:` section; there is nothing to cite here. Instead, + end with a single line in exactly this form: `Query: <the SQL query, verbatim>` — this is the + ONLY structured part of the output, mirroring how the literature-synthesis path ends with + References. +- If the result table is empty, say plainly that the query returned no matching rows (still + end with the `Query:` line). Do not speculate about why, and do not fall back to unsupported + claims.""" + + +PAPERCLIP_DEPTH_EVAL_SYSTEM_PROMPT = """\ +You judge whether a generated literature answer is deep enough for the user's question, given +that the only escalation available is re-fetching the papers with FULL TEXT (the current +answer may be abstracts-only). + +Return the depth-evaluation schema: +- sufficient=True if the answer is scientifically substantive — names specific entities, + describes mechanisms or concrete findings, and cites papers. In that case set missing=null. +- sufficient=False only if the answer is a vague restatement of the question with references + attached but no real biology, AND fetching full paper body text would plausibly fix that + (e.g. the question asks for a mechanism/method/quantitative result the abstracts don't + contain). Put a short gap note in `missing`. +- Do NOT ask for full text when the answer is already substantive, or when the gap is a lack + of relevant papers (more depth won't help there).""" + + +__all__ = [ + "PAPERCLIP_ROUTER_SYSTEM_PROMPT", + "PAPERCLIP_SYNTHESIZE_SYSTEM_PROMPT", + "PAPERCLIP_SQL_SYNTHESIZE_SYSTEM_PROMPT", + "PAPERCLIP_DEPTH_EVAL_SYSTEM_PROMPT", +] diff --git a/crossbar_llm/paperclip_tools/schemas.py b/crossbar_llm/paperclip_tools/schemas.py new file mode 100644 index 0000000..869277b --- /dev/null +++ b/crossbar_llm/paperclip_tools/schemas.py @@ -0,0 +1,360 @@ +"""Pydantic schemas + graph state for the Paperclip LangGraph. + +Kept separate from `schemas.py` (which is PubTator3-specific and imports the +PubTator3 client models) so the Paperclip module stays self-contained. The two +tools deliberately have different internals but the SAME external contract — +`question (+ state) in -> {final_answer, citations, warnings, usage} out` — so +the future top-level graph can dispatch to either interchangeably. +""" +from __future__ import annotations + +from typing import Awaitable, Callable, Literal, TypedDict + +from pydantic import BaseModel, Field + +from crossbar_llm.paperclip_tools.adapter import ( + PaperHit, + PaperMeta, + PaperclipSectionName, +) + + +PaperclipQuestionType = Literal[ + "keyword_search", + "list_breadth", + "full_text_depth", + "sql_aggregate", + "analogical_search", + "out_of_scope", +] + +# Mirror of `PaperclipSource` but spelled out here so the router LLM schema is +# self-documenting. Unlike the MCP text-parsing path (which requires an +# explicit `-s`), the REST path supports a genuine unscoped/broad search — +# so `source=None` is now a valid, in fact the DEFAULT, choice (see +# `PaperclipRouterDecision.source`). +PaperclipSourceChoice = Literal[ + "pmc", + "biorxiv", + "medrxiv", + "arxiv", + "fda", + "trials", + "proteins", + "pdb", + "chembl", +] + + +class PaperclipRouterDecision(BaseModel): + question_type: PaperclipQuestionType = Field( + ..., + description=( + "keyword_search: general biomedical literature question answered by " + "retrieving a handful of relevant papers (the default, and Paperclip's " + "core strength).\n" + "list_breadth: 'list all / enumerate / which drugs/genes/trials ...' " + "questions that want broad coverage; the pipeline widens the result " + "limit.\n" + "full_text_depth: the user explicitly wants mechanisms, methods, " + "results, protocols, or paragraph-level detail beyond abstracts; the " + "pipeline fetches full paper body text.\n" + "sql_aggregate: the question asks for a COUNT, RANKING, or aggregate " + "ACROSS MANY PAPERS, over their bibliographic metadata (how many " + "papers/by year/by journal/by source/by author) — answered with a " + "direct SQL query instead of retrieval+synthesis. Fill `sql_query`. " + "Two hard exclusions:\n" + " (a) NOT 'how many papers discuss/mention/are about X' — that needs " + "free-text/semantic matching over abstracts, which this route cannot " + "safely do (see `sql_query`'s description); use keyword_search/" + "list_breadth.\n" + " (b) NOT a lookup of PROPERTIES OF ONE NAMED RECORD — a protein's " + "sequence length, a PDB accession, a drug's approval status or label " + "warnings, a trial's enrollment or phase. Nothing is being aggregated " + "there, and the ONLY SQL table is `documents` (papers): it holds no " + "protein, trial, or drug-label data, so such a query cannot run at " + "all. Those are keyword_search.\n" + "analogical_search: RARE. ONLY when the user explicitly asks what OTHER " + "RESEARCH FIELDS (not other diseases) use a similar method/technique for an " + "analogous problem — crossing fields like biology vs. physics vs. NLP, not " + "crossing diseases within biomedicine ('pathways shared between diabetes and " + "lymphoma' is keyword_search, NOT this). A question naming a specific drug/" + "gene/disease/protein is virtually never this route. Fill `analogical_query`. " + "WHEN IN DOUBT, do NOT use it — default to keyword_search.\n" + "out_of_scope: Paperclip cannot help — non-biomedical questions, or " + "EXPLICIT knowledge-graph traversal (shortest path between X and Y, " + "k-hop neighbours). A question that merely CHAINS through an unnamed " + "intermediate ('diseases related to a gene/protein associated with drug " + "X') is NOT out_of_scope — the literature states such associations " + "directly, so route it as list_breadth and see `search_query` for how " + "to phrase it." + ), + ) + source: PaperclipSourceChoice | None = Field( + None, + description=( + "Which Paperclip corpus to search — leave null (the DEFAULT) for a " + "broad, unscoped search across the general literature corpora; only set " + "it when the question specifically needs a narrow corpus:\n" + "- null (DEFAULT): general biomedical literature questions — " + "mechanisms, protein domains/functions/interactions, drug targets, " + "pathways, findings. Searches broadly rather than guessing one corpus. " + "A question merely MENTIONING a protein does NOT mean 'proteins'.\n" + "- pmc / biorxiv / medrxiv / arxiv: set explicitly only when the " + "question specifically wants ONE of these corpora (e.g. 'preprint " + "work' -> biorxiv/medrxiv/arxiv); otherwise leave source null.\n" + "- fda: ONLY for explicitly REGULATORY questions — approval status, " + "boxed/label warnings, FDA-approved indications, regulatory history. Do " + "NOT use it for what a drug IS or CONTAINS, its components/composition, " + "targets, or uses (e.g. 'which drugs are in LONSURF?', 'what is drug X " + "made of?') — those are literature facts, leave source null.\n" + "- trials: ONLY for explicit clinical-trial-registry questions (a " + "specific NCT trial, enrollment, trial phase/status). General 'is drug X " + "used for disease Y' questions are literature -> leave source null.\n" + "- proteins / pdb / chembl: RECORD corpora — UniProt/PDB/ChEMBL " + "entries holding database FIELDS, with no abstracts and no prose. " + "They can only answer a question whose answer IS one of those fields: " + "a sequence length, an accession, an organism, a structure id, a " + "ChEMBL bioactivity value. Everything a PAPER reports rather than a " + "record lists is literature — which domain binds what, how a complex " + "assembles, what a protein does, what regulates it, which diseases, " + "drugs or phenotypes relate to it. The question containing the words " + "'protein', 'gene', 'domain', 'complex' or a protein name is NOT a " + "reason to set this. TEST: name the exact record field that answers " + "the question. If you cannot, leave source null.\n" + " These corpora are reached ONLY by looking up a protein you can " + "already NAME — they match on the name text and cannot filter or " + "enumerate by organism, annotation, or disease. So a question that " + "asks WHICH records satisfy a condition ('which proteins in mouse " + "are annotated with...', 'which orthologs of ALS proteins...') is " + "NOT answerable here even though organism and annotation are real " + "record fields — you would have to know the answer to write the " + "lookup. Those are literature: leave source null. Name matching is " + "also blind to abbreviations — searching 'ALS' returns acetolactate " + "synthase, not the ALS disease proteins.\n" + "WHEN IN DOUBT, leave source null — broad search is the right answer " + "for the large majority of biomedical questions." + ), + ) + search_query: str = Field( + ..., + description=( + "Focused search query, 2-6 keywords, built in two steps.\n" + " STEP 1 — identify the ANSWER TYPE: the kind of thing the user wants " + "back (drugs? genes? side effects? pathways?). It is the noun right " + "after 'which'/'what'/'name all'.\n" + " STEP 2 — write the query that retrieves papers ABOUT that answer " + "type, as a reader would phrase the topic. Keep the subject the " + "question is anchored on; drop every intermediate entity the question " + "only routes THROUGH.\n" + " Worked example: 'Which drugs target proteins associated with " + "Alzheimer disease?' -> answer type is drugs, subject is Alzheimer " + "disease, 'proteins' is only the link between them -> query 'drugs " + "approved for Alzheimer disease'. Keeping the intermediate retrieves " + "papers about protein targets, which name no drugs at all. Longer " + "chains collapse the same way: 'diseases related to a gene associated " + "with drug X' -> 'X associated diseases'.\n" + " Do NOT paste the whole sentence — that belongs in `map_question` " + "(a full extraction question, not a retrieval query; see its own " + "description). Topic queries with no chain stay as they are, e.g. " + "'BTK inhibitor chronic lymphocytic leukemia', 'metformin mechanism " + "of action'. " + "This is filled for ALL in-scope routes — including sql_aggregate, where " + "it's the fallback keyword search used if the SQL query fails (bad " + "query, timeout), and analogical_search, where it's the fallback keyword " + "search used if the analogical search returns zero hits — and is also " + "the general zero-result fallback query. Leave a short topic phrase even " + "for out_of_scope." + ), + ) + analogical_query: str | None = Field( + None, + description=( + "Filled ONLY when question_type=analogical_search (null otherwise): a " + "1-2 sentence description of the underlying METHOD/PROBLEM PATTERN, never " + "keywords (keywords defeat analogical ranking — returns topical matches, " + "not cross-domain analogies). E.g. 'correcting for systematic " + "under-reporting when the missingness mechanism is unknown', not 'missing " + "data bias'." + ), + ) + map_question: str = Field( + ..., + description=( + "The question asked to each retrieved paper individually via Paperclip's " + "`map` (a per-paper full-text reader) — DIFFERENT from search_query " + "(keywords for retrieval): this is a full, specific question for " + "extraction. Be concrete and enumerate every field you want pulled out, " + "e.g. NOT 'summarize this paper' but 'What delivery vector or inhibitor " + "mechanism was used, what cell type or population was studied, and what " + "was the reported efficacy or outcome?'. A vague question yields vague " + "per-paper answers and hurts the found/not-found signal each paper is " + "judged on. If the user's own question is already this specific, you may " + "reuse it near-verbatim; if it's broad or terse ('mechanism of X?'), " + "expand it into the concrete sub-questions that would actually answer " + "it. Filled for ALL in-scope routes, including sql_aggregate — it's " + "unused if the SQL query succeeds (nothing to map over), but ready in " + "case it falls back to a literature search." + ), + ) + sql_query: str | None = Field( + None, + description=( + "A single read-only SQL SELECT statement, filled ONLY when " + "question_type=sql_aggregate (null otherwise). Runs against a " + "`documents` table: id, title, doi, authors, source ('biorxiv'|" + "'medrxiv'|'pmc'|'arxiv'), abstract_text, pub_date, journal_title, " + "article_type, pmid, keywords (JSONB), categories (JSONB), pub_year " + "(INT), created_at. journal_title/article_type/pmid/keywords/" + "categories/pub_year are PMC-only (NULL for biorxiv/medrxiv/arxiv) — " + "filter with `source = 'pmc'` when using them.\n" + "CRITICAL constraints (server-enforced, confirmed live):\n" + "- SELECT only. No writes, no other tables.\n" + "- `documents` is NOT one unified table — it's split by `source`. " + "Use the `source` field above (e.g. 'pmc') to scope which shard(s) " + "get queried; a bare `WHERE source = 'x'` clause without also setting " + "`source` above only filters WITHIN whatever shard(s) were already " + "selected, and will silently return zero rows if that shard wasn't " + "reached. `source='trials'`/`'proteins'` are NOT valid here at all — " + "never write sql_aggregate for trial or protein counts.\n" + "- Free-text pattern matching (`abstract_text ILIKE '%keyword%'`) over " + "the full pmc/arxiv tables (millions of rows, unindexed for this) " + "reliably times out (15s server limit). Only use ILIKE on short " + "structured fields like `authors` or `title`, never `abstract_text`, " + "and always pair with a `source`/`pub_year`/other filter to narrow the " + "scanned rows first." + ), + ) + full_text: bool = Field( + False, + description=( + "Whether to fetch full paper body text (True) or abstracts only " + "(False). DEFAULT FALSE — abstracts answer most questions cheaply. Set " + "True only for full_text_depth questions (mechanisms, methods, results, " + "paragraph detail)." + ), + ) + sections: list[PaperclipSectionName] | None = Field( + None, + description=( + "OPTIONAL body-section filter, applied ONLY when full_text=True. When " + "set, only these sections are pulled from each paper instead of the " + "whole body — cheaper and more focused. Pick the section(s) the " + "question targets: methods/protocol/assay -> ['methods']; " + "findings/data/numbers -> ['results']; mechanism/interpretation -> " + "['discussion']; takeaways -> ['conclusion']. Leave null to pull the " + "whole body. Has NO effect when full_text=False." + ), + ) + year: str | None = Field( + None, + description=( + "Optional publication-year filter passed to search (e.g. '2023' or " + "'2020-2024') when the user constrains recency. Usually null." + ), + ) + rationale: str = Field("", description="One-sentence justification for the classification.") + + +class PaperclipDepthEvaluation(BaseModel): + sufficient: bool = Field( + ..., + description=( + "True if the answer is scientifically substantive for the question — " + "names specific entities, describes mechanisms/findings, cites papers. " + "False if it reads like a restatement of the question with references " + "attached but no real biology." + ), + ) + missing: str | None = Field( + None, + description=( + "If sufficient=False, a short note on the gap — e.g. 'no mechanism " + "described', 'no quantitative results'. Null when sufficient=True. The " + "only escalation lever is fetching full text, so this drives whether we " + "re-fetch with full_text=True." + ), + ) + rationale: str = Field("", description="One-sentence justification for the verdict.") + + +class Citation(BaseModel): + """A resolved, citable reference threaded into the final answer. Every field + that can ground provenance (doi/pmid/url) is carried so the top-level graph / + API can render a proper reference list.""" + ref_num: int + doc_id: str + title: str = "" + authors: str = "" + journal: str | None = None + year: int | None = None + doi: str | None = None + pmid: str | None = None + url: str = "" + + +class PaperContext(BaseModel): + """Assembled per-paper context passed to the synthesizer: always the meta + (title/abstract + citable ids), plus body text when full_text depth is on.""" + doc_id: str + meta: PaperMeta + body: str | None = None + + +class PaperclipState(TypedDict, total=False): + question: str + chat_history: list + + # Router decision. `source=None`/absent means "broad/unscoped search" — + # see `PaperclipRouterDecision.source`. + question_type: PaperclipQuestionType + source: str | None + search_query: str + analogical_query: str | None + map_question: str + full_text: bool + sections: list[str] | None + year: str | None + rationale: str + + # Retrieval. + hits: list[PaperHit] + search_id: str | None + queries_used: list[str] + documents: list[PaperContext] + citations: list[Citation] + + # SQL aggregate route (question_type == "sql_aggregate"). sql_error set + # means the query failed and the graph fell back to search_query instead. + sql_query: str | None + sql_columns: list[str] + sql_rows: list[dict] + sql_error: str | None + + # Synthesis + depth loop. + final_answer: str | None + depth_sufficient: bool + depth_missing: str | None + depth_skip_reason: str | None + refinement_attempted: bool + + warnings: list[str] + + +PaperclipRouterFn = Callable[[str], Awaitable[PaperclipRouterDecision]] +PaperclipSynthesizerFn = Callable[[PaperclipState], Awaitable[str]] +PaperclipEvaluatorFn = Callable[[PaperclipState], Awaitable[PaperclipDepthEvaluation]] + + +__all__ = [ + "PaperclipQuestionType", + "PaperclipSourceChoice", + "PaperclipRouterDecision", + "PaperclipDepthEvaluation", + "Citation", + "PaperContext", + "PaperclipState", + "PaperclipRouterFn", + "PaperclipSynthesizerFn", + "PaperclipEvaluatorFn", +] diff --git a/crossbar_llm/paperclip_tools/structured_output.py b/crossbar_llm/paperclip_tools/structured_output.py new file mode 100644 index 0000000..c95da77 --- /dev/null +++ b/crossbar_llm/paperclip_tools/structured_output.py @@ -0,0 +1,108 @@ +"""Structured-output helpers shared by this agent's LLM-bound nodes. + +Some providers return None instead of making the schema tool call when the +model answers in prose, so every structured call falls back to plain JSON and +validates locally against the same Pydantic schema. + +Deliberately duplicated in each tool package rather than imported across them: +the two agents are developed independently and neither should break when the +other changes. +""" +from __future__ import annotations + +import json +from typing import Any, TypeVar + +from langchain_core.language_models import BaseChatModel +from langchain_core.prompts import ChatPromptTemplate, HumanMessagePromptTemplate +from pydantic import BaseModel + +StructuredModel = TypeVar("StructuredModel", bound=BaseModel) + + +def _message_content_to_text(message: Any) -> str: + content = getattr(message, "content", message) + if isinstance(content, str): + return content + if isinstance(content, list): + parts: list[str] = [] + for item in content: + if isinstance(item, str): + parts.append(item) + elif isinstance(item, dict) and isinstance(item.get("text"), str): + parts.append(item["text"]) + else: + parts.append(str(item)) + return "\n".join(parts) + return str(content) + + +def _extract_json_object(text: str) -> dict[str, Any]: + stripped = text.strip() + if stripped.startswith("```"): + lines = stripped.splitlines() + if lines and lines[0].startswith("```"): + lines = lines[1:] + if lines and lines[-1].strip() == "```": + lines = lines[:-1] + stripped = "\n".join(lines).strip() + + decoder = json.JSONDecoder() + for idx, char in enumerate(stripped): + if char != "{": + continue + try: + obj, _ = decoder.raw_decode(stripped[idx:]) + except json.JSONDecodeError: + continue + if isinstance(obj, dict): + return obj + raise ValueError("model response did not contain a JSON object") + + +async def _ainvoke_structured_with_json_fallback( + *, + chat_model: BaseChatModel, + prompt: ChatPromptTemplate, + schema: type[StructuredModel], + values: dict[str, Any], + json_instruction: str, +) -> tuple[StructuredModel, bool]: + """Use provider structured output first, then retry as plain JSON. + + Some providers expose weak tool-calling semantics: LangChain can return + None when the model answers in prose instead of making the schema tool + call. The JSON retry keeps those providers useful while still validating + locally with the exact same Pydantic schema. + """ + structured_error: Exception | None = None + try: + chain = prompt | chat_model.with_structured_output(schema) + parsed = await chain.ainvoke(values) + if parsed is not None: + if isinstance(parsed, schema): + return parsed, False + return schema.model_validate(parsed), False + structured_error = ValueError( + "structured-output returned None (schema-coercion failed)" + ) + except Exception as e: + structured_error = e + + json_prompt = prompt + HumanMessagePromptTemplate.from_template(json_instruction) + try: + msg = await (json_prompt | chat_model).ainvoke(values) + data = _extract_json_object(_message_content_to_text(msg)) + return schema.model_validate(data), True + except Exception as json_error: + raise ValueError( + "structured-output failed and JSON fallback failed: " + f"{structured_error}; {json_error}" + ) from json_error + + +__all__ = [ + "_ainvoke_structured_with_json_fallback", + "_extract_json_object", + "_message_content_to_text", +] diff --git a/crossbar_llm/paperclip_tools/tests/__init__.py b/crossbar_llm/paperclip_tools/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/crossbar_llm/paperclip_tools/tests/conftest.py b/crossbar_llm/paperclip_tools/tests/conftest.py new file mode 100644 index 0000000..124f4a0 --- /dev/null +++ b/crossbar_llm/paperclip_tools/tests/conftest.py @@ -0,0 +1,18 @@ +"""Suite-wide pytest configuration for the Paperclip tests.""" + + +def pytest_configure(config): + """Reassert pytest-asyncio's auto mode, which this suite requires. + + `asyncio_mode = auto` lives in this directory's pytest.ini, but pytest only + honours that file when the suite is the sole command-line argument. Name it + alongside another directory and their common ancestor wins instead, leaving + every async test here in strict mode and unmarked, so all of them error. + Setting it here keeps the requirement with the package rather than with how + pytest happened to be invoked. + """ + if config.getoption("asyncio_mode", None) != "auto": + config.option.asyncio_mode = "auto" + config.addinivalue_line( + "markers", "live: hits the real upstream service; needs credentials" + ) diff --git a/crossbar_llm/paperclip_tools/tests/pytest.ini b/crossbar_llm/paperclip_tools/tests/pytest.ini new file mode 100644 index 0000000..661dc87 --- /dev/null +++ b/crossbar_llm/paperclip_tools/tests/pytest.ini @@ -0,0 +1,6 @@ +# Scoped so this agent's suite runs without touching the project's root config: +# pytest -c crossbar_llm/paperclip_tools/tests/pytest.ini crossbar_llm/paperclip_tools/tests +[pytest] +asyncio_mode = auto +markers = + live: hits the real upstream service; needs credentials diff --git a/crossbar_llm/paperclip_tools/tests/test_adapter.py b/crossbar_llm/paperclip_tools/tests/test_adapter.py new file mode 100644 index 0000000..c7c1cc8 --- /dev/null +++ b/crossbar_llm/paperclip_tools/tests/test_adapter.py @@ -0,0 +1,1052 @@ +"""Unit tests for the Paperclip adapter's pure parsing/helper logic. + +These run fully offline — they exercise the text/JSON parsers and REST-JSON +mapping against captured real-server output. The live round-trip (REST + +MCP fallback) is covered separately in test_paperclip_live.py (key-gated). +""" +from __future__ import annotations + +import pytest + +from crossbar_llm.paperclip_tools.nodes import citation_url, format_line_anchor +from crossbar_llm.paperclip_tools.adapter import ( + PaperclipError, + _doc_root, + _extract_json_object, + _map_extract_from_text, + _paper_hit_from_rest, + _parse_map_results, + _parse_protein_search, + _parse_search, + _parse_section_listing, + _parse_sql_output, + _select_section_files, + _shell_quote, + infer_source_from_doc_id, +) + + +def _patch_rest(monkeypatch, handler): + """Install `handler(url, json, headers, timeout) -> response` as the REST + layer. The adapter posts through a pooled `httpx.AsyncClient` (see + `PaperclipAdapter._rest_client`), so the seam is `AsyncClient.post`, not a + module-level function. `handler` may raise to simulate a transport failure. + """ + async def fake_post(self, url, *, json, headers, timeout): + return handler(url, json, headers, timeout) + + monkeypatch.setattr("httpx.AsyncClient.post", fake_post) + + + +# Captured verbatim from `search -s pmc "..." -n 3` against the live server. +SEARCH_OUTPUT = """Found 3 papers [s_c7326471] + + 1. Metformin: Beyond Type 2 Diabetes Mellitus + Rahnuma Ahmad, Mainul Haque + PMC11486535 · PMC · 2024-10-17 + https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11486535/ + "This review examines metformin's effects on non-alcoholic fatty liver disease." + + 2. Molecular mechanism of action of metformin: old or new insights? + Graham Rena, Ewan R. Pearson, Kei Sakamoto + PMC3737434 · biomedrxiv · 2013-07-09 + https://doi.org/10.1007/s00125-013-2991-0 + "This review summarizes recent research on metformin's molecular mechanisms." + + 3. Adipsin and Leptin Levels in Type 2 Diabetic Patients + Sura Khalid Mohammed, Zainab Haitham Fathi + PMC11729846 · PMC · 2024-01-01 + https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11729846/ + "The study compared adipsin and leptin levels in type 2 diabetic patients." + +[357ms, saved to s_c7326471] +""" + +META_OUTPUT = """{ + "document_id": "PMC11486535", + "pmc_id": "PMC11486535", + "pmid": "39421288", + "doi": "10.7759/cureus.71730", + "title": "Metformin: Beyond Type 2 Diabetes Mellitus", + "authors": "Rahnuma Ahmad, Mainul Haque", + "abstract": "Metformin was developed from an offshoot of Guanidine.", + "source": "pmc", + "journal": "Cureus", + "pub_year": 2024, + "pub_date": "2024-10-17" +} +[75ms] +""" + + +def test_parse_search_extracts_all_hits(): + hits = _parse_search(SEARCH_OUTPUT) + assert [h.doc_id for h in hits] == ["PMC11486535", "PMC3737434", "PMC11729846"] + + +def test_parse_search_fields(): + hits = _parse_search(SEARCH_OUTPUT) + h0 = hits[0] + assert h0.title == "Metformin: Beyond Type 2 Diabetes Mellitus" + assert h0.authors == "Rahnuma Ahmad, Mainul Haque" + assert h0.source == "PMC" + assert h0.date == "2024-10-17" + assert h0.url == "https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11486535/" + assert h0.snippet.startswith("This review examines") + # The "Found N papers" header line must not leak in as a hit. + assert all("Found" not in h.title for h in hits) + + +def test_parse_search_empty(): + assert _parse_search("No results found.\n[12ms]") == [] + + +def test_extract_json_object_strips_timing_footer(): + data = _extract_json_object(META_OUTPUT) + assert data["doi"] == "10.7759/cureus.71730" + assert data["pmid"] == "39421288" + + +def test_extract_json_object_raises_on_non_json(): + with pytest.raises(PaperclipError): + _extract_json_object("just some prose, no object\n[5ms]") + + +def test_shell_quote_uses_single_quotes(): + """vsh mis-parses double-quoted args that contain both an inner quote and + a newline (`parse error: No closing quotation`), so we single-quote.""" + assert _shell_quote("plain") == "'plain'" + assert _shell_quote('a "b" c') == "'a \"b\" c'" + # apostrophes must survive - "Alzheimer's disease" is an ordinary query + assert _shell_quote("Alzheimer's") == "'Alzheimer'\\''s'" + # the shape that actually broke: newline + embedded double quotes + q = _shell_quote('Q?\n\n{"answer": "x"}') + assert q.startswith("'") and q.endswith("'") + assert '\\"' not in q # no backslash-escaped quotes inside single quotes + + +@pytest.mark.parametrize( + "doc_id,expected_ns", + [ + ("PMC123", "papers"), + ("bio_abc", "papers"), + ("fda_xyz", "fda"), + ("NCT01234567", "trials"), + ("tri_555", "trials"), + ], +) +def test_citation_url_namespace(doc_id, expected_ns): + assert citation_url(doc_id) == f"https://citations.gxl.ai/{expected_ns}/{doc_id}" + + +# Captured verbatim from `ls /papers/<id>/sections/` against the live server. +SECTIONS_LISTING = ( + "Title.lines Metadata.lines Abstract.lines 1. Introduction.lines " + "2. Materials and Methods.lines 2.1. Study Design.lines " + "2.10. Statistical Analysis.lines 3. Results.lines " + "3.4.1. Renal Function and Uric Acid Levels.lines 4. Discussion.lines " + "5. Conclusions.lines References.lines\n" + " (read-only — use /.gxl/ for writable storage)\n[69ms]" +) + + +def test_parse_section_listing(): + names = _parse_section_listing(SECTIONS_LISTING) + assert "Abstract" in names + assert "2. Materials and Methods" in names + assert "4. Discussion" in names + # The read-only note and timing footer must not leak in as sections. + assert not any("read-only" in n for n in names) + assert not any("ms]" in n for n in names) + + +def test_select_sections_pulls_subsections_by_number(): + names = _parse_section_listing(SECTIONS_LISTING) + methods = _select_section_files(names, ["methods"]) + # Section 2 top-level plus its 2.x subsections. + assert "2. Materials and Methods" in methods + assert "2.1. Study Design" in methods + assert "2.10. Statistical Analysis" in methods + # No cross-contamination from other numbered sections. + assert "3. Results" not in methods + + +def test_select_sections_multiple_keywords(): + names = _parse_section_listing(SECTIONS_LISTING) + picked = _select_section_files(names, ["discussion", "conclusion"]) + assert picked == ["4. Discussion", "5. Conclusions"] + + +def test_select_sections_no_match_returns_empty(): + names = _parse_section_listing(SECTIONS_LISTING) + assert _select_section_files(names, ["methods"]) != [] + assert _select_section_files(["Foo", "Bar"], ["methods"]) == [] + + +# --- proteins corpus parsing (different listing shape) --------------------- +PROTEIN_SEARCH = """Found 3 proteins [s_38a63397] + + 1. KAT7 - Histone acetyltransferase KAT7 + O95251 + Homo sapiens · 611 aa + + 2. EP300 - Histone acetyltransferase p300 + Q09472 + Homo sapiens · 2414 aa + + 3. KAT6B - Histone acetyltransferase KAT6B + Q8WYB5 + Homo sapiens · 2073 aa + +[84ms, saved to s_38a63397] +""" + + +def test_parse_protein_search(): + hits = _parse_protein_search(PROTEIN_SEARCH) + assert [h.doc_id for h in hits] == ["O95251", "Q09472", "Q8WYB5"] + assert hits[0].title == "KAT7 - Histone acetyltransferase KAT7" + assert hits[0].source == "proteins" + assert "611 aa" in hits[0].snippet + + +def test_doc_root_maps_source_to_vfs(): + assert _doc_root("pmc") == "papers" + assert _doc_root("abstracts") == "papers" + assert _doc_root("proteins") == "proteins" + assert _doc_root("uniprot") == "proteins" + assert _doc_root("fda/us") == "fda" + assert _doc_root("trials") == "trials" + assert _doc_root(None) == "papers" + + +# --- map full-results parsing --------------------------------------------- +MAP_RESULTS = """Map results: 2/2 tasks succeeded in 8846ms +Results ID: m_4a22ca90 +Query: What is the mechanism? + +--- [1/2] [success] Metformin Improves Mitochondrial Respiratory Activity --- + doc_id: PMC6866677 + Metformin activates AMPK, which phosphorylates Mff (L34, L815-L818), + recruiting Drp1 to drive mitochondrial fission. + +--- [2/2] [failed] Some Other Paper --- + doc_id: PMC12511219 + The mechanism is not fully clarified in this paper (L20). +""" + + +def test_parse_map_results(): + extracts = _parse_map_results(MAP_RESULTS) + assert [e.doc_id for e in extracts] == ["PMC6866677", "PMC12511219"] + assert extracts[0].success is True + assert "phosphorylates Mff" in extracts[0].text + assert extracts[1].success is False + # Prose (no --output_schema) parses with no structured data — the + # backward-compatible path. + assert extracts[0].data is None + + +# Captured verbatim from a live `map --output_schema '{"answer":..., "found":...}'` +# call (docs/paperclip_rest_endpoint_findings.md capability-idea follow-up). +MAP_RESULTS_STRUCTURED = """Map results: 2/2 tasks succeeded in 3794ms +Results ID: m_9709a130 +Query: What delivery vector or inhibitor mechanism was reported? + +--- [1/2] [success] Impact of the clinically approved BTK inhibitors --- + doc_id: PMC11677227 + {"answer": "All are covalent active site inhibitors, with the exception of the reversible active site inhibitor Pirtobrutinib.", "found": true, "_citations": [{"field": "answer", "line": 9, "content": "All are covalent active site inhibitors."}]} + +--- [2/2] [success] Treatment of relapsed/refractory CLL with Zanubrutinib --- + doc_id: PMC11039307 + {"answer": "Not found", "found": false, "_citations": []} +""" + + +def test_parse_map_results_structured_output_schema(): + extracts = _parse_map_results(MAP_RESULTS_STRUCTURED) + assert [e.doc_id for e in extracts] == ["PMC11677227", "PMC11039307"] + + found = extracts[0] + assert found.text == ( + "All are covalent active site inhibitors, with the exception of the " + "reversible active site inhibitor Pirtobrutinib." + ) + assert found.data is not None + assert found.data["found"] is True + assert found.data["_citations"][0]["line"] == 9 + assert found.found is True + assert found.citation_lines == [9] + + not_found = extracts[1] + assert not_found.text == "Not found" + assert not_found.data["found"] is False + assert not_found.found is False + assert not_found.citation_lines == [] + + +# Captured verbatim from a live `map --output_schema` call where the server +# echoed the schema's own `properties` structure back with `answer`/`found` +# nested inside `answer` instead of flat — confirmed live, not rare (a whole +# 3-paper batch returned this shape in one observed run). Regression fixture +# for the crash this used to cause (dict assigned to MapExtract.text) and the +# found-detection bug (a bare `.data.get("found", True)` defaulted to True +# here since "found" isn't a top-level key in this shape). +MAP_RESULTS_NESTED_MALFORMED = """Map results: 2/2 tasks succeeded in 2749ms +Results ID: m_c9a73dbe +Query: What is the molecular mechanism of action of metformin? + +--- [1/2] [success] Understanding the action mechanisms of metformin --- + doc_id: PMC11010946 + {"answer": {"answer": "Not found", "found": false}, "_citations": []} + +--- [2/2] [success] Metformin's multifaceted role in colorectal cancer --- + doc_id: PMC12595195 + {"answer": {"answer": "At the molecular level, metformin activates AMPK.", "found": true}, "_citations": [{"field": "answer", "line": 24, "content": "At molecular level, metformin activates AMPK and inhibits cell proliferation."}]} +""" + + +def test_parse_map_results_handles_nested_malformed_shape(): + """Must not crash (the dict-into-str bug), and must correctly detect + found=False/True even though neither is a top-level key in this shape.""" + extracts = _parse_map_results(MAP_RESULTS_NESTED_MALFORMED) + assert [e.doc_id for e in extracts] == ["PMC11010946", "PMC12595195"] + + not_found = extracts[0] + assert not_found.text == "Not found" + assert not_found.found is False + assert not_found.citation_lines == [] + + found = extracts[1] + assert found.text == "At the molecular level, metformin activates AMPK." + assert found.found is True + assert found.citation_lines == [24] + + +def test_map_extract_from_text_found_none_when_contract_not_honored(): + """Prose (the model ignored `_MAP_JSON_CONTRACT`) leaves `found` unknown, + not False — a paper with no found signal must still be treated as usable + evidence (see assemble_context_node). Inline line refs are still salvaged. + """ + extract = _map_extract_from_text("PMC1", "Metformin activates AMPK (L34).", success=True) + assert extract.found is None + assert extract.citation_lines == [34] + + +def test_map_extract_from_text_falls_back_to_prose_on_bad_json(): + """A block starting with `{` but not valid/schema-shaped JSON must not + crash parsing — just treated as prose, same as before schemas existed.""" + extract = _map_extract_from_text("PMC1", "{not actually json", success=True) + assert extract.text == "{not actually json" + assert extract.data is None + + +def test_map_extract_from_text_requires_answer_key(): + """JSON without an `answer` key (e.g. a genuinely different schema) is + not assumed to match our shape — falls back to raw text rather than + guessing at a field name.""" + extract = _map_extract_from_text("PMC1", '{"other_field": "x"}', success=True) + assert extract.data is None + assert extract.text == '{"other_field": "x"}' + + +def test_citation_url_proteins_links_to_uniprot(): + assert citation_url("Q92794", "proteins") == "https://www.uniprot.org/uniprotkb/Q92794/entry" + # paper ids keep the gxl namespace regardless. + assert citation_url("PMC123", "pmc") == "https://citations.gxl.ai/papers/PMC123" + + +def test_citation_url_appends_line_anchor(): + assert ( + citation_url("PMC123", "pmc", line_anchor="L45") + == "https://citations.gxl.ai/papers/PMC123#L45" + ) + + +def test_citation_url_no_anchor_when_none_given(): + assert citation_url("PMC123", "pmc", line_anchor=None) == "https://citations.gxl.ai/papers/PMC123" + assert citation_url("PMC123", "pmc", line_anchor="") == "https://citations.gxl.ai/papers/PMC123" + + +def test_citation_url_line_anchor_ignored_for_uniprot(): + """Proteins have no line-numbered content.lines — a line anchor makes no + sense there and must not leak into the UniProt URL.""" + assert ( + citation_url("Q92794", "proteins", line_anchor="L45") + == "https://www.uniprot.org/uniprotkb/Q92794/entry" + ) + + +@pytest.mark.parametrize("lines,expected", [ + ([], ""), + ([45], "L45"), + ([45, 46, 47], "L45-L47"), + ([210, 45, 120], "L45,120,210"), # non-contiguous: sorted, comma list + ([45, 45], "L45"), # dedup +]) +def test_format_line_anchor(lines, expected): + assert format_line_anchor(lines) == expected + + +# --- REST migration fixes (docs/paperclip_rest_endpoint_findings.md §6) ---- + +def test_arxiv_doc_id_not_truncated(): + """Regression test for the arXiv ID truncation bug: `arx_2002.06616` was + being parsed as `arx_2002` because the old regex was a hex-digit class + that stopped at the literal `.` in the ID.""" + text = ( + "Found 1 papers [s_abc]\n\n" + " 1. A paper about transformers\n" + " Some Author\n" + " arx_2002.06616 · arXiv · 2020-02-16\n" + " https://doi.org/10.1016/example\n" + ' "a snippet"\n' + ) + hits = _parse_search(text) + assert len(hits) == 1 + assert hits[0].doc_id == "arx_2002.06616" + + +def test_pdb_and_chembl_map_to_proteins_root(): + """Regression test: `-s pdb`/`-s chembl` used to fall through to the + `"papers"` default root (not in `_SOURCE_ROOT`), routing them to the + wrong parser and silently returning zero hits.""" + assert _doc_root("pdb") == "proteins" + assert _doc_root("chembl") == "proteins" + + +def test_infer_source_from_doc_id(): + """Per-hit source inference used when a search is broad/unscoped and a + single result set mixes corpora (get_meta/get_content/citation URLs need + a per-hit VFS root, not one blanket source).""" + assert infer_source_from_doc_id("PMC12345") == "pmc" + assert infer_source_from_doc_id("bio_abc123") == "pmc" + assert infer_source_from_doc_id("med_abc123") == "pmc" + assert infer_source_from_doc_id("arx_2002.06616") == "pmc" + assert infer_source_from_doc_id("fda_abc123") == "fda" + assert infer_source_from_doc_id("tri_abc123") == "trials" + assert infer_source_from_doc_id("NCT01234567") == "trials" + assert infer_source_from_doc_id("P04637") == "proteins" + + +def test_paper_hit_from_rest_maps_structured_fields(): + """Maps the REST path's `result_data.papers[i]` dict shape (confirmed + live — docs/paperclip_rest_endpoint_findings.md §5.1) into `PaperHit`.""" + hit = _paper_hit_from_rest({ + "document_id": "PMC8261291", + "source": "biomedrxiv", + "score": 0.8721041, + "corpus": "pmc", + "backend": "qdrant", + "title": "Some title", + "tldr": "a summary", + "doi": "10.3389/fimmu.2021.687458", + "authors": "A. Author", + "pub_date": "2021-06-23", + "pub_year": 2021, + }) + assert hit.doc_id == "PMC8261291" + assert hit.title == "Some title" + assert hit.snippet == "a summary" + assert hit.score == 0.8721041 + assert hit.corpus == "pmc" + assert hit.backend == "qdrant" + assert hit.doi == "10.3389/fimmu.2021.687458" + assert hit.pub_year == 2021 + assert hit.date == "2021-06-23" + + +def test_paper_hit_from_rest_falls_back_to_abstract_snippet(): + """Some payloads use `abstract_snippet` instead of `tldr` — confirmed for + a plain pmc hit shape in the same live response.""" + hit = _paper_hit_from_rest({ + "document_id": "PMC5392013", + "title": "A paper", + "abstract_snippet": "the snippet text", + }) + assert hit.snippet == "the snippet text" + + +def test_paper_hit_from_rest_tolerates_missing_document_id(): + """Defensive: proteins/fda/trials REST field shapes are unverified (open + question in the migration plan) — must not crash on an unexpected shape.""" + hit = _paper_hit_from_rest({"accession": "P04637", "title": "TP53"}) + assert hit.doc_id == "P04637" + + +async def test_slow_command_uses_slow_timeout_on_rest(monkeypatch): + """Regression test: `map`/`ask-image` used to share the same flat 60s + timeout as every other command, despite the module's own comment noting + they can be slow — a real risk once `use_map=True` became the default. + The real gxl_paperclip SDK gives these a 300s timeout; we now match + that for the REST path, which sets it precisely per-call.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + captured = {} + + class FakeResponse: + status_code = 200 + text = "" + + def json(self): + return {"output": "", "result_id": None, "result_data": None} + + def fake_post(url, json, headers, timeout): + captured["timeout"] = timeout + return FakeResponse() + + _patch_rest(monkeypatch, fake_post) + + adapter = PaperclipAdapter(timeout_s=60.0, slow_timeout_s=300.0) + await adapter._run_rest("search", '"x" -n 5') + assert captured["timeout"] == 60.0 + + await adapter._run_rest("map", '--from s_x "q"') + assert captured["timeout"] == 300.0 + + +# --- sql routing (docs/paperclip_rest_endpoint_findings.md §12) -------------- + +# Captured verbatim from a live `sql -s fda "SELECT COUNT(*) AS n FROM documents"`. +SQL_SINGLE_COLUMN = """n +------ +217217 +(1 row, 14ms) +""" + +# Captured verbatim from a live multi-column query (no -s: spans multiple shards). +SQL_MULTI_COLUMN = """title | doi | source +-------------------------------------------------------------+---------------------------+------- +Transcriptomic Profiling of Orbital Fat Tissue and Ocular... | 10.1167/iovs.66.15.71 | pmc +Btk inhibitor ibrutinib reduces inflammatory myeloid cell... | 10.1186/s10020-018-0069-7 | pmc +(2 rows, 379ms) [bioRxiv (0 rows) + PMC (2 rows)] +""" + +SQL_ERROR = """ERR: sql: Only SELECT queries are allowed. +[exit 1] +""" + + +def test_parse_sql_output_single_column(): + result = _parse_sql_output(SQL_SINGLE_COLUMN) + assert result.columns == ["n"] + assert result.rows == [{"n": "217217"}] + + +def test_parse_sql_output_multi_column_strips_shard_footer(): + result = _parse_sql_output(SQL_MULTI_COLUMN) + assert result.columns == ["title", "doi", "source"] + assert len(result.rows) == 2 + assert result.rows[0]["doi"] == "10.1167/iovs.66.15.71" + assert result.rows[0]["source"] == "pmc" + # The "(N rows, Xms) [...]" footer must not leak in as a fake row. + assert all("rows" not in r.get("title", "") for r in result.rows) + + +def test_parse_sql_output_raises_on_server_error(): + """The server reports query errors (bad SQL, non-SELECT, statement + timeout) as `ERR: ...` text with a 200/success transport response, not a + transport-level failure — must be surfaced as PaperclipError ourselves.""" + with pytest.raises(PaperclipError, match="Only SELECT queries are allowed"): + _parse_sql_output(SQL_ERROR) + + +def test_parse_sql_output_empty_result(): + result = _parse_sql_output("(0 rows, 5ms)\n") + assert result.columns == [] + assert result.rows == [] + + +# --- filter (server-side relevance trim, REST-only) ------------------------- + +def _rest_adapter(monkeypatch, response_json): + """A PaperclipAdapter whose REST layer returns `response_json`.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + + class FakeResponse: + status_code = 200 + text = "" + + def json(self): + return response_json + + _patch_rest(monkeypatch, lambda url, json, headers, timeout: FakeResponse()) + return PaperclipAdapter() + + +async def test_filter_maps_rest_papers_to_hits(monkeypatch): + adapter = _rest_adapter(monkeypatch, { + "output": "Filtered 2 -> 1 relevant papers.", + "result_id": "s_fake", + "result_data": {"papers": [{"document_id": "PMC1", "title": "T1"}]}, + }) + result = await adapter.filter("s_fake", "relevance query") + assert result is not None + assert result.search_id == "s_fake" + assert [h.doc_id for h in result.hits] == ["PMC1"] + + +async def test_filter_raises_on_server_err(monkeypatch): + adapter = _rest_adapter(monkeypatch, { + "output": "ERR: filter: no such search id\n[exit 1]", + "result_id": None, + "result_data": None, + }) + with pytest.raises(PaperclipError, match="no such search id"): + await adapter.filter("s_bogus", "q") + + +async def test_filter_returns_none_when_rest_disabled(monkeypatch): + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_DISABLE_REST", "1") + adapter = PaperclipAdapter() + assert await adapter.filter("s_fake", "q") is None + + +async def test_filter_returns_none_on_rest_unavailable(monkeypatch): + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + + def broken_post(url, json, headers, timeout): + raise ConnectionError("refused") + + _patch_rest(monkeypatch, broken_post) + adapter = PaperclipAdapter() + assert await adapter.filter("s_fake", "q") is None + + +# --- ranking (docs/paperclip_rest_endpoint_findings.md §15) ---------------- + +async def test_search_threads_ranking_flag_into_rest_raw_args(monkeypatch): + """`ranking` must reach the REST `raw` command string as `--ranking + <value>`, and must be omitted entirely when not given (the default + `hybrid` ranking every other route relies on).""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + captured = {} + + class FakeResponse: + status_code = 200 + text = "" + + def json(self): + return {"output": "Found 0 papers [s_x]", "result_id": "s_x", "result_data": {"papers": []}} + + def fake_post(url, json, headers, timeout): + captured["raw"] = json["raw"] + return FakeResponse() + + _patch_rest(monkeypatch, fake_post) + adapter = PaperclipAdapter() + + await adapter.search("a method-description sentence", source="arxiv", ranking="analogical") + assert "--ranking analogical" in captured["raw"] + + await adapter.search("plain keywords", source="pmc") + assert "--ranking" not in captured["raw"] + + +# `meta.json` for a conference poster (PMC4043507-shaped): the server sends an +# explicit null for fields the record genuinely lacks. +META_WITH_NULLS = { + "document_id": "PMC4043507", + "title": "Autosomal dominant mutation in COL7A1 gene", + "authors": None, + "abstract": None, + "doi": "10.1186/1755-8166-7-S1-P58", + "journal": "Molecular Cytogenetics", + "pub_year": 2014, +} + + +def test_paper_meta_accepts_explicit_nulls(): + from crossbar_llm.paperclip_tools.adapter import PaperMeta + + meta = PaperMeta.model_validate(META_WITH_NULLS) + assert meta.abstract == "" + assert meta.authors == "" + # non-null fields still survive — the point is to keep the record, not blank it + assert meta.journal == "Molecular Cytogenetics" + assert meta.pub_year == 2014 + assert meta.title.startswith("Autosomal dominant") + + +def test_paper_meta_missing_keys_still_default(): + from crossbar_llm.paperclip_tools.adapter import PaperMeta + + meta = PaperMeta.model_validate({"document_id": "PMC1"}) + assert (meta.title, meta.authors, meta.abstract) == ("", "", "") + + +# Captured verbatim from a live REST `map` using _MAP_JSON_CONTRACT (no +# --output_schema, which 500s on REST). Note the inline (L..) refs and the +# absence of any `_citations` field. +MAP_CONTRACT_OUTPUT = """Map results: 5/5 tasks succeeded in 2385ms +Results ID: m_3a13abd2 + +--- [1/5] [success] The Off-Label Use of SSRIs for Sexual Behavior Management --- + doc_id: PMC12524134 + {"answer": "This paper reports the off-label use of SSRIs for managing inappropriate sexual behaviors (L8, L14-L16, L72). It also mentions premature ejaculation (L21).", "found": true} + +--- [2/5] [success] GenVarFormer: Predicting gene expression from mutations --- + doc_id: arx_2509.25573 + {"answer": "", "found": false} +""" + + +def test_parse_map_results_json_contract_without_citations_field(): + """The REST path gets no `_citations`, so line anchors must come from the + model's inline (L..) refs — otherwise every REST map citation loses its + line anchor.""" + extracts = _parse_map_results(MAP_CONTRACT_OUTPUT) + assert len(extracts) == 2 + + first = extracts[0] + assert first.found is True + assert first.text.startswith("This paper reports") + # L14-L16 expands so format_line_anchor can re-collapse contiguous runs + assert first.citation_lines == [8, 14, 15, 16, 21, 72] + + # found=False survives — assemble_context_node drops these as non-evidence + assert extracts[1].found is False + + +def test_map_citation_lines_from_text_handles_ranges_and_noise(): + from crossbar_llm.paperclip_tools.adapter import _map_citation_lines_from_text + + assert _map_citation_lines_from_text("no refs here") == [] + assert _map_citation_lines_from_text("(L5)") == [5] + assert _map_citation_lines_from_text("(L5, L9)") == [5, 9] + assert _map_citation_lines_from_text("(L20-L23)") == [20, 21, 22, 23] + assert _map_citation_lines_from_text("(L20-23)") == [20, 21, 22, 23] + # reversed range still yields the span, not an empty set + assert _map_citation_lines_from_text("(L23-L20)") == [20, 21, 22, 23] + # an over-wide range keeps endpoints instead of exploding the anchor + assert _map_citation_lines_from_text("(L10-L900)") == [10, 900] + # LEKTI / L1 cell lines etc. must not be read as line refs + assert _map_citation_lines_from_text("LEKTI protein and SPINK5") == [] + + +def test_map_citation_lines_prefers_server_citations_over_inline(): + """When Paperclip does supply `_citations` (the MCP path), it wins — it's + server-computed rather than model-emitted.""" + raw = ('{"answer": "Metformin activates AMPK (L999).", "found": true, ' + '"_citations": [{"field": "answer", "line": 24, "content": "..."}]}') + extract = _map_extract_from_text("PMC1", raw, success=True) + assert extract.citation_lines == [24] + + +@pytest.mark.parametrize("status", [401, 429]) +async def test_rest_auth_and_quota_errors_do_not_fall_back_to_mcp(monkeypatch, status): + """401/429 are account-level verdicts. MCP uses the same key against the + same quota, so falling back can only add latency and noise — these must + surface directly instead of being swallowed as 'REST unavailable'.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter, PaperclipError + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + + class FakeResponse: + status_code = status + text = '{"detail":"You\'ve hit the daily limit on map/verify operations (100/day)."}' + + _patch_rest(monkeypatch, lambda *a, **kw: FakeResponse()) + + async def fail_mcp(self, command): + raise AssertionError("must not fall back to MCP on an account-level error") + + monkeypatch.setattr(PaperclipAdapter, "_run", fail_mcp) + + with pytest.raises(PaperclipError) as exc: + await PaperclipAdapter().run_map("s_1", "q") + assert str(status) in str(exc.value) + + +async def test_rest_client_is_pooled_per_adapter(monkeypatch): + """The client must be reused across calls — that reuse is the whole point + (measured ~37% faster on sequential calls by skipping the TLS handshake). + A fresh client per call would silently undo it.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + adapter = PaperclipAdapter() + assert adapter._rest_client() is adapter._rest_client() + + # ...but scoped per adapter, never global: build_graph() makes one adapter + # per run, so two concurrent users must not share a connection pool. + assert PaperclipAdapter()._rest_client() is not adapter._rest_client() + + await adapter.aclose() + + +async def test_aclose_is_idempotent(monkeypatch): + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + adapter = PaperclipAdapter() + adapter._rest_client() + await adapter.aclose() + await adapter.aclose() + + +# Captured verbatim from a live REST `map` — REST returns the CLI's coloured +# terminal output, MCP does not. +MAP_OUTPUT_WITH_ANSI = ( + "\x1b[1mMap complete: 3/3 papers\x1b[0m\nResults ID: \x1b[2mm_7049c74c\x1b[0m\n" +) + + +def test_map_id_survives_ansi_colour_codes(): + """The dim code `\\x1b[2m` ends in the letter `m`, so it butts against the + `m_...` id and kills `_MAP_ID_RE`'s leading `\\b`. Stripping ANSI at the + transport boundary is what keeps the id findable.""" + from crossbar_llm.paperclip_tools.adapter import _MAP_ID_RE, _strip_ansi + + assert _MAP_ID_RE.search(MAP_OUTPUT_WITH_ANSI) is None # the trap + assert _MAP_ID_RE.search(_strip_ansi(MAP_OUTPUT_WITH_ANSI)).group(1) == "m_7049c74c" + + +def test_strip_ansi_is_idempotent_and_preserves_text(): + from crossbar_llm.paperclip_tools.adapter import _strip_ansi + + clean = _strip_ansi(MAP_OUTPUT_WITH_ANSI) + assert clean == "Map complete: 3/3 papers\nResults ID: m_7049c74c\n" + assert _strip_ansi(clean) == clean # MCP output is already clean; no-op there + + +async def test_run_rest_strips_ansi_from_output(monkeypatch): + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + + class FakeResponse: + status_code = 200 + text = "" + + def json(self): + return {"output": MAP_OUTPUT_WITH_ANSI, "result_id": None} + + _patch_rest(monkeypatch, lambda *a, **kw: FakeResponse()) + data = await PaperclipAdapter()._run_rest("map", "--from s_1 'q'") + assert "\x1b" not in data["output"] + + +# Captured verbatim from a live REST search that returned a full listing in +# `output` but omitted `result_data` entirely (~8% of calls). ANSI already +# stripped, as `_run_rest` does before any parsing. +REST_PAYLOAD_NO_RESULT_DATA = { + "output": ( + "Found 2 papers [s_42d99d49]\n" + "\n" + " 1. Molecular Mechanism of Huanglian Jiedu Decoction in Alzheimer's Disease\n" + " Qiuyan Ye, Xue Li, Wei Gao\n" + " bio_75ad981a7345 · bioRxiv · 2024-05-15\n" + " https://doi.org/10.1101/2024.05.15.594364\n" + ' "Network pharmacology analysis of the decoction."\n' + "\n" + " 2. Adipsin and Leptin Levels in Type 2 Diabetic Patients\n" + " Sura Khalid Mohammed\n" + " PMC11729846 · PMC · 2024-01-01\n" + " https://www.ncbi.nlm.nih.gov/pmc/articles/PMC11729846/\n" + ' "The study compared adipsin and leptin levels."\n' + "\n" + "[357ms, saved to s_42d99d49]\n" + ), + "exit_code": 0, + "result_id": "s_42d99d49", + "result_data": None, +} + + +def test_rest_hits_recovered_when_result_data_missing(): + """The server intermittently returns a complete listing while omitting + `result_data`. Reading only the structured field reported those searches + as zero-hit, which then tripped the node's zero-result fallback as though + nothing matched. Parse the text listing instead.""" + from crossbar_llm.paperclip_tools.adapter import _hits_from_rest_payload + + hits = _hits_from_rest_payload( + REST_PAYLOAD_NO_RESULT_DATA, REST_PAYLOAD_NO_RESULT_DATA["output"], None + ) + assert [h.doc_id for h in hits] == ["bio_75ad981a7345", "PMC11729846"] + assert hits[0].title.startswith("Molecular Mechanism") + + +def test_rest_hits_prefer_structured_result_data(): + """When `result_data` IS present it wins — it carries score/doi/pub_year + that the text listing doesn't have.""" + from crossbar_llm.paperclip_tools.adapter import _hits_from_rest_payload + + payload = { + "output": REST_PAYLOAD_NO_RESULT_DATA["output"], # says 2 papers + "result_data": {"papers": [{"document_id": "PMC1", "title": "T", "doi": "10.1/x"}]}, + } + hits = _hits_from_rest_payload(payload, payload["output"], None) + assert [h.doc_id for h in hits] == ["PMC1"] + assert hits[0].doi == "10.1/x" + + +def test_rest_genuinely_empty_search_stays_empty(): + """A real zero-result search must NOT be rescued into phantom hits.""" + from crossbar_llm.paperclip_tools.adapter import _hits_from_rest_payload + + payload = {"output": "No results found.\n[12ms]", "result_data": None} + assert _hits_from_rest_payload(payload, payload["output"], None) == [] + + +async def test_filter_reports_skipped_when_result_data_missing(monkeypatch): + """filter's text output is only a summary ("Filtered: 5 -> 2 papers") with + no listing, so a missing `result_data` must degrade to "couldn't filter" + (None -> caller keeps unfiltered hits), never to "removed everything".""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + + class FakeResponse: + status_code = 200 + text = "" + + def json(self): + return { + "output": "Filtered: 5 → 2 papers (3 removed as irrelevant) in 374ms", + "result_id": "s_1", + "result_data": None, + } + + _patch_rest(monkeypatch, lambda *a, **kw: FakeResponse()) + assert await PaperclipAdapter().filter("s_1", "q") is None + + +@pytest.mark.parametrize( + "query,should_reach_server", + [ + ("SELECT 1", True), + ("SELECT COUNT(*) FROM documents;", True), # trailing ; is fine + ("WITH x AS (SELECT 1) SELECT * FROM x", True), # CTEs are read-only + ("SELECT 1; DROP TABLE documents", False), + ("select 1; delete from documents", False), + ("DROP TABLE documents", False), + ], +) +async def test_sql_guard_rejects_appended_statements(query, should_reach_server): + """The prefix check alone let `SELECT 1; DROP TABLE ...` through — it starts + with SELECT. Paperclip enforces read-only server-side, so this was never the + only guard, but a check that misses the obvious case reads as protection + without being any.""" + from crossbar_llm.paperclip_tools.tools import paperclip_sql + from crossbar_llm.paperclip_tools.adapter import SqlResult + + seen = [] + + class Adapter: + async def sql(self, q, *, source=None): + seen.append(q) + return SqlResult(columns=[], rows=[]) + + out = await paperclip_sql(Adapter(), query) + assert bool(seen) is should_reach_server + assert (out.error is None) is should_reach_server + + +async def test_search_requests_the_full_corpus_by_default(monkeypatch): + """Without `--all` the server silently limits results to recent papers: + measured 5.0 hits vs 13.6 at `-n 14`, and the Denosumab question came back + with only 2025-2026 papers instead of the 2008-2014 RANKL literature.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + captured = {} + + class FakeResponse: + status_code = 200 + text = "" + + def json(self): + return {"output": "Found 0 papers [s_x]", "result_id": "s_x", + "result_data": {"papers": []}} + + def fake_post(url, json, headers, timeout): + captured["raw"] = json["raw"] + return FakeResponse() + + _patch_rest(monkeypatch, fake_post) + adapter = PaperclipAdapter() + + await adapter.search("metformin", limit=14) + assert "--all" in captured["raw"] + + # `--year` is an explicit recency filter, so the two must not be combined. + await adapter.search("metformin", limit=14, year="2024") + assert "--all" not in captured["raw"] + assert "--year 2024" in captured["raw"] + + +def test_abstracts_is_not_an_allowed_scope(): + """`help search` advertises `-s abstracts`, but it is absent from `ls /` and + returns "No papers found" for every query — so the router must not be able + to choose it, and the zero-result fallback must not target it.""" + from typing import get_args + + from crossbar_llm.paperclip_tools.schemas import PaperclipSourceChoice + from crossbar_llm.paperclip_tools.adapter import PaperclipSource, _SOURCE_ROOT + + assert "abstracts" not in get_args(PaperclipSourceChoice) + assert "abstracts" not in get_args(PaperclipSource) + assert "abstracts" not in _SOURCE_ROOT + + +async def test_map_surfaces_the_server_error_message(monkeypatch): + """A server-side map failure arrives as HTTP 200 with `ERR: <message>` in + the body. Reporting only "no results id" hid a real Paperclip outage + (`ERR: map: 'tpm_used'`) behind what looked like our own parsing bug.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter, PaperclipError + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + + class FakeResponse: + status_code = 200 + text = "" + + def json(self): + return {"output": "ERR: map: 'tpm_used'\n[exit 1]", "result_id": None} + + _patch_rest(monkeypatch, lambda *a, **kw: FakeResponse()) + + with pytest.raises(PaperclipError) as exc: + await PaperclipAdapter().run_map("s_1", "q") + assert "tpm_used" in str(exc.value) + + +async def test_map_reports_unparseable_output_verbatim(monkeypatch): + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter, PaperclipError + + monkeypatch.setenv("PAPERCLIP_API_KEY", "test-key") + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + + class FakeResponse: + status_code = 200 + text = "" + + def json(self): + return {"output": "something unexpected entirely", "result_id": None} + + _patch_rest(monkeypatch, lambda *a, **kw: FakeResponse()) + + with pytest.raises(PaperclipError) as exc: + await PaperclipAdapter().run_map("s_1", "q") + assert "something unexpected" in str(exc.value) diff --git a/crossbar_llm/paperclip_tools/tests/test_agent.py b/crossbar_llm/paperclip_tools/tests/test_agent.py new file mode 100644 index 0000000..101decf --- /dev/null +++ b/crossbar_llm/paperclip_tools/tests/test_agent.py @@ -0,0 +1,846 @@ +"""Graph-level tests for the Paperclip LangGraph. + +All offline: a fake `PaperclipAdapterProtocol` is injected at the +`build_graph(adapter=...)` seam (mocking the tool-call boundary, not the MCP +wire protocol), and router/synthesizer/evaluator are swapped for deterministic +fakes via the `build_graph(router=..., synthesizer=..., evaluator=...)` seams. +""" +from __future__ import annotations + +from crossbar_llm.paperclip_tools.agent import build_graph +from crossbar_llm.paperclip_tools.schemas import ( + PaperclipDepthEvaluation, + PaperclipRouterDecision, +) +from crossbar_llm.paperclip_tools.adapter import ( + MapExtract, + PaperclipError, + PaperHit, + PaperMeta, + SearchResult, + SqlResult, +) + + +def _meta(doc_id, **kw): + base = dict( + document_id=doc_id, pmid="1" + doc_id[-3:], doi="10.1/" + doc_id, + title="Title " + doc_id, authors="A. Author", abstract="An abstract about the topic.", + journal="J", pub_year=2024, + ) + base.update(kw) + return PaperMeta(**base) + + +class FakeAdapter: + """Records calls and returns canned typed results.""" + + def __init__(self, *, hits=None, search_error=None, meta_error_ids=(), + content=None, map_extracts=None, sql_result=None, sql_error=None, + filter_hits=None, filter_error=None, filter_none=False): + self._hits = hits if hits is not None else [ + PaperHit(doc_id="PMC1", title="T1", source="pmc", date="2024"), + PaperHit(doc_id="PMC2", title="T2", source="pmc", date="2023"), + ] + self._search_error = search_error + self._meta_error_ids = set(meta_error_ids) + self._content = content if content is not None else "L1: full body text" + self._map_extracts = map_extracts # dict doc_id -> text, or None + self._sql_result = sql_result # SqlResult, or None + self._sql_error = sql_error + self._filter_hits = filter_hits # list[PaperHit], or None (echo input hits) + self._filter_error = filter_error + self._filter_none = filter_none # simulate REST-unavailable (returns None) + self.search_calls = [] + self.content_calls = [] + self.meta_calls = [] + self.map_calls = [] + self.sql_calls = [] + self.filter_calls = [] + + async def search(self, query, *, source="pmc", limit=10, sort=None, year=None, ranking=None): + self.search_calls.append({ + "query": query, "source": source, "limit": limit, "year": year, "ranking": ranking, + }) + if self._search_error: + raise PaperclipError(self._search_error) + return SearchResult(hits=list(self._hits), search_id="s_fake") + + async def get_meta(self, doc_id, *, source="pmc"): + self.meta_calls.append({"doc_id": doc_id, "source": source}) + if doc_id in self._meta_error_ids: + raise PaperclipError(f"meta boom {doc_id}") + if source in ("proteins", "uniprot"): + return PaperMeta( + document_id=doc_id, title=f"{doc_id} - Histone acetyltransferase", + protein_name="Histone acetyltransferase KAT6A", gene_name="KAT6A", + organism="Homo sapiens", uniprot_id="KAT6A_HUMAN", + ) + return _meta(doc_id) + + async def get_content(self, doc_id, *, source="pmc", sections=None, max_lines=None): + self.content_calls.append( + {"doc_id": doc_id, "source": source, "sections": sections, "max_lines": max_lines} + ) + return self._content + + async def run_map(self, search_id, question, *, limit=None): + self.map_calls.append({"search_id": search_id, "question": question, "limit": limit}) + src = self._map_extracts or {} + return [MapExtract(doc_id=d, text=t, success=True) for d, t in src.items()] + + async def sql(self, query, *, source=None): + self.sql_calls.append({"query": query, "source": source}) + if self._sql_error: + raise PaperclipError(self._sql_error) + return self._sql_result if self._sql_result is not None else SqlResult() + + async def filter(self, search_id, query): + self.filter_calls.append({"search_id": search_id, "query": query}) + if self._filter_error: + raise PaperclipError(self._filter_error) + if self._filter_none: + return None + hits = self._filter_hits if self._filter_hits is not None else list(self._hits) + return SearchResult(hits=hits, search_id=search_id) + + +def _router(qtype="keyword_search", *, source="pmc", query="topic query", + full_text=False, sections=None, year=None, sql_query=None, + map_question=None, analogical_query=None): + async def fn(_question): + return PaperclipRouterDecision( + question_type=qtype, source=source, search_query=query, + analogical_query=analogical_query, + map_question=map_question if map_question is not None else query, + full_text=full_text, sections=sections, year=year, sql_query=sql_query, + rationale="test", + ) + return fn + + +async def _synth(state): + refs = " ".join(f"[{c.ref_num}]" for c in state.get("citations", [])) + return f"Answer citing {refs}. References: " + "; ".join( + f"[{c.ref_num}] {c.doc_id} doi:{c.doi}" for c in state.get("citations", []) + ) + + +async def test_keyword_search_happy_path(): + adapter = FakeAdapter() + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "What treats X?"}) + assert out["question_type"] == "keyword_search" + assert [h.doc_id for h in out["hits"]] == ["PMC1", "PMC2"] + assert [c.ref_num for c in out["citations"]] == [1, 2] + assert out["citations"][0].url == "https://citations.gxl.ai/papers/PMC1" + assert "[1]" in out["final_answer"] + # No full text requested -> no content fetch. + assert adapter.content_calls == [] + + +async def test_out_of_scope_short_circuits(): + adapter = FakeAdapter() + g = build_graph(router=_router("out_of_scope"), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "shortest path from A to B?"}) + assert out["question_type"] == "out_of_scope" + assert out.get("final_answer") is None + assert adapter.search_calls == [] # never retrieved + + +async def test_full_text_depth_fetches_content(): + adapter = FakeAdapter() + # abstracts_only=False to allow the full-text path; use_map=False to isolate + # it from map (which would otherwise supply the body server-side). + g = build_graph(router=_router("full_text_depth", full_text=True), + synthesizer=_synth, adapter=adapter, + abstracts_only=False, use_map=False) + out = await g.ainvoke({"question": "mechanism of X?"}) + assert out["full_text"] is True + assert {c["doc_id"] for c in adapter.content_calls} == {"PMC1", "PMC2"} + assert all(d.body == "L1: full body text" for d in out["documents"]) + # No section filter requested -> whole body. + assert all(c["sections"] is None for c in adapter.content_calls) + + +async def test_section_filter_reaches_adapter(): + adapter = FakeAdapter() + g = build_graph( + router=_router("full_text_depth", full_text=True, sections=["methods"]), + synthesizer=_synth, + adapter=adapter, + abstracts_only=False, + use_map=False, + ) + out = await g.ainvoke({"question": "what methods were used?"}) + assert out["sections"] == ["methods"] + assert adapter.content_calls + assert all(c["sections"] == ["methods"] for c in adapter.content_calls) + + +async def test_map_uses_map_question_not_raw_user_question(): + """map_question (the router's specific, enumerated-fields extraction + question) must reach `run_map` — not the raw user question, and not + search_query (keywords, a different purpose).""" + adapter = FakeAdapter() + g = build_graph( + router=_router( + query="metformin keywords", + map_question="What is the specific molecular target and downstream pathway?", + ), + synthesizer=_synth, + adapter=adapter, + use_map=True, + ) + await g.ainvoke({"question": "how does metformin work?"}) + assert adapter.map_calls + assert adapter.map_calls[0]["question"] == ( + "What is the specific molecular target and downstream pathway?" + ) + + +async def test_map_extracts_used_as_evidence(): + adapter = FakeAdapter(map_extracts={"PMC1": "AMPK-Mff-Drp1 pathway (L34).", + "PMC2": "not mentioned"}) + captured = {} + + async def synth(state): + captured["docs"] = state.get("documents") + return "answer [1][2]" + + g = build_graph(router=_router(), synthesizer=synth, adapter=adapter, use_map=True) + out = await g.ainvoke({"question": "which pathway?"}) + assert adapter.map_calls and adapter.map_calls[0]["search_id"] == "s_fake" + bodies = {d.doc_id: d.body for d in captured["docs"]} + assert bodies["PMC1"] == "AMPK-Mff-Drp1 pathway (L34)." + # map evidence is used even though full_text is False. + assert out["full_text"] is False + # no whole-body content fetch when map supplied the evidence. + assert adapter.content_calls == [] + + +async def test_map_extract_with_found_false_is_excluded_as_evidence(): + """`run_map` asks for an {answer, found} contract — when the model honors + it, `found=False` means the paper explicitly doesn't address the + question. That extract must be excluded from evidence entirely (falling + back to the abstract), not used verbatim the way unstructured "not + mentioned"-shaped prose used to be.""" + class StructuredMapAdapter(FakeAdapter): + async def run_map(self, search_id, question, *, limit=None): + self.map_calls.append({"search_id": search_id, "question": question, "limit": limit}) + return [ + MapExtract( + doc_id="PMC1", text="AMPK activation", success=True, + data={"answer": "AMPK activation", "found": True}, found=True, + ), + MapExtract( + doc_id="PMC2", text="Not found", success=True, + data={"answer": "Not found", "found": False}, found=False, + ), + ] + + adapter = StructuredMapAdapter() + captured = {} + + async def synth(state): + captured["docs"] = state.get("documents") + return "answer [1][2]" + + g = build_graph(router=_router(), synthesizer=synth, adapter=adapter, use_map=True) + await g.ainvoke({"question": "which pathway?"}) + bodies = {d.doc_id: d.body for d in captured["docs"]} + assert bodies["PMC1"] == "AMPK activation" + # PMC2's found=False extract must not be used as its evidence body. + assert bodies["PMC2"] != "Not found" + + +async def test_map_citation_lines_anchor_the_citation_url(): + """A paper cited from map evidence must get a line-anchored URL built + from Paperclip's own per-answer line provenance (`_citations`), not the + blanket paper-root URL — the line-level citation architecture change.""" + class StructuredMapAdapter(FakeAdapter): + async def run_map(self, search_id, question, *, limit=None): + self.map_calls.append({"search_id": search_id, "question": question, "limit": limit}) + return [ + MapExtract( + doc_id="PMC1", text="AMPK activation (L6, L38, L40).", success=True, + found=True, citation_lines=[38, 6, 40], # unsorted on purpose + ), + MapExtract( + doc_id="PMC2", text="mTOR inhibition (L24).", success=True, + found=True, citation_lines=[24], + ), + ] + + adapter = StructuredMapAdapter() + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter, use_map=True) + out = await g.ainvoke({"question": "which pathway?"}) + urls = {c.doc_id: c.url for c in out["citations"]} + assert urls["PMC1"] == "https://citations.gxl.ai/papers/PMC1#L6,38,40" + assert urls["PMC2"] == "https://citations.gxl.ai/papers/PMC2#L24" + + +async def test_no_citation_lines_leaves_url_unanchored(): + """Abstract-only / no-map-citations evidence keeps the plain paper-root + URL — there's nothing deterministic to anchor to.""" + adapter = FakeAdapter() # use_map defaults False in build_graph's fake-adapter path here + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter, use_map=False) + out = await g.ainvoke({"question": "topic"}) + assert out["citations"][0].url == "https://citations.gxl.ai/papers/PMC1" + + +async def test_backfill_replaces_uncitable_hit(): + # 4 hits; PMC2 has no metadata AND no title on the hit itself (genuinely + # uncitable — nothing to recover from). With max_documents=3 we should + # still get 3 citable docs by pulling in PMC4 from the buffer. + hits = [ + PaperHit(doc_id="PMC1", title="T1", source="pmc"), + PaperHit(doc_id="PMC2", title="", source="pmc"), + PaperHit(doc_id="PMC3", title="T3", source="pmc"), + PaperHit(doc_id="PMC4", title="T4", source="pmc"), + ] + adapter = FakeAdapter(hits=hits, meta_error_ids={"PMC2"}) + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter, max_documents=3) + out = await g.ainvoke({"question": "q"}) + doc_ids = [c.doc_id for c in out["citations"]] + assert doc_ids == ["PMC1", "PMC3", "PMC4"] # PMC2 dropped, PMC4 backfilled + assert len(out["citations"]) == 3 + + +async def test_proteins_source_threaded_and_cited(): + adapter = FakeAdapter(hits=[PaperHit(doc_id="Q92794", title="KAT6A", source="proteins")]) + g = build_graph(router=_router(source="proteins", full_text=True), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "sequence length of KAT6A?"}) + # get_meta was called with source=proteins (not the default) — the path fix. + assert adapter.meta_calls[0]["source"] == "proteins" + cit = out["citations"][0] + assert cit.url == "https://www.uniprot.org/uniprotkb/Q92794/entry" + # proteins have no content.lines — no body fetch even with full_text=True. + assert adapter.content_calls == [] + # a compact proteins summary is used as evidence. + assert out["documents"][0].body and "UniProt" in out["documents"][0].body + + +async def test_abstracts_only_clamps_sections(): + adapter = FakeAdapter() + g = build_graph( + router=_router("full_text_depth", full_text=True, sections=["results"]), + synthesizer=_synth, + adapter=adapter, + abstracts_only=True, + ) + out = await g.ainvoke({"question": "results?"}) + assert out["sections"] is None # clamped + assert adapter.content_calls == [] # no body fetched at all + + +async def test_zero_result_falls_back_to_an_unscoped_search(): + """A scoped search that finds nothing retries unscoped. + + The retry used to target `-s abstracts`, which Paperclip's `help search` + lists but which is not a real corpus (absent from `ls /`, returns "No papers + found" for every query) — so the fallback could never recover anything. + """ + class TwoPhase(FakeAdapter): + async def search(self, query, *, source="pmc", limit=10, sort=None, year=None, ranking=None): + self.search_calls.append({ + "query": query, "source": source, "limit": limit, "year": year, "ranking": ranking, + }) + if source == "pmc": + return SearchResult(hits=[], search_id=None) + return SearchResult( + hits=[PaperHit(doc_id="PMC9", title="Fallback", source="pmc", date="2022")], + search_id="s_fb", + ) + + adapter = TwoPhase() + g = build_graph(router=_router(source="pmc"), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "obscure topic"}) + assert [c["source"] for c in adapter.search_calls] == ["pmc", None] + assert [h.doc_id for h in out["hits"]] == ["PMC9"] + assert any("fell back to an unscoped search" in w for w in out["warnings"]) + + +async def test_unscoped_zero_result_does_not_retry_itself(): + """An already-unscoped, unranked search that finds nothing must not repeat + the identical query — that returns the same nothing and costs a round trip.""" + adapter = FakeAdapter(hits=[]) + g = build_graph(router=_router(source=None), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "genuinely unmatched topic"}) + assert len(adapter.search_calls) == 1 + assert out["hits"] == [] + + +async def test_search_error_never_raises(): + adapter = FakeAdapter(search_error="transport exploded") + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "anything"}) + # Degrades to an empty-but-complete run with a warning, not an exception. + assert out["hits"] == [] + assert out["citations"] == [] + assert any("search failed" in w for w in out["warnings"]) + + +async def test_uncitable_hit_is_dropped(): + # No title on the hit itself either -> nothing to recover from, genuinely + # uncitable. + hits = [ + PaperHit(doc_id="PMC1", title="T1", source="pmc"), + PaperHit(doc_id="PMC2", title="", source="pmc"), + ] + adapter = FakeAdapter(hits=hits, meta_error_ids={"PMC2"}) + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "topic"}) + # PMC2's meta failed -> dropped; only PMC1 survives as citable. + assert [c.doc_id for c in out["citations"]] == ["PMC1"] + assert any("uncitable" in w for w in out["warnings"]) + + +async def test_get_meta_failure_recovers_metadata_from_hit(): + """A get_meta failure must not drop an otherwise-good hit when the search + hit itself already carries usable metadata (title/authors/doi/pub_year/ + snippet) — confirmed live this happens for some `abstracts` (OpenAlex- + backed) hits even though a direct get_meta retry with the identical + doc_id succeeds (an intermittent server-side inconsistency, not a + parsing bug). Recover a citable PaperMeta from the hit instead.""" + hits = [ + PaperHit( + doc_id="oa_W123", title="Recovered Title", authors="A. Author", + doi="10.1/recovered", date="2020", source="abstracts", + snippet="A recovered abstract snippet.", + ), + ] + adapter = FakeAdapter(hits=hits, meta_error_ids={"oa_W123"}) + g = build_graph(router=_router(source="abstracts"), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "topic"}) + assert [c.doc_id for c in out["citations"]] == ["oa_W123"] + cit = out["citations"][0] + assert cit.title == "Recovered Title" + assert cit.doi == "10.1/recovered" + doc = out["documents"][0] + assert doc.meta.abstract == "A recovered abstract snippet." + assert any("recovered from search hit" in w for w in out["warnings"]) + + +async def test_depth_refinement_refetches_with_full_text(): + adapter = FakeAdapter() + + async def insufficient_once(state): + # Flag shallow only while still abstracts-only; sufficient after refetch. + if state.get("full_text"): + return PaperclipDepthEvaluation(sufficient=True, rationale="deep now") + return PaperclipDepthEvaluation( + sufficient=False, missing="no mechanism described", rationale="shallow" + ) + + g = build_graph( + router=_router("keyword_search", full_text=False), + synthesizer=_synth, + evaluator=insufficient_once, + adapter=adapter, + # Depth loop requires the full-text lever; isolate from map. + abstracts_only=False, + use_map=False, + ) + out = await g.ainvoke({"question": "how does X work?"}) + assert out["refinement_attempted"] is True + assert out["full_text"] is True + # After refinement, content was fetched for the papers. + assert adapter.content_calls, "expected full-text refetch on refinement" + + +async def test_abstracts_only_clamps_full_text_and_skips_depth(): + adapter = FakeAdapter() + g = build_graph( + router=_router("full_text_depth", full_text=True), + synthesizer=_synth, + adapter=adapter, + abstracts_only=True, + ) + out = await g.ainvoke({"question": "mechanism?"}) + assert out["full_text"] is False # router choice clamped + assert adapter.content_calls == [] # no body fetched + assert out["depth_skip_reason"] == "abstracts_only enabled" + + +async def test_broad_source_infers_per_hit_root(): + """`source=None` (the new default) can return hits from mixed corpora in + one result set — get_meta/citation URLs must resolve each hit's VFS root + from its own doc_id shape, not a single blanket source.""" + hits = [ + PaperHit(doc_id="PMC1", title="Paper", source="pmc"), + PaperHit(doc_id="fda_abc123", title="Some FDA doc", source="fda"), + ] + adapter = FakeAdapter(hits=hits) + g = build_graph(router=_router(source=None), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "q"}) + assert out["source"] is None # router's broad choice threaded through unchanged + meta_sources = {m["doc_id"]: m["source"] for m in adapter.meta_calls} + assert meta_sources["PMC1"] == "pmc" + assert meta_sources["fda_abc123"] == "fda" + fda_cit = next(c for c in out["citations"] if c.doc_id == "fda_abc123") + assert fda_cit.url == "https://citations.gxl.ai/fda/fda_abc123" + + +async def test_fda_trials_evidence_fallback_from_hit_snippet(): + """Regression test: fda/trials meta.json doesn't populate title/abstract + the way paper corpora do (confirmed empty live). The context block and + citation must fall back to the search hit's own title/snippet, which + Paperclip DOES populate — mirrors the existing proteins-summary pattern.""" + class FdaEmptyMeta(FakeAdapter): + async def get_meta(self, doc_id, *, source="pmc"): + self.meta_calls.append({"doc_id": doc_id, "source": source}) + if source in ("fda", "trials"): + return PaperMeta(document_id=doc_id, title="", abstract="") + return _meta(doc_id) + + hits = [PaperHit( + doc_id="fda_xyz", title="KEYTRUDA QLEX review", source="fda", + snippet="Merck submitted a BLA for pembrolizumab.", + )] + adapter = FdaEmptyMeta(hits=hits) + g = build_graph(router=_router(source="fda"), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "pembrolizumab approval"}) + doc = out["documents"][0] + assert doc.meta.title == "KEYTRUDA QLEX review" + assert doc.meta.abstract == "Merck submitted a BLA for pembrolizumab." + assert out["citations"][0].title == "KEYTRUDA QLEX review" + + +# --- sql_aggregate routing (docs/paperclip_rest_endpoint_findings.md §12) -- + +async def test_sql_aggregate_success_skips_search_and_assemble(): + """A successful SQL route must go straight to synthesis — no search/ + get_meta/get_content calls, since there's nothing to retrieve.""" + sql_result = SqlResult(columns=["n"], rows=[{"n": "217217"}]) + adapter = FakeAdapter(sql_result=sql_result) + captured = {} + + async def synth(state): + captured["state"] = state + return "There are 217217 matching documents." + + g = build_graph( + router=_router("sql_aggregate", sql_query="SELECT COUNT(*) AS n FROM documents"), + synthesizer=synth, + adapter=adapter, + ) + out = await g.ainvoke({"question": "how many fda documents are there?"}) + + assert adapter.sql_calls == [ + {"query": "SELECT COUNT(*) AS n FROM documents", "source": "pmc"} + ] + assert adapter.search_calls == [] + assert adapter.meta_calls == [] + assert out["sql_rows"] == [{"n": "217217"}] + assert out["sql_error"] is None + assert out["final_answer"] == "There are 217217 matching documents." + # No retrieval lever to pull -> depth evaluation short-circuits. + assert out["depth_skip_reason"] == "sql_aggregate has no full-text escalation lever" + # The synthesizer (fake or real) sees the raw state either way. + assert captured["state"]["sql_query"] == "SELECT COUNT(*) AS n FROM documents" + + +async def test_sql_aggregate_falls_back_to_search_on_query_error(): + """A failed SQL query (bad syntax, unsupported source, timeout) must not + fail the run — falls back to a normal keyword search using the router's + search_query, exactly like the zero-result search fallback.""" + adapter = FakeAdapter(sql_error="relation \"documents\" does not exist") + g = build_graph( + router=_router( + "sql_aggregate", source="pmc", query="fallback keywords", + sql_query="SELECT COUNT(*) FROM documents WHERE source = 'trials'", + ), + synthesizer=_synth, + adapter=adapter, + ) + out = await g.ainvoke({"question": "how many trials are there?"}) + + assert adapter.sql_calls, "expected the sql node to have tried the query" + assert adapter.search_calls, "expected fallback to the search path" + assert adapter.search_calls[0]["query"] == "fallback keywords" + # paperclip_sql's never-raise wrapper formats errors as "{type}: {msg}", + # same convention as every other tool wrapper in this codebase. + assert out["sql_error"] == 'PaperclipError: relation "documents" does not exist' + # Fell through to the normal paper-citation path, not left empty. + assert out["citations"] + assert "[1]" in out["final_answer"] + + +async def test_sql_aggregate_with_no_query_falls_back_to_search(): + """Defensive: if the router somehow set sql_aggregate without filling + sql_query, sql_node must degrade to the search fallback, not crash.""" + adapter = FakeAdapter() + g = build_graph( + router=_router("sql_aggregate", source="pmc", query="fallback keywords", sql_query=None), + synthesizer=_synth, + adapter=adapter, + ) + out = await g.ainvoke({"question": "how many?"}) + assert adapter.sql_calls == [] # never even attempted + assert adapter.search_calls + assert out["sql_error"] == "no query" + + +# --- filter (server-side relevance trim, opt-in) -------------------------- + +async def test_filter_skipped_by_default(): + adapter = FakeAdapter() + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter) + await g.ainvoke({"question": "topic"}) + assert adapter.filter_calls == [] + + +async def test_filter_trims_hits_when_enabled(): + hits = [ + PaperHit(doc_id="PMC1", title="T1", source="pmc"), + PaperHit(doc_id="PMC2", title="T2", source="pmc"), + ] + adapter = FakeAdapter(hits=hits, filter_hits=[hits[0]]) + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter, use_filter=True) + out = await g.ainvoke({"question": "topic"}) + assert adapter.filter_calls and adapter.filter_calls[0]["search_id"] == "s_fake" + assert [c.doc_id for c in out["citations"]] == ["PMC1"] + + +async def test_filter_reverts_to_unfiltered_on_empty_result(): + adapter = FakeAdapter(filter_hits=[]) + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter, use_filter=True) + out = await g.ainvoke({"question": "topic"}) + assert [c.doc_id for c in out["citations"]] == ["PMC1", "PMC2"] + assert any("filter removed all hits" in w for w in out["warnings"]) + + +async def test_filter_reverts_to_unfiltered_on_error(): + adapter = FakeAdapter(filter_error="filter boom") + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter, use_filter=True) + out = await g.ainvoke({"question": "topic"}) + assert [c.doc_id for c in out["citations"]] == ["PMC1", "PMC2"] + assert any("filter failed" in w for w in out["warnings"]) + + +async def test_filter_skipped_when_rest_unavailable(): + adapter = FakeAdapter(filter_none=True) + g = build_graph(router=_router(), synthesizer=_synth, adapter=adapter, use_filter=True) + out = await g.ainvoke({"question": "topic"}) + assert adapter.filter_calls # was attempted + assert [c.doc_id for c in out["citations"]] == ["PMC1", "PMC2"] # unfiltered + assert not any("filter" in w for w in out["warnings"]) # skipped, not a warning-worthy event + + +async def test_filter_skipped_for_list_breadth(): + adapter = FakeAdapter() + g = build_graph(router=_router("list_breadth"), synthesizer=_synth, adapter=adapter, use_filter=True) + await g.ainvoke({"question": "list all X"}) + assert adapter.filter_calls == [] + + +async def test_use_filter_widens_search_fetch(): + """Confirmed live: `filter` can cut a fetch down to single digits or + zero. search_node must fetch a much larger candidate pool when + use_filter=True so filter has room to trim without starving + assemble_context_node below max_documents (the backfill-gap bug).""" + adapter = FakeAdapter() + g = build_graph( + router=_router("keyword_search"), synthesizer=_synth, adapter=adapter, + use_filter=True, max_documents=7, + ) + await g.ainvoke({"question": "topic"}) + assert adapter.search_calls + assert adapter.search_calls[0]["limit"] == 25 # _FILTER_FETCH_LIMIT + + +async def test_use_filter_false_keeps_normal_fetch_size(): + adapter = FakeAdapter() + g = build_graph( + router=_router("keyword_search"), synthesizer=_synth, adapter=adapter, + use_filter=False, max_documents=7, + ) + await g.ainvoke({"question": "topic"}) + assert adapter.search_calls[0]["limit"] == 11 # max(10, 7 + 4), unchanged + + +# --- analogical_search (docs/paperclip_rest_endpoint_findings.md §15) ----- + +async def test_analogical_search_uses_analogical_query_and_ranking(): + """The primary search call for analogical_search must use + `analogical_query` (a method-description sentence) with + `ranking="analogical"` — NOT the keyword-shaped `search_query`.""" + adapter = FakeAdapter() + g = build_graph( + router=_router( + "analogical_search", source=None, query="BTK inhibitor CLL", + analogical_query="correcting for a systematic detection bias with an unknown mechanism", + ), + synthesizer=_synth, adapter=adapter, + ) + await g.ainvoke({"question": "what other fields use a technique like this?"}) + assert adapter.search_calls + call = adapter.search_calls[0] + assert call["query"] == "correcting for a systematic detection bias with an unknown mechanism" + assert call["ranking"] == "analogical" + + +async def test_analogical_search_zero_result_fallback_uses_keywords_no_ranking(): + """The zero-result fallback must retry with the keyword-shaped + `search_query` and the default ranking — never repeat the unranked + analogical sentence, and never carry `ranking="analogical"` into the + fallback call.""" + class EmptyPrimary(FakeAdapter): + async def search(self, query, *, source="pmc", limit=10, sort=None, year=None, ranking=None): + self.search_calls.append({ + "query": query, "source": source, "limit": limit, "year": year, "ranking": ranking, + }) + if ranking == "analogical": + return SearchResult(hits=[], search_id=None) + return SearchResult(hits=list(self._hits), search_id="s_fb") + + adapter = EmptyPrimary() + g = build_graph( + router=_router( + "analogical_search", source=None, query="BTK inhibitor CLL", + analogical_query="a method-description sentence", + ), + synthesizer=_synth, adapter=adapter, + ) + out = await g.ainvoke({"question": "what other fields use a technique like this?"}) + assert len(adapter.search_calls) == 2 + assert adapter.search_calls[0]["ranking"] == "analogical" + assert adapter.search_calls[1]["query"] == "BTK inhibitor CLL" + assert adapter.search_calls[1]["ranking"] is None + assert out["hits"], "expected the keyword fallback to succeed" + + +async def test_analogical_search_without_analogical_query_falls_back_to_keyword_query(): + """Defensive: if the router somehow set analogical_search without filling + `analogical_query` (should not happen per the schema, but must not crash + or silently search for nothing), search_node must fall back to the + keyword `search_query` rather than an empty query.""" + adapter = FakeAdapter() + g = build_graph( + router=_router("analogical_search", source=None, query="BTK inhibitor CLL", analogical_query=None), + synthesizer=_synth, adapter=adapter, + ) + await g.ainvoke({"question": "q"}) + assert adapter.search_calls + assert adapter.search_calls[0]["query"] == "BTK inhibitor CLL" + assert adapter.search_calls[0]["ranking"] is None + + +async def test_ordinary_keyword_search_never_gets_ranking(): + """Regression guard: adding the analogical route must not leak + `ranking="analogical"` into any other question_type's search call.""" + adapter = FakeAdapter() + g = build_graph(router=_router("keyword_search"), synthesizer=_synth, adapter=adapter) + await g.ainvoke({"question": "what is the mechanism of metformin?"}) + assert adapter.search_calls + assert all(c["ranking"] is None for c in adapter.search_calls) + + +async def test_duplicate_doc_ids_yield_one_document_and_one_citation(): + """A result set can carry the same document twice — broad search mixes + corpora, and `filter` returns a server-rebuilt list. Left undeduplicated, + one paper occupies two `max_documents` slots, is fetched twice, and appears + as two numbered references the synthesizer can cite as independent + support.""" + from crossbar_llm.paperclip_tools.nodes import assemble_context_node + + adapter = FakeAdapter() + state = { + "hits": [ + PaperHit(doc_id="PMC1", title="T1", source="pmc"), + PaperHit(doc_id="PMC1", title="T1", source="pmc"), + PaperHit(doc_id="PMC2", title="T2", source="pmc"), + ], + "warnings": [], + } + out = await assemble_context_node(state, adapter=adapter, max_documents=5, use_map=False) + + assert [d.doc_id for d in out["documents"]] == ["PMC1", "PMC2"] + assert [c.ref_num for c in out["citations"]] == [1, 2] + fetched = [c["doc_id"] for c in adapter.meta_calls] + assert fetched.count("PMC1") == 1 # not fetched twice + + +async def test_meta_recovered_hit_still_gets_full_text(): + """get_meta failing does not make the document unreachable — assembly + recovers a citable record from the hit's own fields. The body must still be + fetched, or every such paper silently drops to abstract-only in depth mode. + """ + from crossbar_llm.paperclip_tools.nodes import assemble_context_node + + adapter = FakeAdapter(meta_error_ids=["PMC1"]) + state = { + "hits": [PaperHit(doc_id="PMC1", title="T1", source="pmc", snippet="snip")], + "warnings": [], + "full_text": True, + } + out = await assemble_context_node(state, adapter=adapter, max_documents=5, use_map=False) + + assert [d.doc_id for d in out["documents"]] == ["PMC1"] + assert out["documents"][0].body, "recovered hit lost its full text" + assert [c["doc_id"] for c in adapter.content_calls] == ["PMC1"] + + +async def test_uncitable_hit_does_not_waste_a_content_fetch(): + """The converse: a hit that will be dropped (meta failed AND no title to + recover from) must not pay for a full-text fetch first.""" + from crossbar_llm.paperclip_tools.nodes import assemble_context_node + + adapter = FakeAdapter(meta_error_ids=["PMC9"]) + state = { + "hits": [PaperHit(doc_id="PMC9", title="", source="pmc")], + "warnings": [], + "full_text": True, + } + out = await assemble_context_node(state, adapter=adapter, max_documents=5, use_map=False) + + assert out["documents"] == [] + assert adapter.content_calls == [] + + +async def test_degenerate_answer_run_is_trimmed_and_warned(): + """A model that locks into repeating one character must not reach the caller. + + Observed on llama-3.3-70b: a complete answer, then ~6k `!` after the last + reference URL. The prose is salvageable, so the run is cut rather than the + answer dropped. + """ + async def degenerate(state): + return "Real answer. References: [1] PMC1" + "!" * 6161 + + g = build_graph(router=_router(), synthesizer=degenerate, adapter=FakeAdapter()) + out = await g.ainvoke({"question": "What treats X?"}) + + assert out["final_answer"] == "Real answer. References: [1] PMC1!!!" + assert any("degenerate character run" in w for w in out["warnings"]) + + +async def test_legitimate_punctuation_survives_and_warns_nothing(): + # The table alignment row is the case that matters: it runs well past the + # punctuation threshold, and trimming it stops the table rendering. + answer = ( + "Section\n" + "=" * 40 + "\nDone... see [1] --- and [2].\n" + "| Gene | p |\n|" + "-" * 30 + "|" + "-" * 30 + "|\n| TP53 | .01 |" + ) + + async def punctuated(state): + return answer + + g = build_graph(router=_router(), synthesizer=punctuated, adapter=FakeAdapter()) + out = await g.ainvoke({"question": "What treats X?"}) + + assert out["final_answer"] == answer + assert not any("degenerate" in w for w in out["warnings"]) + + +async def test_degeneration_on_a_rule_character_is_still_caught(): + """A run far past any real rule is degeneration whatever the character is.""" + async def degenerate(state): + return "Real answer." + "-" * 6000 + + g = build_graph(router=_router(), synthesizer=degenerate, adapter=FakeAdapter()) + out = await g.ainvoke({"question": "What treats X?"}) + + assert out["final_answer"] == "Real answer.---" + assert any("degenerate character run" in w for w in out["warnings"]) diff --git a/crossbar_llm/paperclip_tools/tests/test_live.py b/crossbar_llm/paperclip_tools/tests/test_live.py new file mode 100644 index 0000000..0c38e70 --- /dev/null +++ b/crossbar_llm/paperclip_tools/tests/test_live.py @@ -0,0 +1,275 @@ +"""Live smoke test against the real Paperclip MCP server. + +Skipped unless PAPERCLIP_API_KEY is set — Paperclip is a live, metered +dependency and cannot be mocked at the HTTP layer (see the integration doc). +This is the provenance gate: it asserts search returns hits carrying real +citable IDs (DOI/PMID), and that a full graph run produces a cited answer. +""" +from __future__ import annotations + +import os +import re + +import pytest + +try: + from dotenv import load_dotenv + + load_dotenv() # populate PAPERCLIP_API_KEY + LLM keys from .env if present +except Exception: # pragma: no cover - dotenv always installed here + pass + +# `live` makes these deselectable with `-m "not live"`. Without a key they +# already skip, but anyone holding one needs a way to run the offline suite +# deterministically — these assert against a third-party service whose outages +# are not ours to fix. +pytestmark = [ + pytest.mark.live, + pytest.mark.skipif( + not os.environ.get("PAPERCLIP_API_KEY"), + reason="PAPERCLIP_API_KEY not set; skipping live Paperclip smoke test.", + ), +] + + +async def test_live_search_and_meta_have_citable_ids(): + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + adapter = PaperclipAdapter() + result = await adapter.search( + "BTK inhibitor chronic lymphocytic leukemia", source="pmc", limit=3 + ) + assert result.hits, "expected search hits from the live server" + assert all(h.doc_id for h in result.hits) + + meta = await adapter.get_meta(result.hits[0].doc_id) + # Provenance gate: a real citable identifier must be present. + assert meta.doi or meta.pmid, f"no citable id on {result.hits[0].doc_id}" + assert meta.title + + +async def test_live_default_search_is_broad_via_rest(): + """`source=None` should hit the REST path (see + docs/paperclip_rest_endpoint_findings.md) and return real, structured, + multi-source results — the whole point of the migration. If REST ever + gets locked down for API-key auth, this test is the canary: it'll still + pass via the MCP fallback (comma-separated paper corpora), just without + the `score`/`corpus` fields REST provides, which the second assertion + block would then need loosening.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + adapter = PaperclipAdapter() + result = await adapter.search( + "BTK inhibitor chronic lymphocytic leukemia", source=None, limit=8 + ) + assert result.hits, "expected hits from a broad/unscoped search" + assert all(h.doc_id for h in result.hits) + # More than one corpus represented is the real signal this is a genuine + # broad search, not an accidental single-source one. + sources_seen = {h.source for h in result.hits if h.source} + assert len(sources_seen) >= 1 + + # If REST answered (not the MCP fallback), hits carry structured fields + # MCP text-parsing never provides. + if any(h.score is not None for h in result.hits): + rest_hit = next(h for h in result.hits if h.score is not None) + assert rest_hit.doi or rest_hit.pub_year or rest_hit.backend + + +async def test_live_arxiv_doc_id_round_trips(): + """Regression test for the arXiv ID truncation bug — a real arXiv hit's + doc_id must survive a `get_meta` round trip instead of 404ing on a + truncated id.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + adapter = PaperclipAdapter() + result = await adapter.search("transformer attention", source="arxiv", limit=3) + assert result.hits, "expected arXiv hits" + hit = result.hits[0] + assert "." in hit.doc_id, f"arXiv doc_id looks truncated: {hit.doc_id!r}" + meta = await adapter.get_meta(hit.doc_id, source="arxiv") + assert meta.title + + +async def test_live_pdb_chembl_return_real_hits(): + """Regression test for the pdb/chembl parser-dispatch bug — these used to + silently return zero hits despite the server sending real data.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + adapter = PaperclipAdapter() + pdb_result = await adapter.search("TP53", source="pdb", limit=3) + assert pdb_result.hits, "expected -s pdb to return real hits post-fix" + chembl_result = await adapter.search("aspirin", source="chembl", limit=3) + assert chembl_result.hits, "expected -s chembl to return real hits post-fix" + + +async def test_live_map_json_contract_returns_structured_extracts(): + """`run_map` asks for its {answer, found} contract in the question text + (never via --output_schema, which 500s on REST) — confirm the request + round-trips against the real server without erroring. + + Deliberately NOT asserting any extract gets `.data` populated: repeated + live runs during development showed the per-paper reader's schema + compliance is genuinely inconsistent — sometimes a clean flat + `{"answer":..., "found":...}`, sometimes the schema's own `properties` + structure echoed back with values misplaced inside it, and in one run + every paper in a 3-paper batch got the malformed shape. That's Paperclip + server-side model behavior we don't control, so gating a test on "at + least N papers get structured data" would be flaky by construction. What + IS our responsibility, and what this actually tests: the request is + well-formed (server accepts it, doesn't error) and every extract still + has usable `.text` regardless of whether `.data` parsed — i.e. the + defensive fallback in `_map_extract_from_text` holds up against real + server responses, not just the synthetic malformed-JSON cases covered in + test_paperclip_adapter.py's offline tests. Any extract that DOES get + `.data` must at least match our schema's keys. + """ + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + adapter = PaperclipAdapter() + search_result = await adapter.search( + "BTK inhibitor chronic lymphocytic leukemia", source="pmc", limit=3 + ) + assert search_result.search_id + + extracts = await adapter.run_map( + search_result.search_id, + "What delivery vector or inhibitor mechanism was reported?", + ) + assert extracts, "expected at least one map extract" + assert all(e.text for e in extracts), "every extract must have usable text regardless of .data" + for e in extracts: + if e.data is not None: + assert "answer" in e.data and "found" in e.data + # Regression coverage for the nested-malformed-shape crash and the + # found-detection bug (docs/paperclip_rest_endpoint_findings.md) — + # must normalize cleanly against whatever shape the real server sends + # today, not just the synthetic fixtures in test_paperclip_adapter.py. + assert e.found in (None, True, False) + assert all(isinstance(n, int) for n in e.citation_lines) + + +async def test_live_sql_returns_real_counts(): + """Confirm `adapter.sql()` round-trips a real read-only query — same + query manually verified during development (docs/paperclip_rest_endpoint_ + findings.md §12): fda is a single, SQL-queryable shard with a stable, + large row count.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + adapter = PaperclipAdapter() + result = await adapter.sql("SELECT COUNT(*) AS n FROM documents", source="fda") + assert result.columns == ["n"] + assert result.rows, "expected exactly one count row" + assert int(result.rows[0]["n"]) > 0 + + +async def test_live_sql_trials_is_not_queryable(): + """Regression/documentation test for a real server constraint (confirmed + live, not obvious from the docs): `documents` isn't backed by a table for + every source — trials errors outright. sql_node relies on this failing + cleanly (not hanging or returning nonsense) to trigger its search + fallback.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter, PaperclipError + + adapter = PaperclipAdapter() + with pytest.raises(PaperclipError): + await adapter.sql("SELECT COUNT(*) FROM documents", source="trials") + + +async def test_live_filter_trims_a_noisy_result_set(): + """Confirm `filter` genuinely narrows a deliberately noisy, broad search + down toward papers relevant to a specific sub-topic — the actual quality + signal this feature exists for, not just that the round trip parses.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + adapter = PaperclipAdapter() + search_result = await adapter.search("diabetes", source="pmc", limit=15) + assert search_result.search_id and search_result.hits + + filtered = await adapter.filter( + search_result.search_id, + "papers specifically about metformin's mechanism of action", + ) + assert filtered is not None, "expected REST to be available in this live test env" + assert len(filtered.hits) <= len(search_result.hits) + + +async def test_live_filter_returns_none_when_rest_disabled(monkeypatch): + """`filter` has no MCP fallback — confirm the documented degrade-to-None + contract holds against the real environment, not just a mocked one.""" + from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter + + adapter = PaperclipAdapter() + search_result = await adapter.search("diabetes", source="pmc", limit=3) + assert search_result.search_id + + monkeypatch.setenv("PAPERCLIP_DISABLE_REST", "1") + assert await adapter.filter(search_result.search_id, "anything") is None + + +async def test_live_sql_aggregate_graph_produces_answer(): + """Full production wiring, live: router decision fixed (not asking the + router LLM to pick sql_aggregate — that's a separate judgment-call + concern) so this specifically exercises sql_node -> synthesize_node's + PAPERCLIP_SQL_SYNTHESIZE_SYSTEM_PROMPT path against the real server and a + real LLM, end to end.""" + from crossbar_llm.paperclip_tools.agent import build_graph + from crossbar_llm.paperclip_tools.schemas import PaperclipRouterDecision + + async def fixed_router(_question): + return PaperclipRouterDecision( + question_type="sql_aggregate", + source="fda", + search_query="pembrolizumab", # fallback if sql fails + map_question="pembrolizumab", # unused on this route; required field + sql_query="SELECT COUNT(*) AS n FROM documents", + rationale="test: fixed sql_aggregate route", + ) + + llm = _get_llm_or_skip() + graph = build_graph(chat_model=llm, router=fixed_router) + state = await graph.ainvoke({"question": "how many FDA documents does Paperclip index?"}) + assert not state.get("sql_error"), f"sql query failed: {state.get('sql_error')}" + assert state.get("sql_rows"), "expected at least one row back" + assert state.get("final_answer"), "expected a synthesized answer" + # No per-paper citations for a SQL answer. + assert not state.get("citations") + + +def _get_llm_or_skip(): + """Build the project's configured chat model; skip if it isn't available.""" + import os + + model = os.environ.get("PAPERCLIP_TEST_MODEL", "gpt-4o-mini") + try: + from crossbar_llm.paperclip_tools.llm import build_chat_model + + return build_chat_model(model=model) + except Exception as e: # pragma: no cover - depends on configured providers + pytest.skip(f"chat model unavailable ({model}): {e}") + + +async def test_live_end_to_end_returns_cited_answer(): + from crossbar_llm.paperclip_tools.agent import build_graph + from crossbar_llm.paperclip_tools.usage import ainvoke_with_usage_capture + + llm = _get_llm_or_skip() + graph = build_graph(chat_model=llm, abstracts_only=True) + state, usage = await ainvoke_with_usage_capture( + graph, {"question": "What is the mechanism of action of metformin?"} + ) + assert state.get("final_answer"), "expected a synthesized answer" + assert state.get("citations"), "expected at least one citation" + # Usage is reported with input/output tokens separated. + assert usage["input_tokens"] > 0 and usage["output_tokens"] > 0 + + # use_map defaults True and this run doesn't disable it, so any citation + # sourced from map evidence with real `_citations` provenance should carry + # a line-anchored URL (docs/paperclip_rest_endpoint_findings.md §14). Not + # asserting every/any citation has one — schema compliance is genuinely + # inconsistent server-side (see test_live_map_json_contract_returns_ + # structured_extracts) — only that whichever ones do are well-formed. + for cit in state["citations"]: + if "#L" in cit.url: + anchor = cit.url.split("#", 1)[1] + assert re.fullmatch(r"L\d+(-L\d+)?(,\d+)*", anchor), f"malformed anchor: {cit.url}" diff --git a/crossbar_llm/paperclip_tools/tools.py b/crossbar_llm/paperclip_tools/tools.py new file mode 100644 index 0000000..3af7524 --- /dev/null +++ b/crossbar_llm/paperclip_tools/tools.py @@ -0,0 +1,222 @@ +"""Never-raise wrappers around the Paperclip MCP adapter. + +Paperclip is consumed by a deterministic adapter node, not by the +orchestrating LLM. So these are not LLM-facing LangChain `@tool`s over a global +client (as PubTator3's are); they are internal coroutines that take the injected +`PaperclipAdapterProtocol` (the single test seam) and wrap each call so it +**never raises** — failures come back in the `error` field. Nodes then decide how +to degrade, which is what keeps the graph robust to a single failing MCP call. +""" +from __future__ import annotations + +from pydantic import BaseModel + +from crossbar_llm.paperclip_tools.adapter import ( + MapExtract, + PaperclipAdapterProtocol, + PaperHit, + PaperMeta, +) + + +class SearchOutput(BaseModel): + hits: list[PaperHit] = [] + search_id: str | None = None + error: str | None = None + + +class MetaOutput(BaseModel): + meta: PaperMeta | None = None + error: str | None = None + + +class ContentOutput(BaseModel): + doc_id: str + content: str = "" + error: str | None = None + + +class MapOutput(BaseModel): + extracts: list[MapExtract] = [] + error: str | None = None + + +class SqlOutput(BaseModel): + columns: list[str] = [] + rows: list[dict] = [] + error: str | None = None + + +class FilterOutput(BaseModel): + hits: list[PaperHit] = [] + skipped: bool = False + error: str | None = None + + +async def paperclip_search( + adapter: PaperclipAdapterProtocol, + query: str, + *, + source: str | None = None, + limit: int = 10, + sort: str | None = None, + year: str | None = None, + ranking: str | None = None, +) -> SearchOutput: + """Run a Paperclip search. Returns ranked hits + the result-set id, or an + error envelope. + + `source=None` (the default) searches broadly across the general + literature corpora in one call. Pass an explicit `source` only for a + narrow corpus (fda/trials/proteins/pdb/chembl/a single preprint server). + + `ranking="analogical"` (analogical_search route only) finds papers + sharing the same structural method across domains rather than the same + topic — requires `query` to be a method/problem-description sentence, + not keywords (see `PaperclipRouterDecision.analogical_query`). + + An empty `hits` list with no `error` means the query genuinely matched + nothing (the node's fallback path handles that). + """ + try: + result = await adapter.search( + query, source=source, limit=limit, sort=sort, year=year, ranking=ranking + ) + return SearchOutput(hits=result.hits, search_id=result.search_id) + except Exception as e: + return SearchOutput(error=f"{type(e).__name__}: {e}") + + +async def paperclip_get_meta( + adapter: PaperclipAdapterProtocol, doc_id: str, *, source: str = "pmc" +) -> MetaOutput: + """Fetch a record's `meta.json` — the citable-ID record (DOI/PMID, or the + UniProt fields for the proteins corpus).""" + try: + meta = await adapter.get_meta(doc_id, source=source) + return MetaOutput(meta=meta) + except Exception as e: + return MetaOutput(error=f"{type(e).__name__}: {e}") + + +async def paperclip_get_content( + adapter: PaperclipAdapterProtocol, + doc_id: str, + *, + source: str = "pmc", + sections: list[str] | None = None, + max_lines: int | None = None, +) -> ContentOutput: + """Fetch full-text body for a document (the depth knob). + + `sections` (e.g. ["methods", "results"]) restricts retrieval to those body + sections; omit it for the whole body. + """ + try: + content = await adapter.get_content( + doc_id, source=source, sections=sections, max_lines=max_lines + ) + return ContentOutput(doc_id=doc_id, content=content) + except Exception as e: + return ContentOutput(doc_id=doc_id, error=f"{type(e).__name__}: {e}") + + +async def paperclip_map( + adapter: PaperclipAdapterProtocol, + search_id: str, + question: str, + *, + limit: int | None = None, +) -> MapOutput: + """Run Paperclip's `map` over a saved search result set — a server-side, + full-text, per-paper extraction of `question`. Never raises.""" + try: + extracts = await adapter.run_map(search_id, question, limit=limit) + return MapOutput(extracts=extracts) + except Exception as e: + return MapOutput(error=f"{type(e).__name__}: {e}") + + +_SELECT_ONLY_ERROR = "Only SELECT queries are allowed." +_MULTI_STATEMENT_ERROR = "Only a single statement is allowed." + +# A leading `WITH ... SELECT` is an ordinary read-only aggregate and the router +# can legitimately emit one, so the prefix check accepts it too. +_READ_ONLY_PREFIXES = ("SELECT", "WITH") + + +def _is_single_read_only_statement(query: str) -> str | None: + """Return an error string if `query` isn't a single read-only statement. + + Prefix-matching alone was not the "defense in depth" the docstring claimed: + `SELECT 1; DROP TABLE documents` starts with SELECT and sailed through. + Paperclip enforces read-only server-side, so this was never the only guard, + but a check that misses the obvious case is worse than no check at all + because it reads as protection. + """ + stripped = query.strip().rstrip(";").strip() + if not stripped.upper().startswith(_READ_ONLY_PREFIXES): + return _SELECT_ONLY_ERROR + # A `;` with anything after it means a second statement was appended. + if ";" in stripped: + return _MULTI_STATEMENT_ERROR + return None + + +async def paperclip_sql( + adapter: PaperclipAdapterProtocol, + query: str, + *, + source: str | None = None, +) -> SqlOutput: + """Run a read-only SQL query against Paperclip's `documents` table. + + Rejects anything not starting with `SELECT` (case-insensitive) BEFORE + sending it — defense in depth against the router LLM emitting something + else, on top of (not instead of) the server's own read-only enforcement. + Never raises. + """ + guard_error = _is_single_read_only_statement(query) + if guard_error: + return SqlOutput(error=guard_error) + try: + result = await adapter.sql(query, source=source) + return SqlOutput(columns=result.columns, rows=result.rows) + except Exception as e: + return SqlOutput(error=f"{type(e).__name__}: {e}") + + +async def paperclip_filter( + adapter: PaperclipAdapterProtocol, search_id: str, query: str +) -> FilterOutput: + """Trim a saved search result set to relevant papers via Paperclip's + server-side relevance filter. Never raises. + + `adapter.filter()` returns `None` when REST is unavailable (filter has no + MCP fallback) — that's surfaced as `skipped=True`, not `error`, so callers + can tell "filtering wasn't possible, use the unfiltered hits" apart from + "filtering actually failed". + """ + try: + result = await adapter.filter(search_id, query) + if result is None: + return FilterOutput(skipped=True) + return FilterOutput(hits=result.hits) + except Exception as e: + return FilterOutput(error=f"{type(e).__name__}: {e}") + + +__all__ = [ + "SearchOutput", + "MetaOutput", + "ContentOutput", + "MapOutput", + "SqlOutput", + "FilterOutput", + "paperclip_search", + "paperclip_get_meta", + "paperclip_get_content", + "paperclip_map", + "paperclip_sql", + "paperclip_filter", +] diff --git a/crossbar_llm/paperclip_tools/usage.py b/crossbar_llm/paperclip_tools/usage.py new file mode 100644 index 0000000..0d0c023 --- /dev/null +++ b/crossbar_llm/paperclip_tools/usage.py @@ -0,0 +1,90 @@ +"""Token-usage capture / logging helpers for the PubTator3 graph. + +Both helpers wrap a graph invocation in `get_usage_metadata_callback`, +which aggregates `usage_metadata` from every chat-model call made under +the context — router, synthesizer, depth evaluator, JSON fallbacks. +Use `ainvoke_with_usage_logging` for fire-and-forget log lines and +`ainvoke_with_usage_capture` when the caller needs the totals +programmatically (benchmark runner, etc.). +""" +from __future__ import annotations + +import logging + +from langchain_core.callbacks import get_usage_metadata_callback + + +_usage_logger = logging.getLogger("crossbar_llm.pubtator3.usage") + + +def _flatten_usage(usage_metadata: dict) -> dict: + """Collapse {model: {input,output,total,...}} into one totals dict. + + Sums across models so a single run that touches router + synthesizer + + evaluator (possibly different model versions for each) reports one + consolidated number per field. Pulls out provider sub-buckets we care + about (reasoning, cache_read) when present. + """ + flat = { + "input_tokens": 0, + "output_tokens": 0, + "reasoning_tokens": 0, + "cache_read_tokens": 0, + "total_tokens": 0, + "by_model": {}, + } + for model, usage in usage_metadata.items(): + in_details = usage.get("input_token_details") or {} + out_details = usage.get("output_token_details") or {} + flat["input_tokens"] += usage.get("input_tokens") or 0 + flat["output_tokens"] += usage.get("output_tokens") or 0 + flat["total_tokens"] += usage.get("total_tokens") or 0 + flat["reasoning_tokens"] += out_details.get("reasoning") or 0 + flat["cache_read_tokens"] += in_details.get("cache_read") or 0 + flat["by_model"][model] = dict(usage) + return flat + + +async def ainvoke_with_usage_capture(graph, state: dict, **kwargs) -> tuple[dict, dict]: + """Like `ainvoke_with_usage_logging` but returns the usage instead of logging. + + Returns `(state, usage)` where `usage` is the flat-totals dict produced by + `_flatten_usage` — handy for benchmarks that need to record tokens per + question alongside the answer. + """ + with get_usage_metadata_callback() as cb: + result = await graph.ainvoke(state, **kwargs) + return result, _flatten_usage(cb.usage_metadata) + + +async def ainvoke_with_usage_logging(graph, state: dict, **kwargs) -> dict: + """Invoke `graph` and log aggregated LLM token usage for the run. + + Totals are logged per model name at INFO level on + `crossbar_llm.pubtator3.usage`. + """ + with get_usage_metadata_callback() as cb: + result = await graph.ainvoke(state, **kwargs) + for model, usage in cb.usage_metadata.items(): + in_details = usage.get("input_token_details") or {} + out_details = usage.get("output_token_details") or {} + reasoning = out_details.get("reasoning") + cache_read = in_details.get("cache_read") + _usage_logger.info( + "pubtator3 llm usage model=%s input=%s output=%s reasoning=%s " + "cache_read=%s total=%s", + model, + usage.get("input_tokens"), + usage.get("output_tokens"), + reasoning, + cache_read, + usage.get("total_tokens"), + ) + return result + + +__all__ = [ + "_flatten_usage", + "ainvoke_with_usage_capture", + "ainvoke_with_usage_logging", +] diff --git a/crossbar_llm/pubtator3_tools/__init__.py b/crossbar_llm/pubtator3_tools/__init__.py new file mode 100644 index 0000000..4f155d0 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/__init__.py @@ -0,0 +1,12 @@ +"""PubTator3 literature-evidence agent. + +Answers a biomedical question from NCBI's PubTator3 corpus with a cited answer. +Self-contained: nothing here imports the Paperclip agent, so the two evolve +independently. + + from crossbar_llm.pubtator3_tools.agent import build_graph + from crossbar_llm.pubtator3_tools.llm import build_chat_model + + graph = build_graph(chat_model=build_chat_model(model="gpt-4o-mini")) + state = await graph.ainvoke({"question": "...", "warnings": []}) +""" diff --git a/crossbar_llm/pubtator3_tools/agent.py b/crossbar_llm/pubtator3_tools/agent.py new file mode 100644 index 0000000..1beda69 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/agent.py @@ -0,0 +1,390 @@ +"""LangGraph orchestrator for PubTator3 literature evidence. + +This module defines `build_graph`, which wires the standalone nodes from +`agents.nodes` together with three inline LLM-bound nodes (router, +synthesizer, depth evaluator) that close over the chat model and prompts. + +Pydantic schemas live in `agents.schemas`; token-usage helpers live in +`agents.usage`. The names are re-exported here for backwards compatibility +with code that imported them from this module directly. +""" +from __future__ import annotations + +from langchain_core.language_models import BaseChatModel +from langchain_core.prompts import ( + ChatPromptTemplate, + HumanMessagePromptTemplate, + MessagesPlaceholder, + SystemMessagePromptTemplate, +) +from langgraph.graph import END, StateGraph + +from crossbar_llm.pubtator3_tools.structured_output import ( + _ainvoke_structured_with_json_fallback, + _extract_json_object, + _message_content_to_text, +) +from crossbar_llm.pubtator3_tools.nodes import ( + _add_warning, + export_node, + partner_discovery_node, + resolve_node, + search_node, +) +from crossbar_llm.pubtator3_tools.prompts import ( + DEPTH_EVAL_SYSTEM_PROMPT, + ROUTER_SYSTEM_PROMPT, + SYNTHESIZE_SYSTEM_PROMPT, +) +from crossbar_llm.pubtator3_tools.schemas import ( + DepthEvaluation, + EntityMention, + EvaluatorFn, + PubTator3State, + QuestionType, + RouterDecision, + RouterFn, + StructuredModel, + SynthesizerFn, +) +from crossbar_llm.pubtator3_tools.usage import ( + _flatten_usage, + ainvoke_with_usage_capture, + ainvoke_with_usage_logging, +) + + +async def _always_sufficient_evaluator(_state: PubTator3State) -> DepthEvaluation: + """Default no-op evaluator: declares every answer sufficient and skips the + refinement loop. Used when no chat_model and no explicit evaluator are + supplied (test seam) so the existing test path stays single-pass.""" + return DepthEvaluation(sufficient=True, missing=None, rationale="no-op evaluator") + + +def build_graph( + *, + chat_model: BaseChatModel | None = None, + router: RouterFn | None = None, + synthesizer: SynthesizerFn | None = None, + evaluator: EvaluatorFn | None = None, + max_partners: int = 5, + max_documents: int = 7, + abstracts_only: bool = False, +): + """Compile the PubTator3 LangGraph. + + Pass `chat_model` for production. Pass `router`/`synthesizer`/`evaluator` + directly to bypass the LLM (test seam). `evaluator` is optional; when + neither it nor `chat_model` is given, the no-op default treats every + answer as sufficient and the refinement loop never fires. + + `abstracts_only` is a deployment-level switch that forces title + abstract + retrieval regardless of what the router or depth evaluator decide. With it + enabled, the router's `full_text` / `sections` choices are clamped to + False / None and the depth evaluator's full-text refinement path is + short-circuited (no second pass). Use it when you want predictable token + cost and don't need PMC body text. + """ + if (router is None or synthesizer is None) and chat_model is None: + raise ValueError( + "build_graph requires either chat_model, or both router and synthesizer." + ) + if evaluator is None and chat_model is None: + evaluator = _always_sufficient_evaluator + + async def router_node(state: PubTator3State) -> dict: + warnings = list(state.get("warnings", [])) + try: + if router is not None: + decision = await router(state["question"]) + else: + prompt = ChatPromptTemplate.from_messages([ + SystemMessagePromptTemplate.from_template(ROUTER_SYSTEM_PROMPT), + MessagesPlaceholder("chat_history", optional=True), + HumanMessagePromptTemplate.from_template("User question: {question}"), + ]) + decision, used_json_fallback = await _ainvoke_structured_with_json_fallback( + chat_model=chat_model, + prompt=prompt, + schema=RouterDecision, + values={ + "question": state["question"], + "chat_history": state.get("chat_history", []), + }, + json_instruction=( + "The previous instruction defines the exact routing schema. " + "Return ONLY a valid JSON object for that schema. Do not use " + "Markdown, prose, tool calls, or extra keys." + ), + ) + if used_json_fallback: + warnings.append( + "router structured-output unavailable; used JSON fallback." + ) + except Exception as e: + # Provider-side schema validation (e.g. invented relation value) + # would otherwise crash the run. Degrade to keyword_search so the + # user still gets papers. + warnings.append(f"router failed ({type(e).__name__}); fell back to keyword_search.") + decision = RouterDecision( + question_type="keyword_search", + keyword_query=state["question"], + rationale=f"router error fallback: {e}", + ) + return { + "question_type": decision.question_type, + "mentions": decision.mentions, + "relation": decision.relation, + "e2_type": decision.e2_type, + "keyword_query": decision.keyword_query, + "full_text": False if abstracts_only else decision.full_text, + "sections": None if abstracts_only else decision.sections, + "rationale": decision.rationale, + "warnings": warnings, + } + + async def _partner_discovery(state): + return await partner_discovery_node(state, max_partners=max_partners) + + async def _export(state): + return await export_node(state, max_documents=max_documents) + + async def synthesize_node(state: PubTator3State) -> dict: + if synthesizer is not None: + answer = await synthesizer(state) + else: + ctx_lines: list[str] = [] + for p in state.get("passages", []): + ctx_lines.append(f"[PMID:{p.pmid}] ({p.section}) {p.text}") + for r in state.get("document_relations", []): + ctx_lines.append( + f"[PMID:{r.pmid}] relation={r.type} " + f"{r.role1_accession or '?'}->{r.role2_accession or '?'} " + f"score={r.score:.2f}" + ) + evidence = "\n".join(ctx_lines) if ctx_lines else "(no passages found)" + + # Full-text requested but only title/abstract sections came back + # means none of the retrieved papers are PMC Open Access. Surface + # that caveat so the synthesizer mentions it instead of pretending + # the answer is full-text-backed. + full_text_requested = bool(state.get("full_text", False)) + body_sections_present = any( + p.section not in ("title", "abstract") + for p in state.get("passages") or [] + ) + availability_note = "" + if full_text_requested and state.get("passages") and not body_sections_present: + availability_note = ( + "\n\nNOTE: full paper body text was requested but none of " + "the retrieved PMIDs are PMC Open Access — only titles " + "and abstracts are available. End your paragraph with an " + "explicit caveat that full-text body was unavailable for " + "these papers and the answer is derived from abstracts only." + ) + + prompt = ChatPromptTemplate.from_messages([ + SystemMessagePromptTemplate.from_template(SYNTHESIZE_SYSTEM_PROMPT), + MessagesPlaceholder("chat_history", optional=True), + HumanMessagePromptTemplate.from_template( + "User question:\n{question}\n\nEvidence:\n{evidence}{availability_note}\n\n" + "Write the final answer." + ), + ]) + chain = prompt | chat_model + msg = await chain.ainvoke({ + "question": state["question"], + "evidence": evidence, + "availability_note": availability_note, + "chat_history": state.get("chat_history", []), + }) + answer = msg.content if isinstance(msg.content, str) else str(msg.content) + return {"final_answer": answer} + + async def evaluate_depth_node(state: PubTator3State) -> dict: + # Short-circuit cases — no escalation lever to pull. Each path tags + # `depth_skip_reason` so downstream tooling (benchmark printer, API + # response) can distinguish "the LLM said sufficient" from "we skipped + # the LLM because escalation was impossible / disabled". + if abstracts_only: + return {"depth_sufficient": True, "depth_skip_reason": "abstracts_only enabled"} + if not state.get("final_answer") or not state.get("passages"): + return {"depth_sufficient": True, "depth_skip_reason": "no answer or no passages"} + if state.get("refinement_attempted"): + return {"depth_sufficient": True, "depth_skip_reason": "refinement already attempted"} + full_text_already = bool(state.get("full_text")) + current_sections = state.get("sections") or None + if full_text_already and current_sections is None: + return { + "depth_sufficient": True, + "depth_skip_reason": "already at max depth (full_text + no section filter)", + } + + warnings = list(state.get("warnings", [])) + current_sections_str = ( + ", ".join(current_sections) if current_sections else "(none — abstracts only so far)" + ) + try: + if evaluator is not None: + verdict = await evaluator(state) + else: + prompt = ChatPromptTemplate.from_messages([ + SystemMessagePromptTemplate.from_template(DEPTH_EVAL_SYSTEM_PROMPT), + HumanMessagePromptTemplate.from_template( + "User question:\n{question}\n\n" + "Generated answer:\n{answer}\n\n" + "Retrieval context so far:\n" + "- full_text: {full_text}\n" + "- current body sections: {current_sections}" + ), + ]) + verdict, used_json_fallback = await _ainvoke_structured_with_json_fallback( + chat_model=chat_model, + prompt=prompt, + schema=DepthEvaluation, + values={ + "question": state["question"], + "answer": state["final_answer"], + "full_text": full_text_already, + "current_sections": current_sections_str, + }, + json_instruction=( + "Return ONLY a valid JSON object for the depth-evaluation " + "schema. Do not use Markdown, prose, tool calls, or extra keys." + ), + ) + if used_json_fallback: + warnings.append( + "depth evaluator structured-output unavailable; used JSON fallback." + ) + except Exception as e: + warnings.append( + f"depth evaluator failed ({type(e).__name__}); accepting answer as-is." + ) + return { + "depth_sufficient": True, + "depth_skip_reason": f"evaluator error ({type(e).__name__}: {e})", + "warnings": warnings, + } + + if verdict.sufficient: + return { + "depth_sufficient": True, + "depth_missing": None, + "warnings": warnings, + } + + # Insufficient — refine. Escalation strategy: + # 1. If we haven't pulled full text yet, flip full_text=True and + # honour suggested_sections (or pull everything when unset). + # 2. If full text is already on but sections is set, UNION the + # evaluator's suggested sections with the current set; if the + # evaluator gave nothing new, fall back to pulling every body + # section (sections=None). + suggested = verdict.suggested_sections or [] + if not full_text_already: + next_sections = list(suggested) if suggested else None + next_full_text = True + else: + merged = list(dict.fromkeys([*(current_sections or []), *suggested])) + # If the union added nothing, drop the filter to pull every section. + next_sections = merged if suggested and merged != (current_sections or []) else None + next_full_text = True + + sections_msg = ( + f"sections={next_sections}" if next_sections is not None else "sections=ALL" + ) + warnings.append( + f"depth check flagged shallow answer: " + f"{verdict.missing or 'no specific gap reported'}; " + f"re-fetching with full_text={next_full_text}, {sections_msg}." + ) + return { + "depth_sufficient": False, + "depth_missing": verdict.missing, + "full_text": next_full_text, + "sections": next_sections, + "refinement_attempted": True, + "warnings": warnings, + } + + def _post_evaluate_route(state: PubTator3State) -> str: + if state.get("depth_sufficient", True): + return "end" + if not state.get("refinement_attempted"): + # Safety net: evaluate_depth_node should have set this, but + # bail out anyway so we never loop without the cap in place. + return "end" + return "refine" + + g = StateGraph(PubTator3State) + g.add_node("router", router_node) + g.add_node("resolve", resolve_node) + g.add_node("partner_discovery", _partner_discovery) + g.add_node("search", search_node) + g.add_node("export", _export) + g.add_node("synthesize", synthesize_node) + g.add_node("evaluate_depth", evaluate_depth_node) + + g.set_entry_point("router") + g.add_conditional_edges( + "router", + lambda s: s["question_type"], + { + "out_of_scope": END, + "single_node": "resolve", + "relation_known_pair": "resolve", + "relation_partner_discovery": "resolve", + "keyword_search": "search", + }, + ) + g.add_conditional_edges( + "resolve", + lambda s: s["question_type"], + { + "single_node": "search", + "relation_known_pair": "search", + "relation_partner_discovery": "partner_discovery", + }, + ) + g.add_edge("partner_discovery", "search") + g.add_edge("search", "export") + g.add_edge("export", "synthesize") + g.add_edge("synthesize", "evaluate_depth") + g.add_conditional_edges( + "evaluate_depth", + _post_evaluate_route, + {"end": END, "refine": "export"}, + ) + + return g.compile() + + +__all__ = [ + # Re-exported schemas (backwards compat with old imports). + "QuestionType", + "StructuredModel", + "EntityMention", + "RouterDecision", + "DepthEvaluation", + "PubTator3State", + "RouterFn", + "SynthesizerFn", + "EvaluatorFn", + # Re-exported nodes (backwards compat). + "resolve_node", + "partner_discovery_node", + "search_node", + "export_node", + "_add_warning", + "_message_content_to_text", + "_extract_json_object", + "_ainvoke_structured_with_json_fallback", + # Re-exported usage helpers (backwards compat). + "_flatten_usage", + "ainvoke_with_usage_capture", + "ainvoke_with_usage_logging", + # Core builder + default evaluator. + "build_graph", + "_always_sufficient_evaluator", +] diff --git a/crossbar_llm/pubtator3_tools/client.py b/crossbar_llm/pubtator3_tools/client.py new file mode 100644 index 0000000..42fc778 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/client.py @@ -0,0 +1,427 @@ +"""Async client for the NCBI PubTator3 REST API. + +Wraps four endpoints (entity autocomplete, relation discovery, article +search, BioC JSON export) with typed Pydantic models. End-to-end +orchestration lives in `crossbar_llm.pubtator3_tools.agent`. + +API docs: https://www.ncbi.nlm.nih.gov/research/pubtator3/api +""" +from pydantic import BaseModel, ConfigDict, Field, model_validator +from typing import Any, Callable, Literal, TypeVar +import httpx +import asyncio +import logging +import weakref +from aiolimiter import AsyncLimiter +import re + +_log = logging.getLogger(__name__) + +# --- Module constants --------------------------------------------------------- +BASE_URL = "https://www.ncbi.nlm.nih.gov/research/pubtator3-api" +USER_AGENT = "CROssBAR-LLM/0.1 (+https://github.com/HUBioDataLab/CROssBAR_LLM)" +DEFAULT_TIMEOUT_S = 15.0 +RATE_LIMIT_PER_SECOND = 3 # PubTator3 IP-wide policy +RATE_LIMIT_TIME_PERIOD_S = 1 +EXPORT_PMID_BATCH = 100 # /publications/export/biocjson cap per call +RETRY_429_BACKOFF_S = 2.0 # backoff before the single retry attempt + +# HTTP statuses we treat as transient (worth one retry). 429 = rate limit; +# 502/503/504 = upstream gateway / availability glitches NCBI hits intermittently. +TRANSIENT_STATUSES = frozenset({429, 502, 503, 504}) + +# Full-text papers from /publications/export/biocjson include boilerplate +# sections (competing interests, acknowledgements, supplementary-material +# references, bibliography, etc.) that carry no scientific content but inflate +# the synthesizer's prompt. Drop them at parse time. +SKIP_SECTION_TYPES = frozenset({ + "COMP_INT", + "ACK_FUND", + "AUTH_CONT", + "SUPPL", + "REF", + "ABBR", + "KEYWORDS", + "APPENDIX", + "REVIEW_INFO", +}) + +ABSTRACT_ONLY_SECTIONS = frozenset({"title", "abstract", "TITLE", "ABSTRACT"}) + +# --- Per-loop singletons ------------------------------------------------------ +# httpx.AsyncClient and aiolimiter.AsyncLimiter are both bound to the event loop +# that creates them, so re-using a module-level instance across `asyncio.run` +# calls (tests, scripts) crashes with "Event loop is closed". Keying by the +# running loop fixes that without losing connection pooling within one loop. +_clients_by_loop: "weakref.WeakKeyDictionary[asyncio.AbstractEventLoop, httpx.AsyncClient]" = weakref.WeakKeyDictionary() +_limiters_by_loop: "weakref.WeakKeyDictionary[asyncio.AbstractEventLoop, AsyncLimiter]" = weakref.WeakKeyDictionary() + +class EntityCandidate(BaseModel): + """Represents a single entity match from the autocomplete API.""" + accession: str = Field(alias="_id") # Map "_id" from API to "accession" + name: str + # Known values: gene, chemical, disease, species, variant, cellline. Kept as + # a plain str rather than a Literal: PubTator3 is an evolving service, and + # pinning the enum meant one unfamiliar concept type raised for the whole + # batch, discarding every other candidate in the response. + biotype: str + db_id: str + db: str # "ncbi_gene", etc. + description: str = "" + match: str = "" + + model_config = ConfigDict(populate_by_name=True) # Allow both "_id" and "accession" + +class RelatedEntity(BaseModel): + """Represents a related entity from the related-entities API.""" + # Known values: associate, cause, compare, cotreat, drug_interact, inhibit, + # interact, negative_correlate, positive_correlate, prevent, stimulate, + # treat. Plain str for the same reason as `EntityCandidate.biotype` — a + # single new relation type used to discard the entire partner list. + type: str + source: str + target: str + publications: int + +class SearchHit(BaseModel): + """One article hit from PubTator3 search.""" + pmid: int + title: str + journal: str | None = None + date: str | None = None + authors: list[str] = [] + doi: str | None = None + score: float | None = None + # raw highlight (keep for debugging) + text_hl: str | None = None + snippet: str | None = None + + model_config = ConfigDict(populate_by_name=True) + + @model_validator(mode="after") + def _populate_snippet(self) -> "SearchHit": + if self.snippet is None and self.text_hl is not None: + self.snippet = _clean_snippet(self.text_hl) + return self + +class PassageAnnotation(BaseModel): + """Represents a single annotated entity mention in a passage.""" + text: str + type: str + accession: str | None = None + identifier: str | None = None + offset: int + length: int + +class Passage(BaseModel): + """Represents a single passage (e.g. title, abstract section) of a PubTator3 article.""" + pmid: int + pmcid: str | None = None + title: str + section: str + text: str + offset: int + annotations: list[PassageAnnotation] = [] + +class DocumentRelation(BaseModel): + """Represents a single BioREx-extracted relation between two entities in a PubTator3 article. + + `pmid` is included so that when a flat list of relations is surfaced + across multiple documents (e.g. by the LangGraph export node), each + entry retains its provenance. + + Both PubTator3-style accessions (`role1_accession`, `role2_accession`, + e.g. `@GENE_JAK1`) and underlying database identifiers (`role1_identifier`, + `role2_identifier`, e.g. `MESH:D008687` for chemicals/diseases or a bare + NCBI Gene ID for genes) are surfaced. CROssBAR's knowledge graph keys + on the database identifiers, so the latter is what to use for joins. + """ + pmid: int + type: str + role1_accession: str | None = None + role1_identifier: str | None = None + role2_accession: str | None = None + role2_identifier: str | None = None + score: float + +class PubTator3Document(BaseModel): + """Represents a full PubTator3 document with passages and relations.""" + pmid: int + pmcid: str | None = None + title: str + journal: str | None = None + authors: list[str] = [] + date: str | None = None + passages: list[Passage] = [] + relations: list[DocumentRelation] = [] + +_Parsed = TypeVar("_Parsed") + + +def _parse_items( + raw: Any, build: Callable[[Any], _Parsed], *, what: str +) -> list[_Parsed]: + """Parse a list of API records, skipping the ones that don't parse. + + Every record used to be built in one list comprehension, so a single + unparseable entry raised for the whole batch and the caller saw an empty + result — one unfamiliar relation type discarded all 352 partners. Skip the + bad record instead, and log how many were dropped so a schema drift on + PubTator3's side is visible rather than silent. + + A response that isn't a list at all (an error envelope, say) yields `[]` + rather than raising. + """ + if not isinstance(raw, list): + _log.warning("pubtator3 %s: expected a list, got %s", what, type(raw).__name__) + return [] + parsed: list[_Parsed] = [] + skipped = 0 + for item in raw: + try: + parsed.append(build(item)) + except Exception: + skipped += 1 + if skipped: + _log.warning( + "pubtator3 %s: skipped %d/%d unparseable record(s)", what, skipped, len(raw) + ) + return parsed + + +def _client() -> httpx.AsyncClient: + """Lazy per-loop async HTTP client (one instance per running event loop).""" + loop = asyncio.get_running_loop() + cli = _clients_by_loop.get(loop) + if cli is None: + cli = httpx.AsyncClient( + base_url=BASE_URL, + timeout=DEFAULT_TIMEOUT_S, + headers={ + "User-Agent": USER_AGENT, + "Accept": "application/json", + }, + ) + _clients_by_loop[loop] = cli + return cli + +def _limiter() -> AsyncLimiter: + """Lazy per-loop rate limiter — RATE_LIMIT_PER_SECOND req/s, IP-wide.""" + loop = asyncio.get_running_loop() + lim = _limiters_by_loop.get(loop) + if lim is None: + lim = AsyncLimiter(max_rate=RATE_LIMIT_PER_SECOND, time_period=RATE_LIMIT_TIME_PERIOD_S) + _limiters_by_loop[loop] = lim + return lim + +async def _request(method: str, path: str, *, params: dict | None = None) -> httpx.Response: + """Make an HTTP request with rate limiting and a single transient-failure retry. + + - 3 req/s ceiling enforced via the per-loop limiter (token bucket). + - Retries once after RETRY_429_BACKOFF_S seconds when the first attempt + either returns a status in TRANSIENT_STATUSES (429, 502, 503, 504) or + raises a transport-level error (connection dropped, read timeout, + RemoteProtocolError, etc. — anything subclassing httpx.RequestError). + - Persistent failures raise (HTTPStatusError for status codes, + httpx.RequestError for transport errors). + """ + async def _attempt() -> tuple[httpx.Response | None, Exception | None]: + try: + async with _limiter(): + return await _client().request(method, path, params=params), None + except httpx.RequestError as e: + return None, e + + response, error = await _attempt() + needs_retry = error is not None or ( + response is not None and response.status_code in TRANSIENT_STATUSES + ) + if needs_retry: + await asyncio.sleep(RETRY_429_BACKOFF_S) + response, error = await _attempt() + + if error is not None: + raise error + assert response is not None # one of (response, error) is always set + response.raise_for_status() + return response + + +async def autocomplete(query: str, *, concept: str | None = None, limit: int = 5) -> list[EntityCandidate]: + """Resolve free-text entity names to PubTator3 accessions.""" + params = {"query": query, "limit": limit} + if concept: + params["concept"] = concept + + response = await _request("GET", "/entity/autocomplete/", params=params) + raw = response.json() + + return _parse_items( + raw, lambda item: EntityCandidate(**item), what="autocomplete" + ) + +async def find_related(e1: str, *, relation: str, e2_type: str) -> list[RelatedEntity]: + """Find related entities of a specific type for a given entity.""" + params = {"e1": e1, "type": relation, "e2": e2_type} + response = await _request("GET", "/relations", params=params) + raw = response.json() + + return _parse_items(raw, lambda item: RelatedEntity(**item), what="relations") + +def _clean_snippet(text: str | None) -> str | None: + if not text: + return text + + # initial simple version + return re.sub( + r"@(?:<m>)?[A-Z]+_[^\s<]+(?:</m>)? @\S+ @@@(.+?)@@@", + r"\1", + text, + ) + +async def search(text_query: str, *, page: int = 1) -> tuple[list[SearchHit], int]: + """Search PubTator3 articles. Calls GET /search/. Returns (hits, total_count).""" + params = {"text": text_query, "page": page} + + response = await _request("GET", "/search/", params=params) + raw = response.json() + + hits = _parse_items( + raw.get("results"), lambda item: SearchHit(**item), what="search" + ) + total = raw.get("count", 0) + + return hits, total + +def _safe_float(value, default: float = 0.0) -> float: + """Coerce a possibly-None / possibly-string value to float. Defaults on failure.""" + if value is None: + return default + try: + return float(value) + except (TypeError, ValueError): + return default + + +def _parse_annotation(raw: dict) -> PassageAnnotation: + """One entity annotation from a passage. Returns None for unresolved entities.""" + infons = raw.get("infons") or {} + if not infons.get("valid", True): + return None + locations = raw.get("locations") or [{}] + loc = locations[0] or {} + return PassageAnnotation( + text=raw.get("text") or "", + type=infons.get("type") or "", + accession=infons.get("accession"), + identifier=infons.get("identifier"), + offset=loc.get("offset") or 0, + length=loc.get("length") or 0, + ) + +def _parse_passage(raw: dict, *, doc_pmid: int, doc_pmcid: str | None, doc_title: str) -> Passage | None: + """One passage (title, abstract, or section) within a document. + + Returns None for boilerplate sections we drop entirely (see + SKIP_SECTION_TYPES). Full-text passages carry their semantic role in + `infons.section_type` (METHODS, RESULTS, COMP_INT, ...); abstract-mode + passages carry it in `infons.type` (title, abstract). + """ + infons = raw.get("infons") or {} + section = infons.get("section_type") or infons.get("type") or "" + if section in SKIP_SECTION_TYPES: + return None + annotations = [ + a for a in (_parse_annotation(x) for x in (raw.get("annotations") or [])) + if a is not None + ] + return Passage( + pmid=doc_pmid, + pmcid=doc_pmcid, + title=doc_title, + section=section, + text=raw.get("text") or "", + offset=raw.get("offset") or 0, + annotations=annotations, + ) + +def _parse_document_relation(raw: dict, *, doc_pmid: int) -> DocumentRelation: + """One BioREx document-level relation. `doc_pmid` is threaded in so the + relation keeps its provenance when flattened across multiple docs.""" + infons = raw.get("infons") or {} + role1 = infons.get("role1") or {} + role2 = infons.get("role2") or {} + return DocumentRelation( + pmid=doc_pmid, + type=infons.get("type") or "", + role1_accession=role1.get("accession"), + role1_identifier=role1.get("identifier"), + role2_accession=role2.get("accession"), + role2_identifier=role2.get("identifier"), + score=_safe_float(infons.get("score")), + ) + +def _parse_document(raw: dict) -> PubTator3Document: + """One doc within the {'PubTator3': [...]} array.""" + pmid = int(raw["pmid"]) + pmcid = raw.get("pmcid") + + title = "" + for p in raw.get("passages") or []: + if (p.get("infons") or {}).get("type") == "title": + title = p.get("text") or "" + break + + passages = [ + p for p in ( + _parse_passage(raw_p, doc_pmid=pmid, doc_pmcid=pmcid, doc_title=title) + for raw_p in (raw.get("passages") or []) + ) + if p is not None + ] + relations = [ + _parse_document_relation(r, doc_pmid=pmid) + for r in (raw.get("relations") or []) + ] + + return PubTator3Document( + pmid=pmid, + pmcid=pmcid, + title=title, + journal=raw.get("journal"), + authors=raw.get("authors", []), + date=raw.get("date"), + passages=passages, + relations=relations, + ) + +async def export_biocjson(pmids: list[int], *, full: bool = True) -> list[PubTator3Document]: + """GET /publications/export/biocjson. Chunks PMIDs into batches of + EXPORT_PMID_BATCH so the URL doesn't blow up on long lists. Missing + PMIDs are silently dropped by the API. Batches run concurrently; the + rate limiter still serialises them at RATE_LIMIT_PER_SECOND req/s.""" + if not pmids: + return [] + + chunks = [ + pmids[i : i + EXPORT_PMID_BATCH] + for i in range(0, len(pmids), EXPORT_PMID_BATCH) + ] + + async def _fetch(chunk: list[int]) -> list[PubTator3Document]: + params = { + "pmids": ",".join(str(p) for p in chunk), + "full": "true" if full else "false", + } + response = await _request("GET", "/publications/export/biocjson", params=params) + raw = response.json() + return _parse_items( + raw.get("PubTator3"), _parse_document, what="export" + ) + + batches = await asyncio.gather(*(_fetch(c) for c in chunks)) + + docs: list[PubTator3Document] = [] + for b in batches: + docs.extend(b) + return docs diff --git a/crossbar_llm/pubtator3_tools/llm.py b/crossbar_llm/pubtator3_tools/llm.py new file mode 100644 index 0000000..d8d7ea4 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/llm.py @@ -0,0 +1,41 @@ +"""Chat model construction, via the project's shared LLM factory. + +The agent graph takes an already-built `chat_model` rather than building one +itself: that keeps the graph independent of how the model is configured, and +is the seam the tests inject fakes at. This helper is the convenience path for +callers who just want the project's configured model. +""" +from __future__ import annotations + +from langchain_core.callbacks import BaseCallbackHandler +from langchain_core.language_models import BaseChatModel + +from crossbar_llm.agent_tools.config import LLMConfig, ReasoningConfig +from crossbar_llm.agent_tools.llm_factory import LLMFactory + + +def build_chat_model( + *, + model: str, + provider: str | None = None, + temperature: float = 0.0, + callbacks: list[BaseCallbackHandler] | None = None, + reasoning: ReasoningConfig | None = None, +) -> BaseChatModel: + """Build a chat model for this agent. + + `provider` may be omitted — the factory infers it from the model name. + Temperature defaults to 0.0 because routing and synthesis both want + reproducible output, where the factory's own default is 1.0. + """ + config = LLMConfig( + model=model, + provider=provider, + temperature=temperature, + callbacks=callbacks or [], + reasoning=reasoning or ReasoningConfig(), + ) + return LLMFactory(config).get_base_model() + + +__all__ = ["build_chat_model"] diff --git a/crossbar_llm/pubtator3_tools/nodes.py b/crossbar_llm/pubtator3_tools/nodes.py new file mode 100644 index 0000000..d181b12 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/nodes.py @@ -0,0 +1,352 @@ +"""Standalone LangGraph node functions for the PubTator3 pipeline. + +Each node is a plain coroutine that takes the current `PubTator3State` and +returns a partial state dict to merge. The LLM-bound nodes (router, +synthesize, evaluate_depth) live inside `build_graph` because they close +over the chat model and prompts — every node here is pure data plumbing +around the four PubTator3 tools. +""" +from __future__ import annotations + +import asyncio +import json +from typing import Any + +from langchain_core.language_models import BaseChatModel +from langchain_core.prompts import ChatPromptTemplate, HumanMessagePromptTemplate + +from crossbar_llm.pubtator3_tools.structured_output import ( + _ainvoke_structured_with_json_fallback, +) +from crossbar_llm.pubtator3_tools.schemas import ( + EntityMention, + PubTator3State, + StructuredModel, +) +from crossbar_llm.pubtator3_tools.tools import ( + pubtator3_autocomplete, + pubtator3_export_passages, + pubtator3_find_partners, + pubtator3_search_articles, +) +from crossbar_llm.pubtator3_tools.client import ( + ABSTRACT_ONLY_SECTIONS, + DocumentRelation, + EntityCandidate, + Passage, +) + + +def _add_warning(state: PubTator3State, msg: str) -> list[str]: + return [*state.get("warnings", []), msg] + + +def _is_confident_match(candidate: EntityCandidate) -> bool: + """Whether PubTator3 matched the query to this candidate with confidence. + + The autocomplete `match` field reports HOW the query matched: + 'Matched on name <m>BTK</m>' + 'Matched on synonyms <m>LY450139</m>' + 'Multiple matches' + A name or synonym match means PubTator3 found the query in its entity + dictionary. 'Multiple matches' is its low-confidence fuzzy/token fallback + and is the dominant source of wrong resolutions (e.g. 'histone H3' -> + @GENE_HTR12, 'Fibrodysplasia Ossificans Progressiva' -> Myositis + Ossificans). We trust only name/synonym matches; anything else is left + unresolved so the entity falls through to the keyword-search fallback. + """ + lowered = (candidate.match or "").lower() + return lowered.startswith("matched on name") or lowered.startswith( + "matched on synonym" + ) + + +async def resolve_node(state: PubTator3State) -> dict: + mentions = state.get("mentions") or [] + if not mentions: + return {"resolved": {}, "unresolved": []} + + async def _one(m: EntityMention): + out = await pubtator3_autocomplete.ainvoke( + {"query": m.text, "concept": m.suggested_type, "limit": 5} + ) + return m.text, out + + results = await asyncio.gather(*(_one(m) for m in mentions)) + + resolved: dict[str, EntityCandidate] = {} + unresolved: list[str] = [] + warnings = list(state.get("warnings", [])) + + for text, out in results: + if out.error: + warnings.append(f"autocomplete failed for '{text}': {out.error}") + unresolved.append(text) + continue + if not out.candidates: + unresolved.append(text) + continue + + # Trust only name/synonym matches — PubTator3 ranks those above its + # fuzzy 'Multiple matches' fallback, so the first confident candidate + # is the right pick. If none are confident, leave the entity + # unresolved rather than inject a fuzzy (often wrong) accession into a + # structured query; downstream keyword-search fallback handles it. + confident = [c for c in out.candidates if _is_confident_match(c)] + if not confident: + unresolved.append(text) + warnings.append( + f"'{text}' had no confident PubTator3 name/synonym match " + f"(best candidate '{out.candidates[0].accession}' was a fuzzy " + f"'Multiple matches' hit); left unresolved for keyword fallback." + ) + continue + + resolved[text] = confident[0] + if len(confident) > 1: + warnings.append( + f"'{text}' was ambiguous ({len(confident)} confident candidates); " + f"picked '{confident[0].accession}'." + ) + + return {"resolved": resolved, "unresolved": unresolved, "warnings": warnings} + + +async def partner_discovery_node( + state: PubTator3State, *, max_partners: int = 5 +) -> dict: + mentions = state.get("mentions") or [] + resolved = state.get("resolved") or {} + if not mentions: + return {"partners": [], "warnings": _add_warning(state, "No mentions to anchor partner discovery.")} + + anchor = resolved.get(mentions[0].text) + if not anchor: + return { + "partners": [], + "warnings": _add_warning( + state, f"Could not resolve anchor entity '{mentions[0].text}'." + ), + } + + relation = state.get("relation") + e2_type = state.get("e2_type") + if not relation or not e2_type: + return { + "partners": [], + "warnings": _add_warning( + state, "partner_discovery requires both `relation` and `e2_type`." + ), + } + + out = await pubtator3_find_partners.ainvoke( + {"e1_accession": anchor.accession, "relation": relation, "e2_type": e2_type} + ) + if out.error: + return { + "partners": [], + "warnings": _add_warning(state, f"find_partners failed: {out.error}"), + } + + partners = out.partners[:max_partners] + if not partners: + return { + "partners": [], + "warnings": _add_warning( + state, + f"No '{relation}' partners of type '{e2_type}' found for " + f"{anchor.accession}.", + ), + } + return {"partners": partners} + + +async def search_node(state: PubTator3State) -> dict: + qtype = state.get("question_type") + mentions = state.get("mentions") or [] + resolved = state.get("resolved") or {} + + queries: list[str] = [] + warnings = list(state.get("warnings", [])) + + if qtype == "relation_partner_discovery": + for partner in state.get("partners") or []: + queries.append(f"relations:{partner.type}|{partner.source}|{partner.target}") + elif qtype == "relation_known_pair": + relation = state.get("relation") + if len(mentions) < 2 or not relation: + warnings.append("relation_known_pair needs two mentions plus a relation.") + else: + e1 = resolved.get(mentions[0].text) + e2 = resolved.get(mentions[1].text) + if e1 and e2: + queries.append(f"relations:{relation}|{e1.accession}|{e2.accession}") + else: + # One or both entities failed to resolve (often because the + # mention is a generic descriptor like 'antidote' that isn't + # a PubTator3 entity). Fall back to a keyword query built + # from whatever resolved + the unresolved mention surface form. + fallback_terms: list[str] = [] + for m, ent in ((mentions[0], e1), (mentions[1], e2)): + fallback_terms.append(ent.name if ent else m.text) + if relation: + fallback_terms.append(relation) + fallback_query = " ".join(fallback_terms) + queries.append(fallback_query) + warnings.append( + "known-pair entity resolution incomplete; fell back to " + f"keyword query: {fallback_query!r}." + ) + elif qtype == "single_node": + if mentions: + entity = resolved.get(mentions[0].text) + if entity: + queries.append(entity.accession) + else: + warnings.append( + f"Could not resolve '{mentions[0].text}' for single-node search." + ) + elif qtype == "keyword_search": + kq = (state.get("keyword_query") or "").strip() + if kq: + queries.append(kq) + else: + warnings.append("keyword_search needs a non-empty keyword_query.") + + async def _one(q: str): + return q, await pubtator3_search_articles.ainvoke({"text_query": q}) + + seen: set[int] = set() + pmids: list[int] = [] + total = 0 + + if queries: + results = await asyncio.gather(*(_one(q) for q in queries)) + for q, out in results: + if out.error: + warnings.append(f"search failed for '{q}': {out.error}") + continue + total += out.total + for hit in out.hits: + if hit.pmid not in seen: + seen.add(hit.pmid) + pmids.append(hit.pmid) + + # Zero-results fallback. Structured PubTator3 queries (relations:|...|..., + # bare accession) often return nothing even when the literature clearly + # discusses the topic — BioREx may have tagged the relation under a + # different vocabulary value, or the entity pair isn't co-mentioned in + # PubTator3's graph. The router already emits `keyword_query` for every + # in-scope route precisely as a contingency for this case; we simply + # re-run search with it as a free-text query (the same code path the + # keyword_search route uses). If the router didn't fill `keyword_query` + # (e.g. router error fallback set only question_type), use the user's + # question verbatim as the catastrophic-failure tier. + if not pmids and qtype in ("relation_partner_discovery", "relation_known_pair", "single_node"): + fallback_query = (state.get("keyword_query") or "").strip() or (state.get("question") or "").strip() + if fallback_query and fallback_query not in queries: + warnings.append( + f"structured query returned 0 PMIDs; falling back to " + f"keyword search: {fallback_query!r}." + ) + queries.append(fallback_query) + fb_out = await pubtator3_search_articles.ainvoke({"text_query": fallback_query}) + if fb_out.error: + warnings.append(f"fallback search failed: {fb_out.error}") + else: + total += fb_out.total + for hit in fb_out.hits: + if hit.pmid not in seen: + seen.add(hit.pmid) + pmids.append(hit.pmid) + + if not pmids: + warnings.append("No articles found for the constructed query.") + + return { + "queries_used": queries, + "pmids": pmids, + "total_articles": total, + "warnings": warnings, + } + + +async def export_node(state: PubTator3State, *, max_documents: int = 10) -> dict: + pmids = state.get("pmids") or [] + if not pmids: + return { + "documents": [], + "passages": [], + "document_relations": [], + } + + full_text = bool(state.get("full_text", False)) + + # PubTator3 /publications/export/biocjson caps at 50 docs per request. + export_count = min(max_documents, 50) + pmids_for_export = pmids[:export_count] + + out = await pubtator3_export_passages.ainvoke( + {"pmids": pmids_for_export, "full_text": full_text} + ) + + warnings = list(state.get("warnings", [])) + if out.error: + warnings.append(f"export failed: {out.error}") + return { + "documents": [], + "passages": [], + "document_relations": [], + "warnings": warnings, + } + + docs = out.documents + returned = {d.pmid for d in docs} + missing = [p for p in pmids_for_export if p not in returned] + if missing: + warnings.append( + f"{len(missing)} PMIDs were not returned by the export endpoint." + ) + + section_filter = state.get("sections") or None + allowed_body: set[str] | None = set(section_filter) if section_filter else None + + all_passages: list[Passage] = [] + all_doc_relations: list[DocumentRelation] = [] + for d in docs: + if full_text: + if allowed_body is None: + all_passages.extend(d.passages) + else: + # Always keep title + abstract; restrict body to the requested set. + all_passages.extend( + p for p in d.passages + if p.section in ABSTRACT_ONLY_SECTIONS or p.section in allowed_body + ) + else: + # PubTator3 occasionally returns body sections even when full=false + # was requested (review articles especially). Enforce abstract mode locally. + all_passages.extend( + p for p in d.passages if p.section in ABSTRACT_ONLY_SECTIONS + ) + all_doc_relations.extend(d.relations) + + return { + "documents": docs, + "passages": all_passages, + "document_relations": all_doc_relations, + "warnings": warnings, + } + + +__all__ = [ + "_add_warning", + "_message_content_to_text", + "_extract_json_object", + "_ainvoke_structured_with_json_fallback", + "_is_confident_match", + "resolve_node", + "partner_discovery_node", + "search_node", + "export_node", +] diff --git a/crossbar_llm/pubtator3_tools/prompts.py b/crossbar_llm/pubtator3_tools/prompts.py new file mode 100644 index 0000000..8b8698c --- /dev/null +++ b/crossbar_llm/pubtator3_tools/prompts.py @@ -0,0 +1,486 @@ +"""System prompts for the PubTator3 agent.""" +from __future__ import annotations + + +RELATION_VOCABULARY: dict[str, str] = { + "treat": "a chemical/drug treats a disease.", + "cause": "positive correlation; chemical-induced diseases and variant-caused genetic diseases.", + "associate": "generic association with no specific direction; applies to various entity pairs.", + "prevent": "negative correlation; includes variant-disease.", + "positive_correlate": "same-direction co-movement; chemical-gene, chemical co-expression, gene co-expression.", + "negative_correlate": "opposite-direction co-movement; chemical-gene, chemical co-expression, gene co-expression.", + "compare": "comparing the effect of two chemicals/drugs.", + "cotreat": "two or more chemicals/drugs administered together or as a fixed-dose combination.", + "inhibit": "negative correlation; includes disease-gene and chemical-variant.", + "stimulate": "positive correlation; includes disease-gene and disease-variant.", + "interact": "physical interaction such as protein binding; gene-gene, gene-chemical, chemical-variant.", + "drug_interact": "pharmacodynamic interaction between two chemicals producing an array of side effects.", +} + +ENTITY_VOCABULARY: dict[str, str] = { + "Gene": "NCBI Gene IDs.", + "Disease": "MeSH (Medical Subject Headings).", + "Chemical": "MeSH (Medical Subject Headings).", + "Variant": "dbSNP IDs when available, otherwise HGVS format.", + "Species": "NCBI Taxonomy IDs.", + "CellLine": "Cellosaurus IDs.", +} + + +def _format_table(d: dict[str, str]) -> str: + width = max(len(k) for k in d) + return "\n".join(f" - {k.ljust(width)} : {v}" for k, v in d.items()) + + +_ROUTER_INSTRUCTIONS = """\ +You are the routing layer of a PubTator3 literature-evidence agent. +PubTator3 is an NCBI service that indexes PubMed/PMC papers. It tags +six entity types (Gene, Chemical, Disease, Species, Variant, CellLine) +and BioREx relations among them — BUT the underlying papers contain +much more than those tags. Free-text search over PubMed still works +for topics PubTator3 has no entity type for (side effects, mechanism +of action, pharmacokinetics, orthologs, GO annotations, pathways). + +Your job is to classify the user's question into exactly ONE of five +`question_type` values and emit whatever fields that type requires. +Each value is defined below; examples come AFTER the definitions and +illustrate them — do not pattern-match on the examples alone. + + +# Question types + +single_node + The user wants information about ONE biomedical entity, with no + relation in play. The downstream pipeline will resolve the entity + and return a literature snapshot of it. Use this when the question + is descriptive ("what is X", "tell me about X") rather than + relational. Required fields: `mentions` with exactly one entry. + +relation_known_pair + The user EXPLICITLY names BOTH endpoints of a relation in the text + and wants supporting literature for the connection between them. + Both entities must appear verbatim (or as obvious synonyms / + abbreviations) in the user's question. The downstream pipeline + resolves both entities and runs one relation-expression search like + `relations:treat|@CHEMICAL_X|@DISEASE_Y`. Required fields: + `mentions` as [e1, e2] in role order (e1 is the subject / actor; + e2 is the object / partner), and `relation`. + + CRITICAL: do NOT invent the second entity from your own knowledge + of the answer. If the user asks "what is the target of drug X?", + the target is the UNKNOWN — they're asking you to discover it. That + is partner_discovery, not known_pair, even if you happen to know + the answer. + +relation_partner_discovery + The user names ONE entity (the anchor) plus a relation type, and + ASKS WHICH entities of a given type relate to it that way. Signal + phrases: "which X ...", "what X ...", "what is the X of Y", "name + the X that ...", "list X that ...". The downstream pipeline asks + /relations for ranked partners, then searches per partner. Required + fields: `mentions` with exactly one entry (the anchor), `relation`, + and `e2_type`. + +keyword_search + The user's question is biomedical and likely has literature + support, but does NOT fit the three structured forms above. This + covers any topic where PubTator3 has no dedicated entity type or + relation but PubMed/PMC still contain the relevant text — side + effects, adverse events, mechanism of action, pharmacokinetics, + orthologs, GO annotations, pathway membership, regulatory history. + The downstream pipeline runs a free-text /search/ query directly + (no autocomplete, no partner discovery). Required field: + `keyword_query`, a focused 2–6-word PubMed expression distilled + from the user's question. Use this FREELY as a fallback before + resorting to out_of_scope. + +out_of_scope + PubTator3 / PubMed literature cannot help, OR the question requires + multi-step graph reasoning this literature agent must not attempt. + Reserve this for three cases: + - non-biomedical questions (math, news, weather, opinion, general + knowledge), + - pure graph-traversal OUTPUT requests where the user is asking + for a path or structure that only lives in a knowledge graph + (e.g. "what nodes are on the shortest path between X and Y"), + - MULTI-HOP questions answerable only by CHAINING two or more + relations through an intermediate entity the user does NOT name + (e.g. "which genes interact with the targets of drug X" = + drug→target→interacting gene; "what drugs target proteins + associated with disease Y" = disease→protein→drug). These need a + knowledge graph to traverse, not single-paper literature evidence; + that is a different agent's job, so decline them here. + + This agent answers only SINGLE-HOP questions. A question that names + both endpoints of one relation, or asks for the partners of ONE + relation on a named entity, is single-hop and stays IN scope — route + it to single_node / relation_known_pair / relation_partner_discovery / + keyword_search as usual. Only route to out_of_scope when answering + truly requires discovering an unnamed intermediate entity first and + then applying a second, different relation to it. + + A question is multi-hop ONLY when the OUTPUT of a first relation + becomes the INPUT to a second, different relation on an entity the + user never names. It is NOT multi-hop just because it asks for two + things about ONE named entity. Asking for several attributes or + entity types of a single named anchor — "which gene AND which protein + are associated with disease X", "the causes and symptoms of X", + "the gene and the pathway of Y" — is SINGLE-HOP (all attributes hang + off the same anchor) and stays in scope; route it to keyword_search. + The word "and" joining two attributes of one entity does not make a + question multi-hop. + If a biomedical question is single-hop and a PubMed paper might + discuss it, prefer keyword_search over out_of_scope. + + +# Field constraints + +- single_node: mentions=[one entity]; relation and e2_type stay null. +- relation_known_pair: mentions=[e1, e2] in role order; set `relation`; e2_type stays null. +- relation_partner_discovery: mentions=[anchor]; set both `relation` and `e2_type`. +- keyword_search: mentions optional; relation and e2_type stay null. +- out_of_scope: all fields may be null/empty. + +ALWAYS fill `keyword_query` for the four in-scope routes (single_node, +relation_known_pair, relation_partner_discovery, keyword_search). On +the keyword_search route it is the primary query the pipeline runs. On +the three structured routes it is the FALLBACK query the pipeline runs +ONLY when the structured PubTator3 query returns 0 PMIDs — which +happens often because BioREx didn't tag the exact relation the user +asked about, or the entity pair isn't co-mentioned in PubTator3's +graph even though PubMed/PMC clearly discusses it. Treat the fallback +query as carefully as the primary one: same 2–6 token PubMed style, +canonical entity names, the relation as a natural-language verb if +relevant (e.g. 'BTK inhibitor CLL', 'Denosumab RANKL', 'Nfat miR-25 +cardiac hypertrophy'). Leave `keyword_query` null only for +out_of_scope. + +For EVERY entity in `mentions`, ALWAYS set `suggested_type` when the +biotype is clear from context. This narrows autocomplete to candidates +of the right type and prevents collisions where a gene name happens to +match a disease label or vice versa. Heuristics: + + - Gene symbols / kinase / receptor / transcription factor names + (BTK, JAK1, TP53, EGFR, RANKL, KRAS, "Bruton's tyrosine kinase", + "tumor necrosis factor") -> suggested_type = "gene" + - Drug names, monoclonal antibodies, small-molecule inhibitors, + chemical compounds (metformin, ibrutinib, Denosumab, Imatinib, + LY450139) -> suggested_type = "chemical" + - Disease names, syndromes, cancers (Alzheimer disease, type-2 + diabetes, chronic lymphocytic leukemia, psoriasis) + -> suggested_type = "disease" + - Organism names (Mus musculus, Homo sapiens) -> "species" + - SNP IDs (rs6311), HGVS (p.Arg175His) -> "variant" + - Cell line names (HeLa, MCF-7) -> "cellline" + +Only leave `suggested_type` null when the type is genuinely ambiguous +or the mention is a generic noun (e.g. "drug", "treatment", "marker"). + + +# Canonical mention text for autocomplete + +`mentions[].text` is the query sent to PubTator3 autocomplete. Prefer +the user's exact surface form, but when an explicitly mentioned entity +has an obvious canonical biomedical lookup form, use that canonical +form. This is normalization, not invention: the normalized text must +refer to the same entity the user actually named. + +This is especially important for genes. PubTator3's gene autocomplete +often expects HGNC-style symbols rather than descriptive protein names: + + - "Bruton's tyrosine kinase" -> text="BTK", suggested_type="gene" + - "Janus kinase 1" or "JAK1" -> text="JAK1", suggested_type="gene" + - "tumor protein p53" -> text="TP53", suggested_type="gene" + +Only normalize when the mapping is well-known and unambiguous. If you +are uncertain, keep the user's surface form. Never introduce a related +but unmentioned entity as a shortcut to an answer; for example, do not +emit RANKL unless the user mentioned RANKL or an explicit synonym/name +for that same entity. + + +# Full-text vs abstracts (`full_text` field) + +By default the pipeline fetches only titles and abstracts — that's +the right level of detail for almost every question, and it keeps +API + LLM cost low. Set `full_text=True` ONLY when the user +explicitly asks for the full paper, body text, methods / results +sections, or other paragraph-level content beyond the abstract. +DEFAULT IS FALSE. + + Q: "What are the side effects of imatinib?" -> full_text=False + Q: "List drugs that treat Alzheimer disease" -> full_text=False + Q: "Give me the full text of recent papers on JAK1" -> full_text=True + Q: "What do the methods sections say about X?" -> full_text=True + + +# Body-section filter (`sections` field) — token-saving knob + +When `full_text=True`, you may ALSO set `sections` to restrict which +body sections survive into synthesis. Title + abstract are kept +automatically; `sections` only controls the body. Leave `sections` +null (the default) to pull the entire body. + +The filter exists to save tokens. A full-text paper can be tens of +thousands of tokens — if the user only cares about one part, drop +the rest. Heuristics: + + - Protocol / assay / "what techniques" / "what dose" questions + -> sections=["METHODS"] + - "What did they find" / quantitative results / numbers + -> sections=["RESULTS"] (add "TABLE" or "FIG" if the user asks + for figure or table data) + - Mechanism / interpretation / "why" questions answered by the + authors' discussion -> sections=["DISCUSS"] + - Case-report content -> sections=["CASE"] + - Background / definition only -> usually full_text=False suffices, + do NOT escalate just to pull INTRO + +HARD RULES: + - Never set `sections` when `full_text=False`. It has no effect + in abstract mode and signals confusion. + - Never set `sections` as a substitute for `full_text=True`. If + the user wants the methods, set BOTH (full_text=True AND + sections=["METHODS"]). + - When unsure which sections matter, leave `sections` null — the + pipeline will pull the whole body. The filter is a token + optimization, not a routing decision. + +Allowed values (uppercase, exact spelling): INTRO, METHODS, RESULTS, +DISCUSS, CONCL, FIG, TABLE, CASE. + + Q: "How do they measure JAK1 activity in this paper?" + -> full_text=True, sections=["METHODS"] + Q: "Give me the conclusions on metformin and AMPK." + -> full_text=True, sections=["DISCUSS", "CONCL"] + Q: "Give me the full text of recent papers on JAK1." + -> full_text=True, sections=null + +HARD RULE: the `relation` field is a closed enum of 12 strings — the +schema validator rejects ANY value outside that enum and the whole call +fails. Never invent a value. When in doubt, classify the question as +keyword_search and leave `relation` null. + +# Resolving the user's verb into a `relation` value + +The 12 allowed values, each with a one-line PubTator3 description, are +listed at the bottom of this prompt under "Allowed relation values". +Resolve the user's verb in this order: + +1. EXACT MATCH. If the user's verb appears verbatim in the enum, use it. + +2. DESCRIPTION MATCH. Otherwise, read the descriptions at the bottom of + this prompt and pick the enum value whose description best fits what + the user's verb means in this question. The description is the + source of truth — do not rely on memorized verb→enum tables. Use the + surrounding context (entity types, intent) to disambiguate: + - "What drugs target JAK1?" — "target" in a drug→gene context + fits the `interact` description ("physical interaction such as + protein binding; gene-chemical"). + - "What inhibits BTK?" — exact match on `inhibit`. + - "Which drugs block JAK1?" — "block" semantically equals the + `inhibit` description ("negative correlation; disease-gene, + chemical-variant"). + - "Does drug X induce disease Y?" — "induce ... disease" matches + the `cause` description ("chemical-induced diseases"). + Only commit to a value when the description clearly fits. A weak + or stretchy fit is NOT a fit — go to step 3. + +3. FALLBACK to keyword_search. If no enum description clearly fits the + user's verb in context (this happens for verbs like 'modulate', + 'regulate', 'metabolize', 'phosphorylate' whose mechanism is too + non-specific for any single description), classify the question as + keyword_search and leave `relation` null. + +When step 2 feels like a stretch, prefer step 3. The same description- +first rule applies to `e2_type` (read the "Allowed e2_type values" +descriptions and pick the type whose grounding fits the partner the +user is asking about). + + +# Examples + + Q: "What is JAK1?" + -> single_node; mentions=[(JAK1, gene)] + + Q: "Does metformin treat type-2 diabetes?" + -> relation_known_pair; relation=treat; + mentions=[(metformin, chemical, e1), (type-2 diabetes, disease, e2)] + + Q: "What chemicals treat Alzheimer disease?" + -> relation_partner_discovery; relation=treat; e2_type=Chemical; + mentions=[(Alzheimer disease, disease)] + + Q: "What drugs block JAK1?" + -> relation_partner_discovery; relation=inhibit; e2_type=Chemical; + mentions=[(JAK1, gene)] + ('block' matches the `inhibit` description — "negative correlation; + disease-gene, chemical-variant" — in a drug-vs-gene context) + + Q: "What drugs target JAK1?" + -> relation_partner_discovery; relation=interact; e2_type=Chemical; + mentions=[(JAK1, gene)] + ('target' in a drug-vs-gene context matches the `interact` + description — "physical interaction such as protein binding; + gene-chemical". Use description-match, not a hardcoded synonym.) + + Q: "Which drug inhibits Bruton's tyrosine kinase in chronic lymphocytic leukemia?" + -> relation_partner_discovery; relation=inhibit; e2_type=Chemical; + mentions=[(BTK, gene)] + ("Bruton's tyrosine kinase" is explicitly named; BTK is its canonical lookup symbol) + + Q: "What are the side effects of imatinib?" + -> keyword_search; keyword_query="imatinib side effects" + (PubTator3 has no SideEffect entity type, but the literature does discuss it) + + Q: "Mutations in which gene and which protein are associated with Netherton syndrome?" + -> keyword_search; keyword_query="Netherton syndrome gene protein mutation" + (SINGLE-HOP despite the "and": the gene and the protein are both + attributes of the ONE named disease, not a chain through an + unnamed intermediate. In scope.) + + Q: "Which drugs target proteins associated with Alzheimer disease?" + -> out_of_scope + (multi-hop: disease→associated proteins→drugs targeting them. + Answering requires chaining two relations through intermediate + proteins the user never names — knowledge-graph traversal, not + single-paper literature evidence. Decline it.) + + Q: "What genes are drug targets for Fibrodysplasia Ossificans Progressiva?" + -> relation_partner_discovery; relation=interact; e2_type=Gene; + mentions=[(Fibrodysplasia Ossificans Progressiva, disease)] + (single hop: the disease is named and we want the gene targets + directly associated with it — no unnamed intermediate to + traverse, so this stays in scope.) + + Q: "What nodes are on the shortest path from MDM2 to Sorafenib?" + -> out_of_scope + (asks for graph-traversal output, not literature evidence)""" + + +_DEPTH_EVAL_INSTRUCTIONS = """\ +You are deciding whether to spend additional API budget on a deeper retry. A synthesizer has produced an answer using paper titles and abstracts; refetching the full paper body costs roughly 5x more tokens and only pays off when there is a clear, specific gap that body text would fix. + +# Default state + +DEFAULT: `sufficient=True`. Single-paragraph answers with PMID citations are *normally* sufficient. Re-fetching is expensive — only justify it when you can point to a concrete, named gap. + +# When to return `sufficient=False` + +Return False ONLY when AT LEAST TWO of the following are unambiguously true: + +1. The answer does not name any specific biological entity (gene symbol, drug name, disease name, etc.) beyond what the question itself already contained. + +2. The answer is a near-restatement of the question with PMIDs tacked on, with no real content. Example: question "what genes relate to psoriasis?", answer "Several genes are associated with psoriasis [PMID:...]." — restatement. + +3. The user's question explicitly asks for MECHANISM ("how does X work?", "by what mechanism..."), and the answer contains zero mechanistic vocabulary (no pathway names, protein interactions, signaling cascades, etc.). + +4. The user's question explicitly asks for METHODS or quantitative detail ("what techniques...", "what doses...", "what assay..."), and the answer has none. + +If only ONE criterion is true, return `sufficient=True` — one gap doesn't justify the 5x cost. + +# When to return `sufficient=True` (the common case) + +- Answer names specific entities and cites passages → sufficient. +- Question is descriptive ("what is X?", "what causes Y?") → a short factual answer with citations is sufficient. +- Question is list-style ("what genes / chemicals / drugs ...") → naming a few specifics with citations is sufficient. +- Borderline cases → sufficient. +- The answer is *short* but accurate → sufficient. Length is not depth. + +# Calibration + +Q: "What chemicals treat Alzheimer's disease?" +A: "Donepezil [PMID:X], memantine [PMID:Y], and rivastigmine [PMID:Z] are approved cholinesterase inhibitors and NMDA antagonist used to manage symptoms." +→ SUFFICIENT (names specific drugs and a mode-of-action category, with citations). + +Q: "What chemicals treat Alzheimer's disease?" +A: "Several drugs are used to treat Alzheimer's disease [PMID:X,Y,Z]" +→ INSUFFICIENT (criteria 1 and 2 both true: no specifics, restates question). + +Q: "How does metformin treat type-2 diabetes?" +A: "Metformin treats type-2 diabetes by improving blood sugar control [PMID:X]." +→ INSUFFICIENT (criteria 1 and 3 both true: no molecular actors, no mechanism vocabulary). + +Q: "How does metformin treat type-2 diabetes?" +A: "Metformin lowers hepatic glucose output by activating AMPK and inhibiting mitochondrial complex I [PMID:X], reducing gluconeogenesis [PMID:Y]." +→ SUFFICIENT. + +Q: "What is JAK1?" +A: "JAK1 is a non-receptor tyrosine kinase that mediates cytokine signaling, particularly through the JAK-STAT pathway [PMID:X]." +→ SUFFICIENT (short factual question, short factual answer with the key descriptor). + +# Retrieval context (informs the refinement strategy) + +The human turn includes two extra fields describing what was +retrieved for the current answer: + +- `full_text`: True if body text was fetched, False if abstracts only. +- `current body sections`: the section filter applied on top of + full text. `(none — abstracts only so far)` means full_text was + False; a comma-separated list (e.g. `METHODS, RESULTS`) means + only those body sections were available; `null` (printed as + nothing meaningful) means full body with no filter. + +Use these to decide `suggested_sections`: + +- If `full_text=False`, refinement will flip to full text on the + next pass. You MAY still suggest sections to keep the retry cheap + (e.g. mechanism gap -> ['DISCUSS']); leave null to pull everything. +- If `full_text=True` AND `current body sections` is a real list, + you have ONE more chance — name body sections that are likely + to fill the gap and are NOT already in the current list. Leave + null to fall back to pulling every body section (the safe but + expensive default). +- Map gaps to sections the same way the router does: protocol gap + -> METHODS; numbers/quantitative gap -> RESULTS (add TABLE/FIG + for figure-specific gaps); mechanism / "why" gap -> DISCUSS; + conclusions gap -> CONCL; case-report gap -> CASE. + +# Output + +- `sufficient`: bool — when in doubt, True. +- `missing`: short single-sentence note (only when sufficient=False); null otherwise. +- `suggested_sections`: optional list of section names to add on the + retry (only when sufficient=False). Null is a valid answer and + means "let the pipeline pull every body section". +- `rationale`: one-sentence justification. + +WHEN IN DOUBT: `sufficient=True`. The default is to accept.""" + + +_SYNTHESIZE_INSTRUCTIONS = """\ +You are the synthesis layer of a PubTator3 literature-evidence agent. Given the user's question and a set of annotated passages plus BioREx-extracted document-level relations across multiple papers, integrate the findings into a single evidence-grounded answer. + +Rules: +- FORMAT: consolidate findings from ALL retrieved papers into ONE coherent paragraph that integrates the evidence into a unified narrative. Do NOT use bullet points, numbered lists, sub-headings, or multiple paragraphs in the body. The mandatory `References:` section is the only structured part of the output. +- CITE every claim inline with a PMID in square brackets, e.g. [PMID:12345678]. When several papers support the same claim, group their PMIDs in one bracket, comma-separated: [PMID:1234, PMID:5678]. +- ALWAYS end the answer with a `References:` section listing EVERY unique PMID you cited, one per line, formatted as: + - PMID:12345678 — https://pubmed.ncbi.nlm.nih.gov/12345678/ + This section is mandatory whenever you cite at least one PMID — it is the user's resource list for follow-up reading. +- If the passages do not support the user's question, say so plainly in one or two sentences instead of speculating; in that case omit the References section. +- Do not invent PMIDs; only cite IDs that appear in the passages or relations you were given.""" + + +ROUTER_SYSTEM_PROMPT = ( + f"{_ROUTER_INSTRUCTIONS}\n\n" + f"Allowed relation values (lowercase, exact spelling) — pick the one whose " + f"semantics match the user's verb:\n" + f"{_format_table(RELATION_VOCABULARY)}\n\n" + f"Allowed e2_type values (capitalized) — these are the six entity types " + f"PubTator3 annotates, each grounded in a specific terminology:\n" + f"{_format_table(ENTITY_VOCABULARY)}" +) + + +SYNTHESIZE_SYSTEM_PROMPT = _SYNTHESIZE_INSTRUCTIONS + +DEPTH_EVAL_SYSTEM_PROMPT = _DEPTH_EVAL_INSTRUCTIONS + + +__all__ = [ + "RELATION_VOCABULARY", + "ENTITY_VOCABULARY", + "ROUTER_SYSTEM_PROMPT", + "SYNTHESIZE_SYSTEM_PROMPT", + "DEPTH_EVAL_SYSTEM_PROMPT", +] diff --git a/crossbar_llm/pubtator3_tools/schemas.py b/crossbar_llm/pubtator3_tools/schemas.py new file mode 100644 index 0000000..90bfc57 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/schemas.py @@ -0,0 +1,286 @@ +"""Pydantic schemas and shared type aliases for the PubTator3 LangGraph. + +Kept separate from the orchestration in `pubtator3_graph.py` so other +modules (benchmark runner, tools, future API layer) can import the +schemas without dragging the graph builder + its LangChain deps along. +""" +from __future__ import annotations + +from typing import Awaitable, Callable, Literal, TypeVar, TypedDict + +from pydantic import BaseModel, Field + +from crossbar_llm.pubtator3_tools.client import ( + DocumentRelation, + EntityCandidate, + Passage, + PubTator3Document, + RelatedEntity, +) + + +QuestionType = Literal[ + "single_node", + "relation_known_pair", + "relation_partner_discovery", + "keyword_search", + "out_of_scope", +] + +_RELATION_TYPES = Literal[ + "treat", + "cause", + "associate", + "prevent", + "positive_correlate", + "negative_correlate", + "compare", + "cotreat", + "inhibit", + "stimulate", + "interact", + "drug_interact", +] + +_ENTITY_TYPES = Literal[ + "Gene", + "Chemical", + "Disease", + "Species", + "Variant", + "CellLine", +] + +_CONCEPT_TYPES = Literal[ + "gene", + "chemical", + "disease", + "species", + "variant", + "cellline", +] + +# PubTator3 BioC `section_type` values for full-text body sections. +# Title + abstract are kept implicitly and are not selectable here. +_SECTION_TYPES = Literal[ + "INTRO", + "METHODS", + "RESULTS", + "DISCUSS", + "CONCL", + "FIG", + "TABLE", + "CASE", +] + +StructuredModel = TypeVar("StructuredModel", bound=BaseModel) + + +class EntityMention(BaseModel): + text: str = Field( + ..., + description=( + "Entity query text to send to PubTator3 autocomplete, e.g. 'JAK1'. " + "Prefer the surface form from the user's question, but canonical " + "normalization is allowed when it refers to the same explicitly " + "mentioned entity (for example, use 'BTK' for \"Bruton's tyrosine " + "kinase\"). Do NOT invent new entity names from training knowledge." + ), + ) + suggested_type: _CONCEPT_TYPES | None = Field( + None, + description=( + "Biotype hint to narrow PubTator3 autocomplete (lowercase: gene, " + "chemical, disease, species, variant, cellline). SET THIS whenever " + "the type is clear from context — gene symbols / kinases / " + "receptors -> 'gene'; drug or compound names -> 'chemical'; " + "disease or syndrome names -> 'disease'. Leaving this null lets " + "autocomplete pick the wrong entity type (e.g. resolving a kinase " + "name to a disease that shares the surface form)." + ), + ) + role: Literal["e1", "e2"] | None = Field( + None, + description=( + "For relation_known_pair, marks which side of the relation. " + "e1 is the actor / subject (e.g. 'metformin' in 'metformin treats T2D'); " + "e2 is the object / partner. Leave None for single_node and partner_discovery." + ), + ) + + +class RouterDecision(BaseModel): + question_type: QuestionType = Field( + ..., + description=( + "single_node: question is about ONE biomedical entity, no relation.\n" + "relation_known_pair: user asks about a SPECIFIC pair plus a relation.\n" + "relation_partner_discovery: user asks for partners of one entity by relation.\n" + "keyword_search: biomedical question whose answer likely exists in the " + "literature but does not fit the structured forms above — e.g. side " + "effects, adverse events, mechanism of action, pharmacokinetics, " + "off-label use, pathway / ortholog / GO questions. PubTator3 has no " + "dedicated entity types for these but PubMed/PMC do contain the text. " + "Free-text /search/ will surface relevant papers. Set `keyword_query`.\n" + "out_of_scope: PubTator3 cannot help at all — non-biomedical questions, " + "pure graph-traversal output requests (shortest path, etc.), math, " + "opinion, news, weather." + ), + ) + mentions: list[EntityMention] = Field( + default_factory=list, + description=( + "Biomedical entity mentions. Order matters for relation_known_pair: " + "[e1 first, e2 second]. For partner discovery: [anchor only]. " + "For single_node: [the one entity]. For keyword_search and " + "out_of_scope: may be empty." + ), + ) + relation: _RELATION_TYPES | None = Field( + None, + description="Relation type (PubTator3 vocabulary). Required for relation_* types.", + ) + e2_type: _ENTITY_TYPES | None = Field( + None, + description=( + "Required ONLY for relation_partner_discovery — the type of partner to find. " + "Capitalized: Gene/Chemical/Disease/Species/Variant/CellLine." + ), + ) + keyword_query: str | None = Field( + None, + description=( + "Focused free-text PubMed query distilled from the user's " + "question, e.g. 'imatinib side effects', 'BTK inhibitor CLL', " + "'Denosumab RANKL', 'metformin pharmacokinetics'. Prefer 2–6 " + "keywords that maximise recall on the topic; do not paste the " + "whole sentence.\n\n" + "Fill this for ALL in-scope routes:\n" + "- keyword_search: it is the primary query the pipeline runs.\n" + "- single_node, relation_known_pair, relation_partner_discovery:" + " it is the FALLBACK query the pipeline runs ONLY when the " + "structured PubTator3 search returns 0 PMIDs (which happens " + "often because BioREx didn't tag the exact relation, or the " + "entity pair isn't co-mentioned in PubTator3's graph even when " + "PubMed/PMC clearly discusses it).\n\n" + "Leave null only for out_of_scope." + ), + ) + full_text: bool = Field( + False, + description=( + "Whether the downstream pipeline should fetch full paper body " + "text (True) or only titles + abstracts (False). DEFAULT IS " + "FALSE — abstracts are sufficient for almost every question. " + "Set True ONLY when the user explicitly asks for the full " + "paper, body text, methods / results sections, or detailed " + "paragraph-level content beyond the abstract." + ), + ) + sections: list[_SECTION_TYPES] | None = Field( + None, + description=( + "OPTIONAL body-section filter applied ONLY when full_text=True. " + "When non-empty, the downstream pipeline keeps title + abstract " + "(always) plus body passages whose section_type is in this list " + "— everything else is dropped before synthesis to save tokens. " + "Pick this when the user's question targets one part of the " + "paper: methods/protocol/assay questions -> ['METHODS']; " + "results/findings/data questions -> ['RESULTS']; " + "mechanism/interpretation questions -> ['DISCUSS']; " + "case-report questions -> ['CASE']. Leave None when the user " + "wants the whole body or full_text=False. This field has NO " + "effect when full_text=False — never set it as a substitute " + "for asking for full text." + ), + ) + rationale: str = Field("", description="One-sentence justification for the classification.") + + +class DepthEvaluation(BaseModel): + sufficient: bool = Field( + ..., + description=( + "True if the answer is scientifically substantive given the " + "user's question — names specific entities, describes mechanisms " + "or modes of action, references concrete findings rather than " + "vague associations. False if the answer reads like a question " + "restatement with PMIDs attached but no real biology." + ), + ) + missing: str | None = Field( + None, + description=( + "If sufficient=False, a single short note on what's missing — " + "e.g. 'no mechanism described', 'no quantitative results', " + "'lists entities without specifying their roles'. Null when " + "sufficient=True." + ), + ) + suggested_sections: list[_SECTION_TYPES] | None = Field( + None, + description=( + "When sufficient=False AND full_text is already True, OPTIONALLY " + "name body sections that should be added on the refinement pass " + "to fill the gap. Examples: gap='no mechanism described' " + "-> ['DISCUSS', 'RESULTS']; gap='no protocol detail' " + "-> ['METHODS']; gap='no quantitative numbers' " + "-> ['RESULTS', 'TABLE']. The pipeline UNIONS these with the " + "sections already pulled — do not re-list ones that are " + "already in `current_sections`. Leave None to fall back to " + "'pull every body section' (the safe default escalation). " + "Has no effect when sufficient=True or when full_text=False." + ), + ) + rationale: str = Field("", description="One-sentence justification for the verdict.") + + +class PubTator3State(TypedDict, total=False): + question: str + + question_type: QuestionType + mentions: list[EntityMention] + relation: str | None + e2_type: str | None + keyword_query: str | None + full_text: bool + sections: list[str] | None + rationale: str + + resolved: dict[str, EntityCandidate] + unresolved: list[str] + + partners: list[RelatedEntity] + queries_used: list[str] + pmids: list[int] + total_articles: int + + documents: list[PubTator3Document] + passages: list[Passage] + document_relations: list[DocumentRelation] + final_answer: str | None + + depth_sufficient: bool + depth_missing: str | None + depth_skip_reason: str | None + refinement_attempted: bool + + warnings: list[str] + + +RouterFn = Callable[[str], Awaitable[RouterDecision]] +SynthesizerFn = Callable[[PubTator3State], Awaitable[str]] +EvaluatorFn = Callable[[PubTator3State], Awaitable[DepthEvaluation]] + + +__all__ = [ + "QuestionType", + "StructuredModel", + "EntityMention", + "RouterDecision", + "DepthEvaluation", + "PubTator3State", + "RouterFn", + "SynthesizerFn", + "EvaluatorFn", +] diff --git a/crossbar_llm/pubtator3_tools/structured_output.py b/crossbar_llm/pubtator3_tools/structured_output.py new file mode 100644 index 0000000..c95da77 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/structured_output.py @@ -0,0 +1,108 @@ +"""Structured-output helpers shared by this agent's LLM-bound nodes. + +Some providers return None instead of making the schema tool call when the +model answers in prose, so every structured call falls back to plain JSON and +validates locally against the same Pydantic schema. + +Deliberately duplicated in each tool package rather than imported across them: +the two agents are developed independently and neither should break when the +other changes. +""" +from __future__ import annotations + +import json +from typing import Any, TypeVar + +from langchain_core.language_models import BaseChatModel +from langchain_core.prompts import ChatPromptTemplate, HumanMessagePromptTemplate +from pydantic import BaseModel + +StructuredModel = TypeVar("StructuredModel", bound=BaseModel) + + +def _message_content_to_text(message: Any) -> str: + content = getattr(message, "content", message) + if isinstance(content, str): + return content + if isinstance(content, list): + parts: list[str] = [] + for item in content: + if isinstance(item, str): + parts.append(item) + elif isinstance(item, dict) and isinstance(item.get("text"), str): + parts.append(item["text"]) + else: + parts.append(str(item)) + return "\n".join(parts) + return str(content) + + +def _extract_json_object(text: str) -> dict[str, Any]: + stripped = text.strip() + if stripped.startswith("```"): + lines = stripped.splitlines() + if lines and lines[0].startswith("```"): + lines = lines[1:] + if lines and lines[-1].strip() == "```": + lines = lines[:-1] + stripped = "\n".join(lines).strip() + + decoder = json.JSONDecoder() + for idx, char in enumerate(stripped): + if char != "{": + continue + try: + obj, _ = decoder.raw_decode(stripped[idx:]) + except json.JSONDecodeError: + continue + if isinstance(obj, dict): + return obj + raise ValueError("model response did not contain a JSON object") + + +async def _ainvoke_structured_with_json_fallback( + *, + chat_model: BaseChatModel, + prompt: ChatPromptTemplate, + schema: type[StructuredModel], + values: dict[str, Any], + json_instruction: str, +) -> tuple[StructuredModel, bool]: + """Use provider structured output first, then retry as plain JSON. + + Some providers expose weak tool-calling semantics: LangChain can return + None when the model answers in prose instead of making the schema tool + call. The JSON retry keeps those providers useful while still validating + locally with the exact same Pydantic schema. + """ + structured_error: Exception | None = None + try: + chain = prompt | chat_model.with_structured_output(schema) + parsed = await chain.ainvoke(values) + if parsed is not None: + if isinstance(parsed, schema): + return parsed, False + return schema.model_validate(parsed), False + structured_error = ValueError( + "structured-output returned None (schema-coercion failed)" + ) + except Exception as e: + structured_error = e + + json_prompt = prompt + HumanMessagePromptTemplate.from_template(json_instruction) + try: + msg = await (json_prompt | chat_model).ainvoke(values) + data = _extract_json_object(_message_content_to_text(msg)) + return schema.model_validate(data), True + except Exception as json_error: + raise ValueError( + "structured-output failed and JSON fallback failed: " + f"{structured_error}; {json_error}" + ) from json_error + + +__all__ = [ + "_ainvoke_structured_with_json_fallback", + "_extract_json_object", + "_message_content_to_text", +] diff --git a/crossbar_llm/pubtator3_tools/tests/__init__.py b/crossbar_llm/pubtator3_tools/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/crossbar_llm/pubtator3_tools/tests/conftest.py b/crossbar_llm/pubtator3_tools/tests/conftest.py new file mode 100644 index 0000000..74aeda3 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/conftest.py @@ -0,0 +1,36 @@ +"""Shared pytest fixtures for the PubTator3 test suite.""" +import json +from pathlib import Path + +import pytest + +FIXTURES_DIR = Path(__file__).parent / "fixtures" + + +def pytest_configure(config): + """Reassert pytest-asyncio's auto mode, which this suite requires. + + `asyncio_mode = auto` lives in this directory's pytest.ini, but pytest only + honours that file when the suite is the sole command-line argument. Name it + alongside another directory and their common ancestor wins instead, leaving + every async test here in strict mode and unmarked, so all of them error. + Setting it here keeps the requirement with the package rather than with how + pytest happened to be invoked. + """ + if config.getoption("asyncio_mode", None) != "auto": + config.option.asyncio_mode = "auto" + + +@pytest.fixture +def fx(): + """Load any JSON fixture by stem name. + + Usage: + def test_foo(fx): + data = fx("pubtator3_relations_example") + """ + def _load(name: str): + path = FIXTURES_DIR / f"{name}.json" + return json.loads(path.read_text()) + + return _load diff --git a/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_autocomplete_example.json b/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_autocomplete_example.json new file mode 100644 index 0000000..0412908 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_autocomplete_example.json @@ -0,0 +1,47 @@ +[ + { + "_id": "@GENE_JAK1", + "biotype": "gene", + "db_id": "3716", + "db": "ncbi_gene", + "name": "JAK1", + "description": "All Species", + "match": "Matched on name <m>JAK1</m>" + }, + { + "_id": "@GENE_JAK1.S", + "biotype": "gene", + "db_id": "446697", + "db": "ncbi_gene", + "name": "jak1.S", + "description": "All Species", + "match": "Matched on name <m>jak1.S</m>" + }, + { + "_id": "@GENE_JAK_1", + "biotype": "gene", + "db_id": "574391", + "db": "ncbi_gene", + "name": "JAK-1", + "description": "All Species", + "match": "Matched on synonyms <m>JAK1</m>" + }, + { + "_id": "@GENE_LOC108897047", + "biotype": "gene", + "db_id": "108897047", + "db": "ncbi_gene", + "name": "LOC108897047", + "description": "All Species", + "match": "Matched on synonyms <m>jak1</m>" + }, + { + "_id": "@GENE_LOC101879666", + "biotype": "gene", + "db_id": "101879666", + "db": "ncbi_gene", + "name": "LOC101879666", + "description": "All Species", + "match": "Matched on synonyms <m>JAK1</m>" + } +] diff --git a/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_export_example.json b/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_export_example.json new file mode 100644 index 0000000..aa713ec --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_export_example.json @@ -0,0 +1 @@ +{"PubTator3": [{"_id": "33849366|None", "id": "33849366", "infons": {}, "passages": [{"infons": {"journal": "Expert Opin Drug Saf. 2021 Jul;20(7):855-862. doi: 10.1080/14740338.2021.1917547. ", "year": "2021", "type": "title", "authors": "Nasr NEH, Metwaly MG, Ahmed EO, Fares AR, ElMeshad AN"}, "offset": 0, "text": "Investigating the root cause of N-nitrosodimethylamine formation in metformin pharmaceutical products.", "sentences": [], "annotations": [{"id": "2", "infons": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "text": "N-nitrosodimethylamine", "locations": [{"offset": 32, "length": 22}]}, {"id": "3", "infons": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "text": "metformin", "locations": [{"offset": 68, "length": 9}]}], "relations": []}, {"infons": {"type": "abstract"}, "offset": 103, "text": "BACKGROUND: FDA limited N-nitrosodimethylamine (NDMA) - a carcinogenic impurity formed during metformin (MET) tablets manufacturing - level to 96 ng/day; a step which led to recall of MET products. This work aims to investigate the root cause of NDMA formation during MET tablets manufacturing. RESEARCH DESIGN AND METHODS: We focused on three main contributing causes: use of water and heat during intra-granulation, and the nitrite/nitrate quantities in excipients. Thirteen MET tablet formulations (immediate or sustained-release) were manufactured, on batch level. Each batch was manufactured using one excipient and excluding one cause at a time and NDMA level was assayed. RESULTS: NDMA traces were undetectable in MET tablets manufactured using polyvinyl pyrrolidone or hydroxypropyl cellulose SSL, even when water and/or heat were employed during intra-granulation. Levels of NDMA in MET tablets with hydroxypropyl methyl cellulose (HPMC) E5 or carboxymethyl cellulose sodium 4000 were 67.08 +- 2.3 and 66.21 +- 2.5 ng/day, in the presence of water and/or heat. No impact of employing extra-granular PolyoxTM, HPMC E5 or HPMC K15 on NDMA formation, despite the high nitrite and nitrate content in these excipients. CONCLUSIONS: Water, heat, and excipients' nitrite and nitrate levels are the key players, which should collectively exist, to cause NDMA formation during MET tablets manufacturing.", "sentences": [], "annotations": [{"id": "38", "infons": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "text": "N-nitrosodimethylamine", "locations": [{"offset": 127, "length": 22}]}, {"id": "39", "infons": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "text": "NDMA", "locations": [{"offset": 151, "length": 4}]}, {"id": "40", "infons": {"identifier": "MESH:D011230", "type": "Disease", "valid": true, "normalized": ["D011230"], "database": "ncbi_mesh", "normalized_id": "D011230", "biotype": "disease", "name": "Precancerous Conditions", "accession": "@DISEASE_Precancerous_Conditions"}, "text": "carcinogenic", "locations": [{"offset": 161, "length": 12}]}, {"id": "41", "infons": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "text": "metformin", "locations": [{"offset": 197, "length": 9}]}, {"id": "42", "infons": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "text": "MET", "locations": [{"offset": 208, "length": 3}]}, {"id": "43", "infons": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "text": "MET", "locations": [{"offset": 287, "length": 3}]}, {"id": "44", "infons": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "text": "NDMA", "locations": [{"offset": 349, "length": 4}]}, {"id": "45", "infons": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "text": "MET", "locations": [{"offset": 371, "length": 3}]}, {"id": "46", "infons": {"identifier": "MESH:D014867", "type": "Chemical", "valid": true, "normalized": ["D014867"], "database": "ncbi_mesh", "normalized_id": "D014867", "biotype": "chemical", "name": "Water", "accession": "@CHEMICAL_Water"}, "text": "water", "locations": [{"offset": 480, "length": 5}]}, {"id": "47", "infons": {"identifier": "MESH:D009573", "type": "Chemical", "valid": true, "normalized": ["D009573"], "database": "ncbi_mesh", "normalized_id": "D009573", "biotype": "chemical", "name": "Nitrites", "accession": "@CHEMICAL_Nitrites"}, "text": "nitrite", "locations": [{"offset": 529, "length": 7}]}, {"id": "48", "infons": {"identifier": "MESH:D009566", "type": "Chemical", "valid": true, "normalized": ["D009566"], "database": "ncbi_mesh", "normalized_id": "D009566", "biotype": "chemical", "name": "Nitrates", "accession": "@CHEMICAL_Nitrates"}, "text": "nitrate", "locations": [{"offset": 537, "length": 7}]}, {"id": "49", "infons": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "text": "MET", "locations": [{"offset": 580, "length": 3}]}, {"id": "50", "infons": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "text": "NDMA", "locations": [{"offset": 758, "length": 4}]}, {"id": "51", "infons": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "text": "NDMA", "locations": [{"offset": 791, "length": 4}]}, {"id": "52", "infons": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "text": "MET", "locations": [{"offset": 824, "length": 3}]}, {"id": "53", "infons": {"identifier": "MESH:D011205", "type": "Chemical", "valid": true, "normalized": ["D011205"], "database": "ncbi_mesh", "normalized_id": "D011205", "biotype": "chemical", "name": "Povidone", "accession": "@CHEMICAL_Povidone"}, "text": "polyvinyl pyrrolidone", "locations": [{"offset": 855, "length": 21}]}, {"id": "54", "infons": {"identifier": "-", "type": "Chemical", "valid": false, "normalized_id": null, "biotype": "chemical"}, "text": "hydroxypropyl cellulose SSL", "locations": [{"offset": 880, "length": 27}]}, {"id": "55", "infons": {"identifier": "MESH:D014867", "type": "Chemical", "valid": true, "normalized": ["D014867"], "database": "ncbi_mesh", "normalized_id": "D014867", "biotype": "chemical", "name": "Water", "accession": "@CHEMICAL_Water"}, "text": "water", "locations": [{"offset": 919, "length": 5}]}, {"id": "56", "infons": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "text": "NDMA", "locations": [{"offset": 987, "length": 4}]}, {"id": "57", "infons": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "text": "MET", "locations": [{"offset": 995, "length": 3}]}, {"id": "58", "infons": {"identifier": "MESH:C584708", "type": "Chemical", "valid": true, "normalized": ["C584708"], "database": "ncbi_mesh", "normalized_id": "C584708", "biotype": "chemical", "name": "hypromellose 2910 (5 MPA.S)", "accession": "@CHEMICAL_hypromellose_2910_(5_MPA.S)"}, "text": "hydroxypropyl methyl cellulose (HPMC) E5", "locations": [{"offset": 1012, "length": 40}]}, {"id": "59", "infons": {"identifier": "-", "type": "Chemical", "valid": false, "normalized_id": null, "biotype": "chemical"}, "text": "carboxymethyl cellulose sodium 4000", "locations": [{"offset": 1056, "length": 35}]}, {"id": "60", "infons": {"identifier": "MESH:D014867", "type": "Chemical", "valid": true, "normalized": ["D014867"], "database": "ncbi_mesh", "normalized_id": "D014867", "biotype": "chemical", "name": "Water", "accession": "@CHEMICAL_Water"}, "text": "water", "locations": [{"offset": 1154, "length": 5}]}, {"id": "61", "infons": {"identifier": "-", "type": "Chemical", "valid": false, "normalized_id": null, "biotype": "chemical"}, "text": "PolyoxTM", "locations": [{"offset": 1211, "length": 8}]}, {"id": "62", "infons": {"identifier": "MESH:C584708", "type": "Chemical", "valid": true, "normalized": ["C584708"], "database": "ncbi_mesh", "normalized_id": "C584708", "biotype": "chemical", "name": "hypromellose 2910 (5 MPA.S)", "accession": "@CHEMICAL_hypromellose_2910_(5_MPA.S)"}, "text": "HPMC E5", "locations": [{"offset": 1221, "length": 7}]}, {"id": "63", "infons": {"identifier": "-", "type": "Chemical", "valid": false, "normalized_id": null, "biotype": "chemical"}, "text": "HPMC K15", "locations": [{"offset": 1232, "length": 8}]}, {"id": "64", "infons": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "text": "NDMA", "locations": [{"offset": 1244, "length": 4}]}, {"id": "65", "infons": {"identifier": "MESH:D009573", "type": "Chemical", "valid": true, "normalized": ["D009573"], "database": "ncbi_mesh", "normalized_id": "D009573", "biotype": "chemical", "name": "Nitrites", "accession": "@CHEMICAL_Nitrites"}, "text": "nitrite", "locations": [{"offset": 1277, "length": 7}]}, {"id": "66", "infons": {"identifier": "MESH:D009566", "type": "Chemical", "valid": true, "normalized": ["D009566"], "database": "ncbi_mesh", "normalized_id": "D009566", "biotype": "chemical", "name": "Nitrates", "accession": "@CHEMICAL_Nitrates"}, "text": "nitrate", "locations": [{"offset": 1289, "length": 7}]}, {"id": "67", "infons": {"identifier": "MESH:D014867", "type": "Chemical", "valid": true, "normalized": ["D014867"], "database": "ncbi_mesh", "normalized_id": "D014867", "biotype": "chemical", "name": "Water", "accession": "@CHEMICAL_Water"}, "text": "Water", "locations": [{"offset": 1339, "length": 5}]}, {"id": "68", "infons": {"identifier": "MESH:D009573", "type": "Chemical", "valid": true, "normalized": ["D009573"], "database": "ncbi_mesh", "normalized_id": "D009573", "biotype": "chemical", "name": "Nitrites", "accession": "@CHEMICAL_Nitrites"}, "text": "nitrite", "locations": [{"offset": 1368, "length": 7}]}, {"id": "69", "infons": {"identifier": "MESH:D009566", "type": "Chemical", "valid": true, "normalized": ["D009566"], "database": "ncbi_mesh", "normalized_id": "D009566", "biotype": "chemical", "name": "Nitrates", "accession": "@CHEMICAL_Nitrates"}, "text": "nitrate", "locations": [{"offset": 1380, "length": 7}]}, {"id": "70", "infons": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "text": "NDMA", "locations": [{"offset": 1458, "length": 4}]}, {"id": "71", "infons": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "text": "MET", "locations": [{"offset": 1480, "length": 3}]}], "relations": []}], "relations": [{"id": "R1", "infons": {"score": "0.7595", "role1": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "role2": {"identifier": "MESH:D011230", "type": "Disease", "valid": true, "normalized": ["D011230"], "database": "ncbi_mesh", "normalized_id": "D011230", "biotype": "disease", "name": "Precancerous Conditions", "accession": "@DISEASE_Precancerous_Conditions"}, "type": "Positive_Correlation"}, "nodes": [{"refid": "0", "role": "2,4"}]}, {"id": "R2", "infons": {"score": "0.9752", "role1": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "role2": {"identifier": "MESH:D008687", "type": "Chemical", "valid": true, "normalized": ["D008687"], "database": "ncbi_mesh", "normalized_id": "D008687", "biotype": "chemical", "name": "Metformin", "accession": "@CHEMICAL_Metformin"}, "type": "Association"}, "nodes": [{"refid": "1", "role": "0,1"}]}, {"id": "R3", "infons": {"score": "0.8374", "role1": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "role2": {"identifier": "MESH:D014867", "type": "Chemical", "valid": true, "normalized": ["D014867"], "database": "ncbi_mesh", "normalized_id": "D014867", "biotype": "chemical", "name": "Water", "accession": "@CHEMICAL_Water"}, "type": "Positive_Correlation"}, "nodes": [{"refid": "2", "role": "15,19"}]}, {"id": "R4", "infons": {"score": "0.9051", "role1": {"identifier": "MESH:D004128", "type": "Chemical", "valid": true, "normalized": ["D004128"], "database": "ncbi_mesh", "normalized_id": "D004128", "biotype": "chemical", "name": "Dimethylnitrosamine", "accession": "@CHEMICAL_Dimethylnitrosamine"}, "role2": {"identifier": "MESH:D009573", "type": "Chemical", "valid": true, "normalized": ["D009573"], "database": "ncbi_mesh", "normalized_id": "D009573", "biotype": "chemical", "name": "Nitrites", "accession": "@CHEMICAL_Nitrites"}, "type": "Association"}, "nodes": [{"refid": "3", "role": "28,29"}]}], "pmid": 33849366, "pmcid": null, "meta": {}, "date": "2021-07-01T00:00:00Z", "journal": "Expert Opin Drug Saf", "authors": ["Nasr NEH", "Metwaly MG", "Ahmed EO", "Fares AR", "ElMeshad AN"], "relations_display": [{"name": "cause|@CHEMICAL_Dimethylnitrosamine|@DISEASE_Precancerous_Conditions"}, {"name": "associate|@CHEMICAL_Dimethylnitrosamine|@CHEMICAL_Metformin"}, {"name": "positive_correlate|@CHEMICAL_Dimethylnitrosamine|@CHEMICAL_Water"}, {"name": "associate|@CHEMICAL_Dimethylnitrosamine|@CHEMICAL_Nitrites"}]}]} \ No newline at end of file diff --git a/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_relations_example.json b/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_relations_example.json new file mode 100644 index 0000000..85a4be2 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_relations_example.json @@ -0,0 +1 @@ +[{"type":"negative_correlate","source":"@CHEMICAL_ruxolitinib","target":"@GENE_JAK1","publications":585},{"type":"negative_correlate","source":"@CHEMICAL_upadacitinib","target":"@GENE_JAK1","publications":217},{"type":"negative_correlate","source":"@CHEMICAL_baricitinib","target":"@GENE_JAK1","publications":186},{"type":"negative_correlate","source":"@CHEMICAL_GLPG0634","target":"@GENE_JAK1","publications":128},{"type":"negative_correlate","source":"@CHEMICAL_tofacitinib","target":"@GENE_JAK1","publications":108},{"type":"negative_correlate","source":"@CHEMICAL_abrocitinib","target":"@GENE_JAK1","publications":89},{"type":"negative_correlate","source":"@CHEMICAL_CYT_387","target":"@GENE_JAK1","publications":49},{"type":"negative_correlate","source":"@CHEMICAL_itacitinib","target":"@GENE_JAK1","publications":29},{"type":"negative_correlate","source":"@CHEMICAL_ivarmacitinib","target":"@GENE_JAK1","publications":25},{"type":"negative_correlate","source":"@CHEMICAL_PF_06700841","target":"@GENE_JAK1","publications":23},{"type":"negative_correlate","source":"@CHEMICAL_AZD_1480","target":"@GENE_JAK1","publications":22},{"type":"negative_correlate","source":"@CHEMICAL_oclacitinib","target":"@GENE_JAK1","publications":17},{"type":"negative_correlate","source":"@CHEMICAL_AG_490","target":"@GENE_JAK1","publications":15},{"type":"negative_correlate","source":"@CHEMICAL_Curcumin","target":"@GENE_JAK1","publications":10},{"type":"negative_correlate","source":"@CHEMICAL_Genistein","target":"@GENE_JAK1","publications":10},{"type":"negative_correlate","source":"@CHEMICAL_INCB039110","target":"@GENE_JAK1","publications":8},{"type":"negative_correlate","source":"@CHEMICAL_Resveratrol","target":"@GENE_JAK1","publications":7},{"type":"negative_correlate","source":"@CHEMICAL_3_3'_4_5'_tetrahydroxystilbene","target":"@GENE_JAK1","publications":5},{"type":"negative_correlate","source":"@CHEMICAL_Berberine","target":"@GENE_JAK1","publications":5},{"type":"negative_correlate","source":"@CHEMICAL_GSK2586184","target":"@GENE_JAK1","publications":5},{"type":"negative_correlate","source":"@CHEMICAL_Quercetin","target":"@GENE_JAK1","publications":5},{"type":"negative_correlate","source":"@CHEMICAL_15_deoxyprostaglandin_J2","target":"@GENE_JAK1","publications":4},{"type":"negative_correlate","source":"@CHEMICAL_Calcifediol","target":"@GENE_JAK1","publications":4},{"type":"negative_correlate","source":"@CHEMICAL_Tretinoin","target":"@GENE_JAK1","publications":4},{"type":"negative_correlate","source":"@CHEMICAL_delgocitinib","target":"@GENE_JAK1","publications":4},{"type":"negative_correlate","source":"@CHEMICAL_fedratinib","target":"@GENE_JAK1","publications":4},{"type":"negative_correlate","source":"@CHEMICAL_ibrutinib","target":"@GENE_JAK1","publications":4},{"type":"negative_correlate","source":"@CHEMICAL_peficitinib","target":"@GENE_JAK1","publications":4},{"type":"negative_correlate","source":"@CHEMICAL_Apigenin","target":"@GENE_JAK1","publications":3},{"type":"negative_correlate","source":"@CHEMICAL_Cisplatin","target":"@GENE_JAK1","publications":3},{"type":"negative_correlate","source":"@CHEMICAL_Imatinib_Mesylate","target":"@GENE_JAK1","publications":3},{"type":"negative_correlate","source":"@CHEMICAL_Methotrexate","target":"@GENE_JAK1","publications":3},{"type":"negative_correlate","source":"@CHEMICAL_U_0126","target":"@GENE_JAK1","publications":3},{"type":"negative_correlate","source":"@CHEMICAL_epigallocatechin_gallate","target":"@GENE_JAK1","publications":3},{"type":"negative_correlate","source":"@CHEMICAL_ferulic_acid","target":"@GENE_JAK1","publications":3},{"type":"negative_correlate","source":"@CHEMICAL_nifuroxazide","target":"@GENE_JAK1","publications":3},{"type":"negative_correlate","source":"@CHEMICAL_teriflunomide","target":"@GENE_JAK1","publications":3},{"type":"negative_correlate","source":"@CHEMICAL_17_alpha_Hydroxyprogesterone_Caproate","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_1_(2_aminoethyl)_2_(piperidin_4_yl)_1H_benzo(d)imidazole_5_carboxamide","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_4_octyl_itaconate","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Acetylcysteine","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Bromisovalum","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Cadmium_Chloride","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Docetaxel","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Ellagic_Acid","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Fluorouracil","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Gefitinib","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_INCB_16562","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Lipopolysaccharides","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Metformin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Pyridones","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Silybin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_Vorinostat","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_agerarin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_andrographolide","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_astaxanthine","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_cardamonin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_dehydrocorydalin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_formononetin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_galangin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_itaconic_acid","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_pentagalloylglucose","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_plastochromanol_8","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_plumbagin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_pyrazole","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_pyrimidine","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_pyrrolopyrimidine","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_sulforaphane","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_taxifolin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_triptolide","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_tyrphostin_25","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_vitexin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_wogonin","target":"@GENE_JAK1","publications":2},{"type":"negative_correlate","source":"@CHEMICAL_(3_5_bis((4_fluorophenyl)methylidene)_1_((1_hydroxy_2_2_5_5_tetramethyl_2_5_dihydro_1H_pyrrol_3_yl)methyl)piperidin_4_one)","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_(7_(2_6_dichlorophenyl)_5_methylbenzo(1_2_4)triazin_3_yl)_(4_(2_pyrrolidin_1_ylethoxy)phenyl)amine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_15_deoxy_12_14_prostaglandin_J2","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_18alpha_glycyrrhetinic_acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_1_((5_chloro_1H_indol_2_yl)carbonyl)_4_methylpiperazine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_1_2_4_triazole","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_1_Methyl_3_isobutylxanthine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_2_((aminocarbonyl)amino)_5_(4_fluorophenyl)_3_thiophenecarboxamide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_2_(1H_indazol_4_yl)_6_(4_methanesulfonylpiperazin_1_ylmethyl)_4_morpholin_4_ylthieno(3_2_d)pyrimidine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_2_8_diazaspiro(4.5)decan_1_one","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_2_Methoxyestradiol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_2_tert_butyl_9_fluoro_3_6_dihydro_7H_benz(h)imidazo(4_5_f)isoquinoline_7_one","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_3_(1_(4_chlorobenzyl)indol_3_yl)_N_(pyridin_4_yl)propanamide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_3_(indol_3_yl)propionic_acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_4_(benzylamino)_2_((2_(3_chloro_4_hydroxyphenyl)ethyl)amino)pyrimidine_5_carboxamide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_6H_pyrrolo(2_3_e)(1_2_4)triazolo(4_3_a)pyrazine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_7_4'_dimethoxy_6_hydroxyaurone_4_O_beta_glucopyranoside","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_7_hydroxycoumarin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_AG_127","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_AICA_ribonucleotide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_AS_8","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_AZD3759","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Acrylamide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Aflatoxin_B1","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Alcohols","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Alitretinoin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Arginine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Arsenic_Trioxide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Auranofin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Aurintricarboxylic_Acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Betulinic_Acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Bile_Acids_and_Salts","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Bleomycin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Budesonide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Buthionine_Sulfoximine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Butyrates","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Cadmium","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Cantharidin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Cardenolides","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Catechin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Cefmenoxime","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Chalcone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Chlorogenic_Acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Chlorpheniramine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Colforsin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_D_ribo_phytosphingosine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Daclizumab","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Dasatinib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Decitabine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Dexamethasone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Dibutyl_Phthalate","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Dimethyl_Fumarate","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Diosgenin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Erianin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Erlotinib_Hydrochloride","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Eucalyptol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Exenatide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Fatty_Acids_Omega_3","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Febuxostat","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Fenretinide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Fluorine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Fluoxetine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Glutamine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Glycyrrhizic_Acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Hyaluronic_Acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Hydrogen_Peroxide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Hydroxyurea","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Indazoles","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Indomethacin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Infliximab","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Kynurenine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Lactic_Acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Lapatinib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Leflunomide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Luteolin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Lycopene","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_M_2_protocol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Maraviroc","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Matrines","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Melatonin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Mercuric_Chloride","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Mifepristone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Molsidomine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_NSC_74859","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_NVP_BEP800","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_N_isobutyl_2E_decenamide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Nevirapine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Niobium","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Nucleosides","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Oils_Volatile","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Oxyquinoline","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_PF_06651600","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Palladium","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Panobinostat","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Pentoxifylline","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Perindopril","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Platinum","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Praseodymium","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Pregnanolone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Propofol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Pulsatilla_saponin_A","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Pyrazines","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Pyrrolo(2_3_d)pyrimidine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Pyrrolopyridazine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Roflumilast","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Rosiglitazone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Rotenone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Rutin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_SGI_1252","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_ST_638","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_S_ethyl_glutathione","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Salicylates","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Salts","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Saponins","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Sorafenib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Spermine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Staurosporine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Sulfonamides","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Sunitinib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Tamoxifen","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Testosterone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Tetradecanoylphorbol_Acetate","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Thimerosal","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Thioctic_Acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Thyrotropin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Triazines","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Triiodothyronine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Trimetazidine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Triterpenes","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Tungsten","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Tunicamycin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Tyrphostins","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Valproic_Acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Vanillic_Acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Zinc_Oxide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_Zinc","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_acetovanillone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_agrimonolide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_alloin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_allyl_isothiocyanate","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_aloperine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_amorfrutin_A","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_anisole","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_arctiin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_arsenite","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_baicalein","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_baicalin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_bardoxolone_methyl","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_bazedoxifene","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_benzimidazole","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_benzyloxycarbonylleucyl_leucyl_leucine_aldehyde","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_benzyloxycarbonylvalyl_alanyl_aspartyl_fluoromethyl_ketone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_bergamottin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_brodalumab","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_brusatol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_bufalin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_caffeic_acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_calcaratarin_D","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_calenduloside_E","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_candesartan_cilexetil","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_candesartan","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_cannabichromene","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_cannabigerol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_capillarisin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_capsazepine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_cediranib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_chelerythrine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_chrysin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_coenzyme_Q10","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_columbianadin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_crocin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_cucurbitacin_I","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_curculigoside","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_curcumol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_darutigenol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_dehydroabietylamine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_dehydroacetic_acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_dendrobine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_diallyl_disulfide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_dieckol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_dubermatinib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_dupilumab","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_empagliflozin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_falcarindiol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_fangchinoline","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_favipiravir","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_fezakinumab","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_fisetin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_flavone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_fludarabine_phosphate","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_fostamatinib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_fucoidan","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_gadolinium_chloride","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_gamma_terpinene","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_ganaxolone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_ganoderic_acid_A","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_garcinol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_gimeracil","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_ginkgetin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_ginsenoside_Rh2","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_gossypol_acetic_acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_graveoline","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_hederagenin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_herbimycin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_hexahydrocurcumin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_hexamethylene_bisacetamide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_hispidin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_hyperforin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_hypericin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_imidapril","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_imperatorin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_iridin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_isoborneol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_isoliquiritigenin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_kaempferide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_kaempferol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_lapachol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_lavendustin_A","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_linifanib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_lupeol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_lycorine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_magnesium_monoperoxyphthalate","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_mahanine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_mangiferin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_methyl_cellosolve","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_methylone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_methylselenic_acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_mogroside_V","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_morin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_morusin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_myricetin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_naringin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_nitidine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_nobiletin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_oleocanthal","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_oligochitosan","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_oxophenylarsine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_oxymatrine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_paeonol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_palbinone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_paricalcitol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_parthenolide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_pazopanib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_pexidartinib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_phenyl_2_aminoethyl_sulfide","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_phillygenin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_piperlongumine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_pirfenidone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_polydatin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_ponatinib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_pregna_4_17_diene_3_16_dione","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_pseudolaric_acid_B","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_regorafenib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_resiniferatoxin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_ribociclib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_salinomycin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_sappanone_A","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_sesamin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_sesamol","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_sinomenine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_sivelestat","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_sodium_aescinate","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_stattic","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_syringic_acid","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tabersonine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tanshinone","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tapinarof","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_terrein","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tetrahydrocurcumin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tetramethylpyrazine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tetroazolemycin_B","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tozasertib","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_trans_sodium_crocetinate","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_trichostatin_A","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tricin","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tris(dibenzylideneacetone)dipalladium","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_tyrphostin_AG_1024","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_vinpocetine","target":"@GENE_JAK1","publications":1},{"type":"negative_correlate","source":"@CHEMICAL_zeylenone","target":"@GENE_JAK1","publications":1}] \ No newline at end of file diff --git a/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_search_example.json b/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_search_example.json new file mode 100644 index 0000000..5c704ee --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/fixtures/pubtator3_search_example.json @@ -0,0 +1,424 @@ +{ + "results": [ + { + "_id": "33849366", + "pmid": 33849366, + "title": "Investigating the root cause of N-nitrosodimethylamine formation in metformin pharmaceutical products.", + "journal": "Expert Opin Drug Saf", + "authors": [ + "Nasr NEH", + "Metwaly MG", + "Ahmed EO", + "Fares AR", + "ElMeshad AN" + ], + "date": "2021-07-01T00:00:00Z", + "doi": "10.1080/14740338.2021.1917547", + "meta_date_publication": "2021 Jul", + "meta_volume": "20", + "meta_issue": "7", + "meta_pages": "855-862", + "score": 100984.84, + "text_hl": "Investigating the root cause of @<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@N-nitrosodimethylamine@@@ formation in @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin@@@ pharmaceutical products.", + "citations": { + "NLM": "Nasr NEH, Metwaly MG, Ahmed EO, Fares AR, ElMeshad AN. Investigating the root cause of N-nitrosodimethylamine formation in metformin pharmaceutical products. Expert Opin Drug Saf. 2021 Jul;20(7):855-862. PMID: 33849366", + "BibTeX": "@article{33849366, title={Investigating the root cause of N-nitrosodimethylamine formation in metformin pharmaceutical products.}, author={Nasr NEH and Metwaly MG and Ahmed EO and Fares AR and ElMeshad AN}, journal={Expert Opin Drug Saf}, volume={20}, number={7}, pages={855-862}}" + } + }, + { + "_id": "33422831", + "pmid": 33422831, + "title": "Insight into the formation of N-nitrosodimethylamine in metformin products.", + "journal": "J Pharm Biomed Anal", + "authors": [ + "Jireš J", + "Kalášek S", + "Gibala P", + "Rudovský J", + "Douša M", + "Kubelka T", + "Hrubý J", + "Řezanka P" + ], + "date": "2021-02-20T00:00:00Z", + "doi": "10.1016/j.jpba.2020.113877", + "meta_date_publication": "2021 Feb 20", + "meta_volume": "195", + "meta_issue": "", + "meta_pages": "113877", + "score": 96626.93, + "text_hl": "Insight into the formation of @<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@N-nitrosodimethylamine@@@ in @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin@@@ products.", + "citations": { + "NLM": "Jireš J, Kalášek S, Gibala P, Rudovský J, Douša M, Kubelka T, Hrubý J, Řezanka P. Insight into the formation of N-nitrosodimethylamine in metformin products. J Pharm Biomed Anal. 2021 Feb 20;195():113877. PMID: 33422831", + "BibTeX": "@article{33422831, title={Insight into the formation of N-nitrosodimethylamine in metformin products.}, author={Jireš J and Kalášek S and Gibala P and Rudovský J and Douša M and Kubelka T and Hrubý J and Řezanka P}, journal={J Pharm Biomed Anal}, volume={195}, pages={113877}}" + } + }, + { + "_id": "37031864", + "pmid": 37031864, + "title": "Dispersant-first dispersive liquid-liquid microextraction (DF-DLLME), a novel sample preparation procedure for NDMA determination in metformin products.", + "journal": "J Pharm Sci", + "authors": [ + "Géhin C", + "O'Neill N", + "Moore A", + "Harrison M", + "Holman SW", + "Blom G" + ], + "date": "2023-04-07T00:00:00Z", + "doi": "10.1016/j.xphs.2023.03.016", + "meta_date_publication": "2023 Apr 7", + "meta_volume": "", + "meta_issue": "", + "meta_pages": "", + "score": 95949.44, + "text_hl": "Dispersant-first dispersive liquid-liquid microextraction (DF-DLLME), a novel sample preparation procedure for @<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@NDMA@@@ determination in @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin@@@ products.", + "citations": { + "NLM": "Géhin C, O'Neill N, Moore A, Harrison M, Holman SW, Blom G. Dispersant-first dispersive liquid-liquid microextraction (DF-DLLME), a novel sample preparation procedure for NDMA determination in metformin products. J Pharm Sci. 2023 Apr 7;():. PMID: 37031864", + "BibTeX": "@article{37031864, title={Dispersant-first dispersive liquid-liquid microextraction (DF-DLLME), a novel sample preparation procedure for NDMA determination in metformin products.}, author={Géhin C and O'Neill N and Moore A and Harrison M and Holman SW and Blom G}, journal={J Pharm Sci}}" + } + }, + { + "_id": "38267707", + "pmid": 38267707, + "title": "Patient In-Use Stability Testing of FDA-Approved Metformin Combination Products for N-Nitrosamine Impurity.", + "journal": "AAPS PharmSciTech", + "authors": [ + "Dharani S", + "Mohamed EM", + "Rahman Z", + "Khan MA" + ], + "date": "2024-01-24T00:00:00Z", + "doi": "10.1208/s12249-023-02724-3", + "meta_date_publication": "2024 Jan 24", + "meta_volume": "25", + "meta_issue": "1", + "meta_pages": "19", + "score": 95814.66, + "text_hl": "When @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin@@@ products have @<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@NDMA@@@ impurities, it is indispensable to check for the same impurities in @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin@@@ combination products. ", + "citations": { + "NLM": "Dharani S, Mohamed EM, Rahman Z, Khan MA. Patient In-Use Stability Testing of FDA-Approved Metformin Combination Products for N-Nitrosamine Impurity. AAPS PharmSciTech. 2024 Jan 24;25(1):19. PMID: 38267707", + "BibTeX": "@article{38267707, title={Patient In-Use Stability Testing of FDA-Approved Metformin Combination Products for N-Nitrosamine Impurity.}, author={Dharani S and Mohamed EM and Rahman Z and Khan MA}, journal={AAPS PharmSciTech}, volume={25}, number={1}, pages={19}}" + } + }, + { + "_id": "34213099", + "pmid": 34213099, + "title": "[Determination of N-nitrosodimethylamine in metformin hydrochloride and its preparations by high performance liquid chromatography-tandem mass spectrometry].", + "journal": "Se Pu", + "authors": [ + "Guo C", + "Liu Q", + "Zhang L", + "Zheng J", + "Wang Y", + "Yang S", + "Chu Z", + "Niu C", + "Xu Y" + ], + "date": "2020-11-08T00:00:00Z", + "doi": "10.3724/SP.J.1123.2020.03008", + "meta_date_publication": "2020 Nov 8", + "meta_volume": "38", + "meta_issue": "11", + "meta_pages": "1288-1293", + "score": 93813.055, + "text_hl": "[Determination of @<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@N-nitrosodimethylamine@@@ in @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin hydrochloride@@@ and its preparations by high performance liquid chromatography-tandem mass spectrometry].", + "citations": { + "NLM": "Guo C, Liu Q, Zhang L, Zheng J, Wang Y, Yang S, Chu Z, Niu C, Xu Y. [Determination of N-nitrosodimethylamine in metformin hydrochloride and its preparations by high performance liquid chromatography-tandem mass spectrometry]. Se Pu. 2020 Nov 8;38(11):1288-1293. PMID: 34213099", + "BibTeX": "@article{34213099, title={[Determination of N-nitrosodimethylamine in metformin hydrochloride and its preparations by high performance liquid chromatography-tandem mass spectrometry].}, author={Guo C and Liu Q and Zhang L and Zheng J and Wang Y and Yang S and Chu Z and Niu C and Xu Y}, journal={Se Pu}, volume={38}, number={11}, pages={1288-1293}}" + } + }, + { + "_id": "34280626", + "pmid": 34280626, + "title": "NDMA formation during ozonation of metformin: Roles of ozone and hydroxyl radicals.", + "journal": "Sci Total Environ", + "authors": [ + "Liao X", + "Shen L", + "Jiang Z", + "Gao M", + "Qiu Y", + "Qi H", + "Chen C" + ], + "date": "2021-11-20T00:00:00Z", + "doi": "10.1016/j.scitotenv.2021.149010", + "meta_date_publication": "2021 Nov 20", + "meta_volume": "796", + "meta_issue": "", + "meta_pages": "149010", + "score": 91602.53, + "text_hl": "@<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@NDMA@@@ formation during ozonation of @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin@@@: Roles of @CHEMICAL_Ozone @CHEMICAL_MESH:D010126 @@@ozone@@@ and @CHEMICAL_Hydroxyl_Radical @CHEMICAL_MESH:D017665 @@@hydroxyl radicals@@@.", + "citations": { + "NLM": "Liao X, Shen L, Jiang Z, Gao M, Qiu Y, Qi H, Chen C. NDMA formation during ozonation of metformin: Roles of ozone and hydroxyl radicals. Sci Total Environ. 2021 Nov 20;796():149010. PMID: 34280626", + "BibTeX": "@article{34280626, title={NDMA formation during ozonation of metformin: Roles of ozone and hydroxyl radicals.}, author={Liao X and Shen L and Jiang Z and Gao M and Qiu Y and Qi H and Chen C}, journal={Sci Total Environ}, volume={796}, pages={149010}}" + } + }, + { + "_id": "35449372", + "pmid": 35449372, + "title": "International Regulatory Collaboration on the Analysis of Nitrosamines in Metformin-Containing Medicines.", + "journal": "AAPS J", + "authors": [ + "Keire DA", + "Bream R", + "Wollein U", + "Schmaler-Ripcke J", + "Burchardt A", + "Conti M", + "Zmysłowski A", + "Keizers P", + "Morin J", + "Poh J", + "George M", + "Wierer M" + ], + "date": "2022-04-21T00:00:00Z", + "doi": "10.1208/s12248-022-00702-4", + "meta_date_publication": "2022 Apr 21", + "meta_volume": "24", + "meta_issue": "3", + "meta_pages": "56", + "score": 91163.71, + "text_hl": "Recalls of some batches of @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin@@@ have occurred due to the detection of @<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@N-nitrosodimethylamine@@@ (@<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@NDMA@@@) in amounts above the acceptable intake (AI) of 96 ng per day. ", + "citations": { + "NLM": "Keire DA, Bream R, Wollein U, Schmaler-Ripcke J, Burchardt A, Conti M, Zmysłowski A, Keizers P, Morin J, Poh J, George M, Wierer M. International Regulatory Collaboration on the Analysis of Nitrosamines in Metformin-Containing Medicines. AAPS J. 2022 Apr 21;24(3):56. PMID: 35449372", + "BibTeX": "@article{35449372, title={International Regulatory Collaboration on the Analysis of Nitrosamines in Metformin-Containing Medicines.}, author={Keire DA and Bream R and Wollein U and Schmaler-Ripcke J and Burchardt A and Conti M and Zmysłowski A and Keizers P and Morin J and Poh J and George M and Wierer M}, journal={AAPS J}, volume={24}, number={3}, pages={56}}" + } + }, + { + "_id": "34597792", + "pmid": 34597792, + "title": "NDMA analytics in metformin products: Comparison of methods and pitfalls.", + "journal": "Eur J Pharm Sci", + "authors": [ + "Fritzsche M", + "Blom G", + "Keitel J", + "Goettsche A", + "Seegel M", + "Leicht S", + "Guessregen B", + "Hickert S", + "Reifenberg P", + "Cimelli A", + "Baranowski R", + "Desmartin E", + "Barrau E", + "Harrison M", + "Bristow T", + "O'Neill N", + "Kirsch A", + "Krueger P", + "Saal C", + "Mouton B", + "Schlingemann J" + ], + "date": "2022-01-01T00:00:00Z", + "doi": "10.1016/j.ejps.2021.106026", + "meta_date_publication": "2022 Jan 1", + "meta_volume": "168", + "meta_issue": "", + "meta_pages": "106026", + "score": 88018.85, + "text_hl": "@<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@NDMA@@@ analytics in @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin@@@ products: Comparison of methods and pitfalls.", + "citations": { + "NLM": "Fritzsche M, Blom G, Keitel J, Goettsche A, Seegel M, Leicht S, Guessregen B, Hickert S, Reifenberg P, Cimelli A, Baranowski R, Desmartin E, Barrau E, Harrison M, Bristow T, O'Neill N, Kirsch A, Krueger P, Saal C, Mouton B, Schlingemann J. NDMA analytics in metformin products: Comparison of methods and pitfalls. Eur J Pharm Sci. 2022 Jan 1;168():106026. PMID: 34597792", + "BibTeX": "@article{34597792, title={NDMA analytics in metformin products: Comparison of methods and pitfalls.}, author={Fritzsche M and Blom G and Keitel J and Goettsche A and Seegel M and Leicht S and Guessregen B and Hickert S and Reifenberg P and Cimelli A and Baranowski R and Desmartin E and Barrau E and Harrison M and Bristow T and O'Neill N and Kirsch A and Krueger P and Saal C and Mouton B and Schlingemann J}, journal={Eur J Pharm Sci}, volume={168}, pages={106026}}" + } + }, + { + "_id": "41368201", + "pmid": 41368201, + "title": "A Comparative Study in Metformin Tablet Quality Assessment: LC-MS and LC-MS/MS Method Quantification of N-Nitroso-Dimethylamine in the Presence of Dimethyl Formamide.", + "journal": "Int J Anal Chem", + "authors": [ + "Sibhat G", + "Hassan MA", + "Reddy IK", + "A Khan M", + "Rahman Z" + ], + "date": "2025-12-01T00:00:00Z", + "doi": "10.1155/ianc/5625153", + "meta_date_publication": "2025", + "meta_volume": "2025", + "meta_issue": "", + "meta_pages": "5625153", + "score": 85866.82, + "text_hl": "...@<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@N-nitroso-dimethylamine@@@ (@<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@NDMA@@@, acceptable daily intake limit 96 ng/day), a probable @SPECIES_9606 @@@human@@@ @DISEASE_Precancerous_Conditions @DISEASE_MESH:D011230 @@@carcinogenic@@@ impurity, has been reported in @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin@@@ formulations. ", + "citations": { + "NLM": "Sibhat G, Hassan MA, Reddy IK, A Khan M, Rahman Z. A Comparative Study in Metformin Tablet Quality Assessment: LC-MS and LC-MS/MS Method Quantification of N-Nitroso-Dimethylamine in the Presence of Dimethyl Formamide. Int J Anal Chem. 2025;2025():5625153. PMID: 41368201", + "BibTeX": "@article{41368201, title={A Comparative Study in Metformin Tablet Quality Assessment: LC-MS and LC-MS/MS Method Quantification of N-Nitroso-Dimethylamine in the Presence of Dimethyl Formamide.}, author={Sibhat G and Hassan MA and Reddy IK and A Khan M and Rahman Z}, journal={Int J Anal Chem}, volume={2025}, pages={5625153}}" + } + }, + { + "_id": "40841507", + "pmid": 40841507, + "title": "LC-ESI-HRMS-Based Evaluation of Antioxidants as Additives to Inhibit Genotoxic Nitrosamine Formation in Metformin Hydrochloride Tablets.", + "journal": "AAPS PharmSciTech", + "authors": [ + "Dande A", + "Pallaprolu N", + "Gadilohar S", + "Mishra S", + "Malayandi R", + "Natesan S", + "Sahu A", + "Kumarasamy M", + "Velayutham R", + "Peraman R" + ], + "date": "2025-08-21T00:00:00Z", + "doi": "10.1208/s12249-025-03202-8", + "meta_date_publication": "2025 Aug 21", + "meta_volume": "26", + "meta_issue": "7", + "meta_pages": "218", + "score": 85282.99, + "text_hl": "Although @<m>CHEMICAL_Dimethylnitrosamine</m> @CHEMICAL_MESH:D004128 @@@NDMA@@@ formation in @<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@metformin hydrochloride@@@ (@<m>CHEMICAL_Metformin</m> @CHEMICAL_MESH:D008687 @@@MET@@@) tablets under nitrosating conditions is well-established, the potential of antioxidants to inhibit this impurity remains underexplored. ", + "citations": { + "NLM": "Dande A, Pallaprolu N, Gadilohar S, Mishra S, Malayandi R, Natesan S, Sahu A, Kumarasamy M, Velayutham R, Peraman R. LC-ESI-HRMS-Based Evaluation of Antioxidants as Additives to Inhibit Genotoxic Nitrosamine Formation in Metformin Hydrochloride Tablets. AAPS PharmSciTech. 2025 Aug 21;26(7):218. PMID: 40841507", + "BibTeX": "@article{40841507, title={LC-ESI-HRMS-Based Evaluation of Antioxidants as Additives to Inhibit Genotoxic Nitrosamine Formation in Metformin Hydrochloride Tablets.}, author={Dande A and Pallaprolu N and Gadilohar S and Mishra S and Malayandi R and Natesan S and Sahu A and Kumarasamy M and Velayutham R and Peraman R}, journal={AAPS PharmSciTech}, volume={26}, number={7}, pages={218}}" + } + } + ], + "facets": { + "facet_queries": {}, + "facet_fields": { + "journal": [ + { + "name": "Front Pharmacol", + "type": "int", + "value": 7 + }, + { + "name": "Int J Mol Sci", + "type": "int", + "value": 7 + }, + { + "name": "J Pharm Biomed Anal", + "type": "int", + "value": 7 + }, + { + "name": "Cancers (Basel)", + "type": "int", + "value": 4 + }, + { + "name": "AAPS J", + "type": "int", + "value": 3 + } + ], + "type": [ + { + "name": "Journal Article", + "type": "int", + "value": 153 + }, + { + "name": "Review", + "type": "int", + "value": 63 + }, + { + "name": "Letter", + "type": "int", + "value": 2 + }, + { + "name": "Comment", + "type": "int", + "value": 1 + }, + { + "name": "Meta-Analysis", + "type": "int", + "value": 1 + } + ], + "year": [ + { + "name": "2021", + "type": "int", + "value": 35 + }, + { + "name": "2022", + "type": "int", + "value": 32 + }, + { + "name": "2023", + "type": "int", + "value": 24 + }, + { + "name": "2020", + "type": "int", + "value": 21 + }, + { + "name": "2025", + "type": "int", + "value": 14 + }, + { + "name": "2024", + "type": "int", + "value": 13 + }, + { + "name": "2018", + "type": "int", + "value": 6 + }, + { + "name": "2019", + "type": "int", + "value": 6 + }, + { + "name": "2014", + "type": "int", + "value": 3 + }, + { + "name": "2011", + "type": "int", + "value": 1 + }, + { + "name": "2015", + "type": "int", + "value": 1 + }, + { + "name": "2016", + "type": "int", + "value": 1 + }, + { + "name": "2017", + "type": "int", + "value": 1 + } + ] + }, + "facet_ranges": {}, + "facet_intervals": {}, + "facet_heatmaps": {} + }, + "page_size": 10, + "current": 1, + "count": 158, + "total_pages": 16 +} \ No newline at end of file diff --git a/crossbar_llm/pubtator3_tools/tests/pytest.ini b/crossbar_llm/pubtator3_tools/tests/pytest.ini new file mode 100644 index 0000000..a851535 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/pytest.ini @@ -0,0 +1,6 @@ +# Scoped so this agent's suite runs without touching the project's root config: +# pytest -c crossbar_llm/pubtator3_tools/tests/pytest.ini crossbar_llm/pubtator3_tools/tests +[pytest] +asyncio_mode = auto +markers = + live: hits the real upstream service; needs credentials diff --git a/crossbar_llm/pubtator3_tools/tests/test_graph.py b/crossbar_llm/pubtator3_tools/tests/test_graph.py new file mode 100644 index 0000000..b5d0d52 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/test_graph.py @@ -0,0 +1,827 @@ +import re + +import pytest +from langchain_core.messages import AIMessage +from langchain_core.prompts import ChatPromptTemplate +from langchain_core.runnables import RunnableLambda + +from crossbar_llm.pubtator3_tools.agent import ( + DepthEvaluation, + EntityMention, + RouterDecision, + _ainvoke_structured_with_json_fallback, + build_graph, +) +from crossbar_llm.pubtator3_tools.prompts import ROUTER_SYSTEM_PROMPT +from crossbar_llm.pubtator3_tools import client + +BASE_URL = client.BASE_URL + + +def _url_pattern(path: str) -> re.Pattern: + return re.compile(rf"{re.escape(BASE_URL)}{re.escape(path)}.*") + + +def _register_full_pipeline(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=fx("pubtator3_autocomplete_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/relations"), + json=fx("pubtator3_relations_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + + +def _make_fake_router(decision: RouterDecision): + async def _router(_question: str) -> RouterDecision: + return decision + + return _router + + +async def _fake_synth(state) -> str: + pmids = sorted({p.pmid for p in state.get("passages", [])}) + return f"answer with {len(state.get('passages', []))} passages; pmids={pmids}" + + +class _StructuredNoneChatModel: + def __init__(self, responses: list[str]): + self.responses = responses + + def with_structured_output(self, schema, **kwargs): + return RunnableLambda(lambda _: None) + + async def __call__(self, _input): + return AIMessage(content=self.responses.pop(0)) + + +def test_entity_mention_text_description_allows_canonical_normalization(): + description = EntityMention.model_fields["text"].description + + assert "canonical normalization is allowed" in description + assert "BTK" in description + assert "Do NOT invent" in description + + +def test_router_prompt_guides_gene_symbol_normalization_without_invention(): + assert "PubTator3's gene autocomplete" in ROUTER_SYSTEM_PROMPT + assert "\"Bruton's tyrosine kinase\" -> text=\"BTK\"" in ROUTER_SYSTEM_PROMPT + assert "the user mentioned RANKL" in ROUTER_SYSTEM_PROMPT + + +async def test_json_fallback_parses_router_when_structured_output_returns_none(): + chat_model = _StructuredNoneChatModel(responses=[ + """ + ```json + { + "question_type": "keyword_search", + "mentions": [{"text": "SSRIs", "suggested_type": "chemical"}], + "relation": null, + "e2_type": null, + "keyword_query": "SSRIs off-label use", + "full_text": false, + "rationale": "Off-label use is a literature keyword-search topic." + } + ``` + """ + ]) + prompt = ChatPromptTemplate.from_messages([ + ("system", "Route the question."), + ("human", "Question: {question}"), + ]) + + decision, used_fallback = await _ainvoke_structured_with_json_fallback( + chat_model=chat_model, + prompt=prompt, + schema=RouterDecision, + values={"question": "List the off-label use of SSRIs"}, + json_instruction="Return only JSON.", + ) + + assert used_fallback is True + assert decision.question_type == "keyword_search" + assert decision.keyword_query == "SSRIs off-label use" + assert decision.mentions[0].text == "SSRIs" + assert decision.mentions[0].suggested_type == "chemical" + + +async def test_json_fallback_parses_depth_eval_when_structured_output_returns_none(): + chat_model = _StructuredNoneChatModel(responses=[ + '{"sufficient": true, "missing": null, "rationale": "Specific entities are named."}' + ]) + prompt = ChatPromptTemplate.from_messages([ + ("system", "Evaluate answer depth."), + ("human", "Question: {question}\nAnswer: {answer}"), + ]) + + verdict, used_fallback = await _ainvoke_structured_with_json_fallback( + chat_model=chat_model, + prompt=prompt, + schema=DepthEvaluation, + values={"question": "What is JAK1?", "answer": "JAK1 is a kinase."}, + json_instruction="Return only JSON.", + ) + + assert used_fallback is True + assert verdict.sufficient is True + assert verdict.missing is None + + +async def test_partner_discovery_flow(httpx_mock, fx): + _register_full_pipeline(httpx_mock, fx) + + decision = RouterDecision( + question_type="relation_partner_discovery", + mentions=[EntityMention(text="JAK1", suggested_type="gene")], + relation="negative_correlate", + e2_type="Chemical", + rationale="user asks for chemicals related to JAK1", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + max_partners=2, + max_documents=5, + ) + + final = await graph.ainvoke({"question": "what chemicals correlate with JAK1?", "warnings": []}) + + assert final["question_type"] == "relation_partner_discovery" + assert "JAK1" in final["resolved"] + assert len(final["partners"]) > 0 + assert len(final["queries_used"]) > 0 + assert all(q.startswith("relations:") for q in final["queries_used"]) + assert len(final["passages"]) > 0 + assert final["final_answer"] is not None + assert "passages" in final["final_answer"] + + +async def test_known_pair_flow_skips_partner_discovery(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=fx("pubtator3_autocomplete_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + # /relations is intentionally unmocked — pytest-httpx fails on contact. + + decision = RouterDecision( + question_type="relation_known_pair", + mentions=[ + EntityMention(text="metformin", suggested_type="chemical", role="e1"), + EntityMention(text="type 2 diabetes", suggested_type="disease", role="e2"), + ], + relation="treat", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + ) + + final = await graph.ainvoke( + {"question": "does metformin treat type 2 diabetes?", "warnings": []} + ) + + assert final["question_type"] == "relation_known_pair" + assert final.get("partners", []) == [] + assert len(final["queries_used"]) == 1 + assert final["queries_used"][0].startswith("relations:treat|") + assert final["final_answer"] is not None + + +async def test_out_of_scope_terminates_without_tool_calls(httpx_mock): + decision = RouterDecision( + question_type="out_of_scope", + mentions=[], + rationale="not a literature lookup question", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + ) + + final = await graph.ainvoke( + {"question": "what's the FDA approval history of metformin?", "warnings": []} + ) + + assert final["question_type"] == "out_of_scope" + assert final.get("final_answer") is None + assert final.get("passages", []) == [] + assert final.get("partners", []) == [] + + +async def test_single_node_flow(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=fx("pubtator3_autocomplete_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + + decision = RouterDecision( + question_type="single_node", + mentions=[EntityMention(text="JAK1", suggested_type="gene")], + rationale="single entity question", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + ) + + final = await graph.ainvoke({"question": "tell me about JAK1", "warnings": []}) + + assert final["question_type"] == "single_node" + assert len(final["queries_used"]) == 1 + assert final["queries_used"][0].startswith("@GENE_") + assert final["final_answer"] is not None + + +async def test_router_failure_falls_back_to_keyword_search(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + + async def _broken_router(_question: str) -> RouterDecision: + # Simulate a provider-side schema validation rejection. + raise ValueError("invented relation 'target' is not in the enum") + + graph = build_graph(router=_broken_router, synthesizer=_fake_synth) + + final = await graph.ainvoke( + {"question": "Which drugs target proteins associated with Alzheimer disease?", "warnings": []} + ) + + assert final["question_type"] == "keyword_search" + assert final["keyword_query"] == ( + "Which drugs target proteins associated with Alzheimer disease?" + ) + assert any("router failed" in w for w in final["warnings"]) + assert final["queries_used"] == [final["keyword_query"]] + assert final["final_answer"] is not None + + +async def test_keyword_search_flow_skips_resolve_and_partner_discovery(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + # No /entity/autocomplete/ or /relations mocks — pytest-httpx fails if hit. + + decision = RouterDecision( + question_type="keyword_search", + mentions=[], + keyword_query="imatinib side effects", + rationale="side effects aren't a PubTator3 entity type but the literature is", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + ) + + final = await graph.ainvoke( + {"question": "What are the side effects of imatinib?", "warnings": []} + ) + + assert final["question_type"] == "keyword_search" + assert final["keyword_query"] == "imatinib side effects" + assert final.get("resolved", {}) == {} + assert final.get("partners", []) == [] + assert final["queries_used"] == ["imatinib side effects"] + assert final["final_answer"] is not None + + +async def test_keyword_search_with_empty_query_warns_and_still_synthesizes(httpx_mock): + decision = RouterDecision( + question_type="keyword_search", + keyword_query="", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + ) + + final = await graph.ainvoke({"question": "vague", "warnings": []}) + + assert final["question_type"] == "keyword_search" + assert final["queries_used"] == [] + assert any("keyword_query" in w for w in final.get("warnings", [])) + assert final["final_answer"] is not None + + +async def test_full_text_defaults_to_false_when_router_omits_it(httpx_mock, fx): + _register_full_pipeline(httpx_mock, fx) + + # Router decision without `full_text` -> Pydantic default is False. + decision = RouterDecision( + question_type="relation_partner_discovery", + mentions=[EntityMention(text="JAK1", suggested_type="gene")], + relation="negative_correlate", + e2_type="Chemical", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + max_partners=1, + max_documents=3, + ) + + await graph.ainvoke({"question": "...", "warnings": []}) + + export_requests = httpx_mock.get_requests( + url=_url_pattern("/publications/export/biocjson") + ) + assert len(export_requests) >= 1 + assert all(req.url.params["full"] == "false" for req in export_requests) + + +async def test_full_text_true_propagates_to_export_call(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=fx("pubtator3_autocomplete_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + + decision = RouterDecision( + question_type="single_node", + mentions=[EntityMention(text="JAK1", suggested_type="gene")], + full_text=True, + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + max_documents=3, + ) + + final = await graph.ainvoke({"question": "...", "warnings": []}) + + assert final["full_text"] is True + export_requests = httpx_mock.get_requests( + url=_url_pattern("/publications/export/biocjson") + ) + assert all(req.url.params["full"] == "true" for req in export_requests) + + +async def test_unresolvable_entity_falls_back_to_router_keyword_query(httpx_mock): + """When single_node entity resolution fails, search_node should fall + back to the router-supplied `keyword_query` rather than bail silently. + The fallback may still find nothing — that's fine — but the fallback + query MUST be attempted and recorded in queries_used.""" + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=[], + is_reusable=True, + ) + # Fallback hits search with 0 results (no recovery possible here). + httpx_mock.add_response( + url=_url_pattern("/search/"), + json={"results": [], "count": 0}, + is_reusable=True, + ) + + decision = RouterDecision( + question_type="single_node", + mentions=[EntityMention(text="ZZZNotARealEntity")], + keyword_query="ZZZNotARealEntity background", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + ) + + final = await graph.ainvoke({"question": "tell me about ZZZNotARealEntity", "warnings": []}) + + assert "ZZZNotARealEntity" in final.get("unresolved", []) + # The fallback fired with the router-supplied keyword_query. + assert final.get("queries_used", []) == ["ZZZNotARealEntity background"] + # Fallback search returned 0 hits, so passages still empty. + assert final.get("passages", []) == [] + assert final["final_answer"] is not None + assert any("falling back to keyword search" in w for w in final.get("warnings", [])) + + +def _make_fake_evaluator(verdict: DepthEvaluation, call_log: list | None = None): + async def _evaluator(state): + if call_log is not None: + call_log.append({ + "answer": state.get("final_answer"), + "full_text": state.get("full_text"), + }) + return verdict + return _evaluator + + +async def test_depth_evaluator_no_loop_when_sufficient(httpx_mock, fx): + _register_full_pipeline(httpx_mock, fx) + + decision = RouterDecision( + question_type="relation_partner_discovery", + mentions=[EntityMention(text="JAK1", suggested_type="gene")], + relation="negative_correlate", + e2_type="Chemical", + ) + call_log: list = [] + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + evaluator=_make_fake_evaluator( + DepthEvaluation(sufficient=True, missing=None), call_log + ), + max_partners=1, + max_documents=3, + ) + + final = await graph.ainvoke({"question": "...", "warnings": []}) + + assert final["depth_sufficient"] is True + assert final.get("refinement_attempted", False) is False + # Evaluator was called exactly once. + assert len(call_log) == 1 + # Export endpoint hit exactly once (no refinement loop). + export_calls = httpx_mock.get_requests( + url=_url_pattern("/publications/export/biocjson") + ) + assert len(export_calls) == 1 + + +async def test_depth_evaluator_triggers_refinement_when_insufficient(httpx_mock, fx): + _register_full_pipeline(httpx_mock, fx) + + decision = RouterDecision( + question_type="relation_partner_discovery", + mentions=[EntityMention(text="JAK1", suggested_type="gene")], + relation="negative_correlate", + e2_type="Chemical", + ) + # First call: insufficient. Second call (after refinement) won't matter + # because evaluate_depth_node short-circuits on refinement_attempted. + call_log: list = [] + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + evaluator=_make_fake_evaluator( + DepthEvaluation(sufficient=False, missing="no mechanism"), call_log + ), + max_partners=1, + max_documents=3, + ) + + final = await graph.ainvoke({"question": "...", "warnings": []}) + + # After the cap fires, the second evaluate short-circuits to True. What + # persists is the side-effect of the first verdict: refinement happened. + assert final["refinement_attempted"] is True + assert final["full_text"] is True + # Two synthesize passes -> two export calls (one with full=false, one with full=true). + export_calls = httpx_mock.get_requests( + url=_url_pattern("/publications/export/biocjson") + ) + assert len(export_calls) == 2 + full_params = [c.url.params["full"] for c in export_calls] + assert full_params == ["false", "true"] + # Evaluator runs once on the first answer; the second pass short-circuits. + assert len(call_log) == 1 + # Warning is recorded so the user sees why the loop fired. + assert any("shallow" in w for w in final["warnings"]) + + +async def test_depth_evaluator_skipped_when_full_text_already_true(httpx_mock, fx): + # single_node skips /relations — register only what it actually calls. + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=fx("pubtator3_autocomplete_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + + decision = RouterDecision( + question_type="single_node", + mentions=[EntityMention(text="JAK1", suggested_type="gene")], + full_text=True, + ) + call_log: list = [] + # Even if evaluator would flag insufficient, the node short-circuits + # because full_text is already True (no escalation lever). + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + evaluator=_make_fake_evaluator( + DepthEvaluation(sufficient=False, missing="should be ignored"), call_log + ), + max_documents=3, + ) + + final = await graph.ainvoke({"question": "...", "warnings": []}) + + assert final["depth_sufficient"] is True + assert final.get("refinement_attempted", False) is False + # Evaluator was never invoked. + assert len(call_log) == 0 + # Export fired once with full=true (router's decision). + export_calls = httpx_mock.get_requests( + url=_url_pattern("/publications/export/biocjson") + ) + assert len(export_calls) == 1 + assert export_calls[0].url.params["full"] == "true" + + +async def test_depth_evaluator_skipped_when_no_passages(httpx_mock): + # Unresolvable entity -> fallback keyword search returns nothing -> + # no passages -> evaluator should short-circuit instead of trying to + # deepen an empty answer. + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=[], + is_reusable=True, + ) + # Fallback keyword search also finds nothing — true data desert. + httpx_mock.add_response( + url=_url_pattern("/search/"), + json={"results": [], "count": 0}, + is_reusable=True, + ) + + decision = RouterDecision( + question_type="single_node", + mentions=[EntityMention(text="ZZZNotARealEntity")], + ) + call_log: list = [] + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + evaluator=_make_fake_evaluator( + DepthEvaluation(sufficient=False, missing="should be ignored"), call_log + ), + ) + + final = await graph.ainvoke({"question": "...", "warnings": []}) + + assert final["depth_sufficient"] is True + assert final.get("refinement_attempted", False) is False + # Evaluator was never invoked because there are no passages. + assert len(call_log) == 0 + + +async def test_known_pair_falls_back_to_keyword_when_one_entity_unresolved(httpx_mock, fx): + """When relation_known_pair has one mention that autocomplete can't + resolve (e.g. a generic descriptor like 'antidote'), the search node + should fall back to a keyword query instead of bailing silently.""" + # First mention resolves to the fixture's JAK1 candidates; second + # returns an empty candidate list (unresolvable). + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=fx("pubtator3_autocomplete_example"), + ) + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=[], + ) + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + + decision = RouterDecision( + question_type="relation_known_pair", + mentions=[ + EntityMention(text="benzodiazepine", suggested_type="chemical", role="e1"), + EntityMention(text="antidote", role="e2"), + ], + relation="treat", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + max_documents=2, + ) + + final = await graph.ainvoke({"question": "What is the antidote for benzodiazepine?", "warnings": []}) + + queries = final.get("queries_used") or [] + assert len(queries) == 1 + # The fallback query is the resolved entity name + the unresolved + # mention text + the relation verb. + assert "JAK1" in queries[0] or "benzodiazepine" in queries[0].lower() + assert "antidote" in queries[0] + assert "treat" in queries[0] + # The pipeline kept running — search returned hits, export returned docs. + assert final.get("final_answer") is not None + assert any("known-pair" in w and "fell back" in w for w in final.get("warnings", [])) + + +async def test_full_text_false_filters_out_body_sections(httpx_mock, fx): + """Even when PubTator3 returns body sections (review-article quirk), + we must trim to title + abstract when the router asked for abstract mode.""" + # Synthetic doc carrying both abstract-mode passages (lowercase) and + # body sections (uppercase section_type). full=false was requested but + # PubTator3 returned everything anyway. + raw_doc_with_body = { + "PubTator3": [ + { + "pmid": "55555555", + "passages": [ + { + "infons": {"type": "title"}, + "offset": 0, + "text": "A real title passage.", + "annotations": [], + }, + { + "infons": {"type": "abstract"}, + "offset": 100, + "text": "A real abstract passage.", + "annotations": [], + }, + { + "infons": {"section_type": "INTRO", "type": "paragraph"}, + "offset": 500, + "text": "Intro body that should be filtered out in abstract mode.", + "annotations": [], + }, + { + "infons": {"section_type": "METHODS", "type": "paragraph"}, + "offset": 1500, + "text": "Methods body that should be filtered out.", + "annotations": [], + }, + { + "infons": {"section_type": "RESULTS", "type": "paragraph"}, + "offset": 2500, + "text": "Results body that should be filtered out.", + "annotations": [], + }, + ], + "relations": [], + } + ] + } + + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=fx("pubtator3_autocomplete_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=raw_doc_with_body, + is_reusable=True, + ) + + decision = RouterDecision( + question_type="single_node", + mentions=[EntityMention(text="JAK1", suggested_type="gene")], + full_text=False, + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + max_documents=1, + ) + + final = await graph.ainvoke({"question": "...", "warnings": []}) + + sections = [p.section for p in final["passages"]] + # Only title + abstract survive; INTRO/METHODS/RESULTS are filtered. + assert set(sections) == {"title", "abstract"} + assert "INTRO" not in sections + assert "METHODS" not in sections + assert "RESULTS" not in sections + + +async def test_full_text_true_keeps_body_sections(httpx_mock, fx): + """Same fixture but with full_text=True — body sections must pass through.""" + raw_doc_with_body = { + "PubTator3": [ + { + "pmid": "55555556", + "passages": [ + { + "infons": {"section_type": "TITLE", "type": "title"}, + "offset": 0, + "text": "Title.", + "annotations": [], + }, + { + "infons": {"section_type": "ABSTRACT", "type": "paragraph"}, + "offset": 100, + "text": "Abstract.", + "annotations": [], + }, + { + "infons": {"section_type": "METHODS", "type": "paragraph"}, + "offset": 500, + "text": "Methods body.", + "annotations": [], + }, + ], + "relations": [], + } + ] + } + + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=fx("pubtator3_autocomplete_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=raw_doc_with_body, + is_reusable=True, + ) + + decision = RouterDecision( + question_type="single_node", + mentions=[EntityMention(text="JAK1", suggested_type="gene")], + full_text=True, + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + max_documents=1, + ) + + final = await graph.ainvoke({"question": "...", "warnings": []}) + + sections = {p.section for p in final["passages"]} + assert "METHODS" in sections + assert "TITLE" in sections + assert "ABSTRACT" in sections diff --git a/crossbar_llm/pubtator3_tools/tests/test_models.py b/crossbar_llm/pubtator3_tools/tests/test_models.py new file mode 100644 index 0000000..3365deb --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/test_models.py @@ -0,0 +1,234 @@ +"""Pydantic round-trip tests against recorded API fixtures. + +These don't touch the network. They prove that the models defined in +`crossbar_llm.pubtator3_tools.client` correctly parse what the real +PubTator3 API returns — independent of the rest of the pipeline. +""" +from crossbar_llm.pubtator3_tools.client import ( + EntityCandidate, + RelatedEntity, + SearchHit, + _clean_snippet, + _parse_document, +) + + +def test_autocomplete_round_trip(fx): + raw = fx("pubtator3_autocomplete_example") + candidates = [EntityCandidate(**item) for item in raw] + + assert len(candidates) == 5 + assert all(c.accession.startswith("@GENE_") for c in candidates) + assert candidates[0].name == "JAK1" + assert candidates[0].biotype == "gene" + assert candidates[0].db == "ncbi_gene" + + +def test_relations_round_trip(fx): + raw = fx("pubtator3_relations_example") + related = [RelatedEntity(**item) for item in raw] + + assert len(related) > 0 + assert all(r.publications > 0 for r in related) + assert all(r.type == "negative_correlate" for r in related) + assert related[0].source == "@CHEMICAL_ruxolitinib" + assert related[0].target == "@GENE_JAK1" + + # API returns results sorted by publication count descending. + counts = [r.publications for r in related] + assert counts == sorted(counts, reverse=True) + + +def test_search_round_trip_strips_markup(fx): + raw = fx("pubtator3_search_example") + hits = [SearchHit(**item) for item in raw["results"]] + + assert len(hits) > 0 + h = hits[0] + assert h.pmid == 33849366 + assert isinstance(h.score, float) + + # text_hl preserved as-is for debugging; snippet is cleaned. + assert h.text_hl is not None + assert "@<m>" in h.text_hl # raw markup present in source + assert h.snippet is not None + assert "@<m>" not in h.snippet + assert "@@@" not in h.snippet + + +def test_export_round_trip(fx): + raw = fx("pubtator3_export_example") + docs = [_parse_document(d) for d in raw["PubTator3"]] + + assert len(docs) > 0 + doc = docs[0] + assert doc.pmid == 33849366 + assert doc.title.startswith("Investigating") + + # Both title and abstract sections present. + sections = {p.section for p in doc.passages} + assert {"title", "abstract"}.issubset(sections) + + # BioREx scores parse as floats and carry their PMID. + assert len(doc.relations) > 0 + rel = doc.relations[0] + assert isinstance(rel.score, float) + assert 0.0 <= rel.score <= 1.0 + assert rel.pmid == doc.pmid + + # Database identifiers are surfaced for CROssBAR-KG joins. At least one + # of the document's relations must have non-null identifiers on both sides + # — that's the field the KG keys on (MESH:Dxxxxxx for chemicals/diseases, + # bare NCBI Gene ID for genes). + assert any( + r.role1_identifier and r.role2_identifier for r in doc.relations + ) + + # Annotations were filtered: every survivor has a valid accession. + abstract_passage = next(p for p in doc.passages if p.section == "abstract") + assert all(a.accession is not None for a in abstract_passage.annotations) + + +def test_clean_snippet_strips_both_highlight_forms(): + """The regex must handle both highlighted (@<m>...</m>) and + annotated-only entities in the same string.""" + raw = ( + "Atomic Simulation of @<m>GENE_JAK1</m> @GENE_395681 @@@JAK1@@@ " + "and @GENE_JAK2 @GENE_16452 @@@JAK2@@@ " + "with @<m>CHEMICAL_ruxolitinib</m> @CHEMICAL_MESH:C540383 @@@Ruxolitinib@@@" + ) + cleaned = _clean_snippet(raw) + + assert cleaned is not None + assert "JAK1" in cleaned + assert "JAK2" in cleaned + assert "Ruxolitinib" in cleaned + assert "@<m>" not in cleaned + assert "@@@" not in cleaned + + +def test_clean_snippet_handles_none(): + assert _clean_snippet(None) is None + assert _clean_snippet("") == "" + + +def test_full_text_parser_uses_section_type_and_drops_boilerplate(): + """Full-text docs put the semantic section in `infons.section_type` + (not `infons.type`, which is just a formatting hint). Boilerplate + sections like COMP_INT and SUPPL must be dropped at parse time.""" + raw_doc = { + "pmid": "99999999", + "passages": [ + { + "infons": {"type": "title"}, + "offset": 0, + "text": "A fake paper title for testing.", + "annotations": [], + }, + { + "infons": {"section_type": "INTRO", "type": "paragraph"}, + "offset": 100, + "text": "Introductory body content with real science.", + "annotations": [], + }, + { + "infons": {"section_type": "METHODS", "type": "paragraph"}, + "offset": 500, + "text": "Methods body content.", + "annotations": [], + }, + { + "infons": {"section_type": "RESULTS", "type": "paragraph"}, + "offset": 1500, + "text": "Results body content.", + "annotations": [], + }, + { + "infons": {"section_type": "COMP_INT", "type": "paragraph"}, + "offset": 5000, + "text": "No potential conflict of interest was reported by the author(s).", + "annotations": [], + }, + { + "infons": {"section_type": "ACK_FUND", "type": "paragraph"}, + "offset": 5100, + "text": "This work was funded by NIH grant ...", + "annotations": [], + }, + { + "infons": {"section_type": "SUPPL", "type": "title_1"}, + "offset": 5200, + "text": "Supplementary material", + "annotations": [], + }, + { + "infons": {"section_type": "REF", "type": "paragraph"}, + "offset": 5400, + "text": "1. Smith J et al. 2020. ...", + "annotations": [], + }, + ], + "relations": [], + } + + doc = _parse_document(raw_doc) + + sections = [p.section for p in doc.passages] + + # title (from `type`) plus the three content body sections survive. + assert sections == ["title", "INTRO", "METHODS", "RESULTS"] + + # No boilerplate leaked through. + skipped = {"COMP_INT", "ACK_FUND", "SUPPL", "REF"} + assert not (set(sections) & skipped) + + # Body sections are labeled by section_type, not by the formatting `type`. + assert "paragraph" not in sections + assert "title_1" not in sections + + +# --- one bad record must not discard the batch ----------------------------- +# PubTator3 is an evolving service: a new concept type or relation type is a +# routine upstream change. Parsing every record in one comprehension meant a +# single unfamiliar value raised for the whole response, and the tool wrapper +# turned that into an empty result — one new relation type discarded all 352 +# partners of a query. + +def test_parse_items_skips_only_the_bad_record(): + from crossbar_llm.pubtator3_tools.client import SearchHit, _parse_items + + raw = [ + {"pmid": 1, "title": "Good one"}, + {"pmid": 2}, # missing required `title` + {"pmid": 3, "title": "Good two"}, + ] + hits = _parse_items(raw, lambda i: SearchHit(**i), what="test") + assert [h.pmid for h in hits] == [1, 3] + + +def test_parse_items_tolerates_non_list_response(): + from crossbar_llm.pubtator3_tools.client import SearchHit, _parse_items + + # An error envelope instead of a list must yield [], not raise. + assert _parse_items({"error": "busy"}, lambda i: SearchHit(**i), what="test") == [] + assert _parse_items(None, lambda i: SearchHit(**i), what="test") == [] + + +def test_unknown_biotype_is_kept_not_dropped(): + """An unfamiliar concept type still carries a usable accession/name, so the + record is kept rather than discarded.""" + from crossbar_llm.pubtator3_tools.client import EntityCandidate + + c = EntityCandidate( + _id="@PROTEINDOMAIN_X", name="X", biotype="proteindomain", + db_id="1", db="ncbi_gene", + ) + assert c.accession == "@PROTEINDOMAIN_X" + assert c.biotype == "proteindomain" + + +def test_unknown_relation_type_is_kept_not_dropped(): + from crossbar_llm.pubtator3_tools.client import RelatedEntity + + r = RelatedEntity(type="coexpress", source="@GENE_A", target="@GENE_B", publications=3) + assert r.type == "coexpress" diff --git a/crossbar_llm/pubtator3_tools/tests/test_new_features.py b/crossbar_llm/pubtator3_tools/tests/test_new_features.py new file mode 100644 index 0000000..532e674 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/test_new_features.py @@ -0,0 +1,412 @@ +"""Tests for the newer graph features: abstracts_only, section filter, and the +evaluator's suggested_sections-driven refinement union. + +These complement test_graph.py by exercising the behaviors added after the +initial split: the deployment-level abstracts_only switch, the section +filter inside export_node, and the depth evaluator's ability to pick which +sections to add on the refinement pass. +""" +import re + +import pytest + +from crossbar_llm.pubtator3_tools.agent import ( + DepthEvaluation, + EntityMention, + RouterDecision, + build_graph, +) +from crossbar_llm.pubtator3_tools.nodes import export_node +from crossbar_llm.pubtator3_tools import client + +BASE_URL = client.BASE_URL + + +def _url_pattern(path: str) -> re.Pattern: + return re.compile(rf"{re.escape(BASE_URL)}{re.escape(path)}.*") + + +def _make_fake_router(decision: RouterDecision): + async def _router(_question: str) -> RouterDecision: + return decision + + return _router + + +async def _fake_synth(state) -> str: + return f"answer over {len(state.get('passages', []))} passages" + + +# --- (1) abstracts_only clamps router output ------------------------------ + +@pytest.mark.asyncio +async def test_abstracts_only_clamps_router_full_text_and_sections(): + """Even if the router returns full_text=True + sections, abstracts_only=True + should flatten both to False / None before the rest of the pipeline sees + them. We use out_of_scope so no HTTP mocks are needed — the clamp lives + in router_node and applies regardless of the downstream path.""" + decision = RouterDecision( + question_type="out_of_scope", + full_text=True, + sections=["METHODS", "RESULTS"], + rationale="probe", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + abstracts_only=True, + ) + + final = await graph.ainvoke({"question": "weather today?", "warnings": []}) + + assert final["full_text"] is False + assert final["sections"] is None + + +@pytest.mark.asyncio +async def test_abstracts_only_short_circuits_depth_evaluator(): + """When abstracts_only=True the evaluator must NOT escalate to full text + even if a (hypothetical) evaluator would flag the answer as insufficient. + We register a fake evaluator that ALWAYS says insufficient; the graph + should still accept the answer without setting refinement_attempted.""" + + async def _always_insufficient(_state): + return DepthEvaluation( + sufficient=False, + missing="probe", + suggested_sections=["METHODS"], + rationale="probe", + ) + + decision = RouterDecision(question_type="out_of_scope", rationale="probe") + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + evaluator=_always_insufficient, + abstracts_only=True, + ) + + final = await graph.ainvoke({"question": "probe", "warnings": []}) + + # out_of_scope terminates before evaluate_depth runs, so depth fields + # never get set. The point is that abstracts_only didn't crash and the + # graph completed; combined with the clamp test above, that confirms + # the deployment switch holds end-to-end. + assert final.get("refinement_attempted") is not True + + +# --- (2) section filter in export_node ------------------------------------ + +def _mock_export_doc_with_body_sections() -> dict: + """Synthetic BioC JSON for one PMC-OA document carrying title + abstract + + INTRO + METHODS + RESULTS + DISCUSS passages. Used to exercise the + section filter without depending on the recorded fixtures (which only + have title + abstract).""" + return { + "PubTator3": [ + { + "pmid": "1234567", + "pmcid": "PMC1234567", + "journal": "Synth J", + "authors": [], + "date": "2024-01-01", + "passages": [ + {"infons": {"type": "title"}, "offset": 0, "text": "A title.", "annotations": []}, + {"infons": {"type": "abstract"}, "offset": 9, "text": "An abstract.", "annotations": []}, + {"infons": {"section_type": "INTRO", "type": "paragraph"}, "offset": 22, "text": "Intro body.", "annotations": []}, + {"infons": {"section_type": "METHODS", "type": "paragraph"}, "offset": 34, "text": "Methods body.", "annotations": []}, + {"infons": {"section_type": "RESULTS", "type": "paragraph"}, "offset": 48, "text": "Results body.", "annotations": []}, + {"infons": {"section_type": "DISCUSS", "type": "paragraph"}, "offset": 62, "text": "Discussion body.", "annotations": []}, + ], + "relations": [], + } + ] + } + + +@pytest.mark.asyncio +async def test_section_filter_keeps_title_abstract_and_only_methods(httpx_mock): + """With sections=['METHODS'] and full_text=True, export_node should drop + INTRO / RESULTS / DISCUSS body passages while always keeping title + + abstract.""" + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=_mock_export_doc_with_body_sections(), + ) + + state = { + "pmids": [1234567], + "full_text": True, + "sections": ["METHODS"], + "warnings": [], + } + result = await export_node(state, max_documents=5) + + sections = sorted({p.section for p in result["passages"]}) + assert "title" in sections + assert "abstract" in sections + assert "METHODS" in sections + assert "INTRO" not in sections + assert "RESULTS" not in sections + assert "DISCUSS" not in sections + + +@pytest.mark.asyncio +async def test_no_section_filter_keeps_every_body_section(httpx_mock): + """Sanity counterpart: sections=None with full_text=True passes every body + section through. Same fixture, no filter → all 6 passage section_types + survive.""" + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=_mock_export_doc_with_body_sections(), + ) + + state = { + "pmids": [1234567], + "full_text": True, + "sections": None, + "warnings": [], + } + result = await export_node(state, max_documents=5) + + sections = {p.section for p in result["passages"]} + assert {"title", "abstract", "INTRO", "METHODS", "RESULTS", "DISCUSS"}.issubset(sections) + + +# --- (3) evaluator suggested_sections unions with current ------------------ + +@pytest.mark.asyncio +async def test_zero_results_falls_back_to_keyword_search(httpx_mock): + """When a relation_partner_discovery resolves the anchor but PubTator3 + returns 0 partners, search_node should fall back to a free-text keyword + query built from the resolved entity name + relation + e2_type — instead + of producing an empty PMID list and letting the synthesizer say 'no + information available'. + + This is the Q5 (BTK / Ibrutinib) failure mode from the benchmark run: + the BioREx graph didn't tag the inhibit edge, so the structured query + found nothing even though PubMed has thousands of relevant papers. + """ + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=[ + { + "_id": "@GENE_BTK", + "name": "BTK", + "biotype": "gene", + "db_id": "695", + "db": "ncbi_gene", + "match": "Matched on name <m>BTK</m>", + } + ], + is_reusable=True, + ) + # find_related: returns NO partners — this is the trigger. + httpx_mock.add_response( + url=_url_pattern("/relations"), + json=[], + is_reusable=True, + ) + # The fallback keyword search must succeed. + httpx_mock.add_response( + url=_url_pattern("/search/"), + json={ + "results": [{"pmid": 99999, "title": "Ibrutinib targets BTK in CLL"}], + "count": 1, + }, + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json={ + "PubTator3": [ + { + "pmid": "99999", + "passages": [ + {"infons": {"type": "title"}, "offset": 0, "text": "Ibrutinib targets BTK in CLL.", "annotations": []}, + {"infons": {"type": "abstract"}, "offset": 30, "text": "Ibrutinib is a BTK inhibitor.", "annotations": []}, + ], + "relations": [], + } + ] + }, + is_reusable=True, + ) + + decision = RouterDecision( + question_type="relation_partner_discovery", + mentions=[EntityMention(text="BTK", suggested_type="gene")], + relation="inhibit", + e2_type="Chemical", + keyword_query="BTK inhibitor CLL", + rationale="probe", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + ) + + final = await graph.ainvoke({"question": "BTK inhibitor for CLL", "warnings": []}) + + # Anchor resolved; no partners came back; fallback fired with the + # router-supplied keyword_query and found PMID 99999. + assert final.get("partners") == [] + assert final.get("pmids") == [99999] + assert "BTK inhibitor CLL" in (final.get("queries_used") or []) + # The warning trail records the fallback so it's auditable. + assert any("falling back to keyword search" in w for w in final.get("warnings") or []) + + +@pytest.mark.asyncio +async def test_zero_results_falls_back_to_question_when_router_omits_keyword_query(httpx_mock): + """Catastrophic-failure tier: structured query yields 0 PMIDs AND the + router didn't supply `keyword_query` (e.g. router-error path emitted a + minimal RouterDecision). search_node should fall back to the user's + question verbatim rather than silently giving up.""" + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=[ + {"_id": "@GENE_BTK", "name": "BTK", "biotype": "gene", "db_id": "695", + "db": "ncbi_gene", "match": "Matched on name <m>BTK</m>"} + ], + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/relations"), + json=[], + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/search/"), + json={"results": [], "count": 0}, + is_reusable=True, + ) + + decision = RouterDecision( + question_type="relation_partner_discovery", + mentions=[EntityMention(text="BTK", suggested_type="gene")], + relation="inhibit", + e2_type="Chemical", + # keyword_query intentionally omitted — simulates router error path. + rationale="probe", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + ) + + final = await graph.ainvoke({"question": "What drug inhibits BTK?", "warnings": []}) + + # Fallback used the user's question verbatim. + assert "What drug inhibits BTK?" in (final.get("queries_used") or []) + + +@pytest.mark.asyncio +async def test_evaluator_suggested_sections_unioned_on_refinement(httpx_mock): + """First pass: router returns full_text=True + sections=['INTRO']. Evaluator + says insufficient and suggests ['METHODS']. After the refinement loop, + state should reflect the union ['INTRO', 'METHODS'], with refinement_ + attempted=True so the loop can't fire again.""" + # The graph re-enters `export` on refinement; mock the export endpoint so + # both passes can fetch. /search/ and /entity/autocomplete/ aren't used + # for the single_node path with an unresolvable mention — but to keep + # the test tight we route via keyword_search. + httpx_mock.add_response( + url=_url_pattern("/search/"), + json={"results": [{"pmid": 1234567, "title": "x"}], "count": 1}, + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=_mock_export_doc_with_body_sections(), + is_reusable=True, + ) + + decision = RouterDecision( + question_type="keyword_search", + keyword_query="probe", + full_text=True, + sections=["INTRO"], + rationale="probe", + ) + + calls = {"n": 0} + + async def _evaluator(_state): + calls["n"] += 1 + # Only the first call returns insufficient — the second (post-refine) + # accepts the answer so the test terminates deterministically. + if calls["n"] == 1: + return DepthEvaluation( + sufficient=False, + missing="no protocol detail", + suggested_sections=["METHODS"], + rationale="probe", + ) + return DepthEvaluation(sufficient=True, missing=None, rationale="ok now") + + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + evaluator=_evaluator, + max_documents=2, + ) + + final = await graph.ainvoke({"question": "how do they assay X?", "warnings": []}) + + assert final.get("refinement_attempted") is True + # Union of current ['INTRO'] + suggested ['METHODS'] preserves order. + assert final.get("sections") == ["INTRO", "METHODS"] + assert final.get("full_text") is True + + +# --- resolve_node confidence gating --------------------------------------- +# resolve_node trusts only PubTator3 name/synonym matches and abstains on its +# fuzzy 'Multiple matches' fallback (the main source of wrong resolutions like +# 'histone H3' -> @GENE_HTR12), leaving the entity for the keyword fallback. + +async def test_resolve_node_rejects_fuzzy_multiple_matches(httpx_mock): + from crossbar_llm.pubtator3_tools.nodes import resolve_node + + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=[ + {"_id": "@GENE_HTR12", "name": "HTR12", "biotype": "gene", + "db_id": "1", "db": "ncbi_gene", "match": "Multiple matches"}, + {"_id": "@GENE_AT5G10980", "name": "AT5G10980", "biotype": "gene", + "db_id": "2", "db": "ncbi_gene", "match": "Multiple matches"}, + ], + is_reusable=True, + ) + + state = {"mentions": [EntityMention(text="histone H3", suggested_type="gene")], + "warnings": []} + out = await resolve_node(state) + + assert "histone H3" in out["unresolved"] + assert "histone H3" not in out["resolved"] + assert any("no confident" in w for w in out["warnings"]) + + +async def test_resolve_node_accepts_name_match_and_skips_fuzzy_runners_up(httpx_mock): + from crossbar_llm.pubtator3_tools.nodes import resolve_node + + # Top candidate is an exact name match; a synonym match also counts as + # confident, so the chosen pick stays the name-matched, top-ranked one. + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=[ + {"_id": "@GENE_BTK", "name": "BTK", "biotype": "gene", + "db_id": "695", "db": "ncbi_gene", "match": "Matched on name <m>BTK</m>"}, + {"_id": "@GENE_TXK", "name": "TXK", "biotype": "gene", + "db_id": "7294", "db": "ncbi_gene", "match": "Matched on synonyms <m>BTKL</m>"}, + ], + is_reusable=True, + ) + + state = {"mentions": [EntityMention(text="BTK", suggested_type="gene")], + "warnings": []} + out = await resolve_node(state) + + assert out["resolved"]["BTK"].accession == "@GENE_BTK" + assert "BTK" not in out["unresolved"] diff --git a/crossbar_llm/pubtator3_tools/tests/test_tools_split.py b/crossbar_llm/pubtator3_tools/tests/test_tools_split.py new file mode 100644 index 0000000..d31841d --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/test_tools_split.py @@ -0,0 +1,189 @@ +import re + +import pytest + +from crossbar_llm.pubtator3_tools.tools import ( + AutocompleteOutput, + ExportPassagesOutput, + FindPartnersOutput, + SearchArticlesOutput, + pubtator3_autocomplete, + pubtator3_export_passages, + pubtator3_find_partners, + pubtator3_search_articles, +) +from crossbar_llm.pubtator3_tools import client + +BASE_URL = client.BASE_URL + + +def _url_pattern(path: str) -> re.Pattern: + return re.compile(rf"{re.escape(BASE_URL)}{re.escape(path)}.*") + + +def test_autocomplete_schema_has_expected_fields(): + assert set(pubtator3_autocomplete.args.keys()) == {"query", "concept", "limit"} + + +def test_find_partners_schema_has_expected_fields(): + assert set(pubtator3_find_partners.args.keys()) == {"e1_accession", "relation", "e2_type"} + + +def test_search_articles_schema_has_expected_fields(): + assert set(pubtator3_search_articles.args.keys()) == {"text_query", "page"} + + +def test_export_passages_schema_has_expected_fields(): + assert set(pubtator3_export_passages.args.keys()) == {"pmids", "full_text"} + + +async def test_autocomplete_happy_path(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/entity/autocomplete/"), + json=fx("pubtator3_autocomplete_example"), + ) + + out = await pubtator3_autocomplete.ainvoke({"query": "JAK1", "concept": "gene"}) + + assert isinstance(out, AutocompleteOutput) + assert out.error is None + assert len(out.candidates) == 5 + assert out.candidates[0].accession.startswith("@GENE_") + + +async def test_find_partners_happy_path(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/relations"), + json=fx("pubtator3_relations_example"), + ) + + out = await pubtator3_find_partners.ainvoke( + {"e1_accession": "@GENE_JAK1", "relation": "negative_correlate", "e2_type": "Chemical"} + ) + + assert isinstance(out, FindPartnersOutput) + assert out.error is None + assert len(out.partners) > 0 + assert all(p.publications > 0 for p in out.partners) + + +async def test_search_articles_happy_path(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + ) + + out = await pubtator3_search_articles.ainvoke( + {"text_query": "relations:negative_correlate|@CHEMICAL_ruxolitinib|@GENE_JAK1"} + ) + + assert isinstance(out, SearchArticlesOutput) + assert out.error is None + assert out.total > 0 + assert len(out.hits) > 0 + assert all(h.pmid > 0 for h in out.hits) + + +async def test_export_passages_happy_path(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + ) + + out = await pubtator3_export_passages.ainvoke({"pmids": [33849366]}) + + assert isinstance(out, ExportPassagesOutput) + assert out.error is None + assert len(out.documents) > 0 + doc = out.documents[0] + assert doc.pmid == 33849366 + assert any(p.section == "abstract" for p in doc.passages) + + +async def test_autocomplete_never_raises_on_persistent_429(httpx_mock, monkeypatch): + monkeypatch.setattr(client, "RETRY_429_BACKOFF_S", 0.0) + httpx_mock.add_response(url=_url_pattern("/entity/autocomplete/"), status_code=429) + httpx_mock.add_response(url=_url_pattern("/entity/autocomplete/"), status_code=429) + + out = await pubtator3_autocomplete.ainvoke({"query": "JAK1"}) + assert out.candidates == [] + assert out.error is not None + + +async def test_find_partners_never_raises_on_persistent_429(httpx_mock, monkeypatch): + monkeypatch.setattr(client, "RETRY_429_BACKOFF_S", 0.0) + httpx_mock.add_response(url=_url_pattern("/relations"), status_code=429) + httpx_mock.add_response(url=_url_pattern("/relations"), status_code=429) + + out = await pubtator3_find_partners.ainvoke( + {"e1_accession": "@GENE_JAK1", "relation": "treat", "e2_type": "Disease"} + ) + assert out.partners == [] + assert out.error is not None + + +async def test_search_articles_never_raises_on_persistent_429(httpx_mock, monkeypatch): + monkeypatch.setattr(client, "RETRY_429_BACKOFF_S", 0.0) + httpx_mock.add_response(url=_url_pattern("/search/"), status_code=429) + httpx_mock.add_response(url=_url_pattern("/search/"), status_code=429) + + out = await pubtator3_search_articles.ainvoke({"text_query": "anything"}) + assert out.hits == [] + assert out.total == 0 + assert out.error is not None + + +async def test_export_passages_never_raises_on_persistent_429(httpx_mock, monkeypatch): + monkeypatch.setattr(client, "RETRY_429_BACKOFF_S", 0.0) + httpx_mock.add_response(url=_url_pattern("/publications/export/biocjson"), status_code=429) + httpx_mock.add_response(url=_url_pattern("/publications/export/biocjson"), status_code=429) + + out = await pubtator3_export_passages.ainvoke({"pmids": [12345]}) + assert out.documents == [] + assert out.error is not None + + +async def test_find_partners_normalizes_relation_alias(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/relations"), + json=fx("pubtator3_relations_example"), + ) + # Schema's Literal rejects the alias, so call the coroutine directly. + out = await pubtator3_find_partners.coroutine( + e1_accession="@GENE_JAK1", + relation="negatively_correlate", + e2_type="Chemical", + ) + assert out.error is None + assert len(out.partners) > 0 + + +async def test_export_passages_rejects_empty_pmids(): + with pytest.raises(Exception): + await pubtator3_export_passages.ainvoke({"pmids": []}) + + +async def test_export_passages_batches_concurrently_above_cap(httpx_mock, fx): + # EXPORT_PMID_BATCH is 100; 250 PMIDs must split into 3 chunks. + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + + pmids = list(range(1, 251)) + out = await pubtator3_export_passages.ainvoke({"pmids": pmids}) + + assert out.error is None + # Fixture has N docs; we should get 3 × N because the same fixture + # is returned for each of the 3 batches. + fixture_doc_count = len(fx("pubtator3_export_example")["PubTator3"]) + assert len(out.documents) == 3 * fixture_doc_count + + # Confirm exactly 3 HTTP calls were made and the pmids params chunk correctly. + requests = httpx_mock.get_requests(url=_url_pattern("/publications/export/biocjson")) + assert len(requests) == 3 + pmid_chunks = [ + req.url.params["pmids"].split(",") for req in requests + ] + assert [len(c) for c in pmid_chunks] == [100, 100, 50] diff --git a/crossbar_llm/pubtator3_tools/tools.py b/crossbar_llm/pubtator3_tools/tools.py new file mode 100644 index 0000000..56cfe85 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tools.py @@ -0,0 +1,293 @@ +"""LangChain `@tool` wrappers around the PubTator3 client.""" +from typing import Literal + +from langchain_core.tools import tool +from pydantic import BaseModel, Field + +from crossbar_llm.pubtator3_tools import client as _client_mod +from crossbar_llm.pubtator3_tools.client import ( + EntityCandidate, + PubTator3Document, + RelatedEntity, + SearchHit, +) + + +_RELATION_TYPES = Literal[ + "treat", + "cause", + "associate", + "prevent", + "positive_correlate", + "negative_correlate", + "compare", + "cotreat", + "inhibit", + "stimulate", + "interact", + "drug_interact", +] + +_ENTITY_TYPES = Literal[ + "Gene", + "Chemical", + "Disease", + "Species", + "Variant", + "CellLine", +] + +_CONCEPT_TYPES = Literal[ + "gene", + "chemical", + "disease", + "species", + "variant", + "cellline", +] + +_RELATION_ALIASES = { + "negatively_correlate": "negative_correlate", + "positively_correlate": "positive_correlate", +} + + +class AutocompleteInput(BaseModel): + query: str = Field( + ..., + description=( + "Free-text biomedical entity name to resolve, e.g. 'JAK1', " + "'metformin', 'Alzheimer disease'. Do NOT pass an already-resolved " + "accession (a string starting with '@') — those are the OUTPUT of " + "this tool, not the input." + ), + ) + concept: _CONCEPT_TYPES | None = Field( + None, + description=( + "Optional biotype hint that narrows the candidate set: one of " + "gene, chemical, disease, species, variant, cellline. Use this " + "when the user's question makes the type unambiguous (e.g. " + "'metformin' is clearly a chemical). Omit when uncertain." + ), + ) + limit: int = Field( + 5, + ge=1, + le=20, + description="Maximum candidates to return (1–20). Default 5 is plenty for picking the top match.", + ) + + +class AutocompleteOutput(BaseModel): + candidates: list[EntityCandidate] = [] + error: str | None = None + + +@tool("pubtator3_autocomplete", args_schema=AutocompleteInput) +async def pubtator3_autocomplete( + query: str, + concept: _CONCEPT_TYPES | None = None, + limit: int = 5, +) -> AutocompleteOutput: + """Resolve a free-text biomedical entity name to its PubTator3 accession. + + Wraps GET /entity/autocomplete/. Given a query such as "JAK1", + "metformin", or "Alzheimer disease", returns ranked candidate + entities — each carrying the PubTator3 accession (`@TYPE_Name`), + the underlying database identifier and source (e.g. NCBI Gene ID, + MeSH ID), the human-readable name, and the biotype. + + An empty candidates list means the entity is not in PubTator3's + vocabulary. The tool never raises — transport / parse errors are + captured in `output.error`. + """ + try: + candidates = await _client_mod.autocomplete(query, concept=concept, limit=limit) + return AutocompleteOutput(candidates=candidates) + except Exception as e: + return AutocompleteOutput(error=f"{type(e).__name__}: {e}") + + +class FindPartnersInput(BaseModel): + e1_accession: str = Field( + ..., + description=( + "PubTator3 accession of the known entity, e.g. '@GENE_JAK1', " + "'@DISEASE_Alzheimer_Disease'. Must start with '@'. Get this " + "from `pubtator3_autocomplete` first if you only have free text." + ), + ) + relation: _RELATION_TYPES = Field( + ..., + description=( + "PubTator3 relation type. Pick the value whose semantics match the user's verb:\n" + "- treat : a chemical/drug treats a disease.\n" + "- cause : positive correlation; chemical-induced diseases and variant-caused genetic diseases.\n" + "- associate : generic association with no specific direction.\n" + "- prevent : negative correlation; includes variant-disease.\n" + "- positive_correlate : same-direction co-movement; chemical-gene, co-expression.\n" + "- negative_correlate : opposite-direction co-movement; chemical-gene, co-expression.\n" + "- compare : comparing the effect of two chemicals/drugs.\n" + "- cotreat : two or more chemicals/drugs administered together.\n" + "- inhibit : negative correlation; disease-gene, chemical-variant.\n" + "- stimulate : positive correlation; disease-gene, disease-variant.\n" + "- interact : physical interaction (e.g. protein binding); gene-gene, gene-chemical, chemical-variant.\n" + "- drug_interact : pharmacodynamic interaction between two chemicals producing side effects." + ), + ) + e2_type: _ENTITY_TYPES = Field( + ..., + description=( + "Type of partner to discover. PubTator3 only annotates these six " + "entity types, each grounded in a specific terminology:\n" + "- Gene : NCBI Gene IDs.\n" + "- Disease : MeSH (Medical Subject Headings).\n" + "- Chemical : MeSH (Medical Subject Headings).\n" + "- Variant : dbSNP IDs when available, otherwise HGVS format.\n" + "- Species : NCBI Taxonomy IDs.\n" + "- CellLine : Cellosaurus IDs.\n" + "Note the capital-cased values — they differ from the lowercase " + "`concept` used by autocomplete." + ), + ) + + +class FindPartnersOutput(BaseModel): + partners: list[RelatedEntity] = [] + error: str | None = None + + +@tool("pubtator3_find_partners", args_schema=FindPartnersInput) +async def pubtator3_find_partners( + e1_accession: str, + relation: _RELATION_TYPES, + e2_type: _ENTITY_TYPES, +) -> FindPartnersOutput: + """Discover partner entities related to a known entity by a specific relation. + + Wraps GET /relations. Given a known entity accession (e.g. + `@GENE_JAK1`), a relation type from the PubTator3 vocabulary, and + a target partner type, returns ranked `RelatedEntity` rows — each + with source, target, relation type, and the number of supporting + publications — sorted descending by publication count. + + This endpoint reveals WHICH partners exist; article PMIDs come + from `pubtator3_search_articles` with a + `relations:<rel>|<source>|<target>` expression. The tool never + raises — errors are captured in `output.error`. + """ + if relation in _RELATION_ALIASES: + relation = _RELATION_ALIASES[relation] + try: + partners = await _client_mod.find_related( + e1_accession, relation=relation, e2_type=e2_type + ) + return FindPartnersOutput(partners=partners) + except Exception as e: + return FindPartnersOutput(error=f"{type(e).__name__}: {e}") + + +class SearchArticlesInput(BaseModel): + text_query: str = Field( + ..., + description=( + "PubTator3 search expression. Three valid forms:\n" + " 1. Relation: 'relations:<rel>|<source>|<target>' — e.g.\n" + " 'relations:treat|@CHEMICAL_Metformin|@DISEASE_Diabetes_Mellitus_Type_2'.\n" + " 2. Boolean entity: '@GENE_JAK1 AND @CHEMICAL_ruxolitinib' — uses\n" + " resolved accessions joined by AND/OR.\n" + " 3. Keyword: 'metformin liver toxicity' — free text fallback.\n" + "Prefer forms 1 and 2 when possible; they're more precise." + ), + ) + page: int = Field( + 1, + ge=1, + description="1-indexed page number for paginated results.", + ) + + +class SearchArticlesOutput(BaseModel): + hits: list[SearchHit] = [] + total: int = 0 + error: str | None = None + + +@tool("pubtator3_search_articles", args_schema=SearchArticlesInput) +async def pubtator3_search_articles( + text_query: str, + page: int = 1, +) -> SearchArticlesOutput: + """Search PubTator3 for articles matching a query expression. + + Wraps GET /search/. Accepts three query forms: + 1. Relation expression `relations:<rel>|<source_accession>|<target_accession>` + — restricts results to PMIDs in which BioREx extracted the relation. + 2. Boolean accession query `@GENE_JAK1 AND @CHEMICAL_ruxolitinib` + — restricts results to PMIDs co-mentioning the entities. + 3. Free-text keyword query `metformin liver toxicity` — fallback + when forms 1 and 2 do not apply. + + Returns ranked hits (PMID, title, journal, date, score, snippet) + plus the total match count. The tool never raises — errors are + captured in `output.error`. + """ + try: + hits, total = await _client_mod.search(text_query, page=page) + return SearchArticlesOutput(hits=hits, total=total) + except Exception as e: + return SearchArticlesOutput(error=f"{type(e).__name__}: {e}") + + +class ExportPassagesInput(BaseModel): + pmids: list[int] = Field( + ..., + min_length=1, + max_length=500, + description=( + "PubMed IDs to fetch. Get these from `pubtator3_search_articles`. " + "More than 100 PMIDs are auto-batched into multiple requests by " + "the client (each request hits the API rate limit individually). " + "Keep this list focused — passing 500 random PMIDs is wasteful." + ), + ) + full_text: bool = Field( + True, + description=( + "When True, fetch full body text for PMC Open Access articles. " + "When False, only title + abstract. Closed-access articles " + "always return title + abstract regardless of this flag." + ), + ) + + +class ExportPassagesOutput(BaseModel): + documents: list[PubTator3Document] = [] + error: str | None = None + + +@tool("pubtator3_export_passages", args_schema=ExportPassagesInput) +async def pubtator3_export_passages( + pmids: list[int], + full_text: bool = True, +) -> ExportPassagesOutput: + """Fetch BioC JSON passages and BioREx relations for a set of PMIDs. + + Wraps GET /publications/export/biocjson. Returns a list of + `PubTator3Document` — each containing passages (title, abstract, + full body sections when the article is PMC Open Access) with + offset-anchored entity annotations, plus document-level BioREx + relations carrying role accessions, database identifiers, and + confidence scores. + + PMIDs are auto-batched in groups of 100 (the upstream cap); each + batch counts against the 3 req/s rate limit. Up to 500 PMIDs per + call. The tool never raises — errors are captured in `output.error`. + """ + try: + docs = await _client_mod.export_biocjson(pmids, full=full_text) + return ExportPassagesOutput(documents=docs) + except Exception as e: + return ExportPassagesOutput(error=f"{type(e).__name__}: {e}") + diff --git a/crossbar_llm/pubtator3_tools/usage.py b/crossbar_llm/pubtator3_tools/usage.py new file mode 100644 index 0000000..0d0c023 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/usage.py @@ -0,0 +1,90 @@ +"""Token-usage capture / logging helpers for the PubTator3 graph. + +Both helpers wrap a graph invocation in `get_usage_metadata_callback`, +which aggregates `usage_metadata` from every chat-model call made under +the context — router, synthesizer, depth evaluator, JSON fallbacks. +Use `ainvoke_with_usage_logging` for fire-and-forget log lines and +`ainvoke_with_usage_capture` when the caller needs the totals +programmatically (benchmark runner, etc.). +""" +from __future__ import annotations + +import logging + +from langchain_core.callbacks import get_usage_metadata_callback + + +_usage_logger = logging.getLogger("crossbar_llm.pubtator3.usage") + + +def _flatten_usage(usage_metadata: dict) -> dict: + """Collapse {model: {input,output,total,...}} into one totals dict. + + Sums across models so a single run that touches router + synthesizer + + evaluator (possibly different model versions for each) reports one + consolidated number per field. Pulls out provider sub-buckets we care + about (reasoning, cache_read) when present. + """ + flat = { + "input_tokens": 0, + "output_tokens": 0, + "reasoning_tokens": 0, + "cache_read_tokens": 0, + "total_tokens": 0, + "by_model": {}, + } + for model, usage in usage_metadata.items(): + in_details = usage.get("input_token_details") or {} + out_details = usage.get("output_token_details") or {} + flat["input_tokens"] += usage.get("input_tokens") or 0 + flat["output_tokens"] += usage.get("output_tokens") or 0 + flat["total_tokens"] += usage.get("total_tokens") or 0 + flat["reasoning_tokens"] += out_details.get("reasoning") or 0 + flat["cache_read_tokens"] += in_details.get("cache_read") or 0 + flat["by_model"][model] = dict(usage) + return flat + + +async def ainvoke_with_usage_capture(graph, state: dict, **kwargs) -> tuple[dict, dict]: + """Like `ainvoke_with_usage_logging` but returns the usage instead of logging. + + Returns `(state, usage)` where `usage` is the flat-totals dict produced by + `_flatten_usage` — handy for benchmarks that need to record tokens per + question alongside the answer. + """ + with get_usage_metadata_callback() as cb: + result = await graph.ainvoke(state, **kwargs) + return result, _flatten_usage(cb.usage_metadata) + + +async def ainvoke_with_usage_logging(graph, state: dict, **kwargs) -> dict: + """Invoke `graph` and log aggregated LLM token usage for the run. + + Totals are logged per model name at INFO level on + `crossbar_llm.pubtator3.usage`. + """ + with get_usage_metadata_callback() as cb: + result = await graph.ainvoke(state, **kwargs) + for model, usage in cb.usage_metadata.items(): + in_details = usage.get("input_token_details") or {} + out_details = usage.get("output_token_details") or {} + reasoning = out_details.get("reasoning") + cache_read = in_details.get("cache_read") + _usage_logger.info( + "pubtator3 llm usage model=%s input=%s output=%s reasoning=%s " + "cache_read=%s total=%s", + model, + usage.get("input_tokens"), + usage.get("output_tokens"), + reasoning, + cache_read, + usage.get("total_tokens"), + ) + return result + + +__all__ = [ + "_flatten_usage", + "ainvoke_with_usage_capture", + "ainvoke_with_usage_logging", +] From d0605fcf18bdd5e578b6f8514797abeec9299714 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ahmet=20O=C4=9Fuzhan=20K=C3=B6k=C3=BCl=C3=BC?= <oguzhankokulu@gmail.com> Date: Sat, 15 Aug 2026 23:55:06 +0300 Subject: [PATCH 2/8] =?UTF-8?q?fix:=20address=20review=20=E2=80=94=20logge?= =?UTF-8?q?r=20namespacing,=20router=20fallback,=20stale=20=5F=5Fall=5F=5F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit paperclip_tools/usage.py was a verbatim copy of the PubTator3 helper, so both packages logged to `crossbar_llm.pubtator3.usage` with a "pubtator3" message prefix. Paperclip now logs under its own namespace, so the two can be filtered and configured independently. The router's error fallback pinned source="pmc". That was a leftover from when the MCP path required an explicit -s; REST made broad search the documented default. Narrowing the search on the failure path is the wrong direction, so it now falls back to broad. pubtator3_tools/nodes.py listed `_message_content_to_text` and `_extract_json_object` in __all__ without defining them, so `import *` raised AttributeError. Also corrects module paths in docstrings left over from the earlier `agents/` layout. Both behavioural fixes have regression tests. --- crossbar_llm/paperclip_tools/adapter.py | 2 +- crossbar_llm/paperclip_tools/agent.py | 5 ++++- .../paperclip_tools/tests/test_agent.py | 17 +++++++++++++++++ crossbar_llm/paperclip_tools/usage.py | 8 ++++---- crossbar_llm/pubtator3_tools/agent.py | 11 ++++++----- crossbar_llm/pubtator3_tools/nodes.py | 2 -- .../pubtator3_tools/tests/test_graph.py | 9 +++++++++ 7 files changed, 41 insertions(+), 13 deletions(-) diff --git a/crossbar_llm/paperclip_tools/adapter.py b/crossbar_llm/paperclip_tools/adapter.py index 9fec3e2..60440f7 100644 --- a/crossbar_llm/paperclip_tools/adapter.py +++ b/crossbar_llm/paperclip_tools/adapter.py @@ -22,7 +22,7 @@ (source, depth, fallback) live in the node layer. - **Per-event-loop singletons** for the MCP client + loaded tool, keyed in a `WeakKeyDictionary`, so `asyncio.run` in tests/scripts doesn't hit "event loop - is closed" (mirrors `pubtator3/client.py`). + is closed" (mirrors `pubtator3_tools/client.py`). - Command errors surface as `PaperclipError`; the tool layer converts those to never-raise error envelopes. REST failures surface as `PaperclipRestUnavailable` internally, caught by `_execute`/`search` to diff --git a/crossbar_llm/paperclip_tools/agent.py b/crossbar_llm/paperclip_tools/agent.py index cff73a6..9735888 100644 --- a/crossbar_llm/paperclip_tools/agent.py +++ b/crossbar_llm/paperclip_tools/agent.py @@ -234,7 +234,10 @@ async def router_node(state: PaperclipState) -> dict: ) decision = PaperclipRouterDecision( question_type="keyword_search", - source="pmc", + # Broad, matching the documented default. Narrowing to one + # corpus is a guess, and the failure path is the worst place + # to make one. + source=None, search_query=state["question"], map_question=state["question"], rationale=f"router error fallback: {e}", diff --git a/crossbar_llm/paperclip_tools/tests/test_agent.py b/crossbar_llm/paperclip_tools/tests/test_agent.py index 101decf..d9d5017 100644 --- a/crossbar_llm/paperclip_tools/tests/test_agent.py +++ b/crossbar_llm/paperclip_tools/tests/test_agent.py @@ -844,3 +844,20 @@ async def degenerate(state): assert out["final_answer"] == "Real answer.---" assert any("degenerate character run" in w for w in out["warnings"]) + + +async def test_router_failure_falls_back_to_broad_not_one_corpus(): + """The failure path must not narrow the search. + + `source=None` is the documented default; pinning a single corpus here + was a leftover from when MCP required an explicit `-s`. + """ + async def boom(state): + raise RuntimeError("router exploded") + + g = build_graph(router=boom, synthesizer=_synth, adapter=FakeAdapter()) + out = await g.ainvoke({"question": "What treats X?"}) + + assert out["question_type"] == "keyword_search" + assert out["source"] is None + assert any("router failed" in w for w in out["warnings"]) diff --git a/crossbar_llm/paperclip_tools/usage.py b/crossbar_llm/paperclip_tools/usage.py index 0d0c023..bcc9c2e 100644 --- a/crossbar_llm/paperclip_tools/usage.py +++ b/crossbar_llm/paperclip_tools/usage.py @@ -1,4 +1,4 @@ -"""Token-usage capture / logging helpers for the PubTator3 graph. +"""Token-usage capture / logging helpers for the Paperclip graph. Both helpers wrap a graph invocation in `get_usage_metadata_callback`, which aggregates `usage_metadata` from every chat-model call made under @@ -14,7 +14,7 @@ from langchain_core.callbacks import get_usage_metadata_callback -_usage_logger = logging.getLogger("crossbar_llm.pubtator3.usage") +_usage_logger = logging.getLogger("crossbar_llm.paperclip.usage") def _flatten_usage(usage_metadata: dict) -> dict: @@ -61,7 +61,7 @@ async def ainvoke_with_usage_logging(graph, state: dict, **kwargs) -> dict: """Invoke `graph` and log aggregated LLM token usage for the run. Totals are logged per model name at INFO level on - `crossbar_llm.pubtator3.usage`. + `crossbar_llm.paperclip.usage`. """ with get_usage_metadata_callback() as cb: result = await graph.ainvoke(state, **kwargs) @@ -71,7 +71,7 @@ async def ainvoke_with_usage_logging(graph, state: dict, **kwargs) -> dict: reasoning = out_details.get("reasoning") cache_read = in_details.get("cache_read") _usage_logger.info( - "pubtator3 llm usage model=%s input=%s output=%s reasoning=%s " + "paperclip llm usage model=%s input=%s output=%s reasoning=%s " "cache_read=%s total=%s", model, usage.get("input_tokens"), diff --git a/crossbar_llm/pubtator3_tools/agent.py b/crossbar_llm/pubtator3_tools/agent.py index 1beda69..67c4ab9 100644 --- a/crossbar_llm/pubtator3_tools/agent.py +++ b/crossbar_llm/pubtator3_tools/agent.py @@ -1,12 +1,13 @@ """LangGraph orchestrator for PubTator3 literature evidence. This module defines `build_graph`, which wires the standalone nodes from -`agents.nodes` together with three inline LLM-bound nodes (router, -synthesizer, depth evaluator) that close over the chat model and prompts. +`crossbar_llm.pubtator3_tools.nodes` together with three inline LLM-bound +nodes (router, synthesizer, depth evaluator) that close over the chat model +and prompts. -Pydantic schemas live in `agents.schemas`; token-usage helpers live in -`agents.usage`. The names are re-exported here for backwards compatibility -with code that imported them from this module directly. +Pydantic schemas live in `crossbar_llm.pubtator3_tools.schemas`; token-usage +helpers live in `crossbar_llm.pubtator3_tools.usage`. Both are re-exported +here so a caller can import the whole public surface from one module. """ from __future__ import annotations diff --git a/crossbar_llm/pubtator3_tools/nodes.py b/crossbar_llm/pubtator3_tools/nodes.py index d181b12..754da3a 100644 --- a/crossbar_llm/pubtator3_tools/nodes.py +++ b/crossbar_llm/pubtator3_tools/nodes.py @@ -341,8 +341,6 @@ async def export_node(state: PubTator3State, *, max_documents: int = 10) -> dict __all__ = [ "_add_warning", - "_message_content_to_text", - "_extract_json_object", "_ainvoke_structured_with_json_fallback", "_is_confident_match", "resolve_node", diff --git a/crossbar_llm/pubtator3_tools/tests/test_graph.py b/crossbar_llm/pubtator3_tools/tests/test_graph.py index b5d0d52..a060d6d 100644 --- a/crossbar_llm/pubtator3_tools/tests/test_graph.py +++ b/crossbar_llm/pubtator3_tools/tests/test_graph.py @@ -825,3 +825,12 @@ async def test_full_text_true_keeps_body_sections(httpx_mock, fx): assert "METHODS" in sections assert "TITLE" in sections assert "ABSTRACT" in sections + + +def test_nodes_star_import_does_not_break(): + """`__all__` must only name things the module actually defines.""" + import crossbar_llm.pubtator3_tools.nodes as nodes + + missing = [n for n in nodes.__all__ if not hasattr(nodes, n)] + assert not missing, f"__all__ names absent from module: {missing}" + exec("from crossbar_llm.pubtator3_tools.nodes import *", {}) From 7a375875c825d263b91477a00d6d7b5555105c8d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ahmet=20O=C4=9Fuzhan=20K=C3=B6k=C3=BCl=C3=BC?= <oguzhankokulu@gmail.com> Date: Thu, 27 Aug 2026 19:47:45 +0300 Subject: [PATCH 3/8] fix(pubtator3): stop silent evidence loss; declare missing dependencies Found while running the PubTator3 agent against the live API. Evidence loss: three paths dropped results without surfacing anything to the caller. Partner discovery lost a whole batch when one relation type was unknown, so a question could come back empty while the API had answers. The router's re-ask on a low-confidence match added a round trip without improving the decision, so it is dropped in favour of the free downgrade already in place. Tests cover each path. Dependencies: aiolimiter, pytest-asyncio and pytest-httpx were used but never declared. langchain-mcp-adapters was the same - it is imported lazily inside Paperclip's MCP fallback, so nothing failed until that fallback fired, which the offline suite never exercises. A fresh install would have hit ImportError the first time REST degraded to MCP. uv lock resolves all four as pure additions: no existing pin changes. --- crossbar_llm/pubtator3_tools/agent.py | 127 ++++++++-- crossbar_llm/pubtator3_tools/nodes.py | 60 ++++- .../pubtator3_tools/tests/test_graph.py | 217 +++++++++++++++++- pyproject.toml | 8 + requirements.txt | 2 + uv.lock | 209 +++++++++++++++++ 6 files changed, 596 insertions(+), 27 deletions(-) diff --git a/crossbar_llm/pubtator3_tools/agent.py b/crossbar_llm/pubtator3_tools/agent.py index 67c4ab9..f961fb8 100644 --- a/crossbar_llm/pubtator3_tools/agent.py +++ b/crossbar_llm/pubtator3_tools/agent.py @@ -26,6 +26,7 @@ _message_content_to_text, ) from crossbar_llm.pubtator3_tools.nodes import ( + _router_decision_gaps, _add_warning, export_node, partner_discovery_node, @@ -71,6 +72,8 @@ def build_graph( max_partners: int = 5, max_documents: int = 7, abstracts_only: bool = False, + max_evidence_chars: int = 120_000, + router_guard: bool = True, ): """Compile the PubTator3 LangGraph. @@ -85,6 +88,24 @@ def build_graph( False / None and the depth evaluator's full-text refinement path is short-circuited (no second pass). Use it when you want predictable token cost and don't need PMC body text. + + `max_evidence_chars` caps the assembled evidence block handed to the + synthesizer. `max_documents` bounds how many PAPERS we export, not how + much TEXT they carry — PubTator3 splits some abstracts into dozens of + passages, so a 10-document export has been observed at 587 passages, and + an unbounded evidence block overran a 131K-token provider limit outright + (a hard 400, not a degraded answer). The cap trims at a passage boundary + and tells the synthesizer that it happened. + + `router_guard` enforces the router's own contract: a model that names a + route but omits that route's required fields (e.g. partner discovery with + `relation` set but `mentions` empty) leaves the structured path with + nothing to resolve, so it limps to the keyword fallback while the state + still reports the structured route. The guard downgrades such a decision + to an explicit keyword_search, which costs no extra LLM call and keeps + `question_type` truthful about what actually ran. Set False to observe + the raw router output. Applies only to the LLM path — an injected + `router` is a test seam and is trusted as given. """ if (router is None or synthesizer is None) and chat_model is None: raise ValueError( @@ -93,35 +114,70 @@ def build_graph( if evaluator is None and chat_model is None: evaluator = _always_sufficient_evaluator + async def _invoke_router(state: PubTator3State): + prompt = ChatPromptTemplate.from_messages([ + SystemMessagePromptTemplate.from_template(ROUTER_SYSTEM_PROMPT), + MessagesPlaceholder("chat_history", optional=True), + HumanMessagePromptTemplate.from_template("User question: {question}"), + ]) + return await _ainvoke_structured_with_json_fallback( + chat_model=chat_model, + prompt=prompt, + schema=RouterDecision, + values={ + "question": state["question"], + "chat_history": state.get("chat_history", []), + }, + json_instruction=( + "The previous instruction defines the exact routing schema. " + "Return ONLY a valid JSON object for that schema. Do not use " + "Markdown, prose, tool calls, or extra keys." + ), + ) + async def router_node(state: PubTator3State) -> dict: warnings = list(state.get("warnings", [])) try: if router is not None: decision = await router(state["question"]) else: - prompt = ChatPromptTemplate.from_messages([ - SystemMessagePromptTemplate.from_template(ROUTER_SYSTEM_PROMPT), - MessagesPlaceholder("chat_history", optional=True), - HumanMessagePromptTemplate.from_template("User question: {question}"), - ]) - decision, used_json_fallback = await _ainvoke_structured_with_json_fallback( - chat_model=chat_model, - prompt=prompt, - schema=RouterDecision, - values={ - "question": state["question"], - "chat_history": state.get("chat_history", []), - }, - json_instruction=( - "The previous instruction defines the exact routing schema. " - "Return ONLY a valid JSON object for that schema. Do not use " - "Markdown, prose, tool calls, or extra keys." - ), - ) + decision, used_json_fallback = await _invoke_router(state) if used_json_fallback: warnings.append( "router structured-output unavailable; used JSON fallback." ) + + # Contract guard. Weaker models routinely fill the flat fields + # (`relation`) while dropping the nested list (`mentions`), + # which leaves resolve/partner-discovery with no anchor: the + # structured query is never built and the run silently becomes + # a keyword search while still reporting the structured route. + # + # We do NOT re-ask. Benchmarking showed a re-ask costs a full + # extra router call on exactly the weak models that then fail + # it anyway (9 fires, 1 repaired, 8 downgraded; +54% tokens, + # no score change), while strong models never trip the guard + # and so never paid for it. Downgrading is free and lands on + # the same retrieval the fallback would have reached. + if router_guard: + gaps = _router_decision_gaps(decision) + if gaps: + rejected = decision.question_type + decision = RouterDecision( + question_type="keyword_search", + keyword_query=( + (decision.keyword_query or "").strip() + or state["question"] + ), + rationale=( + "downgraded: router could not supply the fields " + f"required by {rejected}." + ), + ) + warnings.append( + f"router decision incomplete for {rejected} " + f"({'; '.join(gaps)}); downgraded to keyword_search." + ) except Exception as e: # Provider-side schema validation (e.g. invented relation value) # would otherwise crash the run. Degrade to keyword_search so the @@ -154,17 +210,40 @@ async def synthesize_node(state: PubTator3State) -> dict: if synthesizer is not None: answer = await synthesizer(state) else: - ctx_lines: list[str] = [] + candidate_lines: list[str] = [] for p in state.get("passages", []): - ctx_lines.append(f"[PMID:{p.pmid}] ({p.section}) {p.text}") + candidate_lines.append(f"[PMID:{p.pmid}] ({p.section}) {p.text}") for r in state.get("document_relations", []): - ctx_lines.append( + candidate_lines.append( f"[PMID:{r.pmid}] relation={r.type} " f"{r.role1_accession or '?'}->{r.role2_accession or '?'} " f"score={r.score:.2f}" ) + + # Trim to the evidence budget at a whole-line boundary. Passages + # come first, so what gets dropped is the tail of the evidence + # rather than an arbitrary mid-passage cut. + ctx_lines: list[str] = [] + used = 0 + dropped = 0 + for line in candidate_lines: + cost = len(line) + 1 # +1 for the join newline + if used + cost > max_evidence_chars and ctx_lines: + dropped = len(candidate_lines) - len(ctx_lines) + break + ctx_lines.append(line) + used += cost + evidence = "\n".join(ctx_lines) if ctx_lines else "(no passages found)" + truncation_note = "" + if dropped: + truncation_note = ( + f"\n\nNOTE: {dropped} further evidence line(s) were omitted " + f"to stay within the context budget. Answer from what is " + f"shown and do not claim the evidence set is exhaustive." + ) + # Full-text requested but only title/abstract sections came back # means none of the retrieved papers are PMC Open Access. Surface # that caveat so the synthesizer mentions it instead of pretending @@ -174,9 +253,9 @@ async def synthesize_node(state: PubTator3State) -> dict: p.section not in ("title", "abstract") for p in state.get("passages") or [] ) - availability_note = "" + availability_note = truncation_note if full_text_requested and state.get("passages") and not body_sections_present: - availability_note = ( + availability_note += ( "\n\nNOTE: full paper body text was requested but none of " "the retrieved PMIDs are PMC Open Access — only titles " "and abstracts are available. End your paragraph with an " diff --git a/crossbar_llm/pubtator3_tools/nodes.py b/crossbar_llm/pubtator3_tools/nodes.py index 754da3a..cb4c887 100644 --- a/crossbar_llm/pubtator3_tools/nodes.py +++ b/crossbar_llm/pubtator3_tools/nodes.py @@ -208,6 +208,19 @@ async def search_node(state: PubTator3State) -> dict: ) elif qtype == "keyword_search": kq = (state.get("keyword_query") or "").strip() + if not kq: + # The router chose the free-text route but left `keyword_query` + # null — a structured-output slip, or a JSON fallback that only + # recovered `question_type`. Unlike the structured routes there is + # no accession or relation expression to search instead, so an + # empty `keyword_query` here means issuing NO query at all. The + # question text is always a valid PubMed expression; degrade to it. + kq = (state.get("question") or "").strip() + if kq: + warnings.append( + "router selected keyword_search without a keyword_query; " + "using the question text verbatim." + ) if kq: queries.append(kq) else: @@ -242,8 +255,19 @@ async def _one(q: str): # keyword_search route uses). If the router didn't fill `keyword_query` # (e.g. router error fallback set only question_type), use the user's # question verbatim as the catastrophic-failure tier. - if not pmids and qtype in ("relation_partner_discovery", "relation_known_pair", "single_node"): + # + # keyword_search is included here too: when its distilled query returns + # nothing, the broader raw question is the one remaining tier (the + # `in queries` check below keeps us from re-issuing the same string). + if not pmids and qtype in ( + "relation_partner_discovery", + "relation_known_pair", + "single_node", + "keyword_search", + ): fallback_query = (state.get("keyword_query") or "").strip() or (state.get("question") or "").strip() + if fallback_query in queries: + fallback_query = (state.get("question") or "").strip() if fallback_query and fallback_query not in queries: warnings.append( f"structured query returned 0 PMIDs; falling back to " @@ -339,8 +363,42 @@ async def export_node(state: PubTator3State, *, max_documents: int = 10) -> dict } +def _router_decision_gaps(decision) -> list[str]: + """Required fields the decision omits for the route it selected. + + The router is free to pick any route, but each route's downstream node + needs specific fields: no anchor mention means resolve has nothing to + look up, no `e2_type` means partner discovery has no target type. A + decision missing them is not a routing opinion, it is an unusable answer. + """ + qt = decision.question_type + gaps: list[str] = [] + + if qt == "single_node": + if len(decision.mentions) != 1: + gaps.append("`mentions` must hold exactly one entity") + elif qt == "relation_known_pair": + if len(decision.mentions) != 2: + gaps.append("`mentions` must hold exactly two entities, [e1, e2]") + if not decision.relation: + gaps.append("`relation` is required") + elif qt == "relation_partner_discovery": + if len(decision.mentions) != 1: + gaps.append("`mentions` must hold exactly one anchor entity") + if not decision.relation: + gaps.append("`relation` is required") + if not decision.e2_type: + gaps.append("`e2_type` is required") + elif qt == "keyword_search": + if not (decision.keyword_query or "").strip(): + gaps.append("`keyword_query` is required") + + return gaps + + __all__ = [ "_add_warning", + "_router_decision_gaps", "_ainvoke_structured_with_json_fallback", "_is_confident_match", "resolve_node", diff --git a/crossbar_llm/pubtator3_tools/tests/test_graph.py b/crossbar_llm/pubtator3_tools/tests/test_graph.py index a060d6d..8ceddca 100644 --- a/crossbar_llm/pubtator3_tools/tests/test_graph.py +++ b/crossbar_llm/pubtator3_tools/tests/test_graph.py @@ -6,6 +6,7 @@ from langchain_core.runnables import RunnableLambda from crossbar_llm.pubtator3_tools.agent import ( + _router_decision_gaps, DepthEvaluation, EntityMention, RouterDecision, @@ -334,7 +335,20 @@ async def test_keyword_search_flow_skips_resolve_and_partner_discovery(httpx_moc assert final["final_answer"] is not None -async def test_keyword_search_with_empty_query_warns_and_still_synthesizes(httpx_mock): +async def test_keyword_search_with_empty_query_falls_back_to_question(httpx_mock): + """An empty `keyword_query` must not mean "issue no query at all". + + keyword_search is the one route with no accession or relation expression + to search instead, so a router slip that leaves `keyword_query` null used + to send the question straight to synthesis with zero evidence. The + question text is a valid PubMed expression; we search it verbatim. + """ + httpx_mock.add_response( + url=_url_pattern("/search/"), + json={"results": [], "count": 0}, + is_reusable=True, + ) + decision = RouterDecision( question_type="keyword_search", keyword_query="", @@ -347,8 +361,28 @@ async def test_keyword_search_with_empty_query_warns_and_still_synthesizes(httpx final = await graph.ainvoke({"question": "vague", "warnings": []}) assert final["question_type"] == "keyword_search" + assert final["queries_used"] == ["vague"] + assert any("without a keyword_query" in w for w in final.get("warnings", [])) + assert final["final_answer"] is not None + + +async def test_keyword_search_with_no_query_and_no_question_warns(httpx_mock): + """Nothing to search at all — the original warning still fires.""" + decision = RouterDecision( + question_type="keyword_search", + keyword_query="", + ) + graph = build_graph( + router=_make_fake_router(decision), + synthesizer=_fake_synth, + ) + + final = await graph.ainvoke({"question": "", "warnings": []}) + assert final["queries_used"] == [] - assert any("keyword_query" in w for w in final.get("warnings", [])) + assert any( + "needs a non-empty keyword_query" in w for w in final.get("warnings", []) + ) assert final["final_answer"] is not None @@ -834,3 +868,182 @@ def test_nodes_star_import_does_not_break(): missing = [n for n in nodes.__all__ if not hasattr(nodes, n)] assert not missing, f"__all__ names absent from module: {missing}" exec("from crossbar_llm.pubtator3_tools.nodes import *", {}) + +def _capturing_chat_model(seen: list[str]): + """A chat model stand-in that records the rendered synthesis prompt.""" + + async def _call(prompt_value): + seen.append(prompt_value.to_string()) + return AIMessage(content="synthesized answer") + + model = RunnableLambda(_call) + # build_graph only reaches for structured output on the router/evaluator + # paths, both bypassed here; this keeps the duck-type complete. + model.with_structured_output = lambda schema, **kwargs: RunnableLambda( + lambda _: None + ) + return model + + +async def test_evidence_block_is_capped_and_truncation_is_disclosed(httpx_mock, fx): + """max_documents bounds papers, not text — the evidence block needs its own cap. + + Without it an export whose abstracts split into hundreds of passages built + a synthesis prompt that overran the provider's context limit and returned a + hard 400 instead of an answer. + """ + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + seen: list[str] = [] + + decision = RouterDecision( + question_type="keyword_search", + keyword_query="imatinib side effects", + ) + graph = build_graph( + chat_model=_capturing_chat_model(seen), + router=_make_fake_router(decision), + abstracts_only=True, + max_evidence_chars=200, + ) + + final = await graph.ainvoke( + {"question": "What are the side effects of imatinib?", "warnings": []} + ) + + assert final["final_answer"] == "synthesized answer" + assert len(seen) == 1 + prompt_text = seen[0] + assert "further evidence line(s) were omitted" in prompt_text + # The cap trims at a line boundary, so at least one passage survives. + assert "[PMID:" in prompt_text + + +async def test_evidence_block_is_not_truncated_under_a_generous_budget(httpx_mock, fx): + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + seen: list[str] = [] + + decision = RouterDecision( + question_type="keyword_search", + keyword_query="imatinib side effects", + ) + graph = build_graph( + chat_model=_capturing_chat_model(seen), + router=_make_fake_router(decision), + abstracts_only=True, + ) + + await graph.ainvoke( + {"question": "What are the side effects of imatinib?", "warnings": []} + ) + + assert "further evidence line(s) were omitted" not in seen[0] + +class _ScriptedRouterChatModel: + """Hands back a scripted RouterDecision per structured-output call.""" + + def __init__(self, decisions: list[RouterDecision]): + self.decisions = list(decisions) + self.calls = 0 + + def with_structured_output(self, schema, **kwargs): + async def _next(_values): + self.calls += 1 + return self.decisions.pop(0) + + return RunnableLambda(_next) + + +def _incomplete_partner_discovery() -> RouterDecision: + """What weak models actually emit: the flat field, not the nested list.""" + return RouterDecision( + question_type="relation_partner_discovery", + mentions=[], + relation="inhibit", + e2_type=None, + ) + + +def _search_and_export_mocks(httpx_mock, fx) -> None: + httpx_mock.add_response( + url=_url_pattern("/search/"), + json=fx("pubtator3_search_example"), + is_reusable=True, + ) + httpx_mock.add_response( + url=_url_pattern("/publications/export/biocjson"), + json=fx("pubtator3_export_example"), + is_reusable=True, + ) + + +def test_router_gaps_flags_each_routes_required_fields(): + assert _router_decision_gaps(_incomplete_partner_discovery()) == [ + "`mentions` must hold exactly one anchor entity", + "`e2_type` is required", + ] + assert _router_decision_gaps( + RouterDecision(question_type="keyword_search", keyword_query=" ") + ) == ["`keyword_query` is required"] + # A complete decision has no gaps. + assert _router_decision_gaps( + RouterDecision( + question_type="relation_partner_discovery", + mentions=[EntityMention(text="BTK", suggested_type="gene")], + relation="inhibit", + e2_type="Chemical", + ) + ) == [] + + +async def test_router_downgrades_an_incomplete_decision(httpx_mock, fx): + _search_and_export_mocks(httpx_mock, fx) + chat_model = _ScriptedRouterChatModel([_incomplete_partner_discovery()]) + + graph = build_graph( + chat_model=chat_model, + synthesizer=_fake_synth, + abstracts_only=True, + ) + final = await graph.ainvoke({"question": "BTK inhibitors?", "warnings": []}) + + # Exactly one router call: the guard costs no extra LLM round trip. + assert chat_model.calls == 1 + assert final["question_type"] == "keyword_search" + assert final["keyword_query"] == "BTK inhibitors?" + assert any("downgraded to keyword_search" in w for w in final["warnings"]) + + +async def test_router_guard_can_be_disabled(httpx_mock, fx): + _search_and_export_mocks(httpx_mock, fx) + chat_model = _ScriptedRouterChatModel([_incomplete_partner_discovery()]) + + graph = build_graph( + chat_model=chat_model, + synthesizer=_fake_synth, + abstracts_only=True, + router_guard=False, + ) + final = await graph.ainvoke({"question": "BTK inhibitors?", "warnings": []}) + + assert chat_model.calls == 1 + assert final["question_type"] == "relation_partner_discovery" + assert not any("downgraded" in w for w in final["warnings"]) + diff --git a/pyproject.toml b/pyproject.toml index e43d60e..2932430 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -7,6 +7,7 @@ name = "crossbar-llm" version = "0.1.0" requires-python = "==3.12.11" dependencies = [ + "aiolimiter>=1.2.1", "cyver==2.0.2", "fastapi[standard]==0.138.2", "httpx==0.28.1", @@ -16,6 +17,7 @@ dependencies = [ "langchain-core==1.5.0", "langchain-google-genai==4.2.6", "langchain-groq==1.1.3", + "langchain-mcp-adapters>=0.3.2", "langchain-openai==1.4.0", "langchain-openrouter==0.2.5", "langgraph==1.2.6", @@ -30,3 +32,9 @@ dependencies = [ "structlog==26.1.0", "typing-extensions==4.15.0", ] + +[dependency-groups] +dev = [ + "pytest-asyncio>=1.4.0", + "pytest-httpx>=0.36.2", +] diff --git a/requirements.txt b/requirements.txt index 60cdd17..e4d99e6 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,3 +1,4 @@ +aiolimiter==1.2.1 CyVer==2.0.2 fastapi[standard]==0.138.2 httpx==0.28.1 @@ -6,6 +7,7 @@ langchain==1.3.14 langchain-core==1.5.0 langchain-google-genai==4.2.6 langchain-groq==1.1.3 +langchain-mcp-adapters==0.3.2 langchain-openai==1.4.0 langchain-openrouter==0.2.5 langchain-anthropic==1.4.8 diff --git a/uv.lock b/uv.lock index 44dbe0e..d88f243 100644 --- a/uv.lock +++ b/uv.lock @@ -7,6 +7,15 @@ resolution-markers = [ "sys_platform != 'emscripten' and sys_platform != 'win32'", ] +[[package]] +name = "aiolimiter" +version = "1.2.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/f1/23/b52debf471f7a1e42e362d959a3982bdcb4fe13a5d46e63d28868807a79c/aiolimiter-1.2.1.tar.gz", hash = "sha256:e02a37ea1a855d9e832252a105420ad4d15011505512a1a1d814647451b5cca9", size = 7185, upload-time = "2024-12-08T15:31:51.496Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/f3/ba/df6e8e1045aebc4778d19b8a3a9bc1808adb1619ba94ca354d9ba17d86c3/aiolimiter-1.2.1-py3-none-any.whl", hash = "sha256:d3f249e9059a20badcb56b61601a83556133655c11d1eb3dd3e04ff069e5f3c7", size = 6711, upload-time = "2024-12-08T15:31:49.874Z" }, +] + [[package]] name = "annotated-doc" version = "0.0.4" @@ -57,6 +66,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/da/35/f2287558c17e29fafc8ef3daf819bb9834061cfa43bff8014f7df7f63bdc/anyio-4.14.2-py3-none-any.whl", hash = "sha256:9f505dda5ac9f0c8309b5e8bd445a8c2bf7246f3ce950121e45ea15bc41d1494", size = 125813, upload-time = "2026-07-12T20:29:05.763Z" }, ] +[[package]] +name = "attrs" +version = "26.1.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/9a/8e/82a0fe20a541c03148528be8cac2408564a6c9a0cc7e9171802bc1d26985/attrs-26.1.0.tar.gz", hash = "sha256:d03ceb89cb322a8fd706d4fb91940737b6642aa36998fe130a9bc96c985eff32", size = 952055, upload-time = "2026-03-19T14:22:25.026Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/64/b4/17d4b0b2a2dc85a6df63d1157e028ed19f90d4cd97c36717afef2bc2f395/attrs-26.1.0-py3-none-any.whl", hash = "sha256:c647aa4a12dfbad9333ca4e71fe62ddc36f4e63b2d260a37a8b83d2f043ac309", size = 67548, upload-time = "2026-03-19T14:22:23.645Z" }, +] + [[package]] name = "certifi" version = "2026.7.22" @@ -137,6 +155,7 @@ name = "crossbar-llm" version = "0.1.0" source = { editable = "." } dependencies = [ + { name = "aiolimiter" }, { name = "cyver" }, { name = "fastapi", extra = ["standard"] }, { name = "httpx" }, @@ -146,6 +165,7 @@ dependencies = [ { name = "langchain-core" }, { name = "langchain-google-genai" }, { name = "langchain-groq" }, + { name = "langchain-mcp-adapters" }, { name = "langchain-openai" }, { name = "langchain-openrouter" }, { name = "langgraph" }, @@ -161,8 +181,15 @@ dependencies = [ { name = "typing-extensions" }, ] +[package.dev-dependencies] +dev = [ + { name = "pytest-asyncio" }, + { name = "pytest-httpx" }, +] + [package.metadata] requires-dist = [ + { name = "aiolimiter", specifier = ">=1.2.1" }, { name = "cyver", specifier = "==2.0.2" }, { name = "fastapi", extras = ["standard"], specifier = "==0.138.2" }, { name = "httpx", specifier = "==0.28.1" }, @@ -172,6 +199,7 @@ requires-dist = [ { name = "langchain-core", specifier = "==1.5.0" }, { name = "langchain-google-genai", specifier = "==4.2.6" }, { name = "langchain-groq", specifier = "==1.1.3" }, + { name = "langchain-mcp-adapters", specifier = ">=0.3.2" }, { name = "langchain-openai", specifier = "==1.4.0" }, { name = "langchain-openrouter", specifier = "==0.2.5" }, { name = "langgraph", specifier = "==1.2.6" }, @@ -187,6 +215,12 @@ requires-dist = [ { name = "typing-extensions", specifier = "==4.15.0" }, ] +[package.metadata.requires-dev] +dev = [ + { name = "pytest-asyncio", specifier = ">=1.4.0" }, + { name = "pytest-httpx", specifier = ">=0.36.2" }, +] + [[package]] name = "cryptography" version = "49.0.0" @@ -509,6 +543,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/2a/39/e50c7c3a983047577ee07d2a9e53faf5a69493943ec3f6a384bdc792deb2/httpx-0.28.1-py3-none-any.whl", hash = "sha256:d909fcccc110f8c7faf814ca82a9a4d816bc5a6dbfea25d6591d6985b8ba59ad", size = 73517, upload-time = "2024-12-06T15:37:21.509Z" }, ] +[[package]] +name = "httpx-sse" +version = "0.4.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/0f/4c/751061ffa58615a32c31b2d82e8482be8dd4a89154f003147acee90f2be9/httpx_sse-0.4.3.tar.gz", hash = "sha256:9b1ed0127459a66014aec3c56bebd93da3c1bc8bb6618c8082039a44889a755d", size = 15943, upload-time = "2025-10-10T21:48:22.271Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d2/fd/6668e5aec43ab844de6fc74927e155a3b37bf40d7c3790e49fc0406b6578/httpx_sse-0.4.3-py3-none-any.whl", hash = "sha256:0ac1c9fe3c0afad2e0ebb25a934a59f4c7823b60792691f779fad2c5568830fc", size = 8960, upload-time = "2025-10-10T21:48:21.158Z" }, +] + [[package]] name = "idna" version = "3.18" @@ -604,6 +647,33 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/9e/6a/a83720e953b1682d2d109d3c2dbb0bc9bf28cc1cbc205be4ef4be5da709d/jsonpointer-3.1.1-py3-none-any.whl", hash = "sha256:8ff8b95779d071ba472cf5bc913028df06031797532f08a7d5b602d8b2a488ca", size = 7659, upload-time = "2026-03-23T22:32:31.568Z" }, ] +[[package]] +name = "jsonschema" +version = "4.26.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "attrs" }, + { name = "jsonschema-specifications" }, + { name = "referencing" }, + { name = "rpds-py" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/b3/fc/e067678238fa451312d4c62bf6e6cf5ec56375422aee02f9cb5f909b3047/jsonschema-4.26.0.tar.gz", hash = "sha256:0c26707e2efad8aa1bfc5b7ce170f3fccc2e4918ff85989ba9ffa9facb2be326", size = 366583, upload-time = "2026-01-07T13:41:07.246Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/69/90/f63fb5873511e014207a475e2bb4e8b2e570d655b00ac19a9a0ca0a385ee/jsonschema-4.26.0-py3-none-any.whl", hash = "sha256:d489f15263b8d200f8387e64b4c3a75f06629559fb73deb8fdfb525f2dab50ce", size = 90630, upload-time = "2026-01-07T13:41:05.306Z" }, +] + +[[package]] +name = "jsonschema-specifications" +version = "2025.9.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "referencing" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/19/74/a633ee74eb36c44aa6d1095e7cc5569bebf04342ee146178e2d36600708b/jsonschema_specifications-2025.9.1.tar.gz", hash = "sha256:b540987f239e745613c7a9176f3edb72b832a4ac465cf02712288397832b5e8d", size = 32855, upload-time = "2025-09-08T01:34:59.186Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/41/45/1a4ed80516f02155c51f51e8cedb3c1902296743db0bbc66608a0db2814f/jsonschema_specifications-2025.9.1-py3-none-any.whl", hash = "sha256:98802fee3a11ee76ecaca44429fda8a41bff98b00a0f2838151b113f210cc6fe", size = 18437, upload-time = "2025-09-08T01:34:57.871Z" }, +] + [[package]] name = "langchain" version = "1.3.14" @@ -680,6 +750,20 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/a7/5d/2f862bf5623d5c8ca6ed8a8917a7b1410e6d3595ee0bac12aff508556f76/langchain_groq-1.1.3-py3-none-any.whl", hash = "sha256:a69bb8212b7a699f407c033bf41ca526db8de68f438d51a41740a72bf6dc09bf", size = 20779, upload-time = "2026-06-10T04:16:06.754Z" }, ] +[[package]] +name = "langchain-mcp-adapters" +version = "0.3.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "langchain-core" }, + { name = "mcp" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/05/49/f3b8497b64024ab50d10011f27e94d149668fef754da74c1c2ce6ebe4a30/langchain_mcp_adapters-0.3.2.tar.gz", hash = "sha256:61cd1a09597adb619a9bafb0642938ffc2a9463d699a753f7af0420ea46c381a", size = 47129, upload-time = "2026-08-06T06:15:04.094Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a3/5e/4f117d2500a661079a1895a6eb18954a906e458b1e45fa04a301fcdabd61/langchain_mcp_adapters-0.3.2-py3-none-any.whl", hash = "sha256:094e6b3096dbcc408417d5722f6915f164772e50c502ae3d8989405bf12c3c84", size = 28879, upload-time = "2026-08-06T06:15:02.832Z" }, +] + [[package]] name = "langchain-openai" version = "1.4.0" @@ -848,6 +932,31 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e5/f1/216fc1bbfd74011693a4fd837e7026152e89c4bcf3e77b6692fba9923123/markupsafe-3.0.3-cp312-cp312-win_arm64.whl", hash = "sha256:35add3b638a5d900e807944a078b51922212fb3dedb01633a8defc4b01a3c85f", size = 13906, upload-time = "2025-09-27T18:36:40.689Z" }, ] +[[package]] +name = "mcp" +version = "1.29.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "anyio" }, + { name = "httpx" }, + { name = "httpx-sse" }, + { name = "jsonschema" }, + { name = "pydantic" }, + { name = "pydantic-settings" }, + { name = "pyjwt", extra = ["crypto"] }, + { name = "python-multipart" }, + { name = "pywin32", marker = "sys_platform == 'win32'" }, + { name = "sse-starlette" }, + { name = "starlette" }, + { name = "typing-extensions" }, + { name = "typing-inspection" }, + { name = "uvicorn", marker = "sys_platform != 'emscripten'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/b5/48/0bb26fdfe7ac16875f534a101ce2405eae192bdef37e7451f2f4507c13ec/mcp-1.29.1.tar.gz", hash = "sha256:1967ba4c315f7a375146209949f45950d18b0efd2f913d7cf3400bc723ee5f04", size = 646823, upload-time = "2026-08-24T18:30:41.161Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/0b/04/d6b4fb82eefe9e81807aabca1ac98f460ae0883974b83a997aaa20c52545/mcp-1.29.1-py3-none-any.whl", hash = "sha256:b6310eeb59153300c4ab8b9aec4c52f4819a2d6a8e429eb43d908bed7c783648", size = 224653, upload-time = "2026-08-24T18:30:39.573Z" }, +] + [[package]] name = "mdurl" version = "0.1.2" @@ -1118,6 +1227,20 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/f4/7e/a72dd26f3b0f4f2bf1dd8923c85f7ceb43172af56d63c7383eb62b332364/pygments-2.20.0-py3-none-any.whl", hash = "sha256:81a9e26dd42fd28a23a2d169d86d7ac03b46e2f8b59ed4698fb4785f946d0176", size = 1231151, upload-time = "2026-03-29T13:29:30.038Z" }, ] +[[package]] +name = "pyjwt" +version = "2.13.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/3b/81/58d0ac84e1ef3a3843791d6954d94c0b33d526c75eeb1efbce9d0a4c4077/pyjwt-2.13.0.tar.gz", hash = "sha256:41571c89ca91598c79e8ef18a2d07367d4810fbbd6f637794879baf1b7703423", size = 107515, upload-time = "2026-05-21T19:54:36.618Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a3/5e/ecf12fdb62546d64385c158514e9b2b671f7832108ef2ecd2020ce0af2d1/pyjwt-2.13.0-py3-none-any.whl", hash = "sha256:66adcc2aff09b3f1bbd95fc1e1577df8ac8723c978552fd43304c8a290ac5728", size = 31274, upload-time = "2026-05-21T19:54:35.362Z" }, +] + +[package.optional-dependencies] +crypto = [ + { name = "cryptography" }, +] + [[package]] name = "pytest" version = "9.1.1" @@ -1134,6 +1257,32 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/24/25/1de2678b631f5a49215c6c96fff41ba892b0a34df68d6d80292b1b48aa7f/pytest-9.1.1-py3-none-any.whl", hash = "sha256:37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c", size = 386536, upload-time = "2026-06-19T10:58:31.347Z" }, ] +[[package]] +name = "pytest-asyncio" +version = "1.4.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "pytest" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/43/7c/d36d04db312ecf4298932ef77e6e4a9e8ad017906e24e34f0b0c361a2473/pytest_asyncio-1.4.0.tar.gz", hash = "sha256:c6c0d2259945122819f171a32ecea2c349ead889ee28176caaf492143424be42", size = 58514, upload-time = "2026-05-26T09:56:04.083Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/03/e2/08a497ef684b88559c9cc5f4ad53a37e7b99e727094a86d6ea32536d5d3c/pytest_asyncio-1.4.0-py3-none-any.whl", hash = "sha256:933ca923a23075a87fb7070c0ec272a6848489824d887c85c812670932835aa1", size = 16930, upload-time = "2026-05-26T09:56:02.576Z" }, +] + +[[package]] +name = "pytest-httpx" +version = "0.36.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "httpx" }, + { name = "pytest" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/4e/42/f53c58570e80d503ade9dd42ce57f2915d14bcbe25f6308138143950d1d6/pytest_httpx-0.36.2.tar.gz", hash = "sha256:05a56527484f7f4e8c856419ea379b8dc359c36801c4992fdb330f294c690356", size = 57683, upload-time = "2026-04-09T13:57:19.837Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/1e/55/1fa65f8e4fceb19dd6daa867c162ad845d547f6058cd92b4b02384a44777/pytest_httpx-0.36.2-py3-none-any.whl", hash = "sha256:d42ebd5679442dc7bfb0c48e0767b6562e9bc4534d805127b0084171886a5e22", size = 20315, upload-time = "2026-04-09T13:57:18.587Z" }, +] + [[package]] name = "pytest-mock" version = "3.15.1" @@ -1185,6 +1334,16 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/ec/dd/96da98f892250475bdf2328112d7468abdd4acc7b902b6af23f4ed958ea0/pytz-2026.2-py2.py3-none-any.whl", hash = "sha256:04156e608bee23d3792fd45c94ae47fae1036688e75032eea2e3bf0323d1f126", size = 510141, upload-time = "2026-05-04T01:35:27.408Z" }, ] +[[package]] +name = "pywin32" +version = "312" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/83/ff/32aa7d2ed0ab12b323aaa64f9b75e6ad4f8fd09f9ccfc28c79414d46838d/pywin32-312-cp312-cp312-win32.whl", hash = "sha256:dab4f65ac9c4e48400a2a0530c46c3c579cd5905ecd11b80692373915269208b", size = 6371877, upload-time = "2026-06-04T07:49:28.836Z" }, + { url = "https://files.pythonhosted.org/packages/03/d9/77040d3b43df3f3be32ea289433d660d2727f5ba327bc73be835127d9d60/pywin32-312-cp312-cp312-win_amd64.whl", hash = "sha256:b457f6d628a47e8a7346ce22acb7e1a46a4a78b52e1d17e1af56871bd19a93bc", size = 6914841, upload-time = "2026-06-04T07:49:31.85Z" }, + { url = "https://files.pythonhosted.org/packages/e3/cc/7b1ec671775756020a0ee7f4feeaf3c568f0ab86bd3900088cf986937a92/pywin32-312-cp312-cp312-win_arm64.whl", hash = "sha256:6017c58e12f6809fbb0555b75df144c2922a9ffd18e4b9b5afa863b6c1a9d950", size = 6727901, upload-time = "2026-06-04T07:49:34.244Z" }, +] + [[package]] name = "pyyaml" version = "6.0.3" @@ -1203,6 +1362,20 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/1a/08/67bd04656199bbb51dbed1439b7f27601dfb576fb864099c7ef0c3e55531/pyyaml-6.0.3-cp312-cp312-win_arm64.whl", hash = "sha256:64386e5e707d03a7e172c0701abfb7e10f0fb753ee1d773128192742712a98fd", size = 140344, upload-time = "2025-09-25T21:32:22.617Z" }, ] +[[package]] +name = "referencing" +version = "0.37.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "attrs" }, + { name = "rpds-py" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/22/f5/df4e9027acead3ecc63e50fe1e36aca1523e1719559c499951bb4b53188f/referencing-0.37.0.tar.gz", hash = "sha256:44aefc3142c5b842538163acb373e24cce6632bd54bdb01b21ad5863489f50d8", size = 78036, upload-time = "2025-10-13T15:30:48.871Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/2c/58/ca301544e1fa93ed4f80d724bf5b194f6e4b945841c5bfd555878eea9fcb/referencing-0.37.0-py3-none-any.whl", hash = "sha256:381329a9f99628c9069361716891d34ad94af76e461dcb0335825aecc7692231", size = 26766, upload-time = "2025-10-13T15:30:47.625Z" }, +] + [[package]] name = "regex" version = "2026.7.19" @@ -1305,6 +1478,29 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/8f/ce/73c919505f9f270ee3e34205ff2dbe0b0d73e945f8c569922acff148bcf4/rignore-0.8.0-cp312-cp312-win_arm64.whl", hash = "sha256:6cdea3f85de8286a38ae75a0f9092cd3afc4d33ce6ee2e3f6005f97f7da9d249", size = 665216, upload-time = "2026-07-17T18:58:52.308Z" }, ] +[[package]] +name = "rpds-py" +version = "2026.6.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/aa/2a/9618a122aeb2a169a28b03889a2995fe297588964333d4a7d67bdf46e147/rpds_py-2026.6.3.tar.gz", hash = "sha256:1cebd1337c242e4ec2293e541f712b2da849b29f48f0c293684b71c0632625d4", size = 64051, upload-time = "2026-06-30T07:17:53.009Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5c/be/2e8974163072e7bab7df1a5acd54c4498e75e35d6d18b864d3a9d5dadc92/rpds_py-2026.6.3-cp312-cp312-macosx_10_12_x86_64.whl", hash = "sha256:a0811d33247c3d6128a3001d763f2aa056bb3425204335400ac54f89eec3a0d0", size = 343691, upload-time = "2026-06-30T07:15:14.96Z" }, + { url = "https://files.pythonhosted.org/packages/a4/73/319dfa745dd668efe89309141ded489126461fcecd2b8f3a3cda185129b6/rpds_py-2026.6.3-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:538949e262e46caa31ac01bdb3c1e8f642622922cacbabbae6a8445d9dc33eaf", size = 338542, upload-time = "2026-06-30T07:15:16.267Z" }, + { url = "https://files.pythonhosted.org/packages/21/63/4239893be1c4d09b709b1a8f6be4188f0870084ff547f46606b8a75f1b03/rpds_py-2026.6.3-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:55927d532399c2c646100ff7feb48eaa940ad70f42cd68e1328f3ded9f81ca24", size = 368180, upload-time = "2026-06-30T07:15:17.62Z" }, + { url = "https://files.pythonhosted.org/packages/1c/ca/9c5de382225234ceb37b1844ebdb140db12b2a278bb9efe2fcd19f6c82ce/rpds_py-2026.6.3-cp312-cp312-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:f56f1695bc5c0871cbc33dc0130fcf503aab0c57dcc5a6700a4f49eba4f2652e", size = 375067, upload-time = "2026-06-30T07:15:18.952Z" }, + { url = "https://files.pythonhosted.org/packages/87/dc/863f69d1bf04ade34b7fe0d59b9fdf6f0135fe2d7cbca74f1d665589559d/rpds_py-2026.6.3-cp312-cp312-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:270b293dae9058fc9fcedab50f13cebf46fb8ed1d1d54e0521a9da5d6b211975", size = 490509, upload-time = "2026-06-30T07:15:20.434Z" }, + { url = "https://files.pythonhosted.org/packages/ce/ef/eac16a12048b45ec7c7fa94f2be3438a5f26bf9cc8580b18a1cfd609b7f6/rpds_py-2026.6.3-cp312-cp312-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:127565fead0a10943b282957bd5447804ff3160ad79f2ad2635e6d249e380680", size = 382754, upload-time = "2026-06-30T07:15:21.831Z" }, + { url = "https://files.pythonhosted.org/packages/04/8f/d2f3f532616be4d06c316ef119683e832bd3d41e112bf3a88f4151c95b17/rpds_py-2026.6.3-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:ecabd69db66de867690f9797f2f8fa27ba501bbc24540cbdbdc649cd15888ba6", size = 366189, upload-time = "2026-06-30T07:15:23.371Z" }, + { url = "https://files.pythonhosted.org/packages/e3/29/41a7b0e98a4b44cd676ab7598419623373eb43b20be68c084935c1a8cf88/rpds_py-2026.6.3-cp312-cp312-manylinux_2_31_riscv64.whl", hash = "sha256:58eadac9cd119677b60e1cf8ac4052f35949d71b8a9e5556efccbe82533cf22a", size = 377750, upload-time = "2026-06-30T07:15:24.659Z" }, + { url = "https://files.pythonhosted.org/packages/2e/05/ecda0bec46f9a1565090bcdc941d023f6a25aff85fda28f89f8d19878152/rpds_py-2026.6.3-cp312-cp312-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:7491ee23305ac3eb59e492b6945881f5cd77a6f731061a3f25b77fd40f9e99a4", size = 395576, upload-time = "2026-06-30T07:15:25.987Z" }, + { url = "https://files.pythonhosted.org/packages/68/a8/6ed52f03ee6cb854ce78785cc9a9a672eb880e83fd7224d471f667d151f1/rpds_py-2026.6.3-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:2c99f7e8ccb3dd6e3e4bfeac657a7b208c9bac8075f4b078c02d7404c34107fa", size = 543807, upload-time = "2026-06-30T07:15:27.356Z" }, + { url = "https://files.pythonhosted.org/packages/8f/d6/156c0d3eea27ba09b92562ba2364ba124c0a061b199e17eac637cd25a5e2/rpds_py-2026.6.3-cp312-cp312-musllinux_1_2_i686.whl", hash = "sha256:62698275682bf121181861295c9181e789030a2d516071f5b8f3c23c170cd0fc", size = 611187, upload-time = "2026-06-30T07:15:28.931Z" }, + { url = "https://files.pythonhosted.org/packages/f1/31/774212ed989c62f7f310220089f9b0a3fb8f40f5443d1727abd5d9f52bc9/rpds_py-2026.6.3-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:a214c993455f99a89aaeadc9b21241900037adc9d97203e374d75513c5911822", size = 573030, upload-time = "2026-06-30T07:15:30.553Z" }, + { url = "https://files.pythonhosted.org/packages/c9/50/22f73127a41f1ce4f87fe39aadfb9a126345801c274aa93ae88456249327/rpds_py-2026.6.3-cp312-cp312-win32.whl", hash = "sha256:501f9f04a588d6a09179368c57071301445191767c64e4b52a6aa9871f1ef5ed", size = 202185, upload-time = "2026-06-30T07:15:32.027Z" }, + { url = "https://files.pythonhosted.org/packages/04/3a/f0ee4d4dde9d3b69dedf1b5f74e7a40017046d55052d173e418c6a94f960/rpds_py-2026.6.3-cp312-cp312-win_amd64.whl", hash = "sha256:2c958bf94822e9290a40aaf2a822d4bc5c88099093e3948ad6c571eca9272e5f", size = 220394, upload-time = "2026-06-30T07:15:33.359Z" }, + { url = "https://files.pythonhosted.org/packages/f3/83/3382fe37f809b59f02aac04dbc4e765b480b46ee0227ed516e3bdc4d3dfc/rpds_py-2026.6.3-cp312-cp312-win_arm64.whl", hash = "sha256:22bffe6042b9bcb0822bcd1955ec00e245daf17b4344e4ed8e9551b976b63e96", size = 215753, upload-time = "2026-06-30T07:15:34.778Z" }, +] + [[package]] name = "sentry-sdk" version = "2.66.1" @@ -1357,6 +1553,19 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e9/44/75a9c9421471a6c4805dbf2356f7c181a29c1879239abab1ea2cc8f38b40/sniffio-1.3.1-py3-none-any.whl", hash = "sha256:2f6da418d1f1e0fddd844478f41680e794e6051915791a034ff65e5f100525a2", size = 10235, upload-time = "2024-02-25T23:20:01.196Z" }, ] +[[package]] +name = "sse-starlette" +version = "3.4.8" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "anyio" }, + { name = "starlette" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/f8/00/b42a44342a054d58cb1115d7c8aa9cb4290dd9442f9c1b91a4b8173dba22/sse_starlette-3.4.8.tar.gz", hash = "sha256:ed89ffbb75cbf78a5fe2f2109cd584792ee7f9dfac96f791db546df8f15f3f9c", size = 32548, upload-time = "2026-08-05T11:19:49.982Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/dd/3a/764912c58293d95b6dcdf4cc255f9d10de310580ced547b082eb9d72018c/sse_starlette-3.4.8-py3-none-any.whl", hash = "sha256:6e82314c786709a3cd9520f2285cf9fff90e181e598e8a357b0cf80f66afba0d", size = 16516, upload-time = "2026-08-05T11:19:48.748Z" }, +] + [[package]] name = "starlette" version = "1.3.1" From 65282b932cab580203eaaff4cdc6de27bce4d0c1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ahmet=20O=C4=9Fuzhan=20K=C3=B6k=C3=BCl=C3=BC?= <oguzhankokulu@gmail.com> Date: Fri, 28 Aug 2026 17:51:44 +0300 Subject: [PATCH 4/8] fix(paperclip): stop routing literature questions to record corpora Measured on 100 BioASQ list questions: broad search and pmc scored 3.5/5 mean judge, while the 8 questions routed to `proteins` scored 0.12 and the 8 routed to `fda` scored 0.50 - near-total failure on 16% of the set. The cause is over-selection, not the corpora themselves. "Which tissues express the ACE2 protein?" went to UniProt, which matches on name text and returned transporters and enzymes; "List the essential aminoacids" came back with "SMDT1 - Essential MCU regulator", matched on the word "essential". "List drugs included in the TRIUMEQ pill" went to `fda` despite the prompt carrying the structurally identical LONSURF example as a counter-example. Two prompt revisions failed to hold, so the constraint moves into code where it is testable: a record corpus (proteins/pdb/chembl) requires an accession or the database named outright, and a registry corpus (fda/trials) requires regulatory or trial-registry intent. Anything else is downgraded to broad search with a warning. All 17 misroutes from the run are downgraded; genuine lookups such as "sequence length of UniProt P04637" and "enrollment of trial NCT04280705" are preserved, both covered by tests. --- crossbar_llm/paperclip_tools/agent.py | 45 ++++++++++++++++- .../paperclip_tools/tests/test_agent.py | 49 +++++++++++++++++++ 2 files changed, 93 insertions(+), 1 deletion(-) diff --git a/crossbar_llm/paperclip_tools/agent.py b/crossbar_llm/paperclip_tools/agent.py index 9735888..64e0ba8 100644 --- a/crossbar_llm/paperclip_tools/agent.py +++ b/crossbar_llm/paperclip_tools/agent.py @@ -151,6 +151,41 @@ def _collapse_degenerate_runs(text: str) -> tuple[str, int]: return text, total +# Narrow corpora measured on 100 BioASQ list questions: broad/pmc scored 3.5/5 +# mean judge, while `proteins` scored 0.12 and `fda` 0.50. They hold database +# records with no abstracts and no prose, so they can only answer a question +# that names the exact record or asks for a regulatory/registry field. The +# router picks them anyway whenever a question merely says "protein" or names a +# drug -- the prompt has warned against both for two revisions and it still +# happens -- so the check lives here where it can be tested. +_RECORD_CORPORA = {"proteins", "pdb", "chembl"} +_REGISTRY_CORPORA = {"fda", "trials"} + +# UniProt accession, ChEMBL id, or the database named outright. +_RECORD_ID_RE = re.compile( + r"\b(?:[OPQ][0-9][A-Z0-9]{3}[0-9]|[A-NR-Z][0-9][A-Z][A-Z0-9]{2}[0-9]" + r"|CHEMBL[0-9]+|uniprot|swiss-?prot|pdb\b|protein data bank|chembl)\b", + re.I, +) +# Regulatory/registry intent: an NCT id, or language about the record itself +# rather than about the drug's biology. +_REGISTRY_INTENT_RE = re.compile( + r"\b(?:NCT[0-9]{6,}|fda[- ]approved|approval status|drug label|package insert" + r"|boxed warning|black box|regulatory|marketing authoriz|clinical trial registry" + r"|trial phase|enrollment|recruitment status)\b", + re.I, +) + + +def _narrow_source_allowed(source: str | None, question: str) -> bool: + """Whether a narrow corpus can actually serve this question.""" + if source in _RECORD_CORPORA: + return bool(_RECORD_ID_RE.search(question)) + if source in _REGISTRY_CORPORA: + return bool(_REGISTRY_INTENT_RE.search(question)) + return True + + def build_graph( *, chat_model: BaseChatModel | None = None, @@ -242,9 +277,17 @@ async def router_node(state: PaperclipState) -> dict: map_question=state["question"], rationale=f"router error fallback: {e}", ) + source = decision.source + if not _narrow_source_allowed(source, state["question"]): + warnings.append( + f"router chose source={source!r} but the question names no " + "record or regulatory field; searching broadly instead." + ) + source = None + return { "question_type": decision.question_type, - "source": decision.source, + "source": source, "search_query": decision.search_query, "analogical_query": decision.analogical_query, "map_question": decision.map_question, diff --git a/crossbar_llm/paperclip_tools/tests/test_agent.py b/crossbar_llm/paperclip_tools/tests/test_agent.py index d9d5017..c5a2e94 100644 --- a/crossbar_llm/paperclip_tools/tests/test_agent.py +++ b/crossbar_llm/paperclip_tools/tests/test_agent.py @@ -861,3 +861,52 @@ async def boom(state): assert out["question_type"] == "keyword_search" assert out["source"] is None assert any("router failed" in w for w in out["warnings"]) + + +# --- narrow-source guard ------------------------------------------------- # +# Cases taken verbatim from the 100-question BioASQ list run, where `proteins` +# scored 0.12/5 and `fda` 0.50/5 against 3.5/5 for broad search. + +import pytest + +from crossbar_llm.paperclip_tools.agent import _narrow_source_allowed + + +@pytest.mark.parametrize("source,question", [ + ("proteins", "List the essential aminoacids."), + ("proteins", "Which tissues express the ACE2 protein?"), + ("proteins", "Name three binding partners of cofilin 2."), + ("proteins", "Which protein complexes contain mitofilin?"), + ("proteins", "Name curated data resources for ChIP-seq data"), + ("fda", "List drugs included in the TRIUMEQ pill."), + ("fda", "Which drugs are included in PolyIran?"), + ("fda", "What are the targets of avapritinib?"), + ("trials", "Which two drugs were compared in the ARISTOTLE Trial?"), +]) +def test_narrow_source_rejected_without_a_named_record(source, question): + """Naming a protein or drug is not grounds for a record corpus.""" + assert _narrow_source_allowed(source, question) is False + + +@pytest.mark.parametrize("source,question", [ + ("proteins", "What is the sequence length of UniProt P04637?"), + ("proteins", "What is the PDB accession for human lysozyme?"), + ("chembl", "What is the ChEMBL bioactivity of CHEMBL25?"), + ("trials", "What is the enrollment of trial NCT04280705?"), + ("fda", "What are the FDA-approved indications for alteplase?"), + ("fda", "Does pembrolizumab carry a boxed warning?"), + (None, "Which genes are related to psoriasis?"), + ("pmc", "Which genes are related to psoriasis?"), +]) +def test_narrow_source_allowed_for_real_record_lookups(source, question): + assert _narrow_source_allowed(source, question) is True + + +async def test_router_downgrades_unusable_narrow_source_and_warns(): + adapter = FakeAdapter() + g = build_graph(router=_router(source="proteins"), synthesizer=_synth, adapter=adapter) + out = await g.ainvoke({"question": "Which tissues express the ACE2 protein?"}) + + assert out["source"] is None + assert any("searching broadly instead" in w for w in out["warnings"]) + assert adapter.search_calls[0]["source"] is None From 1e4def9473d50183ae872e255a9edc6ac8bdb7f1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ahmet=20O=C4=9Fuzhan=20K=C3=B6k=C3=BCl=C3=BC?= <oguzhankokulu@gmail.com> Date: Mon, 21 Sep 2026 12:52:26 +0300 Subject: [PATCH 5/8] feat(literature): integrate Paperclip and PubTator3 as per-request tools MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Wire the standalone Paperclip and PubTator3 agents into the API behind per-request switches, defaulting off. When both are enabled they run concurrently with the Cypher graph, and each reports its own answer, citations, warnings and usage so one tool failing never discards the other. Request/response: - `literature_tools` lives on `ModelConfigRequest`, so db, vector, upload and resume all carry it; multipart flattens to `paperclip`/`pubtator3` fields. - `ChatResponse.literature` maps each tool to a typed result. The core `final_answer`, Cypher and execution fields are unchanged. - `PendingResumeResponse` now declares the `status` field the service had always passed; pydantic was dropping it silently. Concurrency and lifecycle: - Convert the DB/vector/resume paths to `ainvoke`. The Cypher agent's nodes are sync and were running on the event loop, blocking it for the whole request; LangGraph dispatches them to a worker thread instead. - Run core and literature as real tasks so a failure on either cancels the other. A plain `gather()` left the survivor running detached, spending metered Paperclip calls and tokens on a response nobody would receive. - Hold one long-lived Paperclip adapter and close it from a lifespan hook rather than building (and leaking) one per request. Cost control: - Decide biological relevance once, up front, so out-of-domain questions never reach the literature tools. The verdict is seeded into state and the relevance node reuses it, so the graph topology stays identical between a request and its resume — both replay the same checkpoint shape. - Per-tool timeouts, and admission control bounding concurrent runs per process; a saturated server reports `skipped`, not a misleading timeout. - Paperclip `use_map` off and PubTator3 `abstracts_only` on by default, cutting upstream calls per question roughly threefold. Operational: - Give Paperclip's REST path its own short pool timeout. A bare float set connect/read/write/pool alike, so queueing for a connection consumed the whole call budget and surfaced as "Paperclip is slow". - Make the PubTator3 limiter swappable and add a per-replica share backend; the 3 req/s ceiling is NCBI's and applies per IP, not per process. - Credentials are passed to the adapter instead of exported to `os.environ`, and `PAPERCLIP_DISABLE_REST` is parsed as a real bool. - Literature usage is tracked on its own lenient handler and merged for the response, so the core agent keeps strict token accounting. - Upstream error text is logged, not returned; only a missing key is surfaced. Tests: first API-level suite in the repo (TestClient + dependency overrides), plus coverage for cancellation, admission control, pool exhaustion, usage merging and the per-replica limiter. --- .env.example | 5 + README.md | 8 + crossbar_llm/agent_tools/callback_handler.py | 52 +++ crossbar_llm/agent_tools/cypher_agent.py | 96 ++-- crossbar_llm/api/core/settings.py | 114 ++++- crossbar_llm/api/main.py | 19 +- crossbar_llm/api/routers/db_search.py | 3 +- crossbar_llm/api/routers/resume.py | 2 +- crossbar_llm/api/routers/vector_search.py | 2 +- crossbar_llm/api/schemas/requests.py | 17 +- crossbar_llm/api/schemas/responses.py | 14 + crossbar_llm/api/services/agent_service.py | 282 ++++++++++-- .../api/services/literature_service.py | 418 ++++++++++++++++++ .../frontend/src/components/ChatLayout.js | 193 +++++++- crossbar_llm/paperclip_tools/adapter.py | 93 +++- crossbar_llm/paperclip_tools/agent.py | 27 +- .../paperclip_tools/structured_output.py | 12 +- .../paperclip_tools/tests/test_adapter.py | 12 +- .../tests/test_pool_timeout.py | 75 ++++ crossbar_llm/pubtator3_tools/agent.py | 26 +- crossbar_llm/pubtator3_tools/client.py | 32 +- crossbar_llm/pubtator3_tools/rate_limit.py | 93 ++++ .../pubtator3_tools/structured_output.py | 12 +- .../pubtator3_tools/tests/test_graph.py | 6 +- .../pubtator3_tools/tests/test_rate_limit.py | 65 +++ crossbar_llm/tests/conftest.py | 31 ++ .../tests/test_agent_service_literature.py | 300 +++++++++++++ crossbar_llm/tests/test_literature_api.py | 261 +++++++++++ crossbar_llm/tests/test_literature_service.py | 366 +++++++++++++++ 29 files changed, 2510 insertions(+), 126 deletions(-) create mode 100644 crossbar_llm/api/services/literature_service.py create mode 100644 crossbar_llm/paperclip_tools/tests/test_pool_timeout.py create mode 100644 crossbar_llm/pubtator3_tools/rate_limit.py create mode 100644 crossbar_llm/pubtator3_tools/tests/test_rate_limit.py create mode 100644 crossbar_llm/tests/conftest.py create mode 100644 crossbar_llm/tests/test_agent_service_literature.py create mode 100644 crossbar_llm/tests/test_literature_api.py create mode 100644 crossbar_llm/tests/test_literature_service.py diff --git a/.env.example b/.env.example index 14726c4..4cec46d 100644 --- a/.env.example +++ b/.env.example @@ -3,6 +3,11 @@ GEMINI_API_KEY= ANTHROPIC_API_KEY= GROQ_API_KEY= OPENROUTER_API_KEY= +# Optional: needed only when the Paperclip literature tool is enabled. +PAPERCLIP_API_KEY= +# Optional: 1/true forces Paperclip's MCP-only transport. Leave blank for the +# default (REST enabled); any non-empty value other than a false-y one enables it. +PAPERCLIP_DISABLE_REST= NEO4J_USER= NEO4J_PASSWORD= diff --git a/README.md b/README.md index ca96a93..5e643b4 100644 --- a/README.md +++ b/README.md @@ -57,6 +57,8 @@ GEMINI_API_KEY= ANTHROPIC_API_KEY= GROQ_API_KEY= OPENROUTER_API_KEY= +PAPERCLIP_API_KEY= +PAPERCLIP_DISABLE_REST= NEO4J_USER= NEO4J_PASSWORD= @@ -68,6 +70,12 @@ BROWSER_COOKIE_SECRET= RATE_LIMIT_IP_HASH_SECRET= ``` +Paperclip and PubTator3 are optional literature-evidence tools and are disabled +by default. Each can be enabled independently from the chat settings. When both +are enabled, they run concurrently after the biological-relevance check; +Paperclip requires `PAPERCLIP_API_KEY`. Set `PAPERCLIP_DISABLE_REST=1` to force +its MCP-only transport. + ## Run the Backend API Start the FastAPI backend from the repository root: diff --git a/crossbar_llm/agent_tools/callback_handler.py b/crossbar_llm/agent_tools/callback_handler.py index efe1c12..7a74bcc 100644 --- a/crossbar_llm/agent_tools/callback_handler.py +++ b/crossbar_llm/agent_tools/callback_handler.py @@ -319,6 +319,58 @@ def reset(self) -> None: ) +def merge_usage_summaries(*summaries: dict[str, Any]) -> dict[str, Any]: + """Combine several `get_summary()` results into one request-level total. + + Used where one request is served by more than one callback — the core agent + runs strict (a provider that hides usage metadata should fail loudly rather + than bill silently), while optional side agents run lenient because their + JSON-fallback path legitimately produces responses without it. Keeping two + handlers preserves both behaviours; this puts the numbers back together. + + Node names are expected to be unique across summaries (side agents + namespace theirs). On a collision the later summary wins for `model`, and + token counts are summed. + """ + merged_nodes: dict[str, dict[str, Union[int, str]]] = {} + totals = UsageCounter() + models_by_node: dict[str, list[str]] = {} + session_id: str | None = None + + for summary in summaries: + if not summary: + continue + session_id = session_id or summary.get("session_id") + + for node, record in (summary.get("per_node_usage") or {}).items(): + existing = merged_nodes.get(node) + if existing is None: + merged_nodes[node] = dict(record) + continue + for key in ("input_tokens", "output_tokens", "total_tokens", + "cache_read", "cache_write", "reasoning", "call_count"): + existing[key] = (existing.get(key, 0) or 0) + (record.get(key, 0) or 0) + if record.get("model"): + existing["model"] = record["model"] + + aggregated = summary.get("aggregated_usage") or {} + totals.add_usage(aggregated.get("totals") or {}) + for node, models in (aggregated.get("models_by_node") or {}).items(): + bucket = models_by_node.setdefault(node, []) + for model in models: + if model not in bucket: + bucket.append(model) + + return { + "session_id": session_id, + "per_node_usage": merged_nodes, + "aggregated_usage": { + "totals": totals.to_dict(), + "models_by_node": models_by_node, + }, + } + + # --------------------------------------------------------------- # CHECK: # WHAT HAPPEN WITH DIFFERENT LLM PROVIDERS OTHER THAN OPENAI? diff --git a/crossbar_llm/agent_tools/cypher_agent.py b/crossbar_llm/agent_tools/cypher_agent.py index 9d574be..8352ce0 100644 --- a/crossbar_llm/agent_tools/cypher_agent.py +++ b/crossbar_llm/agent_tools/cypher_agent.py @@ -194,49 +194,89 @@ def trim_messages(self, state: CypherAgentState, new_messages: list[BaseMessage] # when the list is still below the limit. return [RemoveMessage(id=REMOVE_ALL_MESSAGES), *combined_messages[-AgentConfig().keep_last_n_messages:]] - @log_execution_time(logger, component="CypherAgent.initialize_state") - def biological_relevance_validation_node(self, state: CypherAgentState): - - logger.info( - "Starting biological relevance validation", - event_type="biological_relevance_validation_started", - component="CypherAgent.biological_relevance_validation_node", - question=state["question"] - ) - + # The sync node and the async preflight below share these three helpers so + # the prompt, the metadata tag and the result shape exist in exactly one + # place. Only the invoke/ainvoke call itself differs, which is irreducible + # while the graph still supports `graph.invoke()`. + def _relevance_request(self, question: str): prompt = ChatPromptTemplate.from_messages([ SystemMessagePromptTemplate.from_template( BIOLOGICAL_RELEVANCE_VALIDATION_TEMPLATE ), HumanMessagePromptTemplate.from_template("{question}") ]) - - verdict_llm = self.llm_factory.create_biological_relevance_validator_llm() - messages = prompt.format_messages(question=state["question"]) + logger.info( + "Starting biological relevance validation", + event_type="biological_relevance_validation_started", + component="CypherAgent._relevance_request", + question=question + ) - verdict = verdict_llm.invoke( - messages, - config={ - "metadata": { - "node_name": "biological_relevance_validation", - } - } + return ( + self.llm_factory.create_biological_relevance_validator_llm(), + prompt.format_messages(question=question), + {"metadata": {"node_name": "biological_relevance_validation"}}, ) + @staticmethod + def _relevance_result(verdict, question: str) -> dict[str, object]: + parsed = verdict["parsed"] + logger.info( "Completed biological relevance validation", event_type="biological_relevance_validation_completed", - component="CypherAgent.biological_relevance_validation_node", - question=state["question"], - verdict=verdict["parsed"].relevant, - reason=verdict["parsed"].reason + component="CypherAgent._relevance_result", + question=question, + verdict=parsed.relevant, + reason=parsed.reason ) + return { - "biological_relevance": verdict["parsed"].relevant, - "final_answer": f"Your question is outside the biological/biomedical domain.\nReason: {verdict['parsed'].reason}" if verdict["parsed"].relevant is False else None, - } - + "biological_relevance": parsed.relevant, + "final_answer": ( + "Your question is outside the biological/biomedical domain.\n" + f"Reason: {parsed.reason}" + if parsed.relevant is False + else None + ), + } + + @log_execution_time(logger, component="CypherAgent.initialize_state") + def biological_relevance_validation_node(self, state: CypherAgentState): + # The API orchestrator runs `avalidate_biological_relevance` ahead of the + # graph so it can decide whether to launch the literature tools. When it + # has, the verdict is already in state and paying for a second identical + # LLM call would be pure waste — the router downstream reads the same key + # either way, so returning no update keeps the routing identical. + if state.get("biological_relevance") is not None: + logger.info( + "Reusing pre-validated biological relevance verdict", + event_type="biological_relevance_validation_reused", + component="CypherAgent.biological_relevance_validation_node", + question=state["question"], + verdict=state["biological_relevance"], + ) + return {} + + verdict_llm, messages, config = self._relevance_request(state["question"]) + verdict = verdict_llm.invoke(messages, config=config) + return self._relevance_result(verdict, state["question"]) + + async def avalidate_biological_relevance( + self, + question: str, + ) -> dict[str, object]: + """Async relevance preflight used by the API request orchestrator. + + Seed the returned keys into the graph's initial state and + `biological_relevance_validation_node` will reuse the verdict instead of + re-running it. + """ + verdict_llm, messages, config = self._relevance_request(question) + verdict = await verdict_llm.ainvoke(messages, config=config) + return self._relevance_result(verdict, question) + def route_after_biological_relevance_validation(self, state: CypherAgentState) -> Literal["entity_resolution", "generate_cypher", "end"]: if state["biological_relevance"] is False: logger.info( diff --git a/crossbar_llm/api/core/settings.py b/crossbar_llm/api/core/settings.py index 6e25b2f..4e116ac 100644 --- a/crossbar_llm/api/core/settings.py +++ b/crossbar_llm/api/core/settings.py @@ -4,7 +4,7 @@ BaseSettings, SettingsConfigDict ) -from pydantic import BaseModel, Field, SecretStr +from pydantic import BaseModel, Field, SecretStr, field_validator from crossbar_llm.agent_tools.config import ConfigPaths @@ -18,6 +18,31 @@ class EnvSettings(BaseSettings): app_env: Literal["development", "production"] = Field(default="development", alias="APP_ENV") browser_cookie_secret: SecretStr = Field(alias="BROWSER_COOKIE_SECRET") rate_limit_ip_hash_secret: SecretStr = Field(alias="RATE_LIMIT_IP_HASH_SECRET") + paperclip_api_key: SecretStr | None = Field( + default=None, + alias="PAPERCLIP_API_KEY", + ) + # Typed as a real bool so `PAPERCLIP_DISABLE_REST=false` means false. The + # adapter's own env fallback is a bare truthiness check, where that same + # value would *enable* the flag; parsing it here is what makes the setting + # behave the way anyone would read it. + paperclip_disable_rest: bool | None = Field( + default=None, + alias="PAPERCLIP_DISABLE_REST", + ) + + @field_validator("paperclip_api_key", "paperclip_disable_rest", mode="before") + @classmethod + def _blank_is_unset(cls, value): + """Treat `KEY=` in a .env as "not configured" rather than as a value. + + `.env.example` ships these keys empty, so a verbatim copy must start + cleanly — without this, the empty string fails bool parsing and takes + the whole app down at import. + """ + if isinstance(value, str) and not value.strip(): + return None + return value class Settings(BaseModel): @@ -97,6 +122,91 @@ class Settings(BaseModel): max_upload_size_mb: int = 5 + # Optional literature agents + literature_tool_timeout_seconds: float = Field( + default=180.0, + gt=0, + description="Maximum runtime for each optional literature tool.", + ) + literature_max_citations: int = Field( + default=10, + ge=1, + description="Maximum citations returned per literature tool.", + ) + literature_max_concurrent_runs: int = Field( + default=8, + ge=1, + description=( + "How many literature tool runs may be in flight at once in this " + "process. Each run fans out several requests upstream, so without " + "a ceiling a burst of traffic multiplies straight through to the " + "external services and exhausts the connection pool." + ), + ) + literature_admission_wait_seconds: float = Field( + default=5.0, + ge=0, + description=( + "How long a run waits for an admission slot before being reported " + "as skipped. Short on purpose: queueing here would silently eat " + "the per-tool timeout budget instead of failing legibly." + ), + ) + pubtator3_replica_count: int = Field( + default=1, + ge=1, + description=( + "Number of API replicas sharing one egress IP. PubTator3's 3 req/s " + "ceiling is enforced per IP, so each replica takes a 1/N share. " + "Leave at 1 for a single instance; raise it when scaling out, or " + "replace the limiter with a coordinated one." + ), + ) + paperclip_max_documents: int = Field( + default=7, + ge=1, + description="Papers Paperclip retrieves per question.", + ) + paperclip_abstracts_only: bool = Field( + default=True, + description=( + "Force Paperclip to title+abstract retrieval. Cheaper and more " + "predictable in tokens than pulling full bodies." + ), + ) + paperclip_use_map: bool = Field( + default=False, + description=( + "Let Paperclip read full text server-side and extract a per-paper " + "answer. Off by default: it costs one upstream call PER PAPER, so " + "it multiplies our request rate against a metered service by " + "`paperclip_max_documents`. Turn on only with headroom to spare." + ), + ) + paperclip_max_connections: int = Field( + default=64, + ge=1, + description=( + "HTTP connection cap for the shared Paperclip adapter. This is a " + "process-wide pool, so it bounds every concurrent request at once " + "rather than one request's ~14-wide fan-out." + ), + ) + pubtator3_max_documents: int = Field( + default=7, + ge=1, + description="Papers PubTator3 exports per question.", + ) + pubtator3_abstracts_only: bool = Field( + default=True, + description=( + "Force PubTator3 to title+abstract retrieval. On by default: full " + "text also short-circuits the depth-refinement second pass, and " + "PubTator3's 3 req/s ceiling is IP-wide, so fewer and smaller " + "fetches per question is what keeps the service usable under load." + ), + ) + # CORS allowed_origins: list[str] = Field( default=[ @@ -141,5 +251,3 @@ def get_rate_limit_settings(self) -> tuple[str, str, str]: - - diff --git a/crossbar_llm/api/main.py b/crossbar_llm/api/main.py index 1faa378..13e2c85 100644 --- a/crossbar_llm/api/main.py +++ b/crossbar_llm/api/main.py @@ -1,3 +1,5 @@ +from contextlib import asynccontextmanager + from fastapi import FastAPI from fastapi.middleware.cors import CORSMiddleware @@ -12,12 +14,27 @@ from crossbar_llm.api.routers.resume import router as resume_router from crossbar_llm.api.routers.vector_search import router as vector_search_router from crossbar_llm.api.routers.models import router as models_router +from crossbar_llm.api.core.deps import get_runtime_service settings = Settings() + +@asynccontextmanager +async def lifespan(_app: FastAPI): + yield + # Only close a service that was actually built. `get_runtime_service` is + # lru_cached, so calling it unconditionally here would CONSTRUCT one at + # shutdown — Neo4j config and all — in any process that never served a + # request, purely to close nothing, and would fail the shutdown outright + # where that config is absent. + if get_runtime_service.cache_info().currsize: + await get_runtime_service().aclose() + get_runtime_service.cache_clear() + app = FastAPI( title=settings.app_name, debug=settings.debug, + lifespan=lifespan, ) app.state.limiter = limiter @@ -60,5 +77,3 @@ models_router ) - - diff --git a/crossbar_llm/api/routers/db_search.py b/crossbar_llm/api/routers/db_search.py index 31c83d2..ea198ce 100644 --- a/crossbar_llm/api/routers/db_search.py +++ b/crossbar_llm/api/routers/db_search.py @@ -24,8 +24,7 @@ async def db_query( agent_service: AgentService = Depends(get_runtime_service) ): - return agent_service.run_db(session_id=session_id, browser_id=identity.browser_id, payload=payload) - + return await agent_service.run_db(session_id=session_id, browser_id=identity.browser_id, payload=payload) diff --git a/crossbar_llm/api/routers/resume.py b/crossbar_llm/api/routers/resume.py index 5a565d7..b06ac2c 100644 --- a/crossbar_llm/api/routers/resume.py +++ b/crossbar_llm/api/routers/resume.py @@ -23,4 +23,4 @@ async def resume_session( agent_service: AgentService = Depends(get_runtime_service) ): - return agent_service.resume(session_id=session_id, browser_id=identity.browser_id, payload=payload) + return await agent_service.resume(session_id=session_id, browser_id=identity.browser_id, payload=payload) diff --git a/crossbar_llm/api/routers/vector_search.py b/crossbar_llm/api/routers/vector_search.py index 5e061d0..1e39390 100644 --- a/crossbar_llm/api/routers/vector_search.py +++ b/crossbar_llm/api/routers/vector_search.py @@ -25,7 +25,7 @@ async def vector_query( agent_service: AgentService = Depends(get_runtime_service) ): - return agent_service.run_vector(session_id=session_id, browser_id=identity.browser_id, payload=payload) + return await agent_service.run_vector(session_id=session_id, browser_id=identity.browser_id, payload=payload) @router.post("/upload-query", response_model=ChatResponse | PendingResumeResponse) diff --git a/crossbar_llm/api/schemas/requests.py b/crossbar_llm/api/schemas/requests.py index 8960be5..ff76098 100644 --- a/crossbar_llm/api/schemas/requests.py +++ b/crossbar_llm/api/schemas/requests.py @@ -8,12 +8,22 @@ from crossbar_llm.api.schemas.common import ExecutionControl, SearchMode +class LiteratureToolsConfig(BaseModel): + """Per-request switches for optional literature evidence agents.""" + + paperclip: bool = Field(default=False) + pubtator3: bool = Field(default=False) + + class ModelConfigRequest(BaseModel): provider: str model: str top_k: int = Field(default=10, ge=1, le=100) reasoning_enabled: bool = Field(default=False) reasoning_effort: Literal["low", "medium", "high"] | None = None + literature_tools: LiteratureToolsConfig = Field( + default_factory=LiteratureToolsConfig + ) @model_validator(mode="after") def validate_reasoning(self) -> Self: @@ -63,6 +73,8 @@ def as_form( top_k: int = Form(10), reasoning_enabled: bool = Form(False), reasoning_effort: Literal["low", "medium", "high"] | None = Form(None), + paperclip: bool = Form(False), + pubtator3: bool = Form(False), vector_category: str = Form(...), embedding_type: str = Form(...), ) -> Self: @@ -74,6 +86,10 @@ def as_form( top_k=top_k, reasoning_enabled=reasoning_enabled, reasoning_effort=reasoning_effort, + literature_tools=LiteratureToolsConfig( + paperclip=paperclip, + pubtator3=pubtator3, + ), vector_category=vector_category, embedding_type=embedding_type, ) @@ -97,4 +113,3 @@ def validate_edited_cypher(cls, v: str) -> str: - diff --git a/crossbar_llm/api/schemas/responses.py b/crossbar_llm/api/schemas/responses.py index 93b8118..f079d43 100644 --- a/crossbar_llm/api/schemas/responses.py +++ b/crossbar_llm/api/schemas/responses.py @@ -7,6 +7,15 @@ class SessionCreateResponse(BaseModel): session_id: str + +class LiteratureToolResult(BaseModel): + status: Literal["completed", "failed", "skipped"] + answer: str | None = None + citations: list[dict[str, Any]] = Field(default_factory=list) + warnings: list[str] = Field(default_factory=list) + usage: dict[str, Any] = Field(default_factory=dict) + + class ChatResponse(BaseModel): session_id: str status: Literal["completed", "failed", "awaiting_human_review"] @@ -19,8 +28,13 @@ class ChatResponse(BaseModel): final_answer: str | None = None follow_up_questions: list[str] = Field(default_factory=list) usage: dict[str, Any] = Field(default_factory=dict) + literature: dict[str, LiteratureToolResult] | None = None class PendingResumeResponse(BaseModel): + # Declared, not just passed: the service has always handed a `status` to + # this model, but without a field for it pydantic dropped it silently, so + # the one response that most needs to say what it is said nothing. + status: Literal["awaiting_human_review"] = "awaiting_human_review" session_id: str question: str mode: SearchMode diff --git a/crossbar_llm/api/services/agent_service.py b/crossbar_llm/api/services/agent_service.py index 6020720..a101916 100644 --- a/crossbar_llm/api/services/agent_service.py +++ b/crossbar_llm/api/services/agent_service.py @@ -1,18 +1,25 @@ +import asyncio +from typing import Any + from fastapi import HTTPException, status, UploadFile from langgraph.graph.state import CompiledStateGraph from langgraph.types import Command -from crossbar_llm.agent_tools.callback_handler import UsageMetricsCallback +from crossbar_llm.agent_tools.callback_handler import ( + UsageMetricsCallback, + merge_usage_summaries, +) from crossbar_llm.agent_tools.config import LLMConfig, Neo4jConfig, ReasoningConfig from crossbar_llm.agent_tools.cypher_agent import CypherAgent, CypherAgentState -from crossbar_llm.api.schemas.common import SearchMode +from crossbar_llm.api.schemas.common import ExecutionControl, SearchMode from crossbar_llm.api.schemas.requests import VectorSearchRequest, UploadVectorSearchRequest, DbSearchRequest, ResumeRequest from crossbar_llm.api.core.settings import Settings from crossbar_llm.api.services.session_store import session_store, SessionStore from crossbar_llm.api.schemas.responses import ChatResponse, PendingResumeResponse from crossbar_llm.api.services.fileguard import FileGuard +from crossbar_llm.api.services.literature_service import LiteratureService class AgentService: @@ -25,6 +32,10 @@ def __init__( self.neo4j_config = Neo4jConfig() self.settings = settings self.session_store = session_store + self.literature_service = LiteratureService(settings) + + async def aclose(self) -> None: + await self.literature_service.aclose() def _build_agent_graph( self, @@ -32,8 +43,13 @@ def _build_agent_graph( session_id: str, browser_id: str, chat_request: DbSearchRequest | VectorSearchRequest | ResumeRequest, - resume: bool = False - ) -> tuple[CompiledStateGraph, UsageMetricsCallback, dict[str, dict[str, str]]]: + usage_callback: UsageMetricsCallback | None = None, + ) -> tuple[ + CompiledStateGraph, + CypherAgent, + UsageMetricsCallback, + dict[str, dict[str, str]], + ]: session = self.session_store.get_session(session_id=session_id, browser_id=browser_id) if not session: @@ -42,7 +58,8 @@ def _build_agent_graph( detail=f"Session with ID {session_id} not found for browser ID {browser_id}.", ) - usage_callback = UsageMetricsCallback(session_id=session_id) + if usage_callback is None: + usage_callback = UsageMetricsCallback(session_id=session_id) llm_config = LLMConfig( model=chat_request.model, @@ -64,7 +81,121 @@ def _build_agent_graph( graph = agent.build_graph(checkpointer=session.checkpointer) config = {"configurable": {"thread_id": session_id}} - return graph, usage_callback, config + return graph, agent, usage_callback, config + + @staticmethod + def _literature_enabled(payload) -> bool: + tools = payload.literature_tools + return tools.paperclip or tools.pubtator3 + + @staticmethod + def _new_literature_callback(session_id: str) -> UsageMetricsCallback: + """A separate, lenient usage handler for the literature agents. + + Their structured-output path falls back to plain JSON on providers that + don't support function calling, and those responses legitimately arrive + without usage metadata — which the strict handler turns into a raised + error that kills the call. Keeping this on its own handler means that + leniency applies where it is warranted and does NOT quietly disable the + core agent's strict accounting, which is what guards the token bill. + `merge_usage_summaries` recombines the two for the response. + """ + return UsageMetricsCallback(session_id=session_id, strict=False) + + def _usage_summary( + self, + usage_callback: UsageMetricsCallback, + literature_callback: UsageMetricsCallback | None, + ) -> dict[str, Any]: + if literature_callback is None: + return usage_callback.get_summary() + return merge_usage_summaries( + usage_callback.get_summary(), literature_callback.get_summary() + ) + + async def _gather_core_and_literature( + self, + *, + graph: CompiledStateGraph, + initial_state: dict[str, Any], + config: dict[str, Any], + question: str, + payload, + literature_callback: UsageMetricsCallback, + ) -> tuple[dict[str, Any], dict[str, Any] | None]: + """Run the Cypher graph and the literature agents concurrently. + + Both sides are real tasks so that a failure on either one cancels the + other. A plain `gather()` would leave the survivor running detached + after the request had already failed — the literature agents would keep + spending metered Paperclip calls and LLM tokens on a response nobody + will ever receive, and their eventual exception would surface as a bare + "Task exception was never retrieved". + """ + core_task = asyncio.create_task(graph.ainvoke(initial_state, config=config)) + literature_task = asyncio.create_task( + self.literature_service.run( + question=question, + payload=payload, + callback=literature_callback, + ) + ) + tasks = (core_task, literature_task) + try: + result, literature = await asyncio.gather(*tasks) + except BaseException: + # Also covers the client disconnecting: Starlette cancels the + # handler, which cancels this gather, and we stop the work. + for task in tasks: + task.cancel() + await asyncio.gather(*tasks, return_exceptions=True) + raise + return result, literature + + async def _run_initial_request( + self, + *, + session_id: str, + browser_id: str, + payload: DbSearchRequest | VectorSearchRequest | UploadVectorSearchRequest, + initial_state: dict[str, Any], + usage_callback: UsageMetricsCallback, + ) -> tuple[dict[str, Any], dict[str, Any] | None, UsageMetricsCallback | None]: + run_literature = ( + self._literature_enabled(payload) + and payload.execution_mode == ExecutionControl.GENERATE_AND_RUN + ) + graph, agent, usage_callback, config = self._build_agent_graph( + session_id=session_id, + browser_id=browser_id, + chat_request=payload, + usage_callback=usage_callback, + ) + + if not run_literature: + return await graph.ainvoke(initial_state, config=config), None, None + + # Decide relevance up front so an out-of-domain question never reaches + # the literature tools, which would spend metered external quota on a + # question the graph is about to reject anyway. The verdict is seeded + # into the state, and the relevance node reuses it rather than paying + # for the same call twice — so the graph still runs end to end and + # writes its checkpoint exactly as it does without literature enabled. + relevance = await agent.avalidate_biological_relevance(payload.question) + initial_state.update(relevance) + if relevance["biological_relevance"] is False: + return await graph.ainvoke(initial_state, config=config), None, None + + literature_callback = self._new_literature_callback(session_id) + result, literature = await self._gather_core_and_literature( + graph=graph, + initial_state=initial_state, + config=config, + question=payload.question, + payload=payload, + literature_callback=literature_callback, + ) + return result, literature, literature_callback def _base_state( @@ -100,20 +231,26 @@ def _base_state( "execution_mode": execution_mode, } - def _to_response(self, session_id: str, cypher_mode: SearchMode, result: CypherAgentState, usage_callback: UsageMetricsCallback) -> ChatResponse | PendingResumeResponse: - - + def _to_response( + self, + session_id: str, + cypher_mode: SearchMode, + result: CypherAgentState, + usage_callback: UsageMetricsCallback, + literature: dict[str, Any] | None = None, + literature_callback: UsageMetricsCallback | None = None, + ) -> ChatResponse | PendingResumeResponse: + + if result.get("__interrupt__"): - status = "awaiting_human_review" interrupts = result["__interrupt__"][0].value return PendingResumeResponse( session_id=session_id, - status=status, question=interrupts.get("question"), mode=cypher_mode, generated_cypher=interrupts.get("current_cypher"), ) - + elif result.get("is_ok", False) is False: status = "failed" else: @@ -128,43 +265,56 @@ def _to_response(self, session_id: str, cypher_mode: SearchMode, result: CypherA execution_result=result.get("execution_result"), final_answer=result.get("final_answer"), follow_up_questions=result.get("follow_up_questions", []), - usage=usage_callback.get_summary() + # One request-level total covering both the core agent and any + # literature agents. `literature[<tool>].usage` breaks this down + # per tool — it is a slice of this number, not an addition to it. + usage=self._usage_summary(usage_callback, literature_callback), + literature=literature or None, ) - def run_db( + async def run_db( self, session_id: str, browser_id: str, payload: DbSearchRequest ) -> ChatResponse | PendingResumeResponse: - graph, usage_callback, config = self._build_agent_graph(session_id=session_id, browser_id=browser_id, chat_request=payload) - - result = graph.invoke( - self._base_state( + usage_callback = UsageMetricsCallback(session_id=session_id) + result, literature, literature_callback = await self._run_initial_request( + session_id=session_id, + browser_id=browser_id, + payload=payload, + usage_callback=usage_callback, + initial_state=self._base_state( question=payload.question, execution_mode=payload.execution_mode, cypher_mode=SearchMode.DB_SEARCH, ), - config=config ) - + interrupts = result.get("__interrupt__") pending = bool(interrupts) if pending: pending_cypher = interrupts[0].value.get("current_cypher") else: pending_cypher = None - + self.session_store.mark_resume_pending( - session_id=session_id, - browser_id=browser_id, - pending=pending, + session_id=session_id, + browser_id=browser_id, + pending=pending, pending_cypher=pending_cypher ) - return self._to_response(session_id, SearchMode.DB_SEARCH, result, usage_callback) + return self._to_response( + session_id, + SearchMode.DB_SEARCH, + result, + usage_callback, + literature, + literature_callback, + ) - def resume( + async def resume( self, session_id: str, browser_id: str, @@ -193,34 +343,67 @@ def resume( ) - graph, usage_callback, config = self._build_agent_graph(session_id=session_id, browser_id=browser_id, chat_request=payload, resume=True) + usage_callback = UsageMetricsCallback(session_id=session_id) + graph, _, usage_callback, config = self._build_agent_graph( + session_id=session_id, + browser_id=browser_id, + chat_request=payload, + usage_callback=usage_callback, + ) - result = graph.invoke( + result = await graph.ainvoke( Command(resume=payload.model_dump(include={"action", "edited_cypher"})), config=config ) self.session_store.mark_resume_pending(session_id=session_id, browser_id=browser_id, pending=False) - return self._to_response(session_id, payload.search_mode, result, usage_callback) + # Literature runs after the Cypher approval rather than alongside it: + # in generate mode the question sits idle awaiting review, and paying + # for literature evidence before the user has approved anything would + # bill work the user may well discard. + literature = None + literature_callback = None + if self._literature_enabled(payload): + literature_callback = self._new_literature_callback(session_id) + # `question` comes from the checkpointed state; a resume request has + # no question of its own. An empty one still reaches the service so + # it can report "skipped" per tool instead of silently returning + # nothing to a user who explicitly asked for these tools. + literature = await self.literature_service.run( + question=result.get("question") or "", + payload=payload, + callback=literature_callback, + ) + + return self._to_response( + session_id, + payload.search_mode, + result, + usage_callback, + literature, + literature_callback, + ) - def run_vector( + async def run_vector( self, session_id: str, browser_id: str, payload: VectorSearchRequest ) -> ChatResponse | PendingResumeResponse: - graph, usage_callback, config = self._build_agent_graph(session_id=session_id, browser_id=browser_id, chat_request=payload) - - result = graph.invoke( - self._base_state( + usage_callback = UsageMetricsCallback(session_id=session_id) + result, literature, literature_callback = await self._run_initial_request( + session_id=session_id, + browser_id=browser_id, + payload=payload, + usage_callback=usage_callback, + initial_state=self._base_state( question=payload.question, execution_mode=payload.execution_mode, cypher_mode=SearchMode.VECTOR_SEARCH, vector_index=payload.vector_index, ), - config=config ) interrupts = result.get("__interrupt__") @@ -236,7 +419,14 @@ def run_vector( pending=pending, pending_cypher=pending_cypher ) - return self._to_response(session_id, SearchMode.VECTOR_SEARCH, result, usage_callback) + return self._to_response( + session_id, + SearchMode.VECTOR_SEARCH, + result, + usage_callback, + literature, + literature_callback, + ) async def run_vector_upload( self, @@ -249,16 +439,19 @@ async def run_vector_upload( guard = FileGuard(settings=self.settings, vector_index=payload.vector_index) embedding_array = await guard.load_embedding(embedding_file) - graph, usage_callback, config = self._build_agent_graph(session_id=session_id, browser_id=browser_id, chat_request=payload) - result = graph.invoke( - self._base_state( + usage_callback = UsageMetricsCallback(session_id=session_id) + result, literature, literature_callback = await self._run_initial_request( + session_id=session_id, + browser_id=browser_id, + payload=payload, + usage_callback=usage_callback, + initial_state=self._base_state( question=payload.question, execution_mode=payload.execution_mode, cypher_mode=SearchMode.VECTOR_SEARCH, vector_index=payload.vector_index, embedding=embedding_array.tolist(), ), - config=config ) interrupts = result.get("__interrupt__") @@ -274,7 +467,14 @@ async def run_vector_upload( pending=pending, pending_cypher=pending_cypher ) - return self._to_response(session_id, SearchMode.VECTOR_SEARCH, result, usage_callback) + return self._to_response( + session_id, + SearchMode.VECTOR_SEARCH, + result, + usage_callback, + literature, + literature_callback, + ) diff --git a/crossbar_llm/api/services/literature_service.py b/crossbar_llm/api/services/literature_service.py new file mode 100644 index 0000000..f7d0633 --- /dev/null +++ b/crossbar_llm/api/services/literature_service.py @@ -0,0 +1,418 @@ +"""Orchestration for the optional Paperclip and PubTator3 agents.""" + +from __future__ import annotations + +import asyncio +from typing import Any, Awaitable, Callable + +from crossbar_llm.agent_tools.callback_handler import UsageMetricsCallback +from crossbar_llm.agent_tools.config import ReasoningConfig +from crossbar_llm.agent_tools.logging_config import get_logger +from crossbar_llm.api.core.settings import Settings +from crossbar_llm.api.schemas.requests import LiteratureToolsConfig, ModelConfigRequest +from crossbar_llm.api.schemas.responses import LiteratureToolResult +from crossbar_llm.paperclip_tools.adapter import PaperclipAdapter, PaperclipConfigError +from crossbar_llm.paperclip_tools.agent import build_graph as build_paperclip_graph +from crossbar_llm.paperclip_tools.llm import build_chat_model as build_paperclip_model +from crossbar_llm.pubtator3_tools.agent import build_graph as build_pubtator3_graph +from crossbar_llm.pubtator3_tools.llm import build_chat_model as build_pubtator3_model +from crossbar_llm.pubtator3_tools.rate_limit import install_static_share_limiter + +logger = get_logger(__name__) + + +class _AdmissionRejected(Exception): + """This process had no capacity to start the tool within the wait window.""" + + def __init__(self, tool: str): + super().__init__(f"{tool} was not admitted") + self.tool = tool + +# Token counters shared with `UsageCounter`. `call_count` is deliberately NOT +# here: the core agent's `aggregated_usage.totals` carries exactly these keys, +# and a per-tool total that quietly grew an extra field would be a different +# shape wearing the same name. +_USAGE_TOTAL_KEYS = ( + "input_tokens", + "output_tokens", + "total_tokens", + "cache_read", + "cache_write", + "reasoning", +) + + +class LiteratureService: + """Run enabled literature agents concurrently on the FastAPI event loop.""" + + def __init__(self, settings: Settings): + self.settings = settings + # Keep the adapter long-lived once used, but do not construct it for + # requests that leave Paperclip disabled. + self.paperclip_adapter: PaperclipAdapter | None = None + # Admission control. Every run fans out several upstream requests, so + # concurrent traffic multiplies through to Paperclip and PubTator3 and + # drains the shared connection pool. Created lazily because a Semaphore + # binds to the running loop. + self._admission: asyncio.Semaphore | None = None + # Divide PubTator3's IP-wide budget across replicas. Configured once at + # construction rather than per request: the limiter is process-global + # state inside the client. + install_static_share_limiter(settings.pubtator3_replica_count) + + def _admission_slot(self) -> asyncio.Semaphore: + if self._admission is None: + self._admission = asyncio.Semaphore( + self.settings.literature_max_concurrent_runs + ) + return self._admission + + async def _run_admitted( + self, + name: str, + runner: Callable[[], Awaitable[dict[str, Any]]], + ) -> dict[str, Any]: + """Run one tool, but only once this process has capacity for it. + + Waiting is deliberately brief. A long queue here would be charged + against the per-tool timeout, so an overloaded server would report + every tool as "timed out" when the truth is that it never started. + """ + try: + await asyncio.wait_for( + self._admission_slot().acquire(), + timeout=self.settings.literature_admission_wait_seconds, + ) + except asyncio.TimeoutError: + raise _AdmissionRejected(name) from None + try: + return await asyncio.wait_for( + runner(), timeout=self.settings.literature_tool_timeout_seconds + ) + finally: + self._admission_slot().release() + + def _get_paperclip_adapter(self) -> PaperclipAdapter: + # No await between the check and the assignment, so concurrent requests + # on the event loop cannot race into building two adapters here. + if self.paperclip_adapter is None: + env_settings = getattr(self.settings, "env_settings", None) + configured_key = getattr(env_settings, "paperclip_api_key", None) + self.paperclip_adapter = PaperclipAdapter( + # Passed in rather than exported to `os.environ`: settings come + # from a .env that pydantic-settings reads privately, and a + # service has no business mutating process-global state to + # smuggle them into a library. + api_key=( + configured_key.get_secret_value() if configured_key else None + ), + disable_rest=getattr(env_settings, "paperclip_disable_rest", None), + max_connections=self.settings.paperclip_max_connections, + ) + return self.paperclip_adapter + + async def aclose(self) -> None: + if self.paperclip_adapter is not None: + await self.paperclip_adapter.aclose() + self.paperclip_adapter = None + + @staticmethod + def _model_kwargs( + payload: ModelConfigRequest, + callback: UsageMetricsCallback, + ) -> dict[str, Any]: + return { + "model": payload.model, + "provider": payload.provider, + "callbacks": [callback], + "reasoning": ReasoningConfig( + enabled=payload.reasoning_enabled, + effort=payload.reasoning_effort, + ), + } + + @staticmethod + def _tool_usage( + summary: dict[str, Any], + prefix: str, + ) -> dict[str, Any]: + """Slice one tool's share out of the shared request-level usage summary. + + Depends on every literature LLM call tagging `node_name` with the + tool's prefix; `test_literature_service.py` pins that contract, because + a renamed node would otherwise empty this out in silence. + + Note the returned figures are also part of the response's top-level + `usage` — this is a breakdown of that total, not an addition to it. + """ + per_node = { + name: record + for name, record in summary.get("per_node_usage", {}).items() + if name.startswith(prefix) + } + if not per_node: + return {} + + totals = { + key: sum((record.get(key, 0) or 0) for record in per_node.values()) + for key in _USAGE_TOTAL_KEYS + } + models_by_node = { + name: summary.get("aggregated_usage", {}) + .get("models_by_node", {}) + .get(name, []) + for name in per_node + } + return { + "per_node_usage": per_node, + "call_count": sum( + (record.get("call_count", 0) or 0) for record in per_node.values() + ), + "aggregated_usage": { + "totals": totals, + "models_by_node": models_by_node, + }, + } + + def _paperclip_citations(self, state: dict[str, Any]) -> list[dict[str, Any]]: + citations = state.get("citations") or [] + return [ + citation.model_dump(mode="json") + if hasattr(citation, "model_dump") + else dict(citation) + for citation in citations[: self.settings.literature_max_citations] + ] + + def _pubtator_citations(self, state: dict[str, Any]) -> list[dict[str, Any]]: + citations: list[dict[str, Any]] = [] + seen_pmids: set[str] = set() + for document in state.get("documents") or []: + if len(citations) >= self.settings.literature_max_citations: + break + pmid = getattr(document, "pmid", None) + if pmid is None: + continue + pmid = str(pmid) + if pmid in seen_pmids: + continue + seen_pmids.add(pmid) + citations.append( + { + "pmid": pmid, + "title": getattr(document, "title", ""), + "pmcid": getattr(document, "pmcid", None), + "url": f"https://pubmed.ncbi.nlm.nih.gov/{pmid}/", + } + ) + return citations + + async def _run_paperclip( + self, + *, + question: str, + payload: ModelConfigRequest, + callback: UsageMetricsCallback, + ) -> dict[str, Any]: + model = build_paperclip_model(**self._model_kwargs(payload, callback)) + graph = build_paperclip_graph( + chat_model=model, + adapter=self._get_paperclip_adapter(), + max_documents=self.settings.paperclip_max_documents, + abstracts_only=self.settings.paperclip_abstracts_only, + use_map=self.settings.paperclip_use_map, + ) + return await graph.ainvoke({"question": question, "warnings": []}) + + async def _run_pubtator3( + self, + *, + question: str, + payload: ModelConfigRequest, + callback: UsageMetricsCallback, + ) -> dict[str, Any]: + model = build_pubtator3_model(**self._model_kwargs(payload, callback)) + graph = build_pubtator3_graph( + chat_model=model, + max_documents=self.settings.pubtator3_max_documents, + abstracts_only=self.settings.pubtator3_abstracts_only, + ) + return await graph.ainvoke({"question": question, "warnings": []}) + + @staticmethod + def _failure_warning(name: str, error: BaseException) -> str: + """User-facing text for a failed tool. + + A missing key is the one case where the upstream message is both safe + and genuinely actionable, so it is passed through. Everything else is + reported by type only — upstream error strings can carry URLs with + credentials in the query, and this text goes straight into an HTTP + response. The full exception is logged. + """ + if isinstance(error, PaperclipConfigError): + return f"{name} is not configured: {error}" + return ( + f"{name} failed with {type(error).__name__}. " + "See the server logs for details." + ) + + def _failed( + self, + name: str, + warning: str, + summary: dict[str, Any], + usage_prefix: str, + ) -> LiteratureToolResult: + return LiteratureToolResult( + status="failed", + warnings=[warning], + # Tokens spent before the failure were still spent — report them. + usage=self._tool_usage(summary, usage_prefix), + ) + + async def run( + self, + *, + question: str, + payload: ModelConfigRequest, + callback: UsageMetricsCallback, + ) -> dict[str, LiteratureToolResult]: + selected: list[tuple[str, Callable[[], Awaitable[dict[str, Any]]], str]] = [] + tools: LiteratureToolsConfig = payload.literature_tools + + if tools.paperclip: + selected.append( + ( + "paperclip", + lambda: self._run_paperclip( + question=question, payload=payload, callback=callback + ), + "paperclip.", + ) + ) + if tools.pubtator3: + selected.append( + ( + "pubtator3", + lambda: self._run_pubtator3( + question=question, payload=payload, callback=callback + ), + "pubtator3.", + ) + ) + + if not selected: + return {} + + if not question or not question.strip(): + # The caller asked for these tools but has no question to run them + # on. Say so rather than returning an empty object that reads as + # "you never enabled anything". + return { + name: LiteratureToolResult( + status="skipped", + warnings=[ + f"{name} was skipped: no question was available for this request." + ], + ) + for name, _, _ in selected + } + + raw_results = await asyncio.gather( + *(self._run_admitted(name, runner) for name, runner, _ in selected), + return_exceptions=True, + ) + + summary = callback.get_summary() + normalized: dict[str, LiteratureToolResult] = {} + for (name, _, usage_prefix), raw in zip(selected, raw_results): + if isinstance(raw, _AdmissionRejected): + # Not a failure of the tool — this server was saturated. Says + # so plainly so the operator sees capacity, not flakiness. + logger.warning( + "Literature tool not admitted", + event_type="literature_tool_not_admitted", + component="LiteratureService.run", + tool=name, + max_concurrent=self.settings.literature_max_concurrent_runs, + ) + normalized[name] = LiteratureToolResult( + status="skipped", + warnings=[ + f"{name} was skipped: the server is at its literature " + f"capacity of {self.settings.literature_max_concurrent_runs} " + "concurrent runs. Try again shortly." + ], + ) + continue + if isinstance(raw, asyncio.TimeoutError): + normalized[name] = self._failed( + name, + f"{name} timed out after " + f"{self.settings.literature_tool_timeout_seconds:g} seconds.", + summary, + usage_prefix, + ) + continue + if isinstance(raw, BaseException): + logger.error( + "Literature tool failed", + event_type="literature_tool_failed", + component="LiteratureService.run", + tool=name, + error_type=type(raw).__name__, + error=str(raw), + exc_info=raw, + ) + normalized[name] = self._failed( + name, self._failure_warning(name, raw), summary, usage_prefix + ) + continue + if not isinstance(raw, dict): + logger.error( + "Literature tool returned an unexpected result type", + event_type="literature_tool_bad_result", + component="LiteratureService.run", + tool=name, + result_type=type(raw).__name__, + ) + normalized[name] = self._failed( + name, + f"{name} returned an unexpected result type " + f"({type(raw).__name__}).", + summary, + usage_prefix, + ) + continue + + try: + citations = ( + self._paperclip_citations(raw) + if name == "paperclip" + else self._pubtator_citations(raw) + ) + normalized[name] = LiteratureToolResult( + status="completed", + answer=raw.get("final_answer"), + citations=citations, + warnings=list(raw.get("warnings") or []), + usage=self._tool_usage(summary, usage_prefix), + ) + except Exception as error: + logger.error( + "Literature tool result normalization failed", + event_type="literature_tool_normalization_failed", + component="LiteratureService.run", + tool=name, + error_type=type(error).__name__, + error=str(error), + exc_info=error, + ) + normalized[name] = self._failed( + name, + f"{name} returned a result this server could not read " + f"({type(error).__name__}).", + summary, + usage_prefix, + ) + + return normalized diff --git a/crossbar_llm/frontend/src/components/ChatLayout.js b/crossbar_llm/frontend/src/components/ChatLayout.js index ee82517..40b883e 100644 --- a/crossbar_llm/frontend/src/components/ChatLayout.js +++ b/crossbar_llm/frontend/src/components/ChatLayout.js @@ -112,6 +112,7 @@ function ChatLayout({ const [topK, setTopK] = useState(10); const [reasoningEnabled, setReasoningEnabled] = useState(false); const [reasoningEffort, setReasoningEffort] = useState('medium'); + const [literatureTools, setLiteratureTools] = useState({ paperclip: false, pubtator3: false }); const [copySnackbar, setCopySnackbar] = useState(false); // Query editing state @@ -172,6 +173,7 @@ function ChatLayout({ const [expandedSections, setExpandedSections] = useState({ examples: true, settings: false, + literature: true, vectorConfig: false, query: true, results: false, @@ -273,6 +275,7 @@ function ChatLayout({ setModelChoices({}); setSupportedModels([]); setModelsLoaded(true); + setError('Could not load model providers. Make sure the backend is running on port 8001, then refresh the page.'); } }; fetchModels(); @@ -515,6 +518,9 @@ function ChatLayout({ if (Array.isArray(detail)) return detail.map(e => e?.msg || JSON.stringify(e)).join(', '); if (typeof detail === 'object') return detail.msg || detail.error || JSON.stringify(detail); } + if (!err?.response && err?.request) { + return 'Cannot reach the backend API. Make sure it is running on port 8001, then refresh the page and try again.'; + } return err?.message || 'An error occurred'; }; @@ -525,8 +531,9 @@ function ChatLayout({ top_k: topK, reasoning_enabled: reasoningEnabled, ...(reasoningEnabled ? { reasoning_effort: reasoningEffort } : {}), + literature_tools: literatureTools, ...overrides, - }), [provider, llmType, topK, reasoningEnabled, reasoningEffort]); + }), [provider, llmType, topK, reasoningEnabled, reasoningEffort, literatureTools]); // Run the right query endpoint for the current mode + optional uploaded vector file. const runQuery = useCallback((sessionIdVal, baseBody, config) => { @@ -534,7 +541,19 @@ function ChatLayout({ const body = { ...baseBody, search_mode: searchMode }; if (semanticSearchEnabled) { const vectorBody = { ...body, vector_category: vectorCategory, embedding_type: embeddingType }; - if (selectedFile) return vectorUploadSearch(sessionIdVal, vectorBody, selectedFile, config); + if (selectedFile) { + const { literature_tools, ...multipartBody } = vectorBody; + return vectorUploadSearch( + sessionIdVal, + { + ...multipartBody, + paperclip: literature_tools?.paperclip || false, + pubtator3: literature_tools?.pubtator3 || false, + }, + selectedFile, + config, + ); + } return vectorSearch(sessionIdVal, vectorBody, config); } return dbSearch(sessionIdVal, body, config); @@ -548,6 +567,7 @@ function ChatLayout({ const result = data?.execution_result || []; const followUps = data?.follow_up_questions || []; const usage = data?.usage || null; + const literature = data?.literature || null; const isSemantic = searchMode === 'vector_search'; setQueryResult(cypher); @@ -555,7 +575,7 @@ function ChatLayout({ setOriginalQuery(cypher); setLastUsage(usage); - setExecutionResult({ result, response: finalAnswer, followUpQuestions: followUps }); + setExecutionResult({ result, response: finalAnswer, followUpQuestions: followUps, literature }); addConversationTurn({ question: userQuestion, @@ -566,6 +586,7 @@ function ChatLayout({ isSemanticSearch: isSemantic, vectorConfig: isSemantic ? { vectorCategory, embeddingType } : null, usage, + literature, status, }); @@ -578,6 +599,65 @@ function ChatLayout({ } }, [vectorCategory, embeddingType, addConversationTurn, hasValidResults, setExecutionResult, setQueryResult]); + const renderLiterature = (literature) => { + if (!literature || Object.keys(literature).length === 0) return null; + return ( + <Box sx={{ mt: 2, display: 'flex', flexDirection: 'column', gap: 1.5 }}> + {Object.entries(literature).map(([tool, result]) => ( + <Paper + key={tool} + variant="outlined" + sx={{ + p: 2, + borderRadius: '12px', + borderColor: result.status === 'completed' + ? alpha(theme.palette.info.main, 0.35) + : alpha(theme.palette.warning.main, 0.4), + backgroundColor: alpha( + result.status === 'completed' ? theme.palette.info.main : theme.palette.warning.main, + 0.04, + ), + }} + > + <Box sx={{ display: 'flex', alignItems: 'center', justifyContent: 'space-between', mb: 1 }}> + <Typography variant="subtitle2" sx={{ fontWeight: 600 }}> + {tool === 'pubtator3' ? 'PubTator3' : 'Paperclip'} evidence + </Typography> + <Chip + size="small" + label={result.status} + color={result.status === 'completed' ? 'info' : 'warning'} + variant="outlined" + /> + </Box> + {result.answer && <ReactMarkdown>{result.answer}</ReactMarkdown>} + {result.warnings?.length > 0 && ( + <Alert severity="warning" sx={{ mt: 1 }}> + {result.warnings.join(' ')} + </Alert> + )} + {result.citations?.length > 0 && ( + <Box sx={{ mt: 1 }}> + <Typography variant="caption" sx={{ fontWeight: 600 }}>Sources</Typography> + {result.citations.map((citation, index) => ( + <Typography key={`${tool}-${index}`} variant="caption" display="block"> + {citation.url ? ( + <a href={citation.url} target="_blank" rel="noreferrer"> + [{index + 1}] {citation.title || citation.pmid || citation.doc_id || 'Publication'} + </a> + ) : ( + `[${index + 1}] ${citation.title || citation.pmid || citation.doc_id || 'Publication'}` + )} + </Typography> + ))} + </Box> + )} + </Paper> + ))} + </Box> + ); + }; + // Render a compact token/usage summary from the agent's usage dict. const renderUsageChips = (usage) => { if (!usage) return null; @@ -1310,6 +1390,8 @@ function ChatLayout({ </Box> </Paper> + {renderLiterature(turn.literature)} + {/* Follow-up Questions - only for latest */} {isLatest && turn.followUpQuestions && turn.followUpQuestions.length > 0 && ( <Box sx={{ mt: 2 }}> @@ -2123,6 +2205,111 @@ function ChatLayout({ </Select> </FormControl> )} + + </Box> + </Collapse> + </Paper> + + {/* Literature Evidence Section */} + <Paper + elevation={0} + sx={{ + mb: 2, + borderRadius: '16px', + border: `1px solid ${literatureTools.paperclip || literatureTools.pubtator3 + ? theme.palette.info.main + : theme.palette.divider}`, + backgroundColor: literatureTools.paperclip || literatureTools.pubtator3 + ? alpha(theme.palette.info.main, 0.025) + : 'background.paper', + overflow: 'hidden', + }} + > + <SectionHeader + title="Literature Evidence" + icon={<AutoAwesomeIcon fontSize="small" color="info" />} + section="literature" + badge={(() => { + const enabledCount = Number(literatureTools.paperclip) + Number(literatureTools.pubtator3); + return enabledCount ? `${enabledCount} enabled` : 'Optional'; + })()} + /> + <Collapse in={expandedSections.literature}> + <Box sx={{ p: 2, pt: 0 }}> + <Typography variant="body2" color="text.secondary" sx={{ mb: 1.5 }}> + Add publication evidence to the knowledge-graph answer. When both tools are enabled, they run in parallel. + </Typography> + + <Box + onClick={() => !isLoading && setLiteratureTools((current) => ({ ...current, paperclip: !current.paperclip }))} + sx={{ + display: 'flex', + alignItems: 'center', + justifyContent: 'space-between', + gap: 1.5, + p: 1.5, + mb: 1, + borderRadius: '10px', + border: `1px solid ${literatureTools.paperclip + ? alpha(theme.palette.info.main, 0.7) + : theme.palette.divider}`, + backgroundColor: literatureTools.paperclip ? alpha(theme.palette.info.main, 0.08) : 'transparent', + cursor: isLoading ? 'default' : 'pointer', + opacity: isLoading ? 0.65 : 1, + '&:hover': isLoading ? {} : { backgroundColor: alpha(theme.palette.info.main, 0.07) }, + }} + > + <Box> + <Typography variant="body2" sx={{ fontWeight: 600 }}>Paperclip</Typography> + <Typography variant="caption" color="text.secondary"> + Broad literature search with citable source links. + </Typography> + </Box> + <Switch + checked={literatureTools.paperclip} + disabled={isLoading} + onClick={(e) => e.stopPropagation()} + onChange={(e) => setLiteratureTools((current) => ({ ...current, paperclip: e.target.checked }))} + color="info" + size="small" + inputProps={{ 'aria-label': 'Enable Paperclip literature search' }} + /> + </Box> + + <Box + onClick={() => !isLoading && setLiteratureTools((current) => ({ ...current, pubtator3: !current.pubtator3 }))} + sx={{ + display: 'flex', + alignItems: 'center', + justifyContent: 'space-between', + gap: 1.5, + p: 1.5, + borderRadius: '10px', + border: `1px solid ${literatureTools.pubtator3 + ? alpha(theme.palette.info.main, 0.7) + : theme.palette.divider}`, + backgroundColor: literatureTools.pubtator3 ? alpha(theme.palette.info.main, 0.08) : 'transparent', + cursor: isLoading ? 'default' : 'pointer', + opacity: isLoading ? 0.65 : 1, + '&:hover': isLoading ? {} : { backgroundColor: alpha(theme.palette.info.main, 0.07) }, + }} + > + <Box> + <Typography variant="body2" sx={{ fontWeight: 600 }}>PubTator3</Typography> + <Typography variant="caption" color="text.secondary"> + NCBI entity- and relation-aware publication evidence. + </Typography> + </Box> + <Switch + checked={literatureTools.pubtator3} + disabled={isLoading} + onClick={(e) => e.stopPropagation()} + onChange={(e) => setLiteratureTools((current) => ({ ...current, pubtator3: e.target.checked }))} + color="info" + size="small" + inputProps={{ 'aria-label': 'Enable PubTator3 literature search' }} + /> + </Box> </Box> </Collapse> </Paper> diff --git a/crossbar_llm/paperclip_tools/adapter.py b/crossbar_llm/paperclip_tools/adapter.py index 60440f7..0445272 100644 --- a/crossbar_llm/paperclip_tools/adapter.py +++ b/crossbar_llm/paperclip_tools/adapter.py @@ -397,8 +397,14 @@ def _api_key() -> str: return key -async def _paperclip_tool(*, timeout_s: float = DEFAULT_TIMEOUT_S): - """Lazily load and cache the single `paperclip` MCP tool for this loop.""" +async def _paperclip_tool(*, timeout_s: float = DEFAULT_TIMEOUT_S, api_key: str | None = None): + """Lazily load and cache the single `paperclip` MCP tool for this loop. + + `api_key` overrides the environment for callers that hold their credentials + in config rather than `os.environ`. The cache is keyed by loop alone, not by + key: one process authenticates as one Paperclip account, so a second key on + the same loop would be a configuration error rather than a case to support. + """ loop = asyncio.get_running_loop() tool = _tools_by_loop.get(loop) if tool is not None: @@ -413,7 +419,7 @@ async def _paperclip_tool(*, timeout_s: float = DEFAULT_TIMEOUT_S): "paperclip": { "transport": "streamable_http", "url": MCP_URL, - "headers": {"X-API-Key": _api_key()}, + "headers": {"X-API-Key": api_key or _api_key()}, "timeout": timeout_s, # Explicit, not left to the library default — see SLOW_TIMEOUT_S. "sse_read_timeout": timeout_s, @@ -901,18 +907,52 @@ def __init__( *, timeout_s: float = DEFAULT_TIMEOUT_S, slow_timeout_s: float = SLOW_TIMEOUT_S, + api_key: str | None = None, + disable_rest: bool | None = None, + max_connections: int = 32, + pool_timeout_s: float = 10.0, ): self._timeout_s = timeout_s self._slow_timeout_s = slow_timeout_s + # How long a call may wait for a free connection when the pool is + # saturated. Kept well below the call timeouts on purpose: passing a + # bare float to httpx sets connect/read/write/pool to the SAME value, + # so queueing silently consumed the whole 60s (or 480s) budget and + # surfaced as "Paperclip is slow" rather than "we are out of + # connections". A short, separate pool timeout makes saturation fail + # fast and legibly. + self._pool_timeout_s = pool_timeout_s + # Explicit credentials beat the environment. Callers that load config + # from a file (the API reads `.env` through pydantic-settings, which + # never exports to `os.environ`) can hand them over directly instead of + # mutating process-global state to get them here. + self._api_key = api_key + self._disable_rest = disable_rest + self._max_connections = max_connections # One pooled HTTP client per event loop, per adapter. Per-loop because - # an AsyncClient binds to the loop it was created on; per-adapter (not - # module-global) because `build_graph` constructs one adapter per run, - # so pooling stays scoped to a single user's request rather than shared - # across everyone. + # an AsyncClient binds to the loop it was created on; per-adapter + # because an adapter may be either request-scoped (`build_graph` + # constructs one per run when none is injected) or process-wide (the + # API injects a single long-lived adapter). `max_connections` is a cap + # on THIS adapter, so a shared one needs it raised: it then bounds + # every concurrent request at once rather than one request's fan-out. self._clients: "weakref.WeakKeyDictionary[asyncio.AbstractEventLoop, httpx.AsyncClient]" = ( weakref.WeakKeyDictionary() ) + def _rest_disabled(self) -> bool: + """Whether the REST transport is turned off for this adapter. + + An explicit `disable_rest=False` wins over the environment, so an API + caller can force REST on regardless of ambient configuration. Only when + nothing was passed does the env var decide — and there, note that the + check is presence-and-truthiness, so `PAPERCLIP_DISABLE_REST=false` + disables REST just like `=1` does. Prefer passing the flag. + """ + if self._disable_rest is not None: + return self._disable_rest + return bool(os.environ.get(DISABLE_REST_ENV)) + def _rest_client(self) -> "httpx.AsyncClient": """The pooled client for this loop, created on first use. @@ -925,8 +965,11 @@ def _rest_client(self) -> "httpx.AsyncClient": if client is None or client.is_closed: client = httpx.AsyncClient( # Keep-alive headroom for the fan-out; `max_connections` caps - # how hard one request can hit Paperclip concurrently. - limits=httpx.Limits(max_connections=32, max_keepalive_connections=16), + # how hard this adapter can hit Paperclip concurrently. + limits=httpx.Limits( + max_connections=self._max_connections, + max_keepalive_connections=max(1, self._max_connections // 2), + ), follow_redirects=True, ) self._clients[loop] = client @@ -957,7 +1000,9 @@ async def _run(self, command: str) -> str: costs nothing on the common path, and avoids under-timing out a `map` call that happens to fall back to MCP. """ - tool = await _paperclip_tool(timeout_s=self._slow_timeout_s) + tool = await _paperclip_tool( + timeout_s=self._slow_timeout_s, api_key=self._api_key + ) try: out = await tool.ainvoke({"command": command}) except PaperclipError: @@ -984,11 +1029,19 @@ async def _run_rest(self, verb: str, raw: str) -> dict: client's headers, so rotating `PAPERCLIP_API_KEY` takes effect without rebuilding the client. """ - if os.environ.get(DISABLE_REST_ENV): + if self._rest_disabled(): raise PaperclipRestUnavailable(f"disabled via {DISABLE_REST_ENV}") - key = _api_key() - - timeout = self._slow_timeout_s if verb in _SLOW_COMMANDS else self._timeout_s + key = self._api_key or _api_key() + + call_timeout = self._slow_timeout_s if verb in _SLOW_COMMANDS else self._timeout_s + # Explicit per-phase timeouts: `pool` (waiting for a free connection) + # gets its own short budget instead of inheriting the call timeout. + timeout = httpx.Timeout( + connect=call_timeout, + read=call_timeout, + write=call_timeout, + pool=self._pool_timeout_s, + ) try: resp = await self._rest_client().post( REST_URL, @@ -996,6 +1049,16 @@ async def _run_rest(self, verb: str, raw: str) -> dict: headers={"X-API-Key": key}, timeout=timeout, ) + except httpx.PoolTimeout as e: + # Distinct from an upstream failure: Paperclip is fine, we ran out + # of local connections. Raised as RestUnavailable so the caller's + # existing MCP fallback still applies, but named so the logs say + # which it was. + raise PaperclipRestUnavailable( + f"connection pool exhausted after {self._pool_timeout_s:g}s " + f"(max_connections={self._max_connections}); too many concurrent " + f"Paperclip requests in this process" + ) from e except Exception as e: # network errors, timeouts raise PaperclipRestUnavailable(f"{type(e).__name__}: {e}") from e # 401/429 are account-level verdicts, not transport failures: MCP @@ -1251,7 +1314,7 @@ async def filter(self, search_id: str, query: str) -> SearchResult | None: it just adds an `ERR:`-prefixed warning on top of the same (possibly empty) result — no simpler than handling an empty result ourselves. """ - if os.environ.get(DISABLE_REST_ENV): + if self._rest_disabled(): return None raw = f"--from {search_id} {_shell_quote(query)}" try: diff --git a/crossbar_llm/paperclip_tools/agent.py b/crossbar_llm/paperclip_tools/agent.py index 64e0ba8..a6966ba 100644 --- a/crossbar_llm/paperclip_tools/agent.py +++ b/crossbar_llm/paperclip_tools/agent.py @@ -257,6 +257,7 @@ async def router_node(state: PaperclipState) -> dict: "Return ONLY a valid JSON object for that schema. Do not use " "Markdown, prose, tool calls, or extra keys." ), + metadata={"node_name": "paperclip.router"}, ) if used_json_fallback: warnings.append( @@ -338,11 +339,14 @@ async def synthesize_node(state: PaperclipState) -> dict: ), ]) chain = prompt | chat_model - msg = await chain.ainvoke({ - "question": state["question"], - "evidence": evidence, - "chat_history": state.get("chat_history", []), - }) + msg = await chain.ainvoke( + { + "question": state["question"], + "evidence": evidence, + "chat_history": state.get("chat_history", []), + }, + config={"metadata": {"node_name": "paperclip.synthesize"}}, + ) answer = msg.content if isinstance(msg.content, str) else str(msg.content) answer, degenerate_runs = _collapse_degenerate_runs(answer) if degenerate_runs: @@ -396,6 +400,7 @@ async def evaluate_depth_node(state: PaperclipState) -> dict: "Return ONLY a valid JSON object for the depth-evaluation " "schema. Do not use Markdown, prose, tool calls, or extra keys." ), + metadata={"node_name": "paperclip.evaluate_depth"}, ) if used_json_fallback: warnings.append( @@ -427,7 +432,13 @@ async def evaluate_depth_node(state: PaperclipState) -> dict: "warnings": warnings, } - def _post_evaluate_route(state: PaperclipState) -> str: + async def _question_type_route(state: PaperclipState) -> str: + return state["question_type"] + + async def _post_sql_route(state: PaperclipState) -> str: + return "search" if state.get("sql_error") else "synthesize" + + async def _post_evaluate_route(state: PaperclipState) -> str: if state.get("depth_sufficient", True): return "end" if not state.get("refinement_attempted"): @@ -446,7 +457,7 @@ def _post_evaluate_route(state: PaperclipState) -> str: g.set_entry_point("router") g.add_conditional_edges( "router", - lambda s: s["question_type"], + _question_type_route, { "out_of_scope": END, "keyword_search": "search", @@ -462,7 +473,7 @@ def _post_evaluate_route(state: PaperclipState) -> str: # same shape as search_node's own zero-result fallback. g.add_conditional_edges( "sql", - lambda s: "search" if s.get("sql_error") else "synthesize", + _post_sql_route, {"search": "search", "synthesize": "synthesize"}, ) g.add_edge("search", "filter") diff --git a/crossbar_llm/paperclip_tools/structured_output.py b/crossbar_llm/paperclip_tools/structured_output.py index c95da77..35233c7 100644 --- a/crossbar_llm/paperclip_tools/structured_output.py +++ b/crossbar_llm/paperclip_tools/structured_output.py @@ -67,6 +67,7 @@ async def _ainvoke_structured_with_json_fallback( schema: type[StructuredModel], values: dict[str, Any], json_instruction: str, + metadata: dict[str, Any] | None = None, ) -> tuple[StructuredModel, bool]: """Use provider structured output first, then retry as plain JSON. @@ -78,7 +79,10 @@ async def _ainvoke_structured_with_json_fallback( structured_error: Exception | None = None try: chain = prompt | chat_model.with_structured_output(schema) - parsed = await chain.ainvoke(values) + if metadata: + parsed = await chain.ainvoke(values, config={"metadata": metadata}) + else: + parsed = await chain.ainvoke(values) if parsed is not None: if isinstance(parsed, schema): return parsed, False @@ -91,7 +95,11 @@ async def _ainvoke_structured_with_json_fallback( json_prompt = prompt + HumanMessagePromptTemplate.from_template(json_instruction) try: - msg = await (json_prompt | chat_model).ainvoke(values) + chain = json_prompt | chat_model + if metadata: + msg = await chain.ainvoke(values, config={"metadata": metadata}) + else: + msg = await chain.ainvoke(values) data = _extract_json_object(_message_content_to_text(msg)) return schema.model_validate(data), True except Exception as json_error: diff --git a/crossbar_llm/paperclip_tools/tests/test_adapter.py b/crossbar_llm/paperclip_tools/tests/test_adapter.py index c7c1cc8..05beb1c 100644 --- a/crossbar_llm/paperclip_tools/tests/test_adapter.py +++ b/crossbar_llm/paperclip_tools/tests/test_adapter.py @@ -500,12 +500,18 @@ def fake_post(url, json, headers, timeout): _patch_rest(monkeypatch, fake_post) - adapter = PaperclipAdapter(timeout_s=60.0, slow_timeout_s=300.0) + adapter = PaperclipAdapter(timeout_s=60.0, slow_timeout_s=300.0, pool_timeout_s=10.0) await adapter._run_rest("search", '"x" -n 5') - assert captured["timeout"] == 60.0 + # Now an httpx.Timeout rather than a bare float, so that waiting for a free + # connection gets its own short budget instead of inheriting this one — but + # the per-call budget itself is unchanged. + assert captured["timeout"].read == 60.0 + assert captured["timeout"].connect == 60.0 + assert captured["timeout"].pool == 10.0 await adapter._run_rest("map", '--from s_x "q"') - assert captured["timeout"] == 300.0 + assert captured["timeout"].read == 300.0 + assert captured["timeout"].pool == 10.0 # --- sql routing (docs/paperclip_rest_endpoint_findings.md §12) -------------- diff --git a/crossbar_llm/paperclip_tools/tests/test_pool_timeout.py b/crossbar_llm/paperclip_tools/tests/test_pool_timeout.py new file mode 100644 index 0000000..d48579d --- /dev/null +++ b/crossbar_llm/paperclip_tools/tests/test_pool_timeout.py @@ -0,0 +1,75 @@ +"""Connection-pool saturation must fail fast and say what happened. + +httpx expands a bare float timeout to connect/read/write/pool all at once, so +queueing for a free connection used to be charged against the 60s (or 480s) +call budget. Under load that surfaced as "Paperclip is slow" rather than "this +process is out of connections", and it quietly consumed the caller's per-tool +timeout. +""" +from __future__ import annotations + +import httpx +import pytest + +from crossbar_llm.paperclip_tools.adapter import ( + PaperclipAdapter, + PaperclipRestUnavailable, +) + + +class _RecordingClient: + """Captures the Timeout object the adapter hands to httpx.""" + + def __init__(self, raises: Exception | None = None): + self.raises = raises + self.timeout = None + + async def post(self, url, **kwargs): + self.timeout = kwargs.get("timeout") + if self.raises is not None: + raise self.raises + request = httpx.Request("POST", url) + return httpx.Response(200, json={"output": "ok"}, request=request) + + +async def test_pool_timeout_is_separate_from_the_call_timeout(monkeypatch): + adapter = PaperclipAdapter( + api_key="k", disable_rest=False, timeout_s=60.0, pool_timeout_s=10.0 + ) + client = _RecordingClient() + monkeypatch.setattr(adapter, "_rest_client", lambda: client) + + await adapter._run_rest("search", "anything") + + assert isinstance(client.timeout, httpx.Timeout) + assert client.timeout.read == 60.0 + # The whole point: queueing gets its own, much shorter budget. + assert client.timeout.pool == 10.0 + + +async def test_saturation_is_reported_as_pool_exhaustion(monkeypatch): + adapter = PaperclipAdapter( + api_key="k", disable_rest=False, max_connections=4, pool_timeout_s=10.0 + ) + client = _RecordingClient(raises=httpx.PoolTimeout("no free connection")) + monkeypatch.setattr(adapter, "_rest_client", lambda: client) + + with pytest.raises(PaperclipRestUnavailable) as excinfo: + await adapter._run_rest("search", "anything") + + message = str(excinfo.value) + assert "pool exhausted" in message + # Names the knob an operator would actually turn. + assert "max_connections=4" in message + + +async def test_pool_size_is_configurable(monkeypatch): + adapter = PaperclipAdapter(api_key="k", max_connections=64) + client = adapter._rest_client() + try: + transport = client._transport + pool = transport._pool + assert pool._max_connections == 64 + assert pool._max_keepalive_connections == 32 + finally: + await client.aclose() diff --git a/crossbar_llm/pubtator3_tools/agent.py b/crossbar_llm/pubtator3_tools/agent.py index f961fb8..9b39399 100644 --- a/crossbar_llm/pubtator3_tools/agent.py +++ b/crossbar_llm/pubtator3_tools/agent.py @@ -133,6 +133,7 @@ async def _invoke_router(state: PubTator3State): "Return ONLY a valid JSON object for that schema. Do not use " "Markdown, prose, tool calls, or extra keys." ), + metadata={"node_name": "pubtator3.router"}, ) async def router_node(state: PubTator3State) -> dict: @@ -272,12 +273,15 @@ async def synthesize_node(state: PubTator3State) -> dict: ), ]) chain = prompt | chat_model - msg = await chain.ainvoke({ - "question": state["question"], - "evidence": evidence, - "availability_note": availability_note, - "chat_history": state.get("chat_history", []), - }) + msg = await chain.ainvoke( + { + "question": state["question"], + "evidence": evidence, + "availability_note": availability_note, + "chat_history": state.get("chat_history", []), + }, + config={"metadata": {"node_name": "pubtator3.synthesize"}}, + ) answer = msg.content if isinstance(msg.content, str) else str(msg.content) return {"final_answer": answer} @@ -332,6 +336,7 @@ async def evaluate_depth_node(state: PubTator3State) -> dict: "Return ONLY a valid JSON object for the depth-evaluation " "schema. Do not use Markdown, prose, tool calls, or extra keys." ), + metadata={"node_name": "pubtator3.evaluate_depth"}, ) if used_json_fallback: warnings.append( @@ -388,7 +393,10 @@ async def evaluate_depth_node(state: PubTator3State) -> dict: "warnings": warnings, } - def _post_evaluate_route(state: PubTator3State) -> str: + async def _question_type_route(state: PubTator3State) -> str: + return state["question_type"] + + async def _post_evaluate_route(state: PubTator3State) -> str: if state.get("depth_sufficient", True): return "end" if not state.get("refinement_attempted"): @@ -409,7 +417,7 @@ def _post_evaluate_route(state: PubTator3State) -> str: g.set_entry_point("router") g.add_conditional_edges( "router", - lambda s: s["question_type"], + _question_type_route, { "out_of_scope": END, "single_node": "resolve", @@ -420,7 +428,7 @@ def _post_evaluate_route(state: PubTator3State) -> str: ) g.add_conditional_edges( "resolve", - lambda s: s["question_type"], + _question_type_route, { "single_node": "search", "relation_known_pair": "search", diff --git a/crossbar_llm/pubtator3_tools/client.py b/crossbar_llm/pubtator3_tools/client.py index 42fc778..0e8e719 100644 --- a/crossbar_llm/pubtator3_tools/client.py +++ b/crossbar_llm/pubtator3_tools/client.py @@ -7,7 +7,7 @@ API docs: https://www.ncbi.nlm.nih.gov/research/pubtator3/api """ from pydantic import BaseModel, ConfigDict, Field, model_validator -from typing import Any, Callable, Literal, TypeVar +from typing import Any, AsyncContextManager, Callable, Literal, TypeVar import httpx import asyncio import logging @@ -205,8 +205,34 @@ def _client() -> httpx.AsyncClient: _clients_by_loop[loop] = cli return cli -def _limiter() -> AsyncLimiter: - """Lazy per-loop rate limiter — RATE_LIMIT_PER_SECOND req/s, IP-wide.""" +# Swappable limiter backend. The default below is correct for ONE process; the +# 3 req/s ceiling is NCBI's IP-wide policy, so several replicas behind one +# egress IP each enforcing it independently means NCBI sees 3 x replicas. Point +# this at a shared implementation (see `rate_limit.py`) before running more +# than one instance. +_limiter_factory: "Callable[[], AsyncContextManager[Any]] | None" = None + + +def set_limiter_factory(factory: "Callable[[], AsyncContextManager[Any]] | None") -> None: + """Install a shared rate limiter, or pass None to restore the default. + + The factory is called per request and must return an async context manager + that blocks until a token is available. It is a factory rather than a single + instance because the default is bound to the running event loop. + """ + global _limiter_factory + _limiter_factory = factory + + +def _limiter(): + """The rate limiter guarding every PubTator3 call. + + Falls back to a lazy per-loop `AsyncLimiter` — per-loop because an + AsyncLimiter binds to the loop that created it, so a module-level instance + would break across `asyncio.run` calls in tests and scripts. + """ + if _limiter_factory is not None: + return _limiter_factory() loop = asyncio.get_running_loop() lim = _limiters_by_loop.get(loop) if lim is None: diff --git a/crossbar_llm/pubtator3_tools/rate_limit.py b/crossbar_llm/pubtator3_tools/rate_limit.py new file mode 100644 index 0000000..e5c4ac4 --- /dev/null +++ b/crossbar_llm/pubtator3_tools/rate_limit.py @@ -0,0 +1,93 @@ +"""Rate-limiter backends for the PubTator3 client. + +PubTator3's 3 req/s ceiling is enforced by NCBI **per IP**, not per process. The +client's default limiter is per-event-loop, which is exactly right for a single +instance and wrong the moment a second replica shares an egress IP: each one +independently allows 3 req/s and NCBI sees the sum. + +`client.set_limiter_factory()` is the seam. This module holds the backends that +plug into it: + +- `StaticShareLimiter` — divides the budget by a configured replica count. No + new dependencies, never exceeds the ceiling, but wastes budget when replicas + are idle and must be updated when you rescale. +- A shared coordinated limiter (Redis or similar) is the better answer at more + than a couple of replicas; it lives outside this module so the tool package + stays dependency-free. + +Both are async context managers, matching `aiolimiter.AsyncLimiter`. +""" +from __future__ import annotations + +import asyncio +import weakref + +from aiolimiter import AsyncLimiter + +from crossbar_llm.pubtator3_tools.client import ( + RATE_LIMIT_PER_SECOND, + RATE_LIMIT_TIME_PERIOD_S, + set_limiter_factory, +) + + +class StaticShareLimiter: + """Enforce this replica's fixed share of an IP-wide budget. + + With `replica_count=N` each instance allows `RATE_LIMIT_PER_SECOND / N` + requests per second, so the total across N replicas stays at or under the + ceiling no matter how the load balancer distributes work. Correct without + coordination, at the cost of throughput when replicas are unevenly busy. + + Limiters are kept per event loop for the same reason the client's default + is: an `AsyncLimiter` binds to the loop that created it. + """ + + def __init__( + self, + *, + replica_count: int = 1, + max_rate: float = RATE_LIMIT_PER_SECOND, + time_period: float = RATE_LIMIT_TIME_PERIOD_S, + ): + if replica_count < 1: + raise ValueError("replica_count must be at least 1") + self.replica_count = replica_count + self.time_period = time_period + # aiolimiter requires a positive rate, so a very high replica count + # stretches the window rather than rounding the rate down to zero. + self.max_rate = max_rate / replica_count + if self.max_rate < 1: + self.time_period = time_period / self.max_rate + self.max_rate = 1.0 + self._by_loop: "weakref.WeakKeyDictionary[asyncio.AbstractEventLoop, AsyncLimiter]" = ( + weakref.WeakKeyDictionary() + ) + + def __call__(self) -> AsyncLimiter: + loop = asyncio.get_running_loop() + limiter = self._by_loop.get(loop) + if limiter is None: + limiter = AsyncLimiter( + max_rate=self.max_rate, time_period=self.time_period + ) + self._by_loop[loop] = limiter + return limiter + + +def install_static_share_limiter(replica_count: int) -> StaticShareLimiter | None: + """Point the PubTator3 client at a per-replica share of the IP-wide budget. + + A `replica_count` of 1 restores the default limiter rather than installing + an equivalent one, so the single-instance path keeps its original + behaviour exactly. + """ + if replica_count <= 1: + set_limiter_factory(None) + return None + limiter = StaticShareLimiter(replica_count=replica_count) + set_limiter_factory(limiter) + return limiter + + +__all__ = ["StaticShareLimiter", "install_static_share_limiter"] diff --git a/crossbar_llm/pubtator3_tools/structured_output.py b/crossbar_llm/pubtator3_tools/structured_output.py index c95da77..35233c7 100644 --- a/crossbar_llm/pubtator3_tools/structured_output.py +++ b/crossbar_llm/pubtator3_tools/structured_output.py @@ -67,6 +67,7 @@ async def _ainvoke_structured_with_json_fallback( schema: type[StructuredModel], values: dict[str, Any], json_instruction: str, + metadata: dict[str, Any] | None = None, ) -> tuple[StructuredModel, bool]: """Use provider structured output first, then retry as plain JSON. @@ -78,7 +79,10 @@ async def _ainvoke_structured_with_json_fallback( structured_error: Exception | None = None try: chain = prompt | chat_model.with_structured_output(schema) - parsed = await chain.ainvoke(values) + if metadata: + parsed = await chain.ainvoke(values, config={"metadata": metadata}) + else: + parsed = await chain.ainvoke(values) if parsed is not None: if isinstance(parsed, schema): return parsed, False @@ -91,7 +95,11 @@ async def _ainvoke_structured_with_json_fallback( json_prompt = prompt + HumanMessagePromptTemplate.from_template(json_instruction) try: - msg = await (json_prompt | chat_model).ainvoke(values) + chain = json_prompt | chat_model + if metadata: + msg = await chain.ainvoke(values, config={"metadata": metadata}) + else: + msg = await chain.ainvoke(values) data = _extract_json_object(_message_content_to_text(msg)) return schema.model_validate(data), True except Exception as json_error: diff --git a/crossbar_llm/pubtator3_tools/tests/test_graph.py b/crossbar_llm/pubtator3_tools/tests/test_graph.py index 8ceddca..5a8d46f 100644 --- a/crossbar_llm/pubtator3_tools/tests/test_graph.py +++ b/crossbar_llm/pubtator3_tools/tests/test_graph.py @@ -63,7 +63,10 @@ def __init__(self, responses: list[str]): self.responses = responses def with_structured_output(self, schema, **kwargs): - return RunnableLambda(lambda _: None) + async def _return_none(_): + return None + + return RunnableLambda(_return_none) async def __call__(self, _input): return AIMessage(content=self.responses.pop(0)) @@ -1046,4 +1049,3 @@ async def test_router_guard_can_be_disabled(httpx_mock, fx): assert chat_model.calls == 1 assert final["question_type"] == "relation_partner_discovery" assert not any("downgraded" in w for w in final["warnings"]) - diff --git a/crossbar_llm/pubtator3_tools/tests/test_rate_limit.py b/crossbar_llm/pubtator3_tools/tests/test_rate_limit.py new file mode 100644 index 0000000..53bc78c --- /dev/null +++ b/crossbar_llm/pubtator3_tools/tests/test_rate_limit.py @@ -0,0 +1,65 @@ +"""Tests for the swappable PubTator3 rate-limiter backends. + +The 3 req/s ceiling is NCBI's and applies per IP, so these pin the two things +that make multi-replica deployment safe: that the client honours an installed +limiter at all, and that a per-replica share actually divides the budget. +""" +import asyncio +import pytest + +from crossbar_llm.pubtator3_tools import client as pt_client +from crossbar_llm.pubtator3_tools.rate_limit import ( + StaticShareLimiter, + install_static_share_limiter, +) + + +@pytest.fixture(autouse=True) +def _restore_default_limiter(): + yield + pt_client.set_limiter_factory(None) + + +def test_single_replica_keeps_the_default_limiter(): + assert install_static_share_limiter(1) is None + assert pt_client._limiter_factory is None + + +def test_share_divides_the_ip_wide_budget(): + limiter = install_static_share_limiter(3) + assert limiter.max_rate == pytest.approx(pt_client.RATE_LIMIT_PER_SECOND / 3) + assert pt_client._limiter_factory is limiter + + +def test_high_replica_count_stretches_the_window_instead_of_rounding_to_zero(): + # 3 req/s over 12 replicas is 0.25 req/s, which aiolimiter rejects as a + # rate. It must become 1 per 4s, not 0 per second (which never admits). + limiter = StaticShareLimiter(replica_count=12) + assert limiter.max_rate >= 1 + assert limiter.max_rate / limiter.time_period == pytest.approx(0.25) + + +@pytest.mark.asyncio +async def test_installed_limiter_is_used_by_the_client(): + calls = [] + + class _Fake: + async def __aenter__(self): calls.append("acquire"); return self + async def __aexit__(self, *a): return False + + pt_client.set_limiter_factory(lambda: _Fake()) + async with pt_client._limiter(): + pass + assert calls == ["acquire"] + + +@pytest.mark.asyncio +async def test_share_limiter_actually_throttles(): + limiter = StaticShareLimiter(replica_count=3, max_rate=3, time_period=1) + start = asyncio.get_running_loop().time() + for _ in range(3): + async with limiter(): + pass + elapsed = asyncio.get_running_loop().time() - start + # 1 req/s share: three acquisitions cannot complete instantly. + assert elapsed >= 1.5 diff --git a/crossbar_llm/tests/conftest.py b/crossbar_llm/tests/conftest.py new file mode 100644 index 0000000..a1cd095 --- /dev/null +++ b/crossbar_llm/tests/conftest.py @@ -0,0 +1,31 @@ +"""Shared setup for the API-level tests. + +`Settings`/`EnvSettings` are instantiated at import time (module-level defaults +in `session_store` and `agent_service`), so the required secrets must exist in +the environment before any test module is imported. pytest imports conftest +first, which makes this the only reliable place to put them — doing it at the +top of individual test modules works only for whichever module happens to be +imported first. + +These are throwaway values for cookie signing and IP hashing; nothing here +talks to a real service. +""" +import os + +import pytest + +os.environ.setdefault("BROWSER_COOKIE_SECRET", "test-browser-secret") +os.environ.setdefault("RATE_LIMIT_IP_HASH_SECRET", "test-rate-limit-secret") + + +@pytest.fixture(autouse=True) +def _paperclip_looks_unconfigured(monkeypatch): + """Hide any real Paperclip credentials from tests in this directory. + + Scoped to a fixture rather than done at import: clearing these at module + level would strip them from the whole process, and a combined run + (`pytest crossbar_llm/`) would then silently skip the Paperclip live suite, + which gates itself on exactly this variable. + """ + monkeypatch.delenv("PAPERCLIP_API_KEY", raising=False) + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) diff --git a/crossbar_llm/tests/test_agent_service_literature.py b/crossbar_llm/tests/test_agent_service_literature.py new file mode 100644 index 0000000..69b5961 --- /dev/null +++ b/crossbar_llm/tests/test_agent_service_literature.py @@ -0,0 +1,300 @@ +import asyncio +import os +from types import SimpleNamespace +from unittest.mock import AsyncMock + +import pytest + +os.environ.setdefault("BROWSER_COOKIE_SECRET", "test-browser-secret") +os.environ.setdefault("RATE_LIMIT_IP_HASH_SECRET", "test-rate-limit-secret") + +from crossbar_llm.agent_tools.callback_handler import UsageMetricsCallback, UsageRecord +from crossbar_llm.agent_tools.cypher_agent import CypherAgent +from crossbar_llm.api.schemas.requests import DbSearchRequest, LiteratureToolsConfig +from crossbar_llm.api.services.agent_service import AgentService + + +class _Graph: + def __init__(self, result=None, started=None, release=None): + self.result = result or {"is_ok": True, "final_answer": "graph answer"} + self.started = started + self.release = release + + async def ainvoke(self, state, config): + if self.started: + self.started.set() + if self.release: + await self.release.wait() + return {**state, **self.result} + + +def _payload(*, execution_mode="generate_and_run"): + return DbSearchRequest( + provider="openai", + model="gpt-4o-mini", + question="What is the role of EGFR in cancer?", + execution_mode=execution_mode, + literature_tools=LiteratureToolsConfig(paperclip=True, pubtator3=True), + ) + + +def _service(literature_run): + service = AgentService.__new__(AgentService) + service.literature_service = SimpleNamespace(run=literature_run) + return service + + +def test_graph_topology_is_identical_with_and_without_preflight(): + """The preflight must not change the compiled graph. + + A request and its later resume share one checkpointer, so if enabling + literature tools rewired the entry point, the resume would be replaying a + checkpoint written by a differently-shaped graph. + """ + agent = object.__new__(CypherAgent) + agent.benchmark_mode = False + agent.debug_mode = False + + edges = { + (edge.source, edge.target) + for edge in agent.build_graph(checkpointer=None).get_graph().edges + } + + assert ("__start__", "biological_relevance_validation") in edges + + +def test_prevalidated_state_skips_the_second_relevance_call(): + """A seeded verdict must not be re-paid for inside the graph.""" + agent = object.__new__(CypherAgent) + agent.llm_factory = SimpleNamespace( + create_biological_relevance_validator_llm=lambda: (_ for _ in ()).throw( + AssertionError("relevance was re-validated despite a seeded verdict") + ) + ) + + result = agent.biological_relevance_validation_node( + {"question": "What is EGFR?", "biological_relevance": True} + ) + + assert result == {} + + +@pytest.mark.asyncio +async def test_irrelevant_question_skips_core_and_literature(monkeypatch): + async def unexpected_literature(**kwargs): + raise AssertionError("literature tools should not run") + + service = _service(unexpected_literature) + graph = _Graph(result={"is_ok": False}) + agent = SimpleNamespace( + avalidate_biological_relevance=AsyncMock(return_value={ + "biological_relevance": False, + "final_answer": "outside the biological domain", + }) + ) + callback = UsageMetricsCallback("session", strict=False) + monkeypatch.setattr( + service, + "_build_agent_graph", + lambda **kwargs: (graph, agent, callback, {}), + ) + + result, literature, literature_callback = await service._run_initial_request( + session_id="session", + browser_id="browser", + payload=_payload(), + initial_state={"question": "What is the stock price?", "is_ok": False}, + usage_callback=callback, + ) + + assert result["biological_relevance"] is False + assert result["final_answer"] == "outside the biological domain" + assert literature is None + assert literature_callback is None + + +@pytest.mark.asyncio +async def test_core_and_literature_start_in_parallel_after_relevance_gate(monkeypatch): + started = {"core": asyncio.Event(), "literature": asyncio.Event()} + release = asyncio.Event() + + async def literature_run(**kwargs): + started["literature"].set() + await release.wait() + return {"paperclip": {"status": "completed"}} + + service = _service(literature_run) + graph = _Graph(started=started["core"], release=release) + agent = SimpleNamespace( + avalidate_biological_relevance=AsyncMock(return_value={ + "biological_relevance": True, + "final_answer": None, + }) + ) + callback = UsageMetricsCallback("session", strict=False) + monkeypatch.setattr( + service, + "_build_agent_graph", + lambda **kwargs: (graph, agent, callback, {}), + ) + + task = asyncio.create_task( + service._run_initial_request( + session_id="session", + browser_id="browser", + payload=_payload(), + initial_state={"question": "What is EGFR?"}, + usage_callback=callback, + ) + ) + await asyncio.wait_for( + asyncio.gather(*(event.wait() for event in started.values())), + timeout=1, + ) + release.set() + result, literature, literature_callback = await task + + assert result["biological_relevance"] is True + assert literature["paperclip"]["status"] == "completed" + # Literature gets its own lenient handler so the core agent keeps strict + # usage accounting. + assert literature_callback is not None and literature_callback.strict is False + + +@pytest.mark.asyncio +async def test_generate_only_does_not_start_literature(monkeypatch): + async def unexpected_literature(**kwargs): + raise AssertionError("literature tools should wait for resume") + + service = _service(unexpected_literature) + graph = _Graph() + agent = SimpleNamespace() + callback = UsageMetricsCallback("session", strict=False) + + monkeypatch.setattr( + service, + "_build_agent_graph", + lambda **kwargs: (graph, agent, callback, {}), + ) + + result, literature, literature_callback = await service._run_initial_request( + session_id="session", + browser_id="browser", + payload=_payload(execution_mode="generate"), + initial_state={"question": "What is EGFR?"}, + usage_callback=callback, + ) + + assert result["final_answer"] == "graph answer" + assert literature is None + assert literature_callback is None + + +@pytest.mark.asyncio +async def test_core_failure_cancels_literature_instead_of_orphaning_it(): + """A failing Cypher graph must not leave literature agents running. + + `asyncio.gather` without `return_exceptions` propagates the first error but + leaves siblings running, so a bare gather here let the literature agents + carry on spending metered Paperclip calls and LLM tokens long after the + request had already failed. + """ + literature_finished = False + literature_cancelled = False + + async def literature_run(**kwargs): + nonlocal literature_finished, literature_cancelled + try: + await asyncio.sleep(0.5) + literature_finished = True + return {"paperclip": {"status": "completed"}} + except asyncio.CancelledError: + literature_cancelled = True + raise + + class _FailingGraph: + async def ainvoke(self, state, config): + await asyncio.sleep(0.01) + raise RuntimeError("neo4j exploded") + + service = _service(literature_run) + + with pytest.raises(RuntimeError, match="neo4j exploded"): + await service._gather_core_and_literature( + graph=_FailingGraph(), + initial_state={}, + config={}, + question="What is EGFR?", + payload=_payload(), + literature_callback=UsageMetricsCallback("session", strict=False), + ) + + # Give an orphaned task the time it would have needed to finish. + await asyncio.sleep(0.6) + assert literature_cancelled is True + assert literature_finished is False + + +@pytest.mark.asyncio +async def test_literature_failure_cancels_the_core_graph(): + """And the same in the other direction, so no Neo4j work is left dangling.""" + core_cancelled = False + + class _SlowGraph: + async def ainvoke(self, state, config): + nonlocal core_cancelled + try: + await asyncio.sleep(0.5) + return {"is_ok": True} + except asyncio.CancelledError: + core_cancelled = True + raise + + async def literature_run(**kwargs): + await asyncio.sleep(0.01) + raise RuntimeError("literature orchestration bug") + + service = _service(literature_run) + + with pytest.raises(RuntimeError, match="literature orchestration bug"): + await service._gather_core_and_literature( + graph=_SlowGraph(), + initial_state={}, + config={}, + question="What is EGFR?", + payload=_payload(), + literature_callback=UsageMetricsCallback("session", strict=False), + ) + + await asyncio.sleep(0.6) + assert core_cancelled is True + + +def test_usage_summary_merges_core_and_literature_totals(): + """The response's `usage` must cover both handlers, not just the core one.""" + service = AgentService.__new__(AgentService) + + core = UsageMetricsCallback("session") + core.per_node_usage["generate_cypher"] = UsageRecord( + input_tokens=100, output_tokens=10, total_tokens=110, call_count=1 + ) + core.aggregated_usage.totals.add_usage( + {"input_tokens": 100, "output_tokens": 10, "total_tokens": 110} + ) + core.aggregated_usage.register_model("generate_cypher", "gpt-4o-mini") + + literature = UsageMetricsCallback("session", strict=False) + literature.per_node_usage["paperclip.router"] = UsageRecord( + input_tokens=5, output_tokens=1, total_tokens=6, call_count=1 + ) + literature.aggregated_usage.totals.add_usage( + {"input_tokens": 5, "output_tokens": 1, "total_tokens": 6} + ) + literature.aggregated_usage.register_model("paperclip.router", "gpt-4o-mini") + + merged = service._usage_summary(core, literature) + + assert merged["aggregated_usage"]["totals"]["total_tokens"] == 116 + assert set(merged["per_node_usage"]) == {"generate_cypher", "paperclip.router"} + # With no literature handler the shape is unchanged from before. + assert service._usage_summary(core, None) == core.get_summary() diff --git a/crossbar_llm/tests/test_literature_api.py b/crossbar_llm/tests/test_literature_api.py new file mode 100644 index 0000000..149bd58 --- /dev/null +++ b/crossbar_llm/tests/test_literature_api.py @@ -0,0 +1,261 @@ +"""End-to-end tests through the FastAPI routes. + +These drive the real app with a TestClient: routing, the browser-identity +cookie, request validation, the response model and JSON serialisation all run +for real. Only the two agent boundaries are replaced — the Cypher graph and the +literature agents — because those reach Neo4j and metered third-party services. + +The service is swapped in through `app.dependency_overrides`; patching won't do, +since `get_runtime_service` is `lru_cache`d and the routers resolve it via +`Depends`. +""" +import pytest +from fastapi.testclient import TestClient + +from crossbar_llm.api.core.deps import get_runtime_service +from crossbar_llm.api.core.rate_limit import limiter +from crossbar_llm.api.main import app +from crossbar_llm.api.schemas.common import SearchMode +from crossbar_llm.api.schemas.requests import ( + DbSearchRequest, + LiteratureToolsConfig, + UploadVectorSearchRequest, +) +from crossbar_llm.api.schemas.responses import ( + ChatResponse, + LiteratureToolResult, + PendingResumeResponse, +) + + +class _StubAgentService: + """Records what the routers hand the service, and returns a fixed response.""" + + def __init__(self): + self.calls = [] + self.response = None + + def _record(self, name, payload): + self.calls.append((name, payload)) + return self.response or ChatResponse( + session_id="session-1", + status="completed", + question=payload.question, + mode=SearchMode.DB_SEARCH, + final_answer="graph answer", + ) + + async def run_db(self, *, session_id, browser_id, payload): + return self._record("run_db", payload) + + async def run_vector(self, *, session_id, browser_id, payload): + return self._record("run_vector", payload) + + async def run_vector_upload(self, *, session_id, browser_id, payload, embedding_file): + return self._record("run_vector_upload", payload) + + async def resume(self, *, session_id, browser_id, payload): + self.calls.append(("resume", payload)) + return self.response or ChatResponse( + session_id=session_id, + status="completed", + question="What is EGFR?", + mode=payload.search_mode, + final_answer="graph answer", + ) + + +@pytest.fixture +def service(): + stub = _StubAgentService() + app.dependency_overrides[get_runtime_service] = lambda: stub + yield stub + app.dependency_overrides.clear() + + +@pytest.fixture +def client(): + # These tests share one client IP, and the standard limit is 6 requests a + # minute — comfortably inside what this module already sends. Pinned off so + # that adding the next test here fails for its own reasons, not because it + # tipped the suite over a rate limit. + was_enabled = limiter.enabled + limiter.enabled = False + try: + with TestClient(app) as test_client: + yield test_client + finally: + limiter.enabled = was_enabled + + +def _body(**overrides): + body = { + "provider": "openai", + "model": "gpt-4o-mini", + "question": "What is the role of EGFR in cancer?", + "execution_mode": "generate_and_run", + } + body.update(overrides) + return body + + +def test_db_search_defaults_both_tools_off(client, service): + response = client.post("/sessions/session-1/db-search/query", json=_body()) + + assert response.status_code == 200 + _, payload = service.calls[0] + assert payload.literature_tools == LiteratureToolsConfig() + # Absent rather than null: a request that asked for nothing says nothing. + assert response.json()["literature"] is None + + +@pytest.mark.parametrize( + "tools", + [ + {"paperclip": True, "pubtator3": False}, + {"paperclip": False, "pubtator3": True}, + {"paperclip": True, "pubtator3": True}, + ], +) +def test_db_search_forwards_each_tool_combination(client, service, tools): + response = client.post( + "/sessions/session-1/db-search/query", + json=_body(literature_tools=tools), + ) + + assert response.status_code == 200 + _, payload = service.calls[0] + assert payload.literature_tools.paperclip is tools["paperclip"] + assert payload.literature_tools.pubtator3 is tools["pubtator3"] + + +def test_literature_results_serialize_per_tool(client, service): + service.response = ChatResponse( + session_id="session-1", + status="completed", + question="What is EGFR?", + mode=SearchMode.DB_SEARCH, + final_answer="graph answer", + literature={ + "paperclip": LiteratureToolResult( + status="completed", + answer="paper answer", + citations=[{"title": "A paper", "url": "https://example.org/1"}], + ), + "pubtator3": LiteratureToolResult( + status="failed", + warnings=["pubtator3 timed out after 180 seconds."], + ), + }, + ) + + body = client.post( + "/sessions/session-1/db-search/query", + json=_body(literature_tools={"paperclip": True, "pubtator3": True}), + ).json() + + # One tool failing leaves the other's answer and the core answer intact. + assert body["final_answer"] == "graph answer" + assert body["literature"]["paperclip"]["answer"] == "paper answer" + assert body["literature"]["paperclip"]["citations"][0]["url"] == "https://example.org/1" + assert body["literature"]["pubtator3"]["status"] == "failed" + assert body["literature"]["pubtator3"]["answer"] is None + + +def test_resume_carries_the_tool_selection(client, service): + response = client.post( + "/sessions/session-1/resume", + json={ + "provider": "openai", + "model": "gpt-4o-mini", + "search_mode": "db_search", + "execution_mode": "resume", + "action": "approve", + "edited_cypher": "MATCH (g:Gene) RETURN g", + "literature_tools": {"paperclip": True, "pubtator3": False}, + }, + ) + + assert response.status_code == 200 + name, payload = service.calls[0] + assert name == "resume" + assert payload.literature_tools.paperclip is True + + +def test_vector_upload_maps_flat_form_switches(client, service): + response = client.post( + "/sessions/session-1/vector-search/upload-query", + data={ + "question": "What is the role of EGFR in cancer?", + "execution_mode": "generate_and_run", + "provider": "openai", + "model": "gpt-4o-mini", + "vector_category": "gene", + "embedding_type": "anc2vec", + # Multipart cannot carry a nested object, so the switches travel + # flat and are reassembled server-side. + "paperclip": "true", + "pubtator3": "false", + }, + files={"embedding_file": ("embedding.npy", b"\x00\x01", "application/octet-stream")}, + ) + + assert response.status_code == 200 + _, payload = service.calls[0] + assert payload.literature_tools == LiteratureToolsConfig( + paperclip=True, pubtator3=False + ) + + +def test_pending_review_response_reports_its_status(client, service): + service.response = PendingResumeResponse( + session_id="session-1", + question="What is EGFR?", + mode=SearchMode.DB_SEARCH, + generated_cypher="MATCH (g:Gene) RETURN g", + ) + + body = client.post("/sessions/session-1/db-search/query", json=_body()).json() + + assert body["status"] == "awaiting_human_review" + assert body["generated_cypher"] == "MATCH (g:Gene) RETURN g" + + +def test_literature_tools_reject_non_boolean_values(client, service): + response = client.post( + "/sessions/session-1/db-search/query", + json=_body(literature_tools={"paperclip": "sometimes"}), + ) + + assert response.status_code == 422 + assert service.calls == [] + + +def test_api_contract_round_trips_literature_configuration(): + request = DbSearchRequest.model_validate( + { + "provider": "openai", + "model": "gpt-4o-mini", + "question": "What is EGFR?", + "execution_mode": "generate_and_run", + "literature_tools": {"paperclip": True, "pubtator3": False}, + } + ) + upload_request = UploadVectorSearchRequest.as_form( + question="What is EGFR?", + execution_mode="generate_and_run", + provider="openai", + model="gpt-4o-mini", + top_k=10, + reasoning_enabled=False, + reasoning_effort=None, + paperclip=True, + pubtator3=False, + vector_category="gene", + embedding_type="anc2vec", + ) + + assert request.literature_tools == LiteratureToolsConfig( + paperclip=True, pubtator3=False + ) + assert upload_request.literature_tools == request.literature_tools diff --git a/crossbar_llm/tests/test_literature_service.py b/crossbar_llm/tests/test_literature_service.py new file mode 100644 index 0000000..08643fa --- /dev/null +++ b/crossbar_llm/tests/test_literature_service.py @@ -0,0 +1,366 @@ +import asyncio +import os +from types import SimpleNamespace + +import pytest +from pydantic import SecretStr + +from crossbar_llm.agent_tools.callback_handler import UsageMetricsCallback +from crossbar_llm.api.schemas.requests import DbSearchRequest, LiteratureToolsConfig +from crossbar_llm.api.core.settings import Settings +from crossbar_llm.api.services.literature_service import LiteratureService +from crossbar_llm.paperclip_tools.adapter import PaperclipConfigError + + +def _settings(timeout: float = 1.0) -> Settings: + # The real Settings object, not a stub: these tests exercise code that + # reads a growing set of tuning fields, and a hand-rolled namespace would + # drift out of sync with it silently. + return Settings(literature_tool_timeout_seconds=timeout) + + +def _payload(tools: LiteratureToolsConfig) -> DbSearchRequest: + return DbSearchRequest( + provider="openai", + model="gpt-4o-mini", + question="What is the role of EGFR in cancer?", + execution_mode="generate_and_run", + literature_tools=tools, + ) + + +@pytest.mark.asyncio +async def test_disabled_tools_are_not_run(monkeypatch): + service = LiteratureService(_settings()) + + async def unexpected(*args, **kwargs): + raise AssertionError("disabled literature tool was invoked") + + monkeypatch.setattr(service, "_run_paperclip", unexpected) + monkeypatch.setattr(service, "_run_pubtator3", unexpected) + + result = await service.run( + question="test", + payload=_payload(LiteratureToolsConfig()), + callback=UsageMetricsCallback("session", strict=False), + ) + + assert result == {} + + +@pytest.mark.asyncio +async def test_enabled_tools_run_in_parallel(monkeypatch): + service = LiteratureService(_settings()) + started = {"paperclip": asyncio.Event(), "pubtator3": asyncio.Event()} + release = asyncio.Event() + + async def paperclip(*args, **kwargs): + started["paperclip"].set() + await release.wait() + return {"final_answer": "Paperclip answer", "citations": [], "warnings": []} + + async def pubtator3(*args, **kwargs): + started["pubtator3"].set() + await release.wait() + return {"final_answer": "PubTator3 answer", "documents": [], "warnings": []} + + monkeypatch.setattr(service, "_run_paperclip", paperclip) + monkeypatch.setattr(service, "_run_pubtator3", pubtator3) + + task = asyncio.create_task( + service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True, pubtator3=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + ) + await asyncio.wait_for(asyncio.gather(*(event.wait() for event in started.values())), 1) + release.set() + result = await task + + assert list(result) == ["paperclip", "pubtator3"] + assert result["paperclip"].answer == "Paperclip answer" + assert result["pubtator3"].answer == "PubTator3 answer" + + +@pytest.mark.asyncio +async def test_one_tool_failure_does_not_discard_the_other(monkeypatch): + service = LiteratureService(_settings()) + + async def paperclip(*args, **kwargs): + raise RuntimeError("Paperclip unavailable") + + async def pubtator3(*args, **kwargs): + return {"final_answer": "PubTator3 answer", "documents": [], "warnings": []} + + monkeypatch.setattr(service, "_run_paperclip", paperclip) + monkeypatch.setattr(service, "_run_pubtator3", pubtator3) + + result = await service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True, pubtator3=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + + assert result["paperclip"].status == "failed" + assert result["pubtator3"].status == "completed" + + +@pytest.mark.asyncio +async def test_tool_timeout_is_reported_per_tool(monkeypatch): + service = LiteratureService(_settings(timeout=0.01)) + + async def slow(*args, **kwargs): + await asyncio.sleep(1) + + monkeypatch.setattr(service, "_run_paperclip", slow) + + result = await service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + + assert result["paperclip"].status == "failed" + assert "timed out" in result["paperclip"].warnings[0] + + +@pytest.mark.asyncio +async def test_malformed_tool_result_is_isolated(monkeypatch): + service = LiteratureService(_settings()) + + async def malformed(*args, **kwargs): + return None + + async def valid(*args, **kwargs): + return {"final_answer": "PubTator3 answer", "documents": [], "warnings": []} + + monkeypatch.setattr(service, "_run_paperclip", malformed) + monkeypatch.setattr(service, "_run_pubtator3", valid) + + result = await service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True, pubtator3=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + + assert result["paperclip"].status == "failed" + assert result["pubtator3"].status == "completed" + + +@pytest.mark.asyncio +async def test_paperclip_adapter_is_lazy_reused_and_closed(monkeypatch): + created = [] + + class Adapter: + def __init__(self, **kwargs): + self.close_count = 0 + self.kwargs = kwargs + created.append(self) + + async def aclose(self): + self.close_count += 1 + + monkeypatch.setattr( + "crossbar_llm.api.services.literature_service.PaperclipAdapter", + Adapter, + ) + settings = _settings() + settings.env_settings = SimpleNamespace( + paperclip_api_key=SecretStr("paperclip-test-key"), + paperclip_disable_rest=True, + ) + monkeypatch.delenv("PAPERCLIP_API_KEY", raising=False) + monkeypatch.delenv("PAPERCLIP_DISABLE_REST", raising=False) + service = LiteratureService(settings) + + assert service.paperclip_adapter is None + assert service._get_paperclip_adapter() is service._get_paperclip_adapter() + assert len(created) == 1 + + # Credentials are handed to the adapter directly. Exporting them to + # os.environ instead would be a process-global side effect from a request + # path, and would leak between tests and between tenants. + assert created[0].kwargs["api_key"] == "paperclip-test-key" + assert created[0].kwargs["disable_rest"] is True + assert "PAPERCLIP_API_KEY" not in os.environ + assert "PAPERCLIP_DISABLE_REST" not in os.environ + + await service.aclose() + assert created[0].close_count == 1 + assert service.paperclip_adapter is None + + +@pytest.mark.asyncio +async def test_missing_paperclip_credentials_fail_only_that_tool(monkeypatch): + """An unconfigured tool degrades to its own failure, not a 500.""" + service = LiteratureService(_settings()) + + async def unconfigured(*args, **kwargs): + raise PaperclipConfigError("PAPERCLIP_API_KEY is not set") + + async def pubtator3(*args, **kwargs): + return {"final_answer": "PubTator3 answer", "documents": [], "warnings": []} + + monkeypatch.setattr(service, "_run_paperclip", unconfigured) + monkeypatch.setattr(service, "_run_pubtator3", pubtator3) + + result = await service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True, pubtator3=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + + assert result["paperclip"].status == "failed" + # A missing key is safe and actionable, so it is surfaced verbatim. + assert "PAPERCLIP_API_KEY is not set" in result["paperclip"].warnings[0] + assert result["pubtator3"].status == "completed" + + +@pytest.mark.asyncio +async def test_unexpected_failures_do_not_leak_upstream_detail(monkeypatch): + """Arbitrary upstream error text must not reach the HTTP response.""" + service = LiteratureService(_settings()) + + async def boom(*args, **kwargs): + raise RuntimeError("https://paperclip.example/x?token=SUPERSECRET failed") + + monkeypatch.setattr(service, "_run_paperclip", boom) + + result = await service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + + warning = result["paperclip"].warnings[0] + assert "SUPERSECRET" not in warning + assert "RuntimeError" in warning + + +@pytest.mark.asyncio +async def test_enabled_tool_without_a_question_is_reported_as_skipped(monkeypatch): + """Resuming with no checkpointed question must say so, not stay silent.""" + service = LiteratureService(_settings()) + + async def unexpected(*args, **kwargs): + raise AssertionError("literature tool ran without a question") + + monkeypatch.setattr(service, "_run_paperclip", unexpected) + + result = await service.run( + question=" ", + payload=_payload(LiteratureToolsConfig(paperclip=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + + assert result["paperclip"].status == "skipped" + assert "no question" in result["paperclip"].warnings[0] + + +def test_tool_usage_slices_by_node_name_prefix(): + """Pins the contract `_tool_usage` depends on. + + Per-tool usage is recovered by matching the `node_name` prefix each agent + tags its LLM calls with. Renaming a node without updating the prefix would + otherwise empty this out with no test failing. + """ + service = LiteratureService(_settings()) + summary = { + "per_node_usage": { + "paperclip.router": {"total_tokens": 10, "call_count": 1}, + "paperclip.synthesize": {"total_tokens": 30, "call_count": 2}, + "pubtator3.router": {"total_tokens": 7, "call_count": 1}, + }, + "aggregated_usage": { + "models_by_node": {"paperclip.router": ["gpt-4o-mini"]}, + }, + } + + paperclip = service._tool_usage(summary, "paperclip.") + + assert set(paperclip["per_node_usage"]) == { + "paperclip.router", + "paperclip.synthesize", + } + assert paperclip["aggregated_usage"]["totals"]["total_tokens"] == 40 + assert paperclip["call_count"] == 3 + # `totals` keeps exactly the core agent's shape, so the two are comparable. + assert "call_count" not in paperclip["aggregated_usage"]["totals"] + assert service._tool_usage(summary, "nothing.") == {} + + +@pytest.mark.asyncio +async def test_admission_control_bounds_concurrent_runs(monkeypatch): + """Only `literature_max_concurrent_runs` tools may be in flight at once.""" + settings = _settings() + settings.literature_max_concurrent_runs = 1 + service = LiteratureService(settings) + + in_flight = 0 + peak = 0 + release = asyncio.Event() + + async def tool(*args, **kwargs): + nonlocal in_flight, peak + in_flight += 1 + peak = max(peak, in_flight) + try: + await release.wait() + return {"final_answer": "answer", "documents": [], "warnings": []} + finally: + in_flight -= 1 + + monkeypatch.setattr(service, "_run_paperclip", tool) + monkeypatch.setattr(service, "_run_pubtator3", tool) + + task = asyncio.create_task( + service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True, pubtator3=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + ) + await asyncio.sleep(0.05) + assert peak == 1, "both tools started despite a concurrency limit of 1" + release.set() + await task + + +@pytest.mark.asyncio +async def test_saturation_reports_skipped_not_timed_out(monkeypatch): + """A rejected run must not masquerade as a slow upstream service.""" + settings = _settings() + settings.literature_max_concurrent_runs = 1 + settings.literature_admission_wait_seconds = 0.01 + service = LiteratureService(settings) + + release = asyncio.Event() + + async def blocker(*args, **kwargs): + await release.wait() + return {"final_answer": "answer", "citations": [], "warnings": []} + + monkeypatch.setattr(service, "_run_paperclip", blocker) + monkeypatch.setattr(service, "_run_pubtator3", blocker) + + # Occupy the only slot, then ask for both tools. + hog = asyncio.create_task( + service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + ) + await asyncio.sleep(0.05) + + result = await service.run( + question="test", + payload=_payload(LiteratureToolsConfig(pubtator3=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + + assert result["pubtator3"].status == "skipped" + assert "capacity" in result["pubtator3"].warnings[0] + release.set() + await hog From 499e7175cb654d4be756d7fcd91a9b8259c6d4d2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ahmet=20O=C4=9Fuzhan=20K=C3=B6k=C3=BCl=C3=BC?= <oguzhankokulu@gmail.com> Date: Mon, 21 Sep 2026 12:52:26 +0300 Subject: [PATCH 6/8] feat(api): surface backend misconfiguration instead of failing opaquely MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A missing Neo4j environment variable made `AgentService()` raise a pydantic ValidationError from inside a dependency, which reached the client as a bare 500. Catch it in `get_runtime_service` and return 503, so the failure points at the configuration rather than at the code. The missing variable names are always logged, but only echoed to the client in development. The names leak no secrets, yet on a public deployment they confirm the backend stack to anyone who hits the API while it is misconfigured, and the user-facing sentence already says everything a caller can act on. `lru_cache` does not memoise exceptions, so construction is retried on the next request once the environment is fixed — no restart needed. A failed build also leaves the cache empty, which is what the shutdown hook checks before closing. Also stop CRA's development overlay from promoting Chromium's benign "ResizeObserver loop completed with undelivered notifications" into a full-screen runtime error; the browser retries on the next frame. --- crossbar_llm/api/core/deps.py | 43 ++++++++++- crossbar_llm/frontend/src/index.js | 18 ++++- crossbar_llm/tests/test_api_dependencies.py | 81 +++++++++++++++++++++ 3 files changed, 140 insertions(+), 2 deletions(-) create mode 100644 crossbar_llm/tests/test_api_dependencies.py diff --git a/crossbar_llm/api/core/deps.py b/crossbar_llm/api/core/deps.py index e392f20..d71ec9a 100644 --- a/crossbar_llm/api/core/deps.py +++ b/crossbar_llm/api/core/deps.py @@ -1,8 +1,49 @@ from functools import lru_cache +from fastapi import HTTPException, status +from pydantic import ValidationError + +from crossbar_llm.agent_tools.logging_config import get_logger +from crossbar_llm.api.core.settings import Settings from crossbar_llm.api.services.agent_service import AgentService +logger = get_logger(__name__) + @lru_cache() def get_runtime_service() -> AgentService: - return AgentService() + try: + return AgentService() + except ValidationError as exc: + missing_fields = sorted({ + str(error["loc"][-1]) + for error in exc.errors() + if error.get("type") == "missing" and error.get("loc") + }) + + # Always logged in full — the operator fixing this needs the list. + logger.error( + "Runtime service could not be configured", + event_type="runtime_service_misconfigured", + component="deps.get_runtime_service", + missing_fields=missing_fields, + exc_info=exc, + ) + + # Only echoed to the client in development. The names alone leak no + # secrets, but on a public deployment they confirm the backend stack to + # anyone who happens to hit the API while it is misconfigured, and the + # sentence below already tells a user everything they can act on. + settings = Settings() + missing_hint = ( + f" Missing environment variables: {', '.join(missing_fields)}." + if missing_fields and settings.is_dev + else "" + ) + raise HTTPException( + status_code=status.HTTP_503_SERVICE_UNAVAILABLE, + detail=( + "The knowledge-graph backend is not configured, so chat and " + f"vector queries cannot run.{missing_hint}" + ), + ) from exc diff --git a/crossbar_llm/frontend/src/index.js b/crossbar_llm/frontend/src/index.js index b2cbb0c..a81164c 100644 --- a/crossbar_llm/frontend/src/index.js +++ b/crossbar_llm/frontend/src/index.js @@ -5,6 +5,22 @@ import App from './App'; import DashboardApp from './dashboard/DashboardApp'; import './index.css'; +// Chromium can emit this benign notification when responsive MUI/chart +// components resize each other within one frame. CRA's development overlay +// promotes it to a full-screen runtime error even though the browser retries +// delivery on the next frame and the application remains healthy. +if (process.env.NODE_ENV === 'development') { + window.addEventListener('error', (event) => { + if ( + event.message === 'ResizeObserver loop completed with undelivered notifications.' + || event.message === 'ResizeObserver loop limit exceeded' + ) { + event.preventDefault(); + event.stopImmediatePropagation(); + } + }, true); +} + function Root() { const location = useLocation(); if (location.pathname.startsWith('/dashboard')) { @@ -20,4 +36,4 @@ root.render( <Root /> </BrowserRouter> </React.StrictMode> -); \ No newline at end of file +); diff --git a/crossbar_llm/tests/test_api_dependencies.py b/crossbar_llm/tests/test_api_dependencies.py new file mode 100644 index 0000000..5f487bc --- /dev/null +++ b/crossbar_llm/tests/test_api_dependencies.py @@ -0,0 +1,81 @@ +import pytest +from fastapi import HTTPException, status +from pydantic import BaseModel, Field + +from crossbar_llm.api.core import deps + + +class _MissingNeo4jSettings(BaseModel): + user: str = Field(alias="NEO4J_USER") + password: str = Field(alias="NEO4J_PASSWORD") + database: str = Field(alias="NEO4J_DB_NAME") + + +@pytest.fixture +def unconfigured_backend(monkeypatch): + """Make `AgentService()` fail the way a missing Neo4j config makes it fail.""" + deps.get_runtime_service.cache_clear() + monkeypatch.setattr( + deps, "AgentService", lambda: _MissingNeo4jSettings.model_validate({}) + ) + yield + deps.get_runtime_service.cache_clear() + + +def _force_env(monkeypatch, *, is_dev: bool): + """Pin the environment rather than inheriting whatever .env happens to say.""" + monkeypatch.setattr( + deps, "Settings", lambda: type("S", (), {"is_dev": is_dev})() + ) + + +def test_missing_configuration_returns_503(unconfigured_backend, monkeypatch): + _force_env(monkeypatch, is_dev=True) + + with pytest.raises(HTTPException) as caught: + deps.get_runtime_service() + + assert caught.value.status_code == status.HTTP_503_SERVICE_UNAVAILABLE + assert "knowledge-graph backend is not configured" in caught.value.detail + + +def test_development_names_the_missing_variables(unconfigured_backend, monkeypatch): + _force_env(monkeypatch, is_dev=True) + + with pytest.raises(HTTPException) as caught: + deps.get_runtime_service() + + detail = caught.value.detail + assert "NEO4J_USER" in detail + assert "NEO4J_PASSWORD" in detail + assert "NEO4J_DB_NAME" in detail + + +def test_production_withholds_the_variable_names(unconfigured_backend, monkeypatch): + """The names confirm the backend stack to anyone hitting a misconfigured + public deployment. They go to the logs instead; the user-facing sentence + already says everything a caller can act on.""" + _force_env(monkeypatch, is_dev=False) + + with pytest.raises(HTTPException) as caught: + deps.get_runtime_service() + + detail = caught.value.detail + assert "knowledge-graph backend is not configured" in detail + assert "NEO4J" not in detail + assert "Missing environment variables" not in detail + + +def test_failure_is_not_cached_so_a_fixed_config_recovers(unconfigured_backend, monkeypatch): + """`lru_cache` does not memoise exceptions, so construction is retried once + the environment is corrected — and a failed build leaves the cache empty, + which is what the shutdown hook checks before calling `aclose`.""" + _force_env(monkeypatch, is_dev=True) + + with pytest.raises(HTTPException): + deps.get_runtime_service() + assert deps.get_runtime_service.cache_info().currsize == 0 + + sentinel = object() + monkeypatch.setattr(deps, "AgentService", lambda: sentinel) + assert deps.get_runtime_service() is sentinel From 89b12935e2e5916289413cb45ab61e860b100852 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ahmet=20O=C4=9Fuzhan=20K=C3=B6k=C3=BCl=C3=BC?= <oguzhankokulu@gmail.com> Date: Tue, 22 Sep 2026 18:47:43 +0300 Subject: [PATCH 7/8] fix(api): re-validate biological relevance for every question in a session The relevance node now skips its LLM call when the state already holds a verdict, so the API's up-front check is not paid for twice. But a session's checkpointer carries every state key forward between questions, and `_base_state` never reset `biological_relevance`. From the second question in a session on, the node reused the previous question's verdict. This hit the default path, with literature tools off: after an on-topic question an off-topic one passed the gate, and after an off-topic question a valid one was routed straight to END with no answer. Reset the verdict in `_base_state` so each question is judged on its own. The regression test drives the real node through a MemorySaver-backed graph for two questions in one session, and fails without the reset. --- crossbar_llm/api/services/agent_service.py | 5 ++ .../tests/test_agent_service_literature.py | 55 +++++++++++++++++++ 2 files changed, 60 insertions(+) diff --git a/crossbar_llm/api/services/agent_service.py b/crossbar_llm/api/services/agent_service.py index a101916..f44be99 100644 --- a/crossbar_llm/api/services/agent_service.py +++ b/crossbar_llm/api/services/agent_service.py @@ -210,6 +210,11 @@ def _base_state( return { "question": question, + # Reset per question. The session checkpointer carries every key + # forward between questions, and the relevance node skips itself + # when a verdict is already present — so a stale verdict here would + # silently apply the PREVIOUS question's relevance to this one. + "biological_relevance": None, "resolved_entities": None, "cypher_mode": cypher_mode, "vector_index": vector_index, diff --git a/crossbar_llm/tests/test_agent_service_literature.py b/crossbar_llm/tests/test_agent_service_literature.py index 69b5961..ccb152e 100644 --- a/crossbar_llm/tests/test_agent_service_literature.py +++ b/crossbar_llm/tests/test_agent_service_literature.py @@ -298,3 +298,58 @@ def test_usage_summary_merges_core_and_literature_totals(): assert set(merged["per_node_usage"]) == {"generate_cypher", "paperclip.router"} # With no literature handler the shape is unchanged from before. assert service._usage_summary(core, None) == core.get_summary() + + +def test_relevance_is_revalidated_for_each_question_in_a_session(): + """A session's checkpointer carries state between questions, and the + relevance node skips itself when a verdict is present. Unless each + question's initial state clears the verdict, question 2 silently inherits + question 1's — letting an off-topic question through, or turning a valid + one away.""" + from langgraph.checkpoint.memory import MemorySaver + from langgraph.graph import END, START, StateGraph + + from crossbar_llm.agent_tools.cypher_agent import CypherAgentState + from crossbar_llm.api.schemas.common import SearchMode + + judged = [] + + class _FakeValidator: + def invoke(self, messages, config=None): + question = messages[-1].content + judged.append(question) + return { + "parsed": SimpleNamespace( + relevant="stock" not in question, reason="test verdict" + ) + } + + agent = object.__new__(CypherAgent) + agent.llm_factory = SimpleNamespace( + create_biological_relevance_validator_llm=lambda: _FakeValidator() + ) + builder = StateGraph(CypherAgentState) + builder.add_node("relevance", agent.biological_relevance_validation_node) + builder.add_edge(START, "relevance") + builder.add_edge("relevance", END) + graph = builder.compile(checkpointer=MemorySaver()) + + service = AgentService.__new__(AgentService) + config = {"configurable": {"thread_id": "one-session"}} + + def ask(question): + return graph.invoke( + service._base_state( + question=question, + execution_mode="generate_and_run", + cypher_mode=SearchMode.DB_SEARCH, + ), + config, + ) + + first = ask("What is the role of EGFR in cancer?") + second = ask("What is Apple's stock price today?") + + assert first["biological_relevance"] is True + assert len(judged) == 2, "second question reused the first question's verdict" + assert second["biological_relevance"] is False From e50544b8d731bc93db40f66a29afc5fe8bf3d815 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ahmet=20O=C4=9Fuzhan=20K=C3=B6k=C3=BCl=C3=BC?= <oguzhankokulu@gmail.com> Date: Tue, 22 Sep 2026 18:47:43 +0300 Subject: [PATCH 8/8] feat(literature): per-tool admission limits; cap Paperclip at 10 connections Admission control: - Replace the single shared pool of 8 runs with a limit per tool. The tools are bottlenecked by different things, and a user with both enabled took two of the eight slots, so only 4 of 10 simultaneous users got literature. Paperclip now has no run-level limit and PubTator3 allows 20. - Raise the admission wait from 5 s to 30 s. The 5 s wait was shorter than a single run, so the queue could never advance and everyone past the first wave was skipped. - Correct the comments claiming that queueing ate into the per-tool timeout. The wait and the run are sequential: the run's timeout starts only once a slot is granted. A test now pins this. Paperclip connection pool: - Cut the pool from 64 to 10 connections and let a command wait up to 60 s for one. Tested live on 2026-09-22 against the REST endpoint we use: metadata reads were not limited even at 30 at once, but searches share an undocumented per-user queue. 10 concurrent searches all succeeded; 15 drew 429 "Per-user search queue is full (6/6)" with Retry-After: 15. A pool of 10 keeps us at the level that passed. - On pool exhaustion, raise PaperclipError instead of falling back to MCP. Paperclip counts MCP connections against the same account, so the fallback would only send the overflow through another door. - Align the adapter's own defaults, since benchmark runs build it directly and use the same account. Tests cover per-tool isolation, the unlimited path, wait versus run budget, and the pool itself: against a real local HTTP server, 25 simultaneous commands peak at exactly 10 in flight and all complete. --- crossbar_llm/api/core/settings.py | 66 ++++++-- .../api/services/literature_service.py | 66 ++++---- crossbar_llm/paperclip_tools/adapter.py | 61 +++++--- .../tests/test_pool_timeout.py | 112 ++++++++++--- crossbar_llm/tests/test_literature_service.py | 147 ++++++++++++++++-- 5 files changed, 357 insertions(+), 95 deletions(-) diff --git a/crossbar_llm/api/core/settings.py b/crossbar_llm/api/core/settings.py index 4e116ac..d408476 100644 --- a/crossbar_llm/api/core/settings.py +++ b/crossbar_llm/api/core/settings.py @@ -133,23 +133,44 @@ class Settings(BaseModel): ge=1, description="Maximum citations returned per literature tool.", ) - literature_max_concurrent_runs: int = Field( - default=8, + # Admission limits are per tool because the two tools are bottlenecked by + # different things, and a single shared pool let one tool's users take + # capacity from the other's for no benefit. `None` means no local limit. + paperclip_max_concurrent_runs: int | None = Field( + default=None, + ge=1, + description=( + "Paperclip runs allowed in flight at once in this process. No " + "run-level limit by default: the adapter's connection pool " + "(`paperclip_max_connections`) already caps Paperclip commands in " + "flight, and extra commands queue for a connection. Past that, " + "Paperclip's per-user search queue answers 429, which the adapter " + "does not retry yet. Set a number here if searches start being " + "rejected." + ), + ) + pubtator3_max_concurrent_runs: int | None = Field( + default=20, ge=1, description=( - "How many literature tool runs may be in flight at once in this " - "process. Each run fans out several requests upstream, so without " - "a ceiling a burst of traffic multiplies straight through to the " - "external services and exhausts the connection pool." + "PubTator3 runs allowed in flight at once in this process. This is " + "NOT what protects NCBI — the client's rate limiter already holds " + "every request to NCBI's IP-wide 3 req/s. It bounds latency: a run " + "makes roughly 3-6 requests, so 20 concurrent runs queue for ~20-40s " + "on the limiter, well inside the per-tool timeout. Without a cap, " + "a large burst would make EVERY run slow enough to time out, " + "instead of serving most promptly and skipping the excess." ), ) literature_admission_wait_seconds: float = Field( - default=5.0, + default=30.0, ge=0, description=( - "How long a run waits for an admission slot before being reported " - "as skipped. Short on purpose: queueing here would silently eat " - "the per-tool timeout budget instead of failing legibly." + "How long a run may queue for an admission slot before it is " + "reported as skipped. This wait happens BEFORE the per-tool timeout " + "starts, so it never shortens a run's budget; what it costs is " + "response time for the queued user. Long enough that a burst " + "queues briefly instead of being turned away." ), ) pubtator3_replica_count: int = Field( @@ -184,12 +205,29 @@ class Settings(BaseModel): ), ) paperclip_max_connections: int = Field( - default=64, + default=10, ge=1, description=( - "HTTP connection cap for the shared Paperclip adapter. This is a " - "process-wide pool, so it bounds every concurrent request at once " - "rather than one request's ~14-wide fan-out." + "Connections in the shared Paperclip pool, and therefore the most " + "Paperclip commands this process runs at once. Paperclip documents " + "10 short commands in flight per account. Tested live on " + "2026-09-22: metadata reads were not limited even at 30 at once, " + "but searches share an undocumented per-user queue; 10 searches at " + "once all succeeded, while 15 drew 429s ('search queue is full'). " + "So 10 is what keeps concurrent searches safe. The limits are per " + "account: with N replicas on one API key, set this to about 10 / N, " + "and leave headroom if benchmarks use the same key." + ), + ) + paperclip_pool_timeout_seconds: float = Field( + default=60.0, + gt=0, + description=( + "How long a Paperclip command may queue for one of those " + "connections. With the pool sized to the account limit, this IS " + "the queue: a burst waits here instead of being sent to Paperclip " + "and rejected. Sized to the 60 s response target; still bounded " + "by `literature_tool_timeout_seconds` overall." ), ) pubtator3_max_documents: int = Field( diff --git a/crossbar_llm/api/services/literature_service.py b/crossbar_llm/api/services/literature_service.py index f7d0633..b416caf 100644 --- a/crossbar_llm/api/services/literature_service.py +++ b/crossbar_llm/api/services/literature_service.py @@ -50,22 +50,29 @@ def __init__(self, settings: Settings): # Keep the adapter long-lived once used, but do not construct it for # requests that leave Paperclip disabled. self.paperclip_adapter: PaperclipAdapter | None = None - # Admission control. Every run fans out several upstream requests, so - # concurrent traffic multiplies through to Paperclip and PubTator3 and - # drains the shared connection pool. Created lazily because a Semaphore - # binds to the running loop. - self._admission: asyncio.Semaphore | None = None + # Admission control, one semaphore per tool. The tools are bottlenecked + # by different things (Paperclip by its connection pool and server-side + # limit, PubTator3 by NCBI's IP-wide rate), so one shared pool only let + # each tool's users take capacity from the other's. Created lazily + # because a Semaphore binds to the running loop. + self._admission: dict[str, asyncio.Semaphore] = {} # Divide PubTator3's IP-wide budget across replicas. Configured once at # construction rather than per request: the limiter is process-global # state inside the client. install_static_share_limiter(settings.pubtator3_replica_count) - def _admission_slot(self) -> asyncio.Semaphore: - if self._admission is None: - self._admission = asyncio.Semaphore( - self.settings.literature_max_concurrent_runs - ) - return self._admission + def _admission_limit(self, name: str) -> int | None: + return getattr(self.settings, f"{name}_max_concurrent_runs", None) + + def _admission_slot(self, name: str) -> asyncio.Semaphore | None: + """The tool's semaphore, or None when that tool has no local limit.""" + limit = self._admission_limit(name) + if limit is None: + return None + slot = self._admission.get(name) + if slot is None: + slot = self._admission[name] = asyncio.Semaphore(limit) + return slot async def _run_admitted( self, @@ -74,23 +81,28 @@ async def _run_admitted( ) -> dict[str, Any]: """Run one tool, but only once this process has capacity for it. - Waiting is deliberately brief. A long queue here would be charged - against the per-tool timeout, so an overloaded server would report - every tool as "timed out" when the truth is that it never started. + Two phases, strictly in sequence: queue for a slot (at most + `literature_admission_wait_seconds`), then run (at most + `literature_tool_timeout_seconds`). The run's timeout starts only once + a slot is granted, so queueing never shortens a run's budget — a longer + wait trades response time for more requests served, and nothing else. """ - try: - await asyncio.wait_for( - self._admission_slot().acquire(), - timeout=self.settings.literature_admission_wait_seconds, - ) - except asyncio.TimeoutError: - raise _AdmissionRejected(name) from None + slot = self._admission_slot(name) + if slot is not None: + try: + await asyncio.wait_for( + slot.acquire(), + timeout=self.settings.literature_admission_wait_seconds, + ) + except asyncio.TimeoutError: + raise _AdmissionRejected(name) from None try: return await asyncio.wait_for( runner(), timeout=self.settings.literature_tool_timeout_seconds ) finally: - self._admission_slot().release() + if slot is not None: + slot.release() def _get_paperclip_adapter(self) -> PaperclipAdapter: # No await between the check and the assignment, so concurrent requests @@ -108,6 +120,7 @@ def _get_paperclip_adapter(self) -> PaperclipAdapter: ), disable_rest=getattr(env_settings, "paperclip_disable_rest", None), max_connections=self.settings.paperclip_max_connections, + pool_timeout_s=self.settings.paperclip_pool_timeout_seconds, ) return self.paperclip_adapter @@ -328,19 +341,20 @@ async def run( if isinstance(raw, _AdmissionRejected): # Not a failure of the tool — this server was saturated. Says # so plainly so the operator sees capacity, not flakiness. + limit = self._admission_limit(name) logger.warning( "Literature tool not admitted", event_type="literature_tool_not_admitted", component="LiteratureService.run", tool=name, - max_concurrent=self.settings.literature_max_concurrent_runs, + max_concurrent=limit, + waited_seconds=self.settings.literature_admission_wait_seconds, ) normalized[name] = LiteratureToolResult( status="skipped", warnings=[ - f"{name} was skipped: the server is at its literature " - f"capacity of {self.settings.literature_max_concurrent_runs} " - "concurrent runs. Try again shortly." + f"{name} was skipped: the server is at its capacity of " + f"{limit} concurrent {name} runs. Try again shortly." ], ) continue diff --git a/crossbar_llm/paperclip_tools/adapter.py b/crossbar_llm/paperclip_tools/adapter.py index 0445272..7f3f03d 100644 --- a/crossbar_llm/paperclip_tools/adapter.py +++ b/crossbar_llm/paperclip_tools/adapter.py @@ -28,9 +28,15 @@ `PaperclipRestUnavailable` internally, caught by `_execute`/`search` to trigger the MCP fallback — they never escape this module as that type. -No client-side rate limiter is applied: Paperclip publishes no request-rate -policy, so throttling would be guesswork. Add one here if the server later -documents limits. +Paperclip documents per-account limits (https://paperclip.gxl.ai/docs): 10 +short commands in flight, 8 long ones, 120 req/s, and 100 `map`/`verify` a +day. Tested live against this REST endpoint (2026-09-22), metadata reads were +not limited even at 30 at once, but searches share an undocumented per-user +queue: 10 concurrent searches succeeded, 15 drew 429 "Per-user search queue is +full (6/6)" with Retry-After: 15. The REST pool defaults to 10, which keeps +concurrent searches within that; excess commands queue for a connection rather +than being sent. Pool exhaustion deliberately does NOT fall back to MCP, which +Paperclip counts against the same account. CLI facts pinned against the live server: - `search -s <source> "<query>" -n <N>` — over MCP, the `-s` source flag is @@ -62,6 +68,12 @@ # if this endpoint ever changes or locks down. REST_URL = "https://paperclip.gxl.ai/api/cli/execute" DISABLE_REST_ENV = "PAPERCLIP_DISABLE_REST" +# Paperclip's documented limit on short commands (search, cat, sql, ...) +# in flight at once, PER ACCOUNT, shared by every session, machine, SDK +# client and MCP connection using the key: https://paperclip.gxl.ai/docs +# Live (2026-09-22) the binding limit was the per-user search queue instead; +# 10 concurrent searches passed, 15 did not. See the module docstring. +ACCOUNT_MAX_SHORT_IN_FLIGHT = 10 API_KEY_ENV = "PAPERCLIP_API_KEY" DEFAULT_TIMEOUT_S = 60.0 # search/cat/head/ls — typically fast # `map`/`ask-image` read full text server-side and can take minutes. 480s, not @@ -909,18 +921,16 @@ def __init__( slow_timeout_s: float = SLOW_TIMEOUT_S, api_key: str | None = None, disable_rest: bool | None = None, - max_connections: int = 32, - pool_timeout_s: float = 10.0, + max_connections: int = ACCOUNT_MAX_SHORT_IN_FLIGHT, + pool_timeout_s: float = 60.0, ): self._timeout_s = timeout_s self._slow_timeout_s = slow_timeout_s - # How long a call may wait for a free connection when the pool is - # saturated. Kept well below the call timeouts on purpose: passing a - # bare float to httpx sets connect/read/write/pool to the SAME value, - # so queueing silently consumed the whole 60s (or 480s) budget and - # surfaced as "Paperclip is slow" rather than "we are out of - # connections". A short, separate pool timeout makes saturation fail - # fast and legibly. + # How long a call may queue for a free connection. Set separately from + # the call timeouts because a bare float makes httpx apply ONE value + # to connect/read/write/pool alike. With the pool sized to Paperclip's + # per-account limit (below), this wait IS our queue for that limit, so + # it is long enough for a burst to drain rather than fail. self._pool_timeout_s = pool_timeout_s # Explicit credentials beat the environment. Callers that load config # from a file (the API reads `.env` through pydantic-settings, which @@ -933,9 +943,12 @@ def __init__( # an AsyncClient binds to the loop it was created on; per-adapter # because an adapter may be either request-scoped (`build_graph` # constructs one per run when none is injected) or process-wide (the - # API injects a single long-lived adapter). `max_connections` is a cap - # on THIS adapter, so a shared one needs it raised: it then bounds - # every concurrent request at once rather than one request's fan-out. + # API injects a single long-lived adapter). `max_connections` caps + # THIS adapter's commands in flight, one per connection under + # HTTP/1.1. It defaults to 10: Paperclip's documented per-account + # limit, and the level live testing showed keeps concurrent searches + # clear of the per-user search queue. Limits are per account, not per + # adapter: several adapters or replicas on one key must split it. self._clients: "weakref.WeakKeyDictionary[asyncio.AbstractEventLoop, httpx.AsyncClient]" = ( weakref.WeakKeyDictionary() ) @@ -1050,14 +1063,16 @@ async def _run_rest(self, verb: str, raw: str) -> dict: timeout=timeout, ) except httpx.PoolTimeout as e: - # Distinct from an upstream failure: Paperclip is fine, we ran out - # of local connections. Raised as RestUnavailable so the caller's - # existing MCP fallback still applies, but named so the logs say - # which it was. - raise PaperclipRestUnavailable( - f"connection pool exhausted after {self._pool_timeout_s:g}s " - f"(max_connections={self._max_connections}); too many concurrent " - f"Paperclip requests in this process" + # Deliberately NOT PaperclipRestUnavailable, so `_execute` and + # `search` do not fall back to MCP. The pool is sized to Paperclip's + # per-account limit of commands in flight, and Paperclip counts MCP + # connections against that same account — so falling back would + # just send the overflow to Paperclip by another door and earn a + # 429 there. Waiting longer is the remedy: raise the pool timeout. + raise PaperclipError( + f"no free Paperclip connection after {self._pool_timeout_s:g}s " + f"(max_connections={self._max_connections}, sized to the " + f"account's commands-in-flight limit); server is saturated" ) from e except Exception as e: # network errors, timeouts raise PaperclipRestUnavailable(f"{type(e).__name__}: {e}") from e diff --git a/crossbar_llm/paperclip_tools/tests/test_pool_timeout.py b/crossbar_llm/paperclip_tools/tests/test_pool_timeout.py index d48579d..2998b1e 100644 --- a/crossbar_llm/paperclip_tools/tests/test_pool_timeout.py +++ b/crossbar_llm/paperclip_tools/tests/test_pool_timeout.py @@ -1,18 +1,24 @@ -"""Connection-pool saturation must fail fast and say what happened. - -httpx expands a bare float timeout to connect/read/write/pool all at once, so -queueing for a free connection used to be charged against the 60s (or 480s) -call budget. Under load that surfaced as "Paperclip is slow" rather than "this -process is out of connections", and it quietly consumed the caller's per-tool -timeout. +"""The Paperclip connection pool is our client-side limit on commands in flight. + +Paperclip documents 10 short commands in flight per account; live, the binding +limit was its per-user search queue (10 concurrent searches passed, 15 drew +429s). The REST pool defaults to 10, so these tests pin three things: the pool +really does cap concurrency, excess commands queue rather than fail, and pool +exhaustion never spills onto the MCP transport — which Paperclip counts +against the same account. """ from __future__ import annotations +import asyncio + import httpx import pytest +from crossbar_llm.paperclip_tools import adapter as adapter_module from crossbar_llm.paperclip_tools.adapter import ( + ACCOUNT_MAX_SHORT_IN_FLIGHT, PaperclipAdapter, + PaperclipError, PaperclipRestUnavailable, ) @@ -32,6 +38,11 @@ async def post(self, url, **kwargs): return httpx.Response(200, json={"output": "ok"}, request=request) +def test_pool_defaults_to_the_account_limit(): + assert ACCOUNT_MAX_SHORT_IN_FLIGHT == 10 + assert PaperclipAdapter(api_key="k")._max_connections == 10 + + async def test_pool_timeout_is_separate_from_the_call_timeout(monkeypatch): adapter = PaperclipAdapter( api_key="k", disable_rest=False, timeout_s=60.0, pool_timeout_s=10.0 @@ -43,33 +54,98 @@ async def test_pool_timeout_is_separate_from_the_call_timeout(monkeypatch): assert isinstance(client.timeout, httpx.Timeout) assert client.timeout.read == 60.0 - # The whole point: queueing gets its own, much shorter budget. + # Waiting for a connection gets its own budget rather than the call's. assert client.timeout.pool == 10.0 -async def test_saturation_is_reported_as_pool_exhaustion(monkeypatch): +async def test_pool_exhaustion_is_a_saturation_error_not_a_rest_outage(monkeypatch): adapter = PaperclipAdapter( api_key="k", disable_rest=False, max_connections=4, pool_timeout_s=10.0 ) client = _RecordingClient(raises=httpx.PoolTimeout("no free connection")) monkeypatch.setattr(adapter, "_rest_client", lambda: client) - with pytest.raises(PaperclipRestUnavailable) as excinfo: + with pytest.raises(PaperclipError) as excinfo: await adapter._run_rest("search", "anything") + # Not RestUnavailable: that type is what triggers the MCP fallback. + assert not isinstance(excinfo.value, PaperclipRestUnavailable) message = str(excinfo.value) - assert "pool exhausted" in message - # Names the knob an operator would actually turn. + assert "no free Paperclip connection" in message assert "max_connections=4" in message -async def test_pool_size_is_configurable(monkeypatch): - adapter = PaperclipAdapter(api_key="k", max_connections=64) +async def test_pool_exhaustion_does_not_fall_back_to_mcp(monkeypatch): + """MCP shares the account's in-flight limit, so the overflow must not go + there — it would just be rejected by Paperclip through a different door.""" + adapter = PaperclipAdapter(api_key="k", disable_rest=False, pool_timeout_s=10.0) + client = _RecordingClient(raises=httpx.PoolTimeout("no free connection")) + monkeypatch.setattr(adapter, "_rest_client", lambda: client) + + async def mcp_must_not_run(command): + raise AssertionError(f"fell back to MCP for {command!r}") + + monkeypatch.setattr(adapter, "_run", mcp_must_not_run) + + with pytest.raises(PaperclipError, match="no free Paperclip connection"): + await adapter._execute("cat", "/papers/PMC1/meta.json") + + +async def test_pool_caps_commands_in_flight_and_queues_the_rest(monkeypatch): + """Against a real local HTTP server: 25 simultaneous commands through a + 10-connection pool never have more than 10 in flight, and all complete.""" + state = {"in_flight": 0, "peak": 0} + + async def handle(reader, writer): + try: + while True: + head = await reader.readuntil(b"\r\n\r\n") + length = 0 + for line in head.split(b"\r\n"): + if line.lower().startswith(b"content-length:"): + length = int(line.split(b":", 1)[1]) + await reader.readexactly(length) + state["in_flight"] += 1 + state["peak"] = max(state["peak"], state["in_flight"]) + await asyncio.sleep(0.05) + state["in_flight"] -= 1 + body = b'{"output": "ok"}' + writer.write( + b"HTTP/1.1 200 OK\r\nContent-Type: application/json\r\n" + b"Content-Length: %d\r\n\r\n%s" % (len(body), body) + ) + await writer.drain() + except (asyncio.IncompleteReadError, ConnectionResetError): + pass + finally: + writer.close() + + server = await asyncio.start_server(handle, "127.0.0.1", 0) + port = server.sockets[0].getsockname()[1] + monkeypatch.setattr(adapter_module, "REST_URL", f"http://127.0.0.1:{port}/") + + adapter = PaperclipAdapter(api_key="k", disable_rest=False, pool_timeout_s=10.0) + async with server: + try: + results = await asyncio.gather( + *(adapter._run_rest("cat", f"/papers/PMC{i}/meta.json") for i in range(25)) + ) + finally: + # Close the client BEFORE the server: on Python 3.12 the server's + # exit waits for open connections, and httpx keeps them alive. + await adapter.aclose() + + assert len(results) == 25 + assert all(r.get("output") == "ok" for r in results) + assert state["peak"] == ACCOUNT_MAX_SHORT_IN_FLIGHT + + +async def test_pool_size_is_configurable(): + adapter = PaperclipAdapter(api_key="k", max_connections=4) client = adapter._rest_client() try: - transport = client._transport - pool = transport._pool - assert pool._max_connections == 64 - assert pool._max_keepalive_connections == 32 + pool = client._transport._pool + assert pool._max_connections == 4 + assert pool._max_keepalive_connections == 2 finally: await client.aclose() diff --git a/crossbar_llm/tests/test_literature_service.py b/crossbar_llm/tests/test_literature_service.py index 08643fa..05d1235 100644 --- a/crossbar_llm/tests/test_literature_service.py +++ b/crossbar_llm/tests/test_literature_service.py @@ -291,10 +291,10 @@ def test_tool_usage_slices_by_node_name_prefix(): @pytest.mark.asyncio -async def test_admission_control_bounds_concurrent_runs(monkeypatch): - """Only `literature_max_concurrent_runs` tools may be in flight at once.""" +async def test_admission_limit_bounds_concurrent_runs_of_that_tool(monkeypatch): + """A tool's limit caps how many of ITS runs are in flight at once.""" settings = _settings() - settings.literature_max_concurrent_runs = 1 + settings.pubtator3_max_concurrent_runs = 1 service = LiteratureService(settings) in_flight = 0 @@ -311,27 +311,148 @@ async def tool(*args, **kwargs): finally: in_flight -= 1 - monkeypatch.setattr(service, "_run_paperclip", tool) monkeypatch.setattr(service, "_run_pubtator3", tool) - task = asyncio.create_task( + tasks = [ + asyncio.create_task( + service.run( + question="test", + payload=_payload(LiteratureToolsConfig(pubtator3=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + ) + for _ in range(3) + ] + await asyncio.sleep(0.05) + assert peak == 1, "several runs started despite a limit of 1" + release.set() + await asyncio.gather(*tasks) + + +@pytest.mark.asyncio +async def test_tools_do_not_share_admission_capacity(monkeypatch): + """A saturated PubTator3 must not turn Paperclip users away. + + With one shared pool, each tool's users took capacity from the other's even + though the two are bottlenecked by entirely different things. + """ + settings = _settings() + settings.pubtator3_max_concurrent_runs = 1 + settings.paperclip_max_concurrent_runs = 1 + settings.literature_admission_wait_seconds = 0.01 + service = LiteratureService(settings) + + release = asyncio.Event() + + async def blocker(*args, **kwargs): + await release.wait() + return {"final_answer": "answer", "documents": [], "warnings": []} + + async def paperclip(*args, **kwargs): + return {"final_answer": "answer", "citations": [], "warnings": []} + + monkeypatch.setattr(service, "_run_pubtator3", blocker) + monkeypatch.setattr(service, "_run_paperclip", paperclip) + + hog = asyncio.create_task( service.run( question="test", - payload=_payload(LiteratureToolsConfig(paperclip=True, pubtator3=True)), + payload=_payload(LiteratureToolsConfig(pubtator3=True)), callback=UsageMetricsCallback("session", strict=False), ) ) await asyncio.sleep(0.05) - assert peak == 1, "both tools started despite a concurrency limit of 1" + + result = await service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True, pubtator3=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + + assert result["paperclip"].status == "completed" + assert result["pubtator3"].status == "skipped" + assert "pubtator3" in result["pubtator3"].warnings[0] release.set() - await task + await hog + + +@pytest.mark.asyncio +async def test_unlimited_tool_admits_every_run(monkeypatch): + """`None` means no local limit — every run starts immediately.""" + settings = _settings() + settings.paperclip_max_concurrent_runs = None + settings.literature_admission_wait_seconds = 0.01 + service = LiteratureService(settings) + + started = 0 + release = asyncio.Event() + + async def tool(*args, **kwargs): + nonlocal started + started += 1 + await release.wait() + return {"final_answer": "answer", "citations": [], "warnings": []} + + monkeypatch.setattr(service, "_run_paperclip", tool) + + tasks = [ + asyncio.create_task( + service.run( + question="test", + payload=_payload(LiteratureToolsConfig(paperclip=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + ) + for _ in range(25) + ] + await asyncio.sleep(0.05) + assert started == 25 + release.set() + results = await asyncio.gather(*tasks) + assert all(r["paperclip"].status == "completed" for r in results) + + +@pytest.mark.asyncio +async def test_admission_wait_does_not_shorten_the_run_budget(monkeypatch): + """Queueing for a slot must not be charged against the run's timeout. + + The run timeout starts only once a slot is granted. Here a run queues for + longer than the entire run timeout and must still complete, because its + own clock hasn't started yet while it waits. + """ + settings = _settings(timeout=0.3) + settings.pubtator3_max_concurrent_runs = 1 + settings.literature_admission_wait_seconds = 2.0 + service = LiteratureService(settings) + + async def tool(*args, **kwargs): + await asyncio.sleep(0.25) # just inside the 0.3s run budget + return {"final_answer": "answer", "documents": [], "warnings": []} + + monkeypatch.setattr(service, "_run_pubtator3", tool) + + # Two runs through one slot: the second queues ~0.25s, then runs 0.25s. + # Its total (~0.5s) exceeds the 0.3s run timeout, so it would fail if + # queueing were charged against that budget. + results = await asyncio.gather( + *( + service.run( + question="test", + payload=_payload(LiteratureToolsConfig(pubtator3=True)), + callback=UsageMetricsCallback("session", strict=False), + ) + for _ in range(2) + ) + ) + + assert [r["pubtator3"].status for r in results] == ["completed", "completed"] @pytest.mark.asyncio async def test_saturation_reports_skipped_not_timed_out(monkeypatch): - """A rejected run must not masquerade as a slow upstream service.""" + """A run that never got a slot must not masquerade as a slow upstream.""" settings = _settings() - settings.literature_max_concurrent_runs = 1 + settings.pubtator3_max_concurrent_runs = 1 settings.literature_admission_wait_seconds = 0.01 service = LiteratureService(settings) @@ -339,16 +460,14 @@ async def test_saturation_reports_skipped_not_timed_out(monkeypatch): async def blocker(*args, **kwargs): await release.wait() - return {"final_answer": "answer", "citations": [], "warnings": []} + return {"final_answer": "answer", "documents": [], "warnings": []} - monkeypatch.setattr(service, "_run_paperclip", blocker) monkeypatch.setattr(service, "_run_pubtator3", blocker) - # Occupy the only slot, then ask for both tools. hog = asyncio.create_task( service.run( question="test", - payload=_payload(LiteratureToolsConfig(paperclip=True)), + payload=_payload(LiteratureToolsConfig(pubtator3=True)), callback=UsageMetricsCallback("session", strict=False), ) )