Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
70 changes: 70 additions & 0 deletions bin/chat-chainlit.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,9 @@
result_file_kwargs,
run_analysis,
)
from handoff import seed
from handoff.store import handoffs
from handoff.window import acknowledgement, claimed_id
from util.chainlit_helpers import (
PrefixedS3StorageClient,
is_feature_enabled,
Expand All @@ -42,6 +45,8 @@
mounted_secrets,
)

logger = logging.getLogger(__name__)

load_dotenv()
# Before anything reads os.environ. Docker secrets, where mounted, take
# precedence over .env; where not mounted, nothing changes.
Expand Down Expand Up @@ -204,6 +209,71 @@ async def send_file(path: Path) -> None:
await progress.remove()


async def continue_from_handoff(handoff_id: str) -> None:
"""Start this thread from the summary the reader clicked from.

The model gets the summary as its own previous turn plus the data it was
built from, at the tier the reader chose (FR-003); the reader sees their
summary verbatim. An unknown or expired handoff says so (FR-009), rather
than leaving an empty chat the reader will assume has the context.
"""
handoff = handoffs.get(handoff_id)
if handoff is None:
# Expired, never issued, or minted by another process -- the guest
# and logged-in chats do not share this in-memory store.
logger.info("handoff unavailable")
await cl.Message(content=seed.UNAVAILABLE).send()
return

profile: str = (cl.user_session.get("chat_profile") or "").lower()
# `on_chat_start` sets `thread_id` from the session id. A claim arriving
# before it has run would otherwise seed a thread called "None".
thread_id: str = cl.user_session.get("thread_id") or cl.user_session.get("id")
try:
data = await seed.analysis_data(handoff)
seeded = await get_graph().seed_history(
profile, thread_id=thread_id, messages=seed.seeded_turn(handoff, data)
)
except Exception:
# A failed fetch or graph update must not leave the reader in a chat
# that silently lacks the context they came for.
logger.exception("handoff seeding failed")
seeded = False
if not seeded:
logger.warning("handoff not seeded", extra={"profile": profile})
await cl.Message(content=seed.UNAVAILABLE).send()
return

logger.info(
"handoff claimed",
extra={"tier": handoff.tier, "with_data": data is not None},
)
await cl.Message(content=seed.shown_to_reader(handoff)).send()


@cl.on_window_message
async def on_window_message(message: object) -> None:
"""Claim a "Continue in chat" handoff posted by this tab (spec 013).

Chainlit forwards every message posted in the page, including this
server's own acknowledgement, so anything that is not a well-formed claim
is ignored. The tab retries until acknowledged, so one claim can arrive
several times; it is redeemed once per session and acknowledged every
time.
"""
handoff_id = claimed_id(message)
if handoff_id is None:
return

claimed: set[str] = cl.user_session.get("handoff_claimed") or set()
if handoff_id not in claimed:
claimed.add(handoff_id)
cl.user_session.set("handoff_claimed", claimed)
await continue_from_handoff(handoff_id)

await cl.send_window_message(acknowledgement(handoff_id))


@cl.on_message
async def main(message: cl.Message) -> None:
if await message_rate_limited(config):
Expand Down
2 changes: 2 additions & 0 deletions bin/chat-fastapi.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@
from agent.registry import build_graph, set_graph
from api.analysis_summary import router as analysis_summary_router
from api.answer import router as answer_router
from api.handoff import router as handoff_router
from util.caller_token import load_verifying_key
from util.captcha_scope import is_captcha_exempt
from util.embedding_environment import EmbeddingEnvironment
Expand Down Expand Up @@ -72,6 +73,7 @@ async def lifespan(_app: FastAPI) -> AsyncIterator[None]:
# is present -- a stricter bar than the answer endpoint's, because it discloses
# the user's own uploaded analysis rather than public pathway text.
app.include_router(analysis_summary_router, prefix=API_PREFIX)
app.include_router(handoff_router, prefix=API_PREFIX)
CHAINLIT_URL = os.getenv("CHAINLIT_URL")

CLOUDFLARE_SECRET_KEY = get_secret("CLOUDFLARE_SECRET_KEY")
Expand Down
38 changes: 38 additions & 0 deletions public/custom.js
Original file line number Diff line number Diff line change
Expand Up @@ -89,3 +89,41 @@

mo.observe(document.documentElement, { childList: true, subtree: true });
})();

/*
* Continue in chat (spec 013): claim a handoff carried in the URL fragment.
*
* The website opens /chat/guest/#handoff=<id> in a new tab. The fragment
* never reaches a server, so the ID stays out of logs and Referer headers.
* Chainlit forwards every window message to the server over this tab's own
* socket, which binds the handoff to this tab -- a cookie would be shared by
* every tab and could seed the wrong one.
*
* Chainlit drops a message posted before its socket is up, so this retries
* until the server acknowledges, and gives up after a while rather than
* posting forever.
*/
(function () {
const match = /(?:^|&)handoff=([A-Za-z0-9_-]{22,128})(?:&|$)/.exec(
window.location.hash.slice(1)
);
if (!match) return;
const id = match[1];

let acknowledged = false;
window.addEventListener('message', function (event) {
const data = event.data;
if (data && data.type === 'reactome-handoff-ack' && data.id === id) {
acknowledged = true;
}
});

let attempts = 0;
const timer = setInterval(function () {
if (acknowledged || ++attempts > 40) {
clearInterval(timer);
return;
}
window.postMessage({ type: 'reactome-handoff', id: id }, window.location.origin);
}, 500);
})();
167 changes: 167 additions & 0 deletions specs/013-continue-chat-hand/spec.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,167 @@
# Feature Specification: Continue in chat

**Feature Branch**: `013-continue-chat-hand`

**Created**: 2026-09-25

**Status**: Story 1 built and verified end to end; Stories 2 and 3 not started

**Input**: Adam, 2026-09-25: "with the chat on both the search page and the analysis results we want there to be a button to go to the [chat] interface … two options … either the logged in version or the guest version of the chat app and have the context already be set with either the search results or the analysis summary with the summary data." And: "the chat should open in a new tab."

## Context

The website already shows a chatbot-written summary in two places, both
served by this repository:

| where | endpoint | cached? |
|---|---|---|
| search page | `POST /api/answer` (spec 010) | no |
| analysis results | `POST /api/analysis/summary` (spec 011) | yes — `SummaryStore`, keyed `(token, release, tier)` |

The feature is a **"Continue in chat"** button beside each summary. It opens
the chat app in a new tab, guest or logged in, with the conversation already
holding what the user was looking at, so their first message can be a
follow-up rather than a restatement.

## User Scenarios & Testing *(mandatory)*

### User Story 1 — Continue an analysis summary in the guest chat (Priority: P1)

A user reads the summary of their enrichment analysis, clicks "Continue in
chat (guest)", and a new tab opens on the chat with that summary already in
the thread. They ask "which of these pathways involve TP53?" and get an
answer grounded in their analysis.

**Why this priority**: analysis summaries are already cached and carry a
disclosure tier, so the handoff has everything it needs; and the guest chat
is the one path verifiable on beta.

**Independent test**: from a real analysis token, the new tab's thread begins
with the same summary text the website showed, and a follow-up question is
answered using the analysis.

### User Story 2 — Continue a search-page answer (Priority: P2)

The same, from the search page's AI answer and its results.

**Why P2**: `/api/answer` does not keep what it generated, so it needs a
store before the handoff can show the *same* answer (see FR-002).

### User Story 3 — Continue in the logged-in chat (Priority: P2)

The same handoff into `/chat/personal/`, so the conversation is kept in the
user's history.

**Why P2, and why it cannot be verified on beta**: `/chat/personal` is not
deployed on beta — it needs a second container, Postgres, and OAuth whose
redirect URIs are bound to reactome.org. Verifiable in production only.

## Requirements *(mandatory)*

- **FR-001 Pass a reference, never the content.** No search result, summary
text or analysis data in a URL. URLs are written to nginx logs, leak to
other sites through `Referer`, and have length limits. The website asks the
chatbot for a short-lived handoff ID; the link carries only that ID.

- **FR-002 The chat continues the summary the user saw, and does not
regenerate it.** Answers are not reproducible here — the same question on
the same build scores ~0.33 similarity run to run — so a regenerated summary
would greet the user with different text from the one they clicked from.
The handoff hands over the stored text. For analysis summaries it already
exists in `SummaryStore`; for search answers a store is required first.

- **FR-003 The disclosure choice carries over.** A summary produced at the
`aggregate` tier continues at `aggregate`: the chat model receives the same
allow-listed view the summary was built from, and nothing wider. Continuing
in chat must not silently send OpenAI more than the user agreed to on the
website. The `identifiers` tier carries over only if it was chosen there.

- **FR-004 The handoff ID travels in the URL fragment**
(`/chat/guest/#handoff=<id>`). Browsers never send the fragment to a
server, so the ID does not reach nginx logs or `Referer` at all — FR-001
enforced by the browser rather than by us remembering.

- **FR-005 The handoff is bound to the tab, not the browser.** The link opens
in a new tab (FR-008), so a user can open several. A cookie is shared by
every tab: two handoffs in quick succession could overwrite each other
before the first tab connects, and that tab would open on the *wrong*
context. The ID is instead read by a script in the tab itself and sent over
that tab's own connection (FR-006).

- **FR-006 Mechanism (to be prototyped).** Chainlit 2.11 does not give the app
the page URL on connect — its handler reads cookies only. It does offer
`custom_js` (a script injected into the chat page) and `@cl.on_window_message`
(a server hook for messages posted in the page). The script reads the
fragment and posts the ID; the hook redeems it and seeds the thread. The
risk to prove first: the post must arrive after the tab's socket is
connected, or it is lost. The script retries until the server acknowledges.

- **FR-007 Valid for minutes, not once.** A single-use ID would give an empty
chat on a reload. The ID is redeemable for a short window (target: 15
minutes) and then refused. It grants read access to one summary, so the
window is short and IDs are unguessable (≥128 random bits).

- **FR-008 New tab.** The website opens the chat with `target="_blank"` and
`rel="noopener noreferrer"`, so the chat tab cannot reach back into the
website tab and receives no `Referer`.

- **FR-009 An expired or unknown ID says so.** The chat opens normally and
tells the user the context could not be loaded, rather than silently
starting an empty conversation they will assume has the context.

## Who builds what

| | repo |
|---|---|
| the two buttons, the new-tab link, requesting a handoff ID | WebsiteAngular |
| `POST /api/handoff`, the store, the fragment script, the hook that seeds the thread | reactome_chatbot |

## Out of scope

- Seeding the chat with the raw search results or the full analysis result.
The chat receives what the summary was built from, bounded as it already
is, and can fetch more through its own tools.
- Carrying a handoff across devices or accounts.

## Open questions

- Whether the button should be two buttons (guest / logged in, as asked) or
one that the chat resolves after login. Two is what was asked for; recorded
so the choice is visible, not reopened.
- The handoff window length (FR-007) — 15 minutes is a guess to be revisited
once real use shows how long people take.

---

## What was built, 2026-09-25

**Story 1 (analysis summary → guest chat) works end to end**, verified in a
headless browser against a real analysis on beta: the summary comes from the
real endpoint, the handoff is minted for it, the tab opens on *the same text*
the website was given, and a follow-up ("which pathway has the lowest FDR?")
is answered from the analysis -- naming its real top pathway with its exact
FDR. The control, the same question in a chat with no handoff, cannot answer
it at all. An identifiers-tier handoff is refused when the reader chose
aggregate, and an unknown ID says so.

**The transport, measured.** A single `postMessage` in the first second after
load is lost 15 times out of 15 -- Chainlit drops window messages until its
socket is up -- and from two seconds on is claimed 6 of 6. `custom.js`
retries every 500 ms for up to 20 s. Two tabs opened together in one browser
each claimed only their own ID.

**Story 3 needs a shared store before it can work.** The handoff store, like
the summary cache it copies from, is in process memory, and the guest and
logged-in chats are separate processes. A handoff minted by one cannot be
claimed by the other. The endpoint therefore returns only `path`, for the
chat served by the process that minted it; the first version also returned a
`personal_path` from the guest deployment, which could only ever have opened
on "couldn't load the summary". Two ways to do Story 3, to decide later:

- a store both processes share (production already has Postgres for the
logged-in chat's LangGraph checkpoints), with the summary cache moved into
it too, so a handoff can cross processes and still continue the *same*
summary; or
- the website calls the logged-in deployment's own summary and handoff
endpoints -- simpler, but the summary would be generated a second time in
that process, and would not be the one the reader saw (FR-002).
24 changes: 24 additions & 0 deletions src/agent/graph.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
from langchain_core.documents import Document
from langchain_core.embeddings import Embeddings
from langchain_core.language_models.chat_models import BaseChatModel
from langchain_core.messages import BaseMessage
from langchain_core.runnables import RunnableConfig
from langgraph.checkpoint.base import BaseCheckpointSaver
from langgraph.checkpoint.memory import MemorySaver
Expand Down Expand Up @@ -527,6 +528,29 @@ async def astream_answer(
kind="done", state="answered" if answered else "nothing_found"
)

async def seed_history(
self, profile: str, *, thread_id: str, messages: list[BaseMessage]
) -> bool:
"""Put a turn into a thread's history as if it had been asked here.

For spec 013's handoff: the reader's summary becomes the thread's
previous turn, so their first message can be a follow-up. Recorded as
having ended at `postprocess`, the graph's last node, so the next
`ainvoke` starts an ordinary new turn with this history in place.

Returns False, and changes nothing, for an unknown profile.
"""
if self.graph is None:
self.graph = await self.initialize()
if profile not in self.graph:
return False
await self.graph[profile].aupdate_state(
RunnableConfig(configurable={"thread_id": thread_id}),
{"chat_history": messages},
as_node="postprocess",
)
return True

async def ainvoke(
self,
user_input: str,
Expand Down
12 changes: 11 additions & 1 deletion src/api/analysis_summary.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,7 @@
from agent.models import get_llm
from analysis.client import current_release, fetch_not_found, fetch_result
from analysis.disclosure import Tier, for_tier
from analysis.store import SummaryStore
from analysis.store import Stored, SummaryStore
from analysis.summarise import (
INEXACT_COUNT_INSTRUCTION,
NAMED_UNMATCHED_INSTRUCTION,
Expand Down Expand Up @@ -61,6 +61,16 @@
#: a request, as the limiter does.
_store = SummaryStore()


def stored_summary(token: str, release: str, tier: str) -> Stored | None:
"""A summary this endpoint generated and kept, if it still has it.

For spec 013's handoff, which may only continue a summary the reader
has actually been shown.
"""
return _store.get(token, release, tier)


SYSTEM_PROMPT = """
You explain a completed Reactome pathway-analysis result to the researcher who
ran it.
Expand Down
Loading
Loading