Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 0 additions & 7 deletions context-graph/eval/src/context_graph_eval/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -121,12 +121,6 @@ def main(argv: list[str] | None = None) -> int:
"for it to build that this strategy would read. 'hybrid' finds turns by vector and text "
"search and uses the typed graph to find facts and more turns (see hybrid.py).",
)
run.add_argument(
"--question-date",
action="store_true",
help="tell the answering LLM the date each question is asked (#367). Off by default, so runs "
"stay comparable with ones scored without it; compare only runs that agree on it.",
)
run.add_argument(
"--hybrid-user-fact-types-from",
choices=("facts", "names"),
Expand Down Expand Up @@ -478,7 +472,6 @@ def _run(args) -> int:
lanes=tuple(lane for lane in args.hybrid_lanes.split(",") if lane),
user_fact_types_from=args.hybrid_user_fact_types_from,
),
question_date=args.question_date,
),
)
)
Expand Down
18 changes: 9 additions & 9 deletions context-graph/eval/src/context_graph_eval/hybrid.py
Original file line number Diff line number Diff line change
Expand Up @@ -267,13 +267,11 @@ def _turn_rows(graph: ReadOnlyGraph, turn_ids: list[str], chars: int) -> list[st
"RETURN a.action_id AS id, s.session_id AS session, a.timestamp AS ts, a.action_type AS kind, a.text AS text",
{"ids": turn_ids},
)
by_id = {r["id"]: r for r in rows}
out = []
for turn_id in turn_ids:
r = by_id.get(turn_id)
if r:
speaker = "user" if r["kind"] == "user_message" else "assistant"
out.append(f"TURN [session {r['session']}, {str(r['ts'])[:16]}, {speaker}]: {r['text'][:chars]}")
# In time order, like the facts: "which came first" and "what is current" read off the sequence.
for r in sorted(rows, key=lambda r: (str(r["ts"]), r["id"])):
speaker = "user" if r["kind"] == "user_message" else "assistant"
out.append(f"TURN [session {r['session']}, {str(r['ts'])[:16]}, {speaker}]: {r['text'][:chars]}")
return out


Expand Down Expand Up @@ -375,10 +373,12 @@ async def retrieve_hybrid(
if facts:
turn_ids += [f["turn"] for f in facts.values() if f.get("turn")][: config.fact_turns_k]

# Facts in time order: a knowledge-update question wants the latest value,
# a temporal one the sequence.
# Turns first: they are the answer store and the facts an index into them.
# Placed after ~45 fact rows, a turn's evidence was read past (#398). Both
# in time order: a knowledge-update question wants the latest value, a
# temporal one the sequence.
ordered = sorted(facts.values(), key=lambda f: f.get("valid_at") or "")
seen = [_fact(f) for f in ordered] + _turn_rows(graph, list(dict.fromkeys(turn_ids)), config.turn_chars)
seen = _turn_rows(graph, list(dict.fromkeys(turn_ids)), config.turn_chars) + [_fact(f) for f in ordered]
answer = await llm.complete(answer_prompt(question, seen, today))
return Retrieved(
answer=answer.strip(), retrieval_context=seen, queries=queries, latency_seconds=time.monotonic() - started
Expand Down
19 changes: 15 additions & 4 deletions context-graph/eval/src/context_graph_eval/retrieval.py
Original file line number Diff line number Diff line change
Expand Up @@ -543,11 +543,22 @@ def answer_prompt(question: str, seen: list[str], today: str | None = None) -> s
different answering prompts.
"""
rows = "\n".join(seen[:200]) if seen else "(nothing was retrieved)"
# Opt-in (--question-date, #367): "how many days ago" is unanswerable
# without knowing when the question is asked, whatever the graph holds.
# "How many days ago" is unanswerable without knowing when the question is
# asked, whatever the graph holds (#367).
asked = f"The question is being asked on {today}.\n" if today else ""
return (
"Answer the question using only the rows below. Be concise. "
'If the rows do not contain the answer, say exactly "not in memory".\n\n'
"Answer the user's question from their memory: the rows below, retrieved from their past "
"conversations. Each row carries the date it was said or became true.\n"
'- "Today", "yesterday", "last week" inside a row are relative to that row\'s date, not to now.\n'
"- For a count, a total, or the time between events: list each matching item or value with its "
"date, count an item mentioned in several rows once, then compute.\n"
"- When rows disagree about the same thing, the most recent one is current.\n"
"- When the question asks for a recommendation or suggestion, always recommend: use the "
"preferences, interests and possessions the rows show, add your own knowledge where they name "
'nothing specific, and say which preferences you used. Never answer one with "not in memory".\n'
"- Any other question about something the rows never mention, or that assumes something they do "
'not show: say exactly "not in memory" -- do not answer a related question instead.\n'
'- Otherwise, if the rows do not contain the answer, say exactly "not in memory".\n'
"Keep the working short and put the final answer last.\n\n"
f"Rows:\n{rows}\n\n{asked}Question: {question}\nAnswer:"
)
8 changes: 4 additions & 4 deletions context-graph/eval/src/context_graph_eval/runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -102,9 +102,6 @@ class RunPlan:
#: and where its embeddings are cached between runs over the same graph.
hybrid: HybridConfig = field(default_factory=HybridConfig)
hybrid_cache: Path = Path(".cache/context-graph-eval/hybrid-index.npz")
#: Tell the answering LLM when each question is asked (#367). Off by
#: default so runs stay comparable with those scored without it.
question_date: bool = False


@dataclass(frozen=True)
Expand Down Expand Up @@ -296,7 +293,10 @@ async def _retrieve_all(
async def one(golden: "Golden") -> Retrieved:
async with limiter:
started = time.monotonic()
today = (golden.additional_metadata or {}).get("question_date") if plan.question_date else None
# When the question is asked -- the corpus's question_date, as a
# real session's "now" -- without which "how many days ago" and
# "this year" have nothing to count from (#367).
today = (golden.additional_metadata or {}).get("question_date")
try:
if plan.retrieval_strategy == "hybrid":
# Each question is its own user's history (to_session_fixtures).
Expand Down
29 changes: 29 additions & 0 deletions context-graph/eval/tests/test_hybrid.py
Original file line number Diff line number Diff line change
Expand Up @@ -186,3 +186,32 @@ async def context(user_id):
assert "Anna" in scoped
assert "Bob" not in scoped
assert "Bob" in await context(None)


@pytest.mark.asyncio
async def test_turns_come_before_facts_and_both_in_time_order(eval_graph: ActionsGraph, tmp_path):
"""Turns are the answer store: after ~45 fact rows their evidence was read past.
Time order lets "which came first" and "what is current" read off the sequence."""
_plant(eval_graph)
eval_graph.ensure_session(Session(session_id="s0", started_at="2023-01-02T09:00:00+00:00"))
eval_graph.record_message(
session_id="s0",
role=MessageRole.USER,
content="I went to the Museum of Modern Art with Anna.",
timestamp="2023-01-02T09:00:00+00:00",
)
index = ensure_hybrid_index(eval_graph, tmp_path / "index.npz", embedder=_BagOfWords())

result = await retrieve_hybrid(
"Museum of Modern Art",
graph=ReadOnlyGraph(eval_graph.db),
llm=_EchoLLM(),
index=index,
config=HybridConfig(lanes=("turns", "facts")),
)

kinds = [row.split(" ", 1)[0] for row in result.retrieval_context]
assert kinds == sorted(kinds, key=lambda kind: kind != "TURN")
turns = [row for row in result.retrieval_context if row.startswith("TURN")]
assert turns[0].startswith("TURN [session s0, 2023-01-02T09:00")
assert turns == sorted(turns, key=lambda row: row.split(", ")[1])
25 changes: 25 additions & 0 deletions context-graph/eval/tests/test_runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -97,6 +97,31 @@ async def test_bleu_f1_and_latency_are_scored_without_a_judge(eval_graph: Action
assert scored.latency_seconds > 0.0


async def test_every_question_is_answered_knowing_when_it_is_asked(eval_graph: ActionsGraph):
"""'How many days ago' has nothing to count from without the question date,
so the runner always passes it -- there is no run that should go without (#367)."""

class _Recording(_StubLLM):
def __init__(self):
super().__init__()
self.prompts: list[str] = []

async def complete(self, prompt: str) -> str:
self.prompts.append(prompt)
return await super().complete(prompt)

llm = _Recording()
await run_batch(
[to_golden(_record("q1"))],
records=[_record("q1")],
graph=eval_graph,
llm=llm,
plan=RunPlan(reconcile=False, judge=None),
)

assert "The question is being asked on 2023/06/15 (Thu) 09:12." in llm.prompts[-1]


async def test_fixtures_are_injected_before_retrieval_runs(eval_graph: ActionsGraph):
"""Ordering is the runner's whole job: retrieving before injection would
query an empty graph and score every question as a miss."""
Expand Down
13 changes: 12 additions & 1 deletion context-graph/eval/tests/test_text_search.py
Original file line number Diff line number Diff line change
Expand Up @@ -190,7 +190,7 @@ def test_safe_query_keeps_meaningful_short_words():

@pytest.mark.asyncio
async def test_the_question_date_reaches_the_answer_prompt_only_when_given(eval_graph: ActionsGraph):
"""#367: 'how many days ago' needs the day the question is asked; opt-in so default runs stay comparable."""
"""#367: 'how many days ago' needs the day the question is asked."""
_plant(eval_graph, "s1", role=MessageRole.USER, content="I adopted a beagle named Max")
ensure_turn_text_index(eval_graph)
llm = _EchoLLM()
Expand All @@ -202,3 +202,14 @@ async def test_the_question_date_reaches_the_answer_prompt_only_when_given(eval_

assert "is being asked on" not in llm.prompts[0]
assert "The question is being asked on 2023/06/15 (Thu) 09:12." in llm.prompts[1]


def test_a_recommendation_question_is_never_refused_for_lack_of_a_specific_item():
"""'Recommend a conference' names no conference in memory; declining it under
the false-premise rule threw away the user's stated interests (#404)."""
from context_graph_eval.retrieval import answer_prompt

prompt = answer_prompt("Can you recommend a conference?", ["TURN [...]: I work on medical imaging."])

assert 'Never answer one with "not in memory"' in prompt
assert prompt.index("recommendation or suggestion") < prompt.index("Any other question")
Loading