Skip to content

Commit f8852fa

Browse files
Fix remote eval reruns by preserving upsert IDs (#763)
### Why Fixes [COR-84](https://linear.app/braintrustdata/issue/COR-84/remote-eval-playground-re-running-a-row-leaves-a-duplicate-grid-row). The playground sends a stable `upsert_id` for each row/eval column so reruns replace the existing result. Python's `EvalCase.from_dict()` drops this field, and the eval runner creates a fresh root record on every run. The resulting records appear as duplicate grid rows with stale outputs. The JS SDK already honors `upsert_id`. This addresses the SDK cause identified during review of the earlier [UI workaround, braintrust#20332](braintrustdata/braintrust#20332). ### Repro 1. Configure a dataset-backed playground with two Python remote evals. 2. Run a row, then rerun it in either eval column. 3. Although the request supplies the same `upsert_id`, Python writes a different root record ID, leaving both results in the grid. The regression test reproduces this by running an evaluator twice with the same `upsert_id` and checking the logged root record IDs. It fails before the fix. ### Fix Preserve optional `upsert_id` in `EvalCase` and its input TypedDicts. The first trial uses that value as its root record ID; additional trials derive deterministic UUIDs from the upsert ID and trial index. This keeps the single-trial playground behavior and lets each additional trial replace its own result on rerun without overwriting other trials. Both experiment-backed and parent-context evaluations use this logic. Missing or empty IDs continue to generate fresh record IDs; dataset row IDs and origins retain their existing meaning. This prevents future duplicates after SDK upgrade; it does not remove historical duplicate records. This PR changes Python only; the analogous JS multi-trial behavior needs a separate fix. ### Test - All 48 rerun regression cases pass, covering dictionary/dataclass inputs, experiment/parent-context execution, supplied/missing/empty upsert IDs, and global/per-row trial counts. The tests verify distinct stable IDs per trial, updated output, preserved origin, and matching root/child outputs. The multi-trial cases fail with the bare upsert ID reused across trials. - Full core suite: 850 passed, 63 skipped, 12 expected failures. - All 35 type tests pass, including the optional field on dataclass and TypedDict inputs; pyright and mypy pass. - Ruff lint, formatting, and commit hooks pass. - No manual browser run with the patched SDK yet.
1 parent 3f8ba89 commit f8852fa

4 files changed

Lines changed: 79 additions & 1 deletion

File tree

‎py/src/braintrust/framework.py‎

Lines changed: 8 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -7,6 +7,7 @@
77
import re
88
import sys
99
import traceback
10+
import uuid
1011
import warnings
1112
from collections import defaultdict
1213
from collections.abc import Awaitable, Callable, Coroutine, Iterable, Iterator, Mapping, Sequence
@@ -104,6 +105,7 @@ class EvalCase(SerializableDataClass, Generic[Input, Expected]):
104105
_xact_id: str | None = None
105106
created: str | None = None
106107
origin: ObjectReference | None = None
108+
upsert_id: str | None = None
107109

108110

109111
# Inheritance doesn't quite work for dataclasses, so we redefine the fields
@@ -1648,6 +1650,12 @@ async def run_evaluator_task(datum, trial_index=0):
16481650
tags=tags,
16491651
**({"origin": origin} if origin is not None else {}),
16501652
)
1653+
if datum.upsert_id:
1654+
base_event["id"] = (
1655+
datum.upsert_id
1656+
if trial_index == 0
1657+
else str(uuid.uuid5(uuid.NAMESPACE_URL, f"braintrust:eval:{datum.upsert_id}:trial:{trial_index}"))
1658+
)
16511659

16521660
if experiment:
16531661
root_span = experiment.start_span(**base_event)

‎py/src/braintrust/test_framework.py‎

Lines changed: 54 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -4,7 +4,7 @@
44
from unittest.mock import MagicMock, patch
55

66
import pytest
7-
from braintrust.logger import BraintrustState, Dataset, ObjectMetadata, ProjectDatasetMetadata
7+
from braintrust.logger import BraintrustState, Dataset, ObjectMetadata, ProjectDatasetMetadata, parent_context
88
from braintrust.util import LazyValue
99

1010
from .framework import (
@@ -69,6 +69,59 @@ def test_eval_case_from_dict_preserves_valid_origin():
6969
assert EvalCase.from_dict({"input": 1, "origin": SOURCE_ORIGIN}).origin == SOURCE_ORIGIN
7070

7171

72+
@pytest.mark.parametrize("upsert_id", ["eval-row", None, ""])
73+
@pytest.mark.parametrize("as_dict", [True, False], ids=["dict", "dataclass"])
74+
@pytest.mark.parametrize("use_experiment", [True, False], ids=["experiment", "parent-context"])
75+
@pytest.mark.parametrize(("trial_count", "row_trial_count"), [(1, None), (3, None), (1, 3), (3, 1)])
76+
@pytest.mark.asyncio
77+
async def test_run_evaluator_upsert_id(
78+
upsert_id, as_dict, use_experiment, trial_count, row_trial_count, with_memory_logger, with_simulate_login
79+
):
80+
experiment = init_test_exp("test-upsert", "test-project")
81+
expected_trials = row_trial_count if row_trial_count is not None else trial_count
82+
root_ids = []
83+
for input_value in [1, 2]:
84+
data = {"input": input_value, "id": "dataset-row", "origin": SOURCE_ORIGIN, "trial_count": row_trial_count}
85+
if upsert_id is not None:
86+
data["upsert_id"] = upsert_id
87+
evaluator = Evaluator(
88+
project_name="test-project",
89+
eval_name="test-upsert",
90+
data=[data if as_dict else EvalCase(**data)],
91+
task=lambda input_value, hooks: [input_value * 2, hooks.trial_index],
92+
scores=[],
93+
experiment_name=None,
94+
metadata=None,
95+
summarize_scores=False,
96+
trial_count=trial_count,
97+
)
98+
with parent_context(experiment.export()):
99+
await run_evaluator(
100+
experiment=experiment if use_experiment else None,
101+
evaluator=evaluator,
102+
position=None,
103+
filters=[],
104+
)
105+
logs = with_memory_logger.pop()
106+
roots = [log for log in logs if not log["span_parents"]]
107+
children = [log for log in logs if log["span_parents"]]
108+
assert len(roots) == len(children) == expected_trials
109+
assert sorted(root["output"] for root in roots) == [
110+
[input_value * 2, trial_index] for trial_index in range(expected_trials)
111+
]
112+
assert all(root["origin"] == SOURCE_ORIGIN for root in roots)
113+
root_outputs = {root["root_span_id"]: root["output"] for root in roots}
114+
assert all(child["output"] == root_outputs[child["root_span_id"]] for child in children)
115+
root_ids.append({root["output"][1]: root["id"] for root in roots})
116+
117+
if upsert_id:
118+
assert root_ids[0] == root_ids[1]
119+
assert root_ids[0][0] == upsert_id
120+
else:
121+
assert set(root_ids[0].values()).isdisjoint(root_ids[1].values())
122+
assert all("dataset-row" not in ids.values() for ids in root_ids)
123+
124+
72125
@pytest.mark.parametrize(
73126
("inline_origin", "expected_origin"),
74127
[(SOURCE_ORIGIN, SOURCE_ORIGIN), (INVALID_ORIGIN, None)],

‎py/src/braintrust/type_tests/test_eval_generics.py‎

Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -16,6 +16,7 @@
1616
from braintrust.framework import EvalAsync, EvalCase, EvalResultWithSummary
1717
from braintrust.generated_types import ObjectReference
1818
from braintrust.score import Score
19+
from braintrust.types._eval import EvalCaseDict, EvalCaseDictNoOutput
1920

2021

2122
# --- Domain types for testing ---
@@ -141,3 +142,17 @@ async def test_eval_origin_types():
141142
no_send_logs=True,
142143
)
143144
assert result.results[0].origin == origin
145+
146+
147+
def test_eval_upsert_id_types() -> None:
148+
eval_case: EvalCase[str, str] = EvalCase(input="case", upsert_id="case-root")
149+
assert eval_case.upsert_id == "case-root"
150+
dict_case: EvalCaseDictNoOutput[str] = {"input": "dictionary", "upsert_id": "dict-root"}
151+
expected_case: EvalCaseDict[str, str] = {
152+
"input": "expected",
153+
"expected": "expected",
154+
"upsert_id": "expected-root",
155+
}
156+
157+
assert EvalCase.from_dict(dict(dict_case)).upsert_id == "dict-root"
158+
assert EvalCase.from_dict(dict(expected_case)).upsert_id == "expected-root"

‎py/src/braintrust/types/_eval.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -34,6 +34,7 @@ class EvalCaseDictNoOutput(Generic[Input], TypedDict):
3434
_xact_id: NotRequired[str | None]
3535
created: NotRequired[str | None]
3636
origin: NotRequired[ObjectReference | None]
37+
upsert_id: NotRequired[str | None]
3738

3839

3940
class EvalCaseDict(Generic[Input, Expected], EvalCaseDictNoOutput[Input]):
@@ -57,3 +58,4 @@ class ExperimentDatasetEvent(TypedDict):
5758
tags: Sequence[str] | None
5859
created: NotRequired[str | None]
5960
origin: NotRequired[ObjectReference | None]
61+
upsert_id: NotRequired[str | None]

0 commit comments

Comments
 (0)