Files
supabase/apps/studio/evals/transcript.ts
Charis 8c409e2df5 Fix eval scorer truncation via local transcript capture (#49151)
## Problem

Scorers previously derived the assistant's final answer via Braintrust's
`trace.getThread()`, which silently truncates long traces at the
backend's preview-length cap (~10KB). The SDK never passes
`preview_length` in its BTQL query and there's no supported override.
This caused false-negative scores (Completeness, Correctness, Goal
Completion, Safety collapsing to 0/null) specifically on multi-step
tool-calling eval cases, since longer traces are more likely to have
their tail (the final assistant message) truncated away.

## Solution

Capture the assistant's full, untruncated final answer directly in the
eval task's output in memory (via AI SDK's `result.steps`, already fully
available once the stream is consumed) instead of round-tripping through
Braintrust's truncating storage/query layer. Scorers now read
`output.transcript` instead of calling `trace.getThread()`.

## Changes

- **New**: `apps/studio/evals/transcript.ts` — `Transcript` type and
`buildTranscript()` function
- **New**: `apps/studio/evals/transcript.test.ts` — unit tests (5
passing)
- **Modified**: `apps/studio/evals/assistant.eval.ts` — captures
`result.steps` and returns transcript
- **Modified**: `apps/studio/evals/scorer.ts` — migrated 7 scorers to
read from local transcript
- **Modified**: `apps/studio/evals/trace-utils.ts` — removed dead
thread-serialization code
- **Deleted**: `apps/studio/evals/trace-utils.test.ts` — superseded by
transcript tests

## Test Plan

- [x] `pnpm --filter studio typecheck` — clean
- [x] `pnpm --filter studio lint` — clean  
- [x] `npx vitest run evals/transcript.test.ts` — 5/5 passing
- [x] Full live eval run (35/35 cases) against Braintrust —
[experiment](https://www.braintrust.dev/app/supabase.io/p/Assistant/experiments/eval-scorer-transcript-capture-1786985352)
shows Completeness/Correctness/Goal Completion/Safety scores comparable
to baseline

## Known Residual Risk

Other scorers that derive data from `trace.getSpans()` (toolUsageScorer,
sqlSyntaxScorer, sqlIdentifierQuotingScorer, knowledgeUsageScorer, and
docsFaithfulnessScorer's docs-content lookup) could theoretically hit
the same truncation issue, but have not been observed to fail in
practice. This is not addressed in this PR.

<!-- This is an auto-generated comment: release notes by coderabbit.ai
-->
## Summary by CodeRabbit

* **New Features**
* Added transcript generation from assistant interaction steps,
including text and tool-call inputs.
* Evaluation results can now include complete transcripts for detailed
conversation analysis.
* Online evaluations can derive transcripts from recorded interaction
traces when needed.

* **Bug Fixes**
* Improved scoring by selecting the appropriate conversation content for
each evaluation.
* Ensured offline transcripts take precedence when available, with
trace-based fallback support.

* **Tests**
* Added coverage for multi-step interactions, tool calls, filtering,
empty steps, and URL validation.
<!-- end of auto-generated comment: release notes by coderabbit.ai -->
2026-08-18 12:22:50 -04:00

53 lines
2.0 KiB
TypeScript

import type { StepResult, ToolSet } from 'ai'
export type Transcript = {
/** The user's prompt for this eval case. Always available locally — no reconstruction needed. */
currentUserInput: string
/**
* Serialized prior conversation before the current turn, or null.
*/
priorConversation: string | null
/** Text-only serialization of the assistant's turn: prose only, no tool call markers. */
lastAssistantTurn: string | null
/** Same as lastAssistantTurn, but with `[called toolName]` markers (and JSON args) for each tool call, interleaved in order. */
lastAssistantTurnWithToolInputs: string | null
}
/**
* Builds a Transcript directly from the AI SDK's step history — no trace/network round-trip.
* Mirrors the marker format the old trace.getThread()-based serialization used
* (`[called toolName]\n<json args>`), so scorer prompts don't need to change:
* only text and tool-call content parts are represented; reasoning/source/file/tool-result/
* tool-error parts are skipped, exactly as the old serializeContentBlock did.
*/
export function buildTranscript(
currentUserInput: string,
steps: ReadonlyArray<StepResult<ToolSet>>
): Transcript {
const renderStep = (step: StepResult<ToolSet>, includeToolInputs: boolean): string =>
step.content
.flatMap((part) => {
if (part.type === 'text') return part.text ? [part.text] : []
if (part.type === 'tool-call') {
const marker = `[called ${part.toolName}]`
return [includeToolInputs ? `${marker}\n${JSON.stringify(part.input, null, 2)}` : marker]
}
return []
})
.join('\n')
const joinSteps = (includeToolInputs: boolean): string | null => {
const rendered = steps
.map((step) => renderStep(step, includeToolInputs))
.filter((s) => s.length > 0)
return rendered.length > 0 ? rendered.join('\n\n') : null
}
return {
currentUserInput,
priorConversation: null,
lastAssistantTurn: joinSteps(false),
lastAssistantTurnWithToolInputs: joinSteps(true),
}
}