mirror of
https://github.com/supabase/supabase.git
synced 2026-09-06 09:59:03 +08:00
## Problem Scorers previously derived the assistant's final answer via Braintrust's `trace.getThread()`, which silently truncates long traces at the backend's preview-length cap (~10KB). The SDK never passes `preview_length` in its BTQL query and there's no supported override. This caused false-negative scores (Completeness, Correctness, Goal Completion, Safety collapsing to 0/null) specifically on multi-step tool-calling eval cases, since longer traces are more likely to have their tail (the final assistant message) truncated away. ## Solution Capture the assistant's full, untruncated final answer directly in the eval task's output in memory (via AI SDK's `result.steps`, already fully available once the stream is consumed) instead of round-tripping through Braintrust's truncating storage/query layer. Scorers now read `output.transcript` instead of calling `trace.getThread()`. ## Changes - **New**: `apps/studio/evals/transcript.ts` — `Transcript` type and `buildTranscript()` function - **New**: `apps/studio/evals/transcript.test.ts` — unit tests (5 passing) - **Modified**: `apps/studio/evals/assistant.eval.ts` — captures `result.steps` and returns transcript - **Modified**: `apps/studio/evals/scorer.ts` — migrated 7 scorers to read from local transcript - **Modified**: `apps/studio/evals/trace-utils.ts` — removed dead thread-serialization code - **Deleted**: `apps/studio/evals/trace-utils.test.ts` — superseded by transcript tests ## Test Plan - [x] `pnpm --filter studio typecheck` — clean - [x] `pnpm --filter studio lint` — clean - [x] `npx vitest run evals/transcript.test.ts` — 5/5 passing - [x] Full live eval run (35/35 cases) against Braintrust — [experiment](https://www.braintrust.dev/app/supabase.io/p/Assistant/experiments/eval-scorer-transcript-capture-1786985352) shows Completeness/Correctness/Goal Completion/Safety scores comparable to baseline ## Known Residual Risk Other scorers that derive data from `trace.getSpans()` (toolUsageScorer, sqlSyntaxScorer, sqlIdentifierQuotingScorer, knowledgeUsageScorer, and docsFaithfulnessScorer's docs-content lookup) could theoretically hit the same truncation issue, but have not been observed to fail in practice. This is not addressed in this PR. <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit * **New Features** * Added transcript generation from assistant interaction steps, including text and tool-call inputs. * Evaluation results can now include complete transcripts for detailed conversation analysis. * Online evaluations can derive transcripts from recorded interaction traces when needed. * **Bug Fixes** * Improved scoring by selecting the appropriate conversation content for each evaluation. * Ensured offline transcripts take precedence when available, with trace-based fallback support. * **Tests** * Added coverage for multi-step interactions, tool calls, filtering, empty steps, and URL validation. <!-- end of auto-generated comment: release notes by coderabbit.ai -->
53 lines
2.0 KiB
TypeScript
53 lines
2.0 KiB
TypeScript
import type { StepResult, ToolSet } from 'ai'
|
|
|
|
export type Transcript = {
|
|
/** The user's prompt for this eval case. Always available locally — no reconstruction needed. */
|
|
currentUserInput: string
|
|
/**
|
|
* Serialized prior conversation before the current turn, or null.
|
|
*/
|
|
priorConversation: string | null
|
|
/** Text-only serialization of the assistant's turn: prose only, no tool call markers. */
|
|
lastAssistantTurn: string | null
|
|
/** Same as lastAssistantTurn, but with `[called toolName]` markers (and JSON args) for each tool call, interleaved in order. */
|
|
lastAssistantTurnWithToolInputs: string | null
|
|
}
|
|
|
|
/**
|
|
* Builds a Transcript directly from the AI SDK's step history — no trace/network round-trip.
|
|
* Mirrors the marker format the old trace.getThread()-based serialization used
|
|
* (`[called toolName]\n<json args>`), so scorer prompts don't need to change:
|
|
* only text and tool-call content parts are represented; reasoning/source/file/tool-result/
|
|
* tool-error parts are skipped, exactly as the old serializeContentBlock did.
|
|
*/
|
|
export function buildTranscript(
|
|
currentUserInput: string,
|
|
steps: ReadonlyArray<StepResult<ToolSet>>
|
|
): Transcript {
|
|
const renderStep = (step: StepResult<ToolSet>, includeToolInputs: boolean): string =>
|
|
step.content
|
|
.flatMap((part) => {
|
|
if (part.type === 'text') return part.text ? [part.text] : []
|
|
if (part.type === 'tool-call') {
|
|
const marker = `[called ${part.toolName}]`
|
|
return [includeToolInputs ? `${marker}\n${JSON.stringify(part.input, null, 2)}` : marker]
|
|
}
|
|
return []
|
|
})
|
|
.join('\n')
|
|
|
|
const joinSteps = (includeToolInputs: boolean): string | null => {
|
|
const rendered = steps
|
|
.map((step) => renderStep(step, includeToolInputs))
|
|
.filter((s) => s.length > 0)
|
|
return rendered.length > 0 ? rendered.join('\n\n') : null
|
|
}
|
|
|
|
return {
|
|
currentUserInput,
|
|
priorConversation: null,
|
|
lastAssistantTurn: joinSteps(false),
|
|
lastAssistantTurnWithToolInputs: joinSteps(true),
|
|
}
|
|
}
|