11'use client'
22
33import { useState } from 'react'
4- import { Chip , ChipConfirmModal , ChipInput , ChipModalError , ChipTextarea } from '@sim/emcn'
4+ import {
5+ Chip ,
6+ ChipConfirmModal ,
7+ ChipInput ,
8+ ChipModalError ,
9+ ChipSelect ,
10+ ChipTextarea ,
11+ } from '@sim/emcn'
512import { Download , Trash } from '@sim/emcn/icons'
613import { useQueryStates } from 'nuqs'
714import type {
815 BenchmarkCase ,
16+ RunBenchmarkComparisonBody ,
917 RunBenchmarkStageBody ,
1018 UpdateBenchmarkBody ,
1119} from '@/lib/api/contracts/benchmarks'
20+ import {
21+ type BenchmarkModelConfig ,
22+ DEFAULT_BENCHMARK_EVALUATOR ,
23+ DEFAULT_BENCHMARK_PLANNER ,
24+ } from '@/lib/benchmarks/models'
25+ import { MOTHERSHIP_MODEL_OPTIONS } from '@/lib/mothership/model-options'
1226import { saveBlob } from '@/lib/uploads/client/download'
1327import { BenchmarkHistory } from '@/app/o/[organizationId]/benchmark/components/benchmark-history'
28+ import { BenchmarkModelPicker } from '@/app/o/[organizationId]/benchmark/components/benchmark-model-picker'
1429import { BenchmarkReference } from '@/app/o/[organizationId]/benchmark/components/benchmark-reference'
1530import { BenchmarkResults } from '@/app/o/[organizationId]/benchmark/components/benchmark-results'
1631import { BenchmarkStep } from '@/app/o/[organizationId]/benchmark/components/benchmark-step'
@@ -22,6 +37,7 @@ import {
2237 useBenchmark ,
2338 useBenchmarkWorkspaces ,
2439 useDeleteBenchmark ,
40+ useRunBenchmarkComparison ,
2541 useRunBenchmarkStage ,
2642 useUpdateBenchmark ,
2743} from '@/hooks/queries/benchmarks'
@@ -41,7 +57,13 @@ interface BenchmarkEditorProps {
4157 saving : boolean
4258 stage : RunBenchmarkStageBody [ 'stage' ] | null
4359 onUpdate : ( body : UpdateBenchmarkBody , onSaved : ( ) => void ) => void
44- onRun : ( stage : RunBenchmarkStageBody [ 'stage' ] , runLabel ?: string ) => void
60+ comparing : boolean
61+ onCompare : ( body : Omit < RunBenchmarkComparisonBody , 'version' > ) => void
62+ onRun : (
63+ stage : RunBenchmarkStageBody [ 'stage' ] ,
64+ runLabel ?: string ,
65+ model ?: BenchmarkModelConfig
66+ ) => void
4567}
4668
4769function BenchmarkEditor ( {
@@ -52,6 +74,8 @@ function BenchmarkEditor({
5274 stage,
5375 onUpdate,
5476 onRun,
77+ comparing,
78+ onCompare,
5579} : BenchmarkEditorProps ) {
5680 const [ edit , setEdit ] = useState < {
5781 version : number
@@ -64,6 +88,15 @@ function BenchmarkEditor({
6488 artifacts : { ...( current ?. artifacts ?? benchmark . artifacts ) , ...patch } ,
6589 } ) )
6690 const [ runLabel , setRunLabel ] = useState ( '' )
91+ const [ planner , setPlanner ] = useState (
92+ benchmark . artifacts . modelRuns ?. plan ?. config ?? DEFAULT_BENCHMARK_PLANNER
93+ )
94+ const [ evaluator , setEvaluator ] = useState (
95+ benchmark . artifacts . modelRuns ?. reconstruct ?. config ?? DEFAULT_BENCHMARK_EVALUATOR
96+ )
97+ const [ comparisonModels , setComparisonModels ] = useState < string [ ] > (
98+ MOTHERSHIP_MODEL_OPTIONS . map ( ( option ) => option . value )
99+ )
67100 const { artifacts } = benchmark
68101 const referenceDirty =
69102 draft . taskBrief !== artifacts . taskBrief || draft . referenceSpec !== artifacts . referenceSpec
@@ -113,7 +146,7 @@ function BenchmarkEditor({
113146 ( ) => setEdit ( ( current ) => ( current === edit ? null : current ) )
114147 )
115148 } }
116- onRun = { onRun }
149+ onRun = { ( nextStage ) => onRun ( nextStage , undefined , evaluator ) }
117150 />
118151 < BenchmarkStep
119152 number = { 2 }
@@ -124,12 +157,73 @@ function BenchmarkEditor({
124157 < Chip
125158 variant = 'primary'
126159 disabled = { busy || dirty || ! canPlan || ! hasPlannerInputs }
127- onClick = { ( ) => onRun ( 'plan' ) }
160+ onClick = { ( ) => onRun ( 'plan' , undefined , planner ) }
128161 >
129162 { stage === 'plan' ? 'Planning…' : artifacts . generatedSpec ? 'Run again' : 'Run planner' }
130163 </ Chip >
131164 }
132165 >
166+ < div className = 'flex flex-wrap gap-6' >
167+ < BenchmarkModelPicker
168+ label = 'Planner'
169+ value = { planner }
170+ disabled = { busy }
171+ onChange = { setPlanner }
172+ />
173+ < BenchmarkModelPicker
174+ label = 'Evaluator'
175+ value = { evaluator }
176+ disabled = { busy }
177+ onChange = { setEvaluator }
178+ />
179+ </ div >
180+ < div className = 'flex flex-col gap-3 rounded-lg border border-[var(--border)] p-4' >
181+ < h3 className = 'text-[var(--text-primary)] text-base' > Compare models</ h3 >
182+ < p className = 'text-[var(--text-muted)] text-small' >
183+ Each model plans, reconstructs, and grades in fresh conversations. All planners use{ ' ' }
184+ { planner . effort } effort and the same evaluator for reconstruction and grading. Completed
185+ results are saved as each model finishes.
186+ </ p >
187+ < div className = 'flex flex-wrap items-center gap-2' >
188+ < ChipSelect
189+ aria-label = 'Models to compare'
190+ multiSelect
191+ options = { MOTHERSHIP_MODEL_OPTIONS }
192+ multiSelectValues = { comparisonModels }
193+ onMultiSelectChange = { setComparisonModels }
194+ showAllOption = { false }
195+ placeholder = 'Select models'
196+ disabled = { busy }
197+ />
198+ < Chip
199+ variant = 'primary'
200+ disabled = {
201+ busy || dirty || ! canPlan || ! hasPlannerInputs || comparisonModels . length === 0
202+ }
203+ onClick = { ( ) =>
204+ onCompare ( {
205+ planners : MOTHERSHIP_MODEL_OPTIONS . filter ( ( option ) =>
206+ comparisonModels . includes ( option . value )
207+ ) . map ( ( option ) => ( {
208+ modelSelection : { model : option . value , fastMode : false } ,
209+ effort : planner . effort ,
210+ } ) ) ,
211+ evaluator,
212+ runLabel,
213+ } )
214+ }
215+ >
216+ { comparing
217+ ? 'Running comparison…'
218+ : `Run and grade ${ comparisonModels . length } ${ comparisonModels . length === 1 ? 'model' : 'models' } ` }
219+ </ Chip >
220+ </ div >
221+ { comparing && (
222+ < p role = 'status' className = 'text-[var(--text-muted)] text-small' >
223+ Keep this page open. { stage ? `Current step: ${ stage } .` : 'Starting the next model…' }
224+ </ p >
225+ ) }
226+ </ div >
133227 { ! canPlan && (
134228 < p className = 'text-[var(--text-muted)] text-small' >
135229 The selected user needs Plan mode access and permission to create organization
@@ -158,7 +252,7 @@ function BenchmarkEditor({
158252 action = {
159253 < Chip
160254 disabled = { busy || dirty || ! artifacts . generatedSpec }
161- onClick = { ( ) => onRun ( 'reconstruct' ) }
255+ onClick = { ( ) => onRun ( 'reconstruct' , undefined , evaluator ) }
162256 >
163257 { stage === 'reconstruct' ? 'Reconstructing…' : 'Reconstruct' }
164258 </ Chip >
@@ -174,7 +268,7 @@ function BenchmarkEditor({
174268 action = {
175269 < Chip
176270 disabled = { busy || dirty || ! artifacts . reconstruction }
177- onClick = { ( ) => onRun ( 'grade' , runLabel ) }
271+ onClick = { ( ) => onRun ( 'grade' , runLabel , evaluator ) }
178272 >
179273 { stage === 'grade' ? 'Grading…' : 'Grade' }
180274 </ Chip >
@@ -219,7 +313,8 @@ export function BenchmarkDetail({
219313 onDeleted,
220314} : BenchmarkDetailProps ) {
221315 const [ { benchmarkView } , setParams ] = useQueryStates ( benchmarkParams , benchmarkUrlOptions )
222- const benchmarkQuery = useBenchmark ( organizationId , benchmarkId )
316+ const comparison = useRunBenchmarkComparison ( organizationId , benchmarkId )
317+ const benchmarkQuery = useBenchmark ( organizationId , benchmarkId , comparison . isPending )
223318 const updateBenchmark = useUpdateBenchmark ( organizationId , benchmarkId )
224319 const runStage = useRunBenchmarkStage ( organizationId , benchmarkId )
225320 const deleteBenchmark = useDeleteBenchmark ( organizationId , benchmarkId )
@@ -251,13 +346,21 @@ export function BenchmarkDetail({
251346 benchmark . leaseExpiresAt !== null &&
252347 Date . parse ( benchmark . leaseExpiresAt ) > Date . now ( )
253348 const busy =
254- activeLease || runStage . isPending || updateBenchmark . isPending || deleteBenchmark . isPending
349+ activeLease ||
350+ comparison . isPending ||
351+ runStage . isPending ||
352+ updateBenchmark . isPending ||
353+ deleteBenchmark . isPending
255354 const stage = runStage . isPending
256355 ? runStage . variables . stage
257356 : activeLease
258357 ? benchmark . runningStage
259358 : null
260- const error = runStage . error ?. message ?? updateBenchmark . error ?. message ?? benchmark . error
359+ const error =
360+ comparison . error ?. message ??
361+ runStage . error ?. message ??
362+ updateBenchmark . error ?. message ??
363+ benchmark . error
261364
262365 return (
263366 < div aria-busy = { benchmarkQuery . isFetching } className = 'flex flex-col gap-6' >
@@ -297,7 +400,12 @@ export function BenchmarkDetail({
297400 { error }
298401 </ p >
299402 ) }
300- { benchmark . runningStage && ! activeLease && ! runStage . isPending && (
403+ { comparison . error && (
404+ < p className = 'text-[var(--text-muted)] text-small' >
405+ The comparison stopped. Completed models remain in Run history.
406+ </ p >
407+ ) }
408+ { benchmark . runningStage && ! activeLease && ! runStage . isPending && ! comparison . isPending && (
301409 < p role = 'status' className = 'text-[var(--text-muted)] text-small' >
302410 The previous attempt expired. You can run that step again.
303411 </ p >
@@ -326,14 +434,33 @@ export function BenchmarkDetail({
326434 busy = { busy }
327435 saving = { updateBenchmark . isPending }
328436 stage = { stage }
437+ comparing = { comparison . isPending }
438+ onCompare = { ( body ) => {
439+ runStage . reset ( )
440+ updateBenchmark . reset ( )
441+ comparison . mutate (
442+ { ...body , version : benchmark . version } ,
443+ {
444+ onSuccess : ( ) =>
445+ setParams ( {
446+ benchmarkView : 'history' ,
447+ runId : null ,
448+ compareRunId : null ,
449+ runsCursor : null ,
450+ } ) ,
451+ }
452+ )
453+ } }
329454 onUpdate = { ( body , onSaved ) => {
330455 runStage . reset ( )
456+ comparison . reset ( )
331457 updateBenchmark . mutate ( body , { onSuccess : onSaved } )
332458 } }
333- onRun = { ( nextStage , runLabel ) => {
459+ onRun = { ( nextStage , runLabel , model ) => {
460+ comparison . reset ( )
334461 updateBenchmark . reset ( )
335462 runStage . mutate (
336- { version : benchmark . version , stage : nextStage , runLabel } ,
463+ { version : benchmark . version , stage : nextStage , runLabel, model } ,
337464 {
338465 onSuccess : ( ) => {
339466 if ( nextStage === 'grade' )
0 commit comments