From fc166715ff95cdcfe8d55cf952ecd1a81bed7f56 Mon Sep 17 00:00:00 2001 From: Silas Date: Thu, 20 Aug 2026 16:30:52 +0530 Subject: [PATCH 01/10] feat(ui): interactive peer review studio and evaluation benchmark dashboard (#54) * fix(docker): use native npm build in frontend stage for maximum stability * feat(ui): interactive peer review studio and evaluation benchmark dashboard - Add PeerReviewStudio component with conference rubric selector (ICLR, NeurIPS, ICML, CVPR, ACL), 3-reviewer deliberation cards, meta-reviewer consensus summary, and Markdown report export. - Add EvalBenchmarkDashboard component with benchmark suite runner, category/difficulty filtering, metric pass rates, speedup radars, and custom reproduction & optimization task registration modal. - Connect Review and Eval tabs to ProjectContext and ChatContainer with lazy loading. - Add comprehensive Vitest test coverage for PeerReviewStudio and EvalBenchmarkDashboard. --- frontend/src/api.ts | 13 + .../src/components/chat/ChatContainer.tsx | 42 +++ .../src/components/eval/CustomTaskModal.tsx | 335 ++++++++++++++++++ .../eval/EvalBenchmarkDashboard.test.tsx | 187 ++++++++++ .../eval/EvalBenchmarkDashboard.tsx | 248 +++++++++++++ .../src/components/eval/SuiteRunSummary.tsx | 192 ++++++++++ frontend/src/components/eval/TaskCard.tsx | 151 ++++++++ frontend/src/components/eval/types.ts | 15 + .../components/review/MetaReviewSummary.tsx | 157 ++++++++ .../review/PeerReviewStudio.test.tsx | 164 +++++++++ .../components/review/PeerReviewStudio.tsx | 329 +++++++++++++++++ .../src/components/review/ReviewerCard.tsx | 162 +++++++++ .../components/review/RubricViewerModal.tsx | 118 ++++++ frontend/src/components/review/types.ts | 15 + frontend/src/context/ProjectContext.tsx | 2 +- frontend/src/types.ts | 85 +++++ 16 files changed, 2214 insertions(+), 1 deletion(-) create mode 100644 frontend/src/components/eval/CustomTaskModal.tsx create mode 100644 frontend/src/components/eval/EvalBenchmarkDashboard.test.tsx create mode 100644 frontend/src/components/eval/EvalBenchmarkDashboard.tsx create mode 100644 frontend/src/components/eval/SuiteRunSummary.tsx create mode 100644 frontend/src/components/eval/TaskCard.tsx create mode 100644 frontend/src/components/eval/types.ts create mode 100644 frontend/src/components/review/MetaReviewSummary.tsx create mode 100644 frontend/src/components/review/PeerReviewStudio.test.tsx create mode 100644 frontend/src/components/review/PeerReviewStudio.tsx create mode 100644 frontend/src/components/review/ReviewerCard.tsx create mode 100644 frontend/src/components/review/RubricViewerModal.tsx create mode 100644 frontend/src/components/review/types.ts diff --git a/frontend/src/api.ts b/frontend/src/api.ts index 276867b..6b0a8c0 100644 --- a/frontend/src/api.ts +++ b/frontend/src/api.ts @@ -170,6 +170,19 @@ export const api = { testMcpServer: (url: string, headers?: Record, params?: Record) => post('/api/mcp/test', { url, headers: headers || null, params: params || null }), + // Peer Review Simulation + getReviewRubrics: () => get('/api/review/rubrics'), + evaluateSubmission: (body: { + submission_text: string; + venue?: string; + title?: string; + context?: Record; + }) => post('/api/review/evaluate', body), + reviewProjectWorkspace: ( + projectId: number, + body: { venue?: string; include_latex?: boolean; include_notes?: boolean } + ) => post(`/api/projects/${projectId}/review`, body), + // Evaluation & Benchmark Harness listEvalSuites: () => get('/api/eval/suites'), listEvalTasks: (category?: string) => diff --git a/frontend/src/components/chat/ChatContainer.tsx b/frontend/src/components/chat/ChatContainer.tsx index efea101..509ad67 100644 --- a/frontend/src/components/chat/ChatContainer.tsx +++ b/frontend/src/components/chat/ChatContainer.tsx @@ -19,6 +19,12 @@ const CitationGraph = lazy(() => const RunDashboard = lazy(() => import('../experiments/RunDashboard').then((m) => ({ default: m.RunDashboard })) ); +const PeerReviewStudio = lazy(() => + import('../review/PeerReviewStudio').then((m) => ({ default: m.PeerReviewStudio })) +); +const EvalBenchmarkDashboard = lazy(() => + import('../eval/EvalBenchmarkDashboard').then((m) => ({ default: m.EvalBenchmarkDashboard })) +); export function ChatContainer() { const { @@ -132,6 +138,28 @@ export function ChatContainer() { > Experiments + + {/* Closable Image tab */} {imageTab && (
+ {/* Peer Review tab */} +
+ Loading Peer Review Studio...
}> + + + + + {/* Evaluation Benchmark tab */} +
+ Loading Benchmark Harness...
}> + + + + {/* Image tab */} {mainTab === 'image' && imageTab && ( diff --git a/frontend/src/components/eval/CustomTaskModal.tsx b/frontend/src/components/eval/CustomTaskModal.tsx new file mode 100644 index 0000000..ca4efbe --- /dev/null +++ b/frontend/src/components/eval/CustomTaskModal.tsx @@ -0,0 +1,335 @@ +import { useState } from 'react'; +import { X, PlusCircle, FileCheck, Zap, AlertCircle } from 'lucide-react'; +import { api } from '../../api'; + +interface Props { + onClose: () => void; + onCreated: () => void; +} + +export function CustomTaskModal({ onClose, onCreated }: Readonly) { + const [taskType, setTaskType] = useState<'reproduction' | 'optimization'>('reproduction'); + const [loading, setLoading] = useState(false); + const [error, setError] = useState(null); + + // Common + const [taskId, setTaskId] = useState(''); + const [name, setName] = useState(''); + const [description, setDescription] = useState(''); + const [difficulty, setDifficulty] = useState('medium'); + const [timeoutSeconds, setTimeoutSeconds] = useState(300); + + // Reproduction specific + const [paperTitle, setPaperTitle] = useState(''); + const [arxivId, setArxivId] = useState(''); + const [datasetName, setDatasetName] = useState('custom'); + const [targetMetricsJson, setTargetMetricsJson] = useState('{"accuracy": 0.85, "loss": 0.25}'); + + // Optimization specific + const [kernelName, setKernelName] = useState(''); + const [framework, setFramework] = useState('triton'); + const [baselineLatencyMs, setBaselineLatencyMs] = useState(25.0); + const [targetSpeedup, setTargetSpeedup] = useState(1.5); + + const handleSubmit = async (e: React.FormEvent) => { + e.preventDefault(); + setError(null); + setLoading(true); + + try { + if (taskType === 'reproduction') { + let metrics: Record = {}; + try { + metrics = JSON.parse(targetMetricsJson); + } catch { + throw new Error('Target metrics must be a valid JSON object of numbers (e.g. {"accuracy": 0.85})'); + } + + await api.registerCustomReproductionTask({ + task_id: taskId, + name, + description, + paper_title: paperTitle, + arxiv_id: arxivId, + dataset_name: datasetName, + target_metrics: metrics, + difficulty, + timeout_seconds: timeoutSeconds, + }); + } else { + await api.registerCustomOptimizationTask({ + task_id: taskId, + name, + description, + kernel_name: kernelName, + framework, + baseline_latency_ms: baselineLatencyMs, + target_speedup: targetSpeedup, + difficulty, + timeout_seconds: timeoutSeconds, + }); + } + + onCreated(); + onClose(); + } catch (err: unknown) { + const errMsg = err instanceof Error ? err.message : 'Failed to register benchmark task.'; + setError(errMsg); + } finally { + setLoading(false); + } + }; + + return ( +
+
+ {/* Header */} +
+
+ +

Register Custom Benchmark Task

+
+ +
+ + {/* Form content */} +
+ {/* Type Toggle */} +
+ + +
+ + {error && ( +
+ + {error} +
+ )} + + {/* Common fields */} +
+
+ + setTaskId(e.target.value)} + placeholder="e.g. repro-flashattention-3" + className="w-full bg-bg border border-border rounded-xl px-3 py-1.5 text-text focus:border-primary focus:outline-none" + /> +
+
+ + setName(e.target.value)} + placeholder="e.g. FlashAttention-3 Hopper FP8" + className="w-full bg-bg border border-border rounded-xl px-3 py-1.5 text-text focus:border-primary focus:outline-none" + /> +
+
+ +
+
+ + +
+
+ + setTimeoutSeconds(parseInt(e.target.value) || 300)} + className="w-full bg-bg border border-border rounded-xl px-3 py-1.5 text-text focus:border-primary focus:outline-none" + /> +
+
+ +
+ +