diff --git a/.github/workflows/tutor-quality-real-provider.yml b/.github/workflows/tutor-quality-real-provider.yml new file mode 100644 index 00000000..29262606 --- /dev/null +++ b/.github/workflows/tutor-quality-real-provider.yml @@ -0,0 +1,56 @@ +name: Tutor Quality Real-Provider Evaluation + +on: + workflow_dispatch: + inputs: + model: + description: Fixed provider model ID (auto is rejected) + required: true + type: string + samples: + description: Samples per anchor scenario + required: true + default: "1" + type: string + max_calls: + description: Maximum provider calls, including model-list calls + required: true + default: "30" + type: string + +permissions: + contents: read + +jobs: + evaluate: + name: Stage 2A observation run + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-node@v4 + with: + node-version-file: .nvmrc + cache: npm + + - run: npm ci + + - name: Run optional real-provider evaluation + env: + UNOSIM_TUTOR_EVAL_CREDENTIAL: ${{ secrets.UNOSIM_TUTOR_EVAL_CREDENTIAL }} + run: | + npm run eval:tutor-quality:real -- \ + --model "${{ inputs.model }}" \ + --samples "${{ inputs.samples }}" \ + --max-calls "${{ inputs.max_calls }}" \ + --credential-env UNOSIM_TUTOR_EVAL_CREDENTIAL \ + --output-dir .tutor-quality-output + + - name: Upload Stage 2A artifacts + if: always() + uses: actions/upload-artifact@v4 + with: + name: tutor-quality-stage-2a-${{ github.run_id }} + path: .tutor-quality-output/ + if-no-files-found: error + retention-days: 14 diff --git a/.gitignore b/.gitignore index b90d9998..c40c3ae4 100644 --- a/.gitignore +++ b/.gitignore @@ -46,6 +46,7 @@ storage/binaries/ !/temp/.gitkeep /cache/ /test-results/ +.tutor-quality-output/ /playwright-report/ # /screenshots/ .scannerwork/ diff --git a/docs/plan-tutor-quality-stage-2a.md b/docs/plan-tutor-quality-stage-2a.md new file mode 100644 index 00000000..950c8ac4 --- /dev/null +++ b/docs/plan-tutor-quality-stage-2a.md @@ -0,0 +1,97 @@ +# Tutor Quality – Stage 2A Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Build a small, reviewable real-provider evaluation foundation that runs the normal `TutorService` path, records secret-free transcripts, reapplies deterministic Stage-1 checks, and reports technical outcomes separately from Tutor invariant violations. + +**Architecture:** Keep Stage 2A as an explicitly invoked observation layer. A repository-owned YAML anchor manifest resolves sketch and typed Course Content fixtures. A testable runner receives an injected `LLMProvider`, clones scenario state per sample, invokes `TutorService` and `CurriculumTutorAdapter`, captures the parsed provider result before service validation/repair, writes allow-listed JSON artifacts, and aggregates deterministic categories. The CLI constructs the real `KiconnectProvider`, performs clean-worktree/credential/model/call-budget preflight, and is never imported by the application runtime or required PR workflows. + +**Tech Stack:** TypeScript/ESM, existing `TutorService`/`LLMProvider`/`KiconnectProvider`, `yaml`, Node `crypto`/`fs`, Vitest, `tsx`, GitHub Actions `workflow_dispatch`. + +**Spec:** `ssot/ssot_function_definition_TutorQualityStage2A.md` + +## Global Constraints + +- Preserve all Stage-1 hard gates and existing Tutor trust boundaries. +- Do not call a provider directly from the evaluator; use `TutorService` and the existing provider implementation. +- Do not modify normal provider fallback behavior. Stage 2A requires an explicit model and marks missing/mismatched model metadata `invalid`. +- Never accept or print a credential value, authorization header, `process.env`, or an uploaded diff. Accept only a credential environment-variable name. +- Any tracked or indexed Git change is an invalid preflight and issues no provider call. Untracked files are invalidating only under versioned evaluation/runtime input roots; unrelated editor files, protected local SSOT files, and ignored output directories are excluded. +- No real-provider call is made by unit tests, pull-request CI, or required checks. Missing credentials produce `not-run` output. +- No semantic grades, LLM-as-Judge, adaptive strategy, fact-extractor expansion, learner profiles, or learning-effect claims. +- Keep artifact schemas allow-listed, bounded, deterministic, and disposable. Do not commit generated run output. +- Use TDD: write a focused failing test, run it red, implement the smallest change, run it green, then commit each coherent task. + +## Review Focus + +- Verify the runner records both `executionStatus` and `invariantViolations`; a repaired response can be completed while retaining raw invariant violations. +- Verify raw means the parsed `ProviderQuestionResult.result` captured before TutorService validation/repair, never an HTTP envelope. +- Verify `evaluationIdentity` includes Git SHA, Course Content revision, corpus/version, provider/model, prompt revision and effective-template digest, and all bounded parameters. +- Verify prompt revision is app-owned/versioned and changes when the effective system/user templates change. +- Verify every multi-turn scripted answer has an explicit preceding-question binding; otherwise the scenario is invalid. +- Verify `TQ-REG-001` checks no variables-topic activation and no exact/Stage-1-heuristic repeat, not an unimplemented semantic judgement. +- Verify reports distinguish invalid metadata, not-run preflight, technical provider failures, invariant violations, and completed observations with explicit denominators. + +--- + +## Task 1 – Lock the corpus and fixture contract with tests + +- [x] Add a failing loader/contract test for a manifest with `corpusId`, `corpusVersion`, stable scenario IDs, sketch references, course fixture/free-Tutor mode, synthetic answers, explicit turn bindings, and structural expectations. +- [x] Add a failing test that rejects duplicate IDs, missing fixture references, `auto` model declarations, and unbound continuation answers. +- [x] Add a separate corpus-evolution validator and failing tests for `compare(previousCorpus, currentCorpus)`: a parsed/digest-changing add, removal, or semantic edit requires a higher `corpusVersion`; formatting-only changes may keep the version only when parsed content and digest are unchanged. Keep historical comparison out of the current-corpus loader. +- [x] Add `evals/tutor-quality/anchor-corpus.yaml` at version 1 with the ten approved anchors: `TQ-REG-001` (no variables-topic activation and no exact or Stage-1-heuristic repeat), simple variable, Serial prediction, incorrect answer, partial answer, strong answer/progression, unmatched/free Tutor, LEARN→DEEPEN, EXPAND, and off-topic answer. +- [x] Add only the small required `.ino` fixtures under `evals/tutor-quality/fixtures/`; reuse the existing PWM fixture where possible rather than copying production content. +- [x] Add a typed `anchor-course-content.ts` fixture factory for the small valid Course Content snapshots and seeded progression states required by topic activation, LEARN/DEEPEN, and EXPAND cases. Keep revision strings and question IDs explicit and reviewable. +- [x] Run the focused corpus tests red before implementation and green after the loader/factory exists. + +## Task 2 – Make prompt revision metadata explicit without changing prompts + +- [x] Add a failing unit test asserting a versioned prompt revision identifier and SHA-256 digest are stable, contain the effective system/initial-user/dialog-user template sources before scenario substitution, and change when a template source changes. +- [x] Refactor only the prompt-source declarations needed by `TutorService` so existing generated prompt text remains byte-for-byte compatible; export the revision descriptor for the evaluator. +- [x] Do not add a new prompt, quality rule, or runtime strategy. The revision helper is metadata only. +- [x] Run existing Tutor prompt tests plus the new revision test. + +## Task 3 – Implement the injectable evaluation runner and transcript model + +- [x] Add failing tests for fresh state/history cloning per sample, normal `TutorService` initial/dialog invocation, provider capture before validation/repair, shared diagnostic repeat/solution/schema checks, state-before/state-after snapshots, and explicit expected structural checks. +- [x] Add `server/services/tutor/evaluation/real-provider-evaluation.ts` with small typed contracts for corpus scenarios, invocation options, sample metadata, logical turns, deterministic check records, transcript artifacts, and aggregate reports. +- [x] Inject the provider, clock, random suffix, Git metadata, and output writer seams so tests never need credentials, network, or a mutable repository. +- [x] Wrap the provider to capture requests and parsed `ProviderQuestionResult` values before `TutorService` receives them. Keep the raw capture allow-listed and exclude transport envelopes/headers. +- [x] Invoke `TutorService` with `CurriculumTutorAdapter` when a scenario declares Course Content; invoke the same service without planning for free-Tutor cases. +- [x] Split the existing pure learning-question validation internally into one shared diagnostic function that returns granular deterministic violation records and keep `validateLearningQuestion` as the existing throw/normalization wrapper. Export the diagnostic function for Stage 2A; add no new rule and preserve runtime behavior. +- [x] Reuse that diagnostic function and `isSemanticallyRepeatedQuestion` for deterministic checks. Record raw complete-solution/repeat violations even when TutorService rejects or repairs the response; never turn these into semantic scores. +- [x] Enforce fixed requested model, preflight model availability, returned-model equality, explicit sample limits, and a provider-call budget covering *all* external calls, including `listModels()` and generation. Report `providerCalls`, `modelListCalls`, and `generationCalls` separately; stop before any call that would exceed the budget. +- [x] Implement the two status axes from the SSOT: `executionStatus` (`completed`, `invalid`, `technical-failure`, `not-run`) and `invariantViolations` (array). Classify missing credentials/zero preflight budget as `not-run`; provider errors, timeout, malformed responses, and mid-run budget exhaustion as technical failures; metadata/model/binding problems as invalid. +- [x] Implement canonical JSON hashing for `evaluationIdentity`, run IDs with UTC timestamp plus collision-resistant suffix, and secret-free allow-listed JSON transcript writing. +- [x] Implement aggregate counts/rates per scenario and overall with explicit denominators, exact provider-call counts, separate model-list/generation counts, budget exhaustion, and cost `unavailable` when the provider supplies no cost data. Always write a run-level `not-run` report for missing credentials with `reason: missing-credential`, zero provider calls, and no sample transcripts. +- [x] Run the focused evaluator tests red before implementation and green after each runner slice. + +## Task 4 – Add adversarial fake-provider coverage + +- [x] Add fake-provider tests for exact repeated questions, Stage-1 heuristic repeats, invalid schema, complete solution, forged/wrong planning metadata, provider error before commit, timeout, returned-model mismatch, and call-budget exhaustion. +- [x] Assert repaired final planning metadata remains application-owned and state commits only after a successful TutorService request. +- [x] Assert technical failures do not mutate progression state and are not counted as Tutor-quality violations unless a separate raw deterministic violation was observed. +- [x] Assert no credential value appears in serialized transcript, report, thrown error, or logger input. + +## Task 5 – Add the explicit CLI and local execution contract + +- [x] Add `scripts/tutor-quality-real-provider-eval.ts` as a thin CLI around the runner. Require a fixed `--model`, bounded `--samples` and `--max-calls`, `--output-dir`, corpus selection, and a credential environment-variable name; reject `auto` and any credential value flag. +- [x] Add a package script such as `eval:tutor-quality:real` that is not referenced by `test`, `test:unit`, `test:tutor-quality`, or normal build gates. +- [x] Resolve the repository Git SHA/clean state and configured Course Content/corpus revisions before execution. Abort as `invalid` without provider calls when preflight identity cannot be proven. +- [x] Add a short operator document with a local command, expected output paths, missing-credential behavior, call-budget example, and explicit warning that Stage 2A observes deterministic integrity rather than learning effect. +- [x] Test CLI argument validation and missing-credential `not-run` behavior without contacting a provider. + +## Task 6 – Add a manual-only workflow + +- [x] Add `.github/workflows/tutor-quality-real-provider.yml` with only `workflow_dispatch`, explicit model/sample/call-budget inputs, Node version from `.nvmrc`, and a repository secret exposed only to the invoked process through the configured environment variable. +- [x] Upload bounded transcript/report artifacts with retention; never print the secret or use a `pull_request`/required-check trigger. +- [x] Make missing secret a visible skipped/not-run result, not a failing PR gate. +- [x] Document that this workflow is optional/manual first; do not add nightly scheduling until cost and stability are known. + +## Task 7 – Verification and review handoff + +- [x] Run the Node-version check, focused Stage-2A tests, existing `npm run test:tutor-quality`, `npm run check`, `npm run check:docs`, and `git diff --check`. +- [x] Run the CLI in no-credential mode and verify it writes the documented run-level `not-run` report with `reason: missing-credential`, zero provider calls, no sample transcripts, and no secret-like data. +- [x] Confirm generated transcripts/reports are ignored or written only to caller-selected disposable directories. +- [x] Review the diff for accidental production behavior changes, duplicated Stage-1 validation, direct HTTP access, semantic scoring, and CI hard-gate coupling. +- [x] Commit the implementation in coherent commits and report branch/base/HEAD, files, anchors, deterministic metrics, excluded Stage-2B dimensions, tests, and the absence of a real-provider run if credentials are unavailable. diff --git a/docs/tutor-quality-stage-2a.md b/docs/tutor-quality-stage-2a.md new file mode 100644 index 00000000..7a349856 --- /dev/null +++ b/docs/tutor-quality-stage-2a.md @@ -0,0 +1,55 @@ +# Tutor Quality Stage 2A – Real-Provider Evaluation + +Stage 2A is an explicitly invoked observation run. It uses the versioned +anchor corpus and the normal `TutorService`/`KiconnectProvider` path. It does +not run in unit tests or pull-request gates and does not measure learning +effect or assign semantic quality scores. + +## Local run + +Use a clean relevant Git state, a fixed model ID, and a caller-selected +disposable output directory: + +```sh +npm run eval:tutor-quality:real -- \ + --model \ + --credential-env UNOSIM_TUTOR_EVAL_CREDENTIAL \ + --samples 2 \ + --max-calls 30 \ + --output-dir /tmp/unosim-tutor-quality-stage-2a +``` + +The credential is read from the named environment variable only. Never pass a +credential value as a command-line argument. The runner never writes the +credential, authorization headers, or `process.env` to an artifact. + +The call budget includes model-list and generation calls. The report exposes +both categories separately. If the credential is absent, the command writes a +run-level `report.json` with `runStatus: "not-run"`, +`reason: "missing-credential"`, zero provider calls, and no sample +transcripts. + +Each invocation is bounded to at most 20 samples per scenario and 500 total +provider calls. The CLI rejects larger values before any provider call. + +## Artifacts + +`report.json` contains the run identity, corpus/course/prompt/provider/model +metadata, explicit status counts and rates, provider-call counts, and per- +scenario aggregates. One allow-listed JSON transcript is written per sample +when execution starts. Transcripts retain synthetic inputs, Tutor request +prompts, parsed provider results before service validation/repair, normalized +Tutor results, state snapshots, deterministic checks, and technical errors. + +`completed` observations may still contain deterministic invariant violations; +technical failures and invalid evaluation metadata remain separate categories. +Generated output is disposable and must not be committed. + +## Scope boundary + +Stage 2A reports schema/response validity, question-count and complete-solution +guards, bounded question-repeat checks, application-owned planning/state +consistency, provider failures, and execution counts. It deliberately does +not judge clarity, scaffolding quality, difficulty, learning support, or +learning progress. Those require the Stage 2B semantic and human-calibration +workflows. diff --git a/evals/tutor-quality/anchor-corpus.yaml b/evals/tutor-quality/anchor-corpus.yaml new file mode 100644 index 00000000..f4149255 --- /dev/null +++ b/evals/tutor-quality/anchor-corpus.yaml @@ -0,0 +1,104 @@ +corpusId: unosim-tutor-quality-anchor +corpusVersion: 1 +scenarios: + - id: TQ-REG-001 + sketch: tests/fixtures/tutor-quality/TQ-REG-001-pwm.ino + courseContent: variables + turns: + - kind: dialog + question: Welchen Wert verwendet der Sketch an der betrachteten Integer-Variablen? + answer: brightness läuft in Fünferschritten zwischen 0 und 255 und bestimmt den PWM-Tastgrad. + bindsToQuestion: Welchen Wert verwendet der Sketch an der betrachteten Integer-Variablen? + difficulty: 35 + expected: + topicIdAbsent: variables-and-serial + stateUnchanged: true + questionNotRepeat: exact-or-heuristic + + - id: simple-variable + sketch: evals/tutor-quality/fixtures/simple-variable.ino + courseContent: variables + turns: + - kind: initial + difficulty: 20 + expected: + topicId: variables-and-serial + + - id: serial-output-prediction + sketch: evals/tutor-quality/fixtures/serial-output.ino + courseContent: variables + turns: + - kind: initial + difficulty: 35 + expected: + topicId: variables-and-serial + + - id: incorrect-answer-remediation + sketch: evals/tutor-quality/fixtures/simple-variable.ino + courseContent: free + turns: + - kind: dialog + question: Welchen Wert gibt der Sketch über Serial aus? + answer: Der Sketch gibt immer 99 aus. + bindsToQuestion: Welchen Wert gibt der Sketch über Serial aus? + difficulty: 25 + + - id: partial-answer-follow-up + sketch: evals/tutor-quality/fixtures/simple-variable.ino + courseContent: free + turns: + - kind: dialog + question: Wie hängt counter mit der Schleife zusammen? + answer: counter ist eine Zahl. + bindsToQuestion: Wie hängt counter mit der Schleife zusammen? + difficulty: 30 + + - id: strong-answer-progression + sketch: evals/tutor-quality/fixtures/serial-output.ino + courseContent: free + turns: + - kind: dialog + question: Welche Ausgabe erzeugt Serial.println im Sketch? + answer: Die Funktion schreibt den aktuellen Wert von counter als Zeile in die serielle Ausgabe. + bindsToQuestion: Welche Ausgabe erzeugt Serial.println im Sketch? + difficulty: 40 + + - id: unmatched-topic-free-tutor + sketch: tests/fixtures/tutor-quality/TQ-REG-001-pwm.ino + courseContent: variables + turns: + - kind: initial + difficulty: 35 + expected: + topicIdAbsent: variables-and-serial + + - id: learn-to-deepen + sketch: evals/tutor-quality/fixtures/simple-variable.ino + courseContent: progression-learn + turns: + - kind: dialog + question: Welche Rolle spielt der Integer-Datentyp im aktuellen Sketch? + answer: int speichert den ganzzahligen Wert von counter, der anschließend im Sketch verwendet wird. + bindsToQuestion: Welche Rolle spielt der Integer-Datentyp im aktuellen Sketch? + difficulty: 30 + expected: + learningPhase: DEEPEN + + - id: expand + sketch: evals/tutor-quality/fixtures/serial-output.ino + courseContent: progression-expand + turns: + - kind: initial + difficulty: 55 + expected: + learningPhase: EXPAND + + - id: off-topic-answer + sketch: evals/tutor-quality/fixtures/simple-variable.ino + courseContent: free + turns: + - kind: dialog + question: Welche konkrete Beobachtung zeigt der Sketch? + answer: Ich möchte lieber über Fußball und das Wetter sprechen. + bindsToQuestion: Welche konkrete Beobachtung zeigt der Sketch? + difficulty: 20 diff --git a/evals/tutor-quality/fixtures/serial-output.ino b/evals/tutor-quality/fixtures/serial-output.ino new file mode 100644 index 00000000..d1320f9e --- /dev/null +++ b/evals/tutor-quality/fixtures/serial-output.ino @@ -0,0 +1,11 @@ +int counter = 3; + +void setup() { + Serial.begin(9600); + Serial.println(counter); +} + +void loop() { + Serial.println(counter); + delay(1000); +} diff --git a/evals/tutor-quality/fixtures/simple-variable.ino b/evals/tutor-quality/fixtures/simple-variable.ino new file mode 100644 index 00000000..37139ec1 --- /dev/null +++ b/evals/tutor-quality/fixtures/simple-variable.ino @@ -0,0 +1,11 @@ +int counter = 3; + +void setup() { + Serial.begin(9600); + Serial.println(counter); +} + +void loop() { + counter += 1; + delay(1000); +} diff --git a/package.json b/package.json index c0c85044..820bc749 100644 --- a/package.json +++ b/package.json @@ -46,6 +46,7 @@ "test:security:inputs": "TEST_BUDGET_SUITE=security-inputs TEST_BUDGET_MS=15000 LOG_LEVEL=warn vitest run --project=unit-node --project=unit-node-http tests/shared/input-limits.test.ts tests/shared/websocket-direction-schemas.test.ts tests/server/security/safe-paths.test.ts tests/server/routes/compiler.routes.test.ts tests/integration/simulation-state-sequence.test.ts --reporter=default --reporter=./scripts/test-budget-reporter.mjs", "test:tutor-quality": "TEST_BUDGET_SUITE=tutor-quality TEST_BUDGET_MS=15000 LOG_LEVEL=warn vitest run --project=unit-node tests/server/services/tutor tests/server/services/course-content --reporter=default --reporter=./scripts/test-budget-reporter.mjs", "validate:tutor-course-content": "tsx scripts/validate-tutor-course-content.mjs", + "eval:tutor-quality:real": "tsx scripts/tutor-quality-real-provider-eval.ts", "test:all": "npm run test:unit && npm run test:integration && npm run test:docker", "test:watch": "LOG_LEVEL=info vitest --project=unit-client --project=unit-node --project=unit-node-http", "test:coverage": "node scripts/coverage-guard.mjs prepare && TEST_BUDGET_SUITE=unit-coverage TEST_BUDGET_WARN_MS=75000 TEST_BUDGET_FAIL_MS=90000 LOG_LEVEL=warn vitest run --coverage --project=unit-client --project=unit-node --project=unit-node-http --reporter=default --reporter=./scripts/test-budget-reporter.mjs && node scripts/coverage-guard.mjs validate", diff --git a/scripts/tutor-quality-real-provider-eval.ts b/scripts/tutor-quality-real-provider-eval.ts new file mode 100644 index 00000000..a70164d7 --- /dev/null +++ b/scripts/tutor-quality-real-provider-eval.ts @@ -0,0 +1,239 @@ +import { execFileSync } from "node:child_process"; +import { access, readFile } from "node:fs/promises"; +import { accessSync, constants as fsConstants } from "node:fs"; +import path from "node:path"; +import { pathToFileURL } from "node:url"; +import { parse as parseYaml } from "yaml"; +import { createAnchorCourseContent, ANCHOR_COURSE_CONTENT_FIXTURE_IDS, type AnchorCourseContentFixtureId } from "../server/services/tutor/evaluation/anchor-course-content"; +import { + parseTutorQualityCorpus, + type TutorQualityCorpusSource, + type TutorQualityHistoryEntrySource, + type TutorQualityTurnSource, +} from "../server/services/tutor/evaluation/anchor-corpus"; +import { + MAX_TUTOR_QUALITY_CALLS, + MAX_TUTOR_QUALITY_SAMPLES, + runTutorQualityEvaluation, + type TutorQualityEvaluationScenario, + type TutorQualityGitState, + type TutorQualityTurn, +} from "../server/services/tutor/evaluation/real-provider-evaluation"; +import { KiconnectProvider } from "../server/services/tutor/kiconnect-provider"; +import type { LLMProvider } from "../server/services/tutor/llm-provider"; + +export interface TutorQualityCliOptions { + readonly model: string; + readonly samples: number; + readonly maxCalls: number; + readonly outputDir: string; + readonly corpusPath: string; + readonly credentialEnv: string; +} + +export interface TutorQualityCliDependencies { + readonly cwd?: string; + readonly environment?: NodeJS.ProcessEnv; + readonly provider?: LLMProvider; + readonly git?: TutorQualityGitState; +} + +const DEFAULT_CORPUS_PATH = "evals/tutor-quality/anchor-corpus.yaml"; +const DEFAULT_CREDENTIAL_ENV = "UNOSIM_TUTOR_EVAL_CREDENTIAL"; + +function argumentPairs(argv: readonly string[]): readonly (readonly [string, string])[] { + if (argv.length % 2 !== 0) throw new Error("Tutor Quality evaluation options require values"); + return Array.from({ length: argv.length / 2 }, (_, pairIndex) => { + const index = pairIndex * 2; + const flag = argv[index]; + const value = argv[index + 1]; + if (!flag?.startsWith("--")) throw new Error(`Unknown Tutor Quality evaluation option: ${flag ?? ""}`); + if (!value || value.startsWith("--")) throw new Error(`${flag} requires a value`); + return [flag, value] as const; + }); +} + +function parsePositiveInteger(value: string, flag: string, allowZero = false, maximum?: number): number { + if (!/^\d+$/.test(value)) throw new Error(`${flag} must be an integer`); + const parsed = Number(value); + if (!Number.isSafeInteger(parsed) || (allowZero ? parsed < 0 : parsed < 1)) throw new Error(`${flag} is out of range`); + if (maximum !== undefined && parsed > maximum) throw new Error(`${flag} exceeds maximum ${maximum}`); + return parsed; +} + +function parseCredentialEnvironmentName(value: string): string { + if (!/^[A-Z][A-Z0-9_]{1,63}$/.test(value)) throw new Error("--credential-env must be an environment-variable name"); + return value; +} + +export function parseTutorQualityCliArgs(argv: readonly string[]): TutorQualityCliOptions { + let model: string | undefined; + let samples = 1; + let maxCalls = 20; + let outputDir: string | undefined; + let corpusPath = DEFAULT_CORPUS_PATH; + let credentialEnv = DEFAULT_CREDENTIAL_ENV; + for (const [flag, value] of argumentPairs(argv)) { + switch (flag) { + case "--model": + model = value; + break; + case "--samples": + samples = parsePositiveInteger(value, flag, false, MAX_TUTOR_QUALITY_SAMPLES); + break; + case "--max-calls": + maxCalls = parsePositiveInteger(value, flag, true, MAX_TUTOR_QUALITY_CALLS); + break; + case "--output-dir": + outputDir = value; + break; + case "--corpus": + corpusPath = value; + break; + case "--credential-env": + credentialEnv = parseCredentialEnvironmentName(value); + break; + default: + throw new Error(`Unknown Tutor Quality evaluation option: ${flag}`); + } + } + if (!model || model === "auto") throw new Error("--model requires a fixed model id; auto is not allowed"); + if (!outputDir) throw new Error("--output-dir is required"); + if (!/^[A-Za-z0-9][A-Za-z0-9._:/-]{0,127}$/.test(model)) throw new Error("--model is invalid"); + return { model, samples, maxCalls, outputDir, corpusPath, credentialEnv }; +} + +const GIT_EXECUTABLE_CANDIDATES = process.platform === "win32" + ? [String.raw`C:\Program Files\Git\cmd\git.exe`, String.raw`C:\Program Files\Git\bin\git.exe`] + : ["/usr/bin/git", "/opt/homebrew/bin/git", "/usr/local/bin/git"]; + +function resolveGitExecutable(): string { + const executable = GIT_EXECUTABLE_CANDIDATES.find((candidate) => { + try { + accessSync(candidate, fsConstants.X_OK); + return true; + } catch { + return false; + } + }); + if (!executable) throw new Error("git executable not found in fixed system locations"); + return executable; +} + +function historyEntry(entry: TutorQualityHistoryEntrySource): import("../shared/tutor").TutorDialogTurn { + return { + question: entry.question, + answer: entry.answer ?? "synthetic scripted history", + responseStyle: entry.responseStyle ?? "normal", + ...(entry.answerRating === undefined ? {} : { answerRating: entry.answerRating }), + ...(entry.questionId === undefined ? {} : { questionId: entry.questionId }), + }; +} + +function materializeTurn(turn: TutorQualityTurnSource): TutorQualityTurn { + if (turn.kind === "initial") return turn; + return { + ...turn, + ...(turn.history === undefined ? {} : { history: turn.history.map(historyEntry) }), + }; +} + +export async function loadTutorQualityEvaluationScenarios( + cwd: string, + corpusPath: string, +): Promise { + const manifestPath = path.resolve(cwd, corpusPath); + const source = parseYaml(await readFile(manifestPath, "utf8")) as TutorQualityCorpusSource; + const sketchRefs = new Set(); + for (const { sketch } of source.scenarios) { + try { + await access(path.resolve(cwd, sketch)); + sketchRefs.add(sketch); + } catch { + // Keep the reference absent so the corpus validator reports the contract error. + } + } + const courseContentFixtures = new Set(ANCHOR_COURSE_CONTENT_FIXTURE_IDS); + const corpus = parseTutorQualityCorpus(source, { sketches: sketchRefs, courseContentFixtures }); + return Promise.all(corpus.scenarios.map(async (scenario) => ({ + id: scenario.id, + corpusId: corpus.corpusId, + corpusVersion: corpus.corpusVersion, + sketchRef: scenario.sketch, + sketch: await readFile(path.resolve(cwd, scenario.sketch), "utf8"), + ...(scenario.courseContent === "free" + ? {} + : { courseContent: createAnchorCourseContent(scenario.courseContent as AnchorCourseContentFixtureId) }), + turns: scenario.turns.map(materializeTurn), + ...(scenario.expected === undefined ? {} : { expected: scenario.expected }), + }))); +} + +function statusLines(cwd: string): readonly string[] { + const output = execFileSync(resolveGitExecutable(), ["status", "--porcelain=v1", "-uall"], { cwd, encoding: "utf8" }); + return output.split("\n").map((line) => line.trimEnd()).filter(Boolean); +} + +export function isTutorQualityRelevantUntrackedPath(relativePath: string, outputDir: string): boolean { + const normalized = relativePath.replaceAll("\\", "/"); + const normalizedOutput = outputDir.replaceAll("\\", "/").replace(/\/$/, ""); + if (normalizedOutput && (normalized === normalizedOutput || normalized.startsWith(`${normalizedOutput}/`))) return false; + return normalized.startsWith("evals/tutor-quality/") + || normalized.startsWith("server/") + || normalized.startsWith("shared/") + || normalized.startsWith("scripts/") + || ["package.json", "package-lock.json", ".nvmrc"].includes(normalized); +} + +export function readTutorQualityGitState(cwd: string, outputDir: string): TutorQualityGitState { + const sha = execFileSync(resolveGitExecutable(), ["rev-parse", "HEAD"], { cwd, encoding: "utf8" }).trim(); + const lines = statusLines(cwd); + const trackedClean = lines.every((line) => line.slice(0, 2) === "??"); + const outputRelative = path.relative(cwd, path.resolve(cwd, outputDir)); + const relevantUntrackedClean = lines.every((line) => { + if (line.slice(0, 2) !== "??") return true; + return !isTutorQualityRelevantUntrackedPath(line.slice(3).trim(), outputRelative); + }); + return { sha, trackedClean, relevantUntrackedClean }; +} + +export async function runTutorQualityCli( + argv: readonly string[], + dependencies: TutorQualityCliDependencies = {}, +) { + const options = parseTutorQualityCliArgs(argv); + const cwd = dependencies.cwd ?? process.cwd(); + const environment = dependencies.environment ?? process.env; + const scenarios = await loadTutorQualityEvaluationScenarios(cwd, options.corpusPath); + const git = dependencies.git ?? readTutorQualityGitState(cwd, options.outputDir); + const provider = dependencies.provider ?? new KiconnectProvider(); + const timeoutMs = Number(environment.UNOSIM_LLM_TIMEOUT_MS ?? 30_000); + const credential = environment[options.credentialEnv]; + return runTutorQualityEvaluation({ + scenarios, + provider, + providerId: "kiconnect", + endpointOrigin: environment.UNOSIM_LLM_BASE_URL ? new URL(environment.UNOSIM_LLM_BASE_URL).origin : undefined, + ...(credential ? { credential } : {}), + requestedModel: options.model, + samples: options.samples, + maxCalls: options.maxCalls, + timeoutMs, + temperature: 0.2, + outputDir: path.resolve(cwd, options.outputDir), + git, + }); +} + +const invokedDirectly = process.argv[1] !== undefined + && import.meta.url === pathToFileURL(path.resolve(process.argv[1])).href; + +if (invokedDirectly) { + try { + const { report } = await runTutorQualityCli(process.argv.slice(2)); + console.log(JSON.stringify({ runId: report.runId, runStatus: report.runStatus, reason: report.reason, providerCalls: report.providerCalls })); + } catch (error: unknown) { + console.error(error instanceof Error ? error.message : "Tutor Quality evaluation failed"); + process.exitCode = 1; + } +} diff --git a/server/services/tutor/evaluation/anchor-corpus.ts b/server/services/tutor/evaluation/anchor-corpus.ts new file mode 100644 index 00000000..ac26c0c6 --- /dev/null +++ b/server/services/tutor/evaluation/anchor-corpus.ts @@ -0,0 +1,245 @@ +import { createHash } from "node:crypto"; + +export interface TutorQualityHistoryEntrySource { + readonly question: string; + readonly answer?: string; + readonly responseStyle?: "normal" | "philosophical"; + readonly answerRating?: 1 | 2 | 3 | 4 | 5; + readonly questionId?: string; +} + +export type TutorQualityTurnSource = + | { + readonly kind: "initial"; + readonly difficulty?: number; + } + | { + readonly kind: "dialog"; + readonly question: string; + readonly answer: string; + readonly bindsToQuestion: string; + readonly continuationOf?: number; + readonly difficulty?: number; + readonly history?: readonly TutorQualityHistoryEntrySource[]; + }; + +export interface TutorQualityScenarioSource { + readonly id: string; + readonly sketch: string; + readonly courseContent: string; + readonly model?: string; + readonly turns: readonly TutorQualityTurnSource[]; + readonly expected?: { + readonly topicId?: string; + readonly topicIdAbsent?: string; + readonly learningPhase?: "LEARN" | "DEEPEN" | "EXPAND"; + readonly stateUnchanged?: boolean; + readonly questionNotRepeat?: "exact-or-heuristic"; + }; +} + +export interface TutorQualityCorpusSource { + readonly corpusId: string; + readonly corpusVersion: number; + readonly scenarios: readonly TutorQualityScenarioSource[]; +} + +export interface TutorQualityCorpusReferences { + readonly sketches: ReadonlySet; + readonly courseContentFixtures: ReadonlySet; +} + +export interface TutorQualityCorpus extends TutorQualityCorpusSource { + readonly digest: string; +} + +export interface CorpusEvolutionComparison { + readonly valid: boolean; + readonly reason?: "version-not-increased" | "version-changed-without-semantic-change"; +} + +function fail(message: string): never { + throw new Error(`Invalid Tutor Quality corpus: ${message}`); +} + +function assertObject(value: unknown, label: string): asserts value is Record { + if (value === null || typeof value !== "object" || Array.isArray(value)) fail(`${label} must be an object`); +} + +function assertNonEmptyString(value: unknown, label: string): asserts value is string { + if (typeof value !== "string" || value.trim().length === 0) fail(`${label} must be a non-empty string`); +} + +function optionalInteger(value: unknown, label: string): number | undefined { + if (value === undefined) return undefined; + if (typeof value !== "number" || !Number.isInteger(value) || value < 1 || value > 100) { + fail(`${label} must be an integer from 1 to 100`); + } + return value; +} + +function parseHistory(value: unknown, label: string): readonly TutorQualityHistoryEntrySource[] | undefined { + if (value === undefined) return undefined; + if (!Array.isArray(value)) fail(`${label}.history must be an array`); + for (const [index, entry] of value.entries()) { + const historyLabel = `${label}.history[${index}]`; + assertObject(entry, historyLabel); + assertNonEmptyString(entry.question, `${historyLabel}.question`); + if (entry.answer !== undefined) assertNonEmptyString(entry.answer, `${historyLabel}.answer`); + if (entry.responseStyle !== undefined && entry.responseStyle !== "normal" && entry.responseStyle !== "philosophical") { + fail(`${historyLabel}.responseStyle is invalid`); + } + if (entry.answerRating !== undefined && ![1, 2, 3, 4, 5].includes(entry.answerRating as 1 | 2 | 3 | 4 | 5)) { + fail(`${historyLabel}.answerRating is invalid`); + } + if (entry.questionId !== undefined) assertNonEmptyString(entry.questionId, `${historyLabel}.questionId`); + } + return value as readonly TutorQualityHistoryEntrySource[]; +} + +function parseInitialTurn(value: Record, label: string): TutorQualityTurnSource | undefined { + if (value.kind !== "initial") return undefined; + const difficulty = optionalInteger(value.difficulty, `${label}.difficulty`); + return { kind: "initial", ...(difficulty === undefined ? {} : { difficulty }) }; +} + +function parseDialogTurn(value: Record, label: string): TutorQualityTurnSource { + if (value.kind !== "dialog") fail(`${label}.kind must be initial or dialog`); + assertNonEmptyString(value.question, `${label}.question`); + assertNonEmptyString(value.answer, `${label}.answer`); + assertNonEmptyString(value.bindsToQuestion, `${label}.bindsToQuestion`); + if (value.question !== value.bindsToQuestion) fail(`${label} bindsToQuestion must equal question`); + const continuationOf = value.continuationOf; + if (continuationOf !== undefined && (typeof continuationOf !== "number" || !Number.isInteger(continuationOf) || continuationOf < 0)) { + fail(`${label}.continuationOf must reference a preceding turn`); + } + const difficulty = optionalInteger(value.difficulty, `${label}.difficulty`); + const history = parseHistory(value.history, label); + return { + kind: "dialog", + question: value.question, + answer: value.answer, + bindsToQuestion: value.bindsToQuestion, + ...(continuationOf === undefined ? {} : { continuationOf }), + ...(difficulty === undefined ? {} : { difficulty }), + ...(history === undefined ? {} : { history }), + }; +} + +function parseTurn(value: unknown, label: string): TutorQualityTurnSource { + assertObject(value, label); + return parseInitialTurn(value, label) ?? parseDialogTurn(value, label); +} + +function normalizedCorpusSource(source: TutorQualityCorpusSource): TutorQualityCorpusSource { + return { + corpusId: source.corpusId, + corpusVersion: source.corpusVersion, + scenarios: source.scenarios.map((scenario) => ({ + ...scenario, + turns: scenario.turns.map((turn) => ({ ...turn })), + })), + }; +} + +function stableJson(value: unknown): string { + if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`; + if (value !== null && typeof value === "object") { + return `{${Object.entries(value as Record) + .sort(([left], [right]) => left.localeCompare(right)) + .map(([key, item]) => `${JSON.stringify(key)}:${stableJson(item)}`) + .join(",")}}`; + } + return JSON.stringify(value); +} + +export function tutorQualityCorpusDigest(source: TutorQualityCorpusSource): string { + return createHash("sha256").update(stableJson(normalizedCorpusSource(source))).digest("hex"); +} + +function parseExpected(value: unknown, label: string): TutorQualityScenarioSource["expected"] | undefined { + if (value === undefined) return undefined; + assertObject(value, `${label}.expected`); + if (value.learningPhase !== undefined && !["LEARN", "DEEPEN", "EXPAND"].includes(value.learningPhase as string)) { + fail(`${label}.expected.learningPhase is invalid`); + } + if (value.questionNotRepeat !== undefined && value.questionNotRepeat !== "exact-or-heuristic") { + fail(`${label}.expected.questionNotRepeat is invalid`); + } + return value as TutorQualityScenarioSource["expected"]; +} + +function parseTurns(value: unknown, label: string): readonly TutorQualityTurnSource[] { + if (!Array.isArray(value) || value.length === 0) fail(`${label}.turns must be a non-empty array`); + const turns = value.map((turn, turnIndex) => parseTurn(turn, `${label}.turns[${turnIndex}]`)); + turns.forEach((turn, turnIndex) => { + if (turn.kind === "dialog" && turn.continuationOf !== undefined && turn.continuationOf >= turnIndex) { + fail(`${label}.turns[${turnIndex}].continuationOf must reference a preceding turn`); + } + }); + return turns; +} + +function parseScenario( + rawScenario: unknown, + index: number, + references: TutorQualityCorpusReferences, + ids: Set, +): TutorQualityScenarioSource { + const label = `scenarios[${index}]`; + assertObject(rawScenario, label); + assertNonEmptyString(rawScenario.id, `${label}.id`); + if (!/^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/.test(rawScenario.id)) fail(`${label}.id is not stable-safe`); + if (ids.has(rawScenario.id)) fail(`duplicate scenario id ${rawScenario.id}`); + ids.add(rawScenario.id); + assertNonEmptyString(rawScenario.sketch, `${label}.sketch`); + if (!references.sketches.has(rawScenario.sketch)) fail(`${label}.sketch is not a registered fixture`); + assertNonEmptyString(rawScenario.courseContent, `${label}.courseContent`); + if (rawScenario.courseContent !== "free" && !references.courseContentFixtures.has(rawScenario.courseContent)) { + fail(`${label}.courseContent is not a registered fixture`); + } + if (rawScenario.model !== undefined) { + assertNonEmptyString(rawScenario.model, `${label}.model`); + if (rawScenario.model === "auto") fail(`${label}.model must be a fixed model id`); + } + const turns = parseTurns(rawScenario.turns, label); + const expected = parseExpected(rawScenario.expected, label); + return { + id: rawScenario.id, + sketch: rawScenario.sketch, + courseContent: rawScenario.courseContent, + ...(rawScenario.model === undefined ? {} : { model: rawScenario.model }), + turns, + ...(expected === undefined ? {} : { expected }), + } satisfies TutorQualityScenarioSource; +} + +export function parseTutorQualityCorpus( + source: TutorQualityCorpusSource, + references: TutorQualityCorpusReferences, +): TutorQualityCorpus { + assertObject(source, "corpus"); + assertNonEmptyString(source.corpusId, "corpusId"); + if (!Number.isInteger(source.corpusVersion) || source.corpusVersion < 1) fail("corpusVersion must be a positive integer"); + if (!Array.isArray(source.scenarios) || source.scenarios.length === 0) fail("scenarios must be a non-empty array"); + + const ids = new Set(); + const scenarios = source.scenarios.map((rawScenario, index) => parseScenario(rawScenario, index, references, ids)); + + const normalized = { corpusId: source.corpusId, corpusVersion: source.corpusVersion, scenarios } satisfies TutorQualityCorpusSource; + return { ...normalized, digest: tutorQualityCorpusDigest(normalized) }; +} + +export function compareTutorQualityCorpusVersions( + previous: TutorQualityCorpus, + current: TutorQualityCorpus, +): CorpusEvolutionComparison { + const changed = previous.digest !== current.digest; + if (!changed && previous.corpusVersion !== current.corpusVersion) { + return { valid: false, reason: "version-changed-without-semantic-change" }; + } + if (changed && current.corpusVersion <= previous.corpusVersion) { + return { valid: false, reason: "version-not-increased" }; + } + return { valid: true }; +} diff --git a/server/services/tutor/evaluation/anchor-course-content.ts b/server/services/tutor/evaluation/anchor-course-content.ts new file mode 100644 index 00000000..4a16961e --- /dev/null +++ b/server/services/tutor/evaluation/anchor-course-content.ts @@ -0,0 +1,185 @@ +import type { TutorCapability } from "../../course-content/course-content-loader"; +import { curriculumTopicSchema, validateCurriculumTopic, type CurriculumTopic } from "../curriculum/curriculum-schema"; +import { createTutorProgressionState, type TutorProgressionState } from "../curriculum/progression-state"; +import type { TutorPlanningContentContext } from "../tutor-planning"; + +export const ANCHOR_COURSE_CONTENT_FIXTURE_IDS = [ + "variables", + "progression-learn", + "progression-expand", +] as const; + +export type AnchorCourseContentFixtureId = typeof ANCHOR_COURSE_CONTENT_FIXTURE_IDS[number]; + +const REVISION = "2".repeat(40); + +function variableTopic(schemaVersion: 1 | 2 = 1): CurriculumTopic { + const raw = { + schemaVersion, + id: "variables-and-serial", + title: "Variablen und Serial-Ausgabe", + locale: "de-DE", + activation: { any: [{ fact: "serial-call", values: ["print"] }] }, + concepts: [{ + id: "variable-values", + title: "Variablenwerte", + objective: "Den Zusammenhang zwischen einem deklarierten Integerwert und seiner Verwendung erklären.", + prerequisites: [], + difficulty: { entry: [1, 50], transfer: [20, 80] }, + misconceptions: [], + indicators: [{ id: "relates-value-to-use", description: "Ordnet einen Integerwert seiner Verwendung im Sketch zu." }], + mastery: { + minimumSuccessfulProbes: 1, + successRatingAtLeast: 3, + requiredIndicators: ["relates-value-to-use"], + minimumDistinctQuestionKinds: 1, + recentWeakAnswersAllowed: 0, + }, + }], + questions: [ + { + id: "variable-value-recall", + concept: "variable-values", + indicator: "relates-value-to-use", + kind: "recall", + difficulty: [1, 50], + requires: [{ fact: "type-used", values: ["int"] }], + text: "Welche Rolle spielt der Integer-Datentyp im aktuellen Sketch?", + }, + { + id: "serial-output-prediction", + concept: "variable-values", + indicator: "relates-value-to-use", + kind: "prediction", + difficulty: [20, 70], + requires: [{ fact: "serial-call", values: ["print"] }], + text: "Welche Ausgabe erzeugt Serial.println im aktuellen Sketch?", + }, + { + id: "variable-output-transfer", + concept: "variable-values", + indicator: "relates-value-to-use", + kind: "transfer", + difficulty: [30, 90], + requires: [{ fact: "type-used", values: ["int"] }, { fact: "serial-call", values: ["print"] }], + text: "Wie würdest du die Veränderung von counter an der seriellen Ausgabe überprüfen?", + }, + ], + scaffolds: [], + progression: { + entryConcepts: ["variable-values"], + preferredOrder: ["variable-values"], + onRating: { + "1-2": "remediate", + "3": "clarify-same-indicator", + "4": "probe-missing-indicator", + "5": "evaluate-mastery-and-advance", + }, + }, + ...(schemaVersion === 2 ? { + deepening: { + minimumSuccessfulProbes: 1, + successRatingAtLeast: 4, + requiredQuestionKinds: ["transfer"], + recentWeakAnswersAllowed: 0, + }, + extensions: [{ topic: "serial-output", objective: "Eine weitere serielle Beobachtung am Sketch ableiten." }], + } : {}), + }; + return validateCurriculumTopic(curriculumTopicSchema.parse(raw)); +} + +function serialOutputTopic(): CurriculumTopic { + return validateCurriculumTopic(curriculumTopicSchema.parse({ + schemaVersion: 1, + id: "serial-output", + title: "Serielle Ausgabe", + locale: "de-DE", + activation: { any: [{ fact: "type-used", values: ["float"] }] }, + concepts: [{ + id: "serial-observation", + title: "Serielle Beobachtung", + objective: "Eine serielle Ausgabe als Beobachtung beschreiben.", + prerequisites: [], + difficulty: { entry: [1, 60], transfer: [20, 80] }, + misconceptions: [], + indicators: [{ id: "observes-output", description: "Beschreibt eine serielle Ausgabe." }], + mastery: { + minimumSuccessfulProbes: 1, + successRatingAtLeast: 3, + requiredIndicators: ["observes-output"], + minimumDistinctQuestionKinds: 1, + recentWeakAnswersAllowed: 0, + }, + }], + questions: [{ + id: "serial-output-observe", + concept: "serial-observation", + indicator: "observes-output", + kind: "prediction", + difficulty: [1, 60], + requires: [{ fact: "type-used", values: ["float"] }], + text: "Welche serielle Beobachtung erwartest du?", + }], + scaffolds: [], + progression: { + entryConcepts: ["serial-observation"], + preferredOrder: ["serial-observation"], + onRating: { + "1-2": "remediate", + "3": "clarify-same-indicator", + "4": "probe-missing-indicator", + "5": "evaluate-mastery-and-advance", + }, + }, + })); +} + +function tutorCapability(topics: readonly CurriculumTopic[]): TutorCapability { + return { + status: "valid", + manifest: { + schemaVersion: 2, + topics: [], + strategies: [], + }, + topics, + strategies: [], + }; +} + +function baseState(): TutorProgressionState { + return createTutorProgressionState(REVISION); +} + +function variablesCourseContent(): TutorPlanningContentContext { + return { + revision: REVISION, + tutor: tutorCapability([variableTopic()]), + progressionState: baseState(), + }; +} + +export function createAnchorCourseContent(fixtureId: AnchorCourseContentFixtureId): TutorPlanningContentContext { + switch (fixtureId) { + case "variables": + case "progression-learn": + return variablesCourseContent(); + case "progression-expand": { + const state = baseState(); + state.activeTopicId = "variables-and-serial"; + state.phase = "EXPAND"; + state.masteredTopicIds = ["variables-and-serial"]; + state.retainedPhases = { "variables-and-serial": "EXPAND" }; + return { + revision: REVISION, + tutor: tutorCapability([variableTopic(2), serialOutputTopic()]), + progressionState: state, + }; + } + } +} + +export function anchorCourseContentRevision(): string { + return REVISION; +} diff --git a/server/services/tutor/evaluation/real-provider-evaluation.ts b/server/services/tutor/evaluation/real-provider-evaluation.ts new file mode 100644 index 00000000..bb1614f7 --- /dev/null +++ b/server/services/tutor/evaluation/real-provider-evaluation.ts @@ -0,0 +1,1164 @@ +import { createHash, randomUUID } from "node:crypto"; +import { mkdir, writeFile } from "node:fs/promises"; +import type { TutorContentResult, TutorDialogTurn } from "@shared/tutor"; +import { + inspectLearningQuestion, + isSemanticallyRepeatedQuestion, + TUTOR_PROMPT_REVISION, + TutorService, +} from "../tutor-service"; +import { + TutorProviderError, + type LLMProvider, + type LLMProviderRequest, + type ProviderQuestionResult, +} from "../llm-provider"; +import { CurriculumTutorAdapter } from "../curriculum-tutor-adapter"; +import type { TutorPlanningContentContext } from "../tutor-planning"; +import type { TutorProgressionState } from "../curriculum/progression-state"; + +export type TutorQualityExecutionStatus = "completed" | "invalid" | "technical-failure" | "not-run"; + +export const MAX_TUTOR_QUALITY_SAMPLES = 20; +export const MAX_TUTOR_QUALITY_CALLS = 500; +const EMPTY_PROVIDER_CALLS: TutorQualityProviderCallCounts = { + total: 0, + modelListCalls: 0, + generationCalls: 0, +}; + +export type TutorQualityTurn = + | { + readonly kind: "initial"; + readonly difficulty?: number; + } + | { + readonly kind: "dialog"; + readonly question: string; + readonly answer: string; + readonly bindsToQuestion: string; + readonly continuationOf?: number; + readonly difficulty?: number; + readonly history?: readonly TutorDialogTurn[]; + }; + +export interface TutorQualityEvaluationScenario { + readonly id: string; + readonly corpusId: string; + readonly corpusVersion: number; + readonly sketchRef: string; + readonly sketch: string; + readonly courseContent?: TutorPlanningContentContext; + readonly turns: readonly TutorQualityTurn[]; + readonly expected?: { + readonly topicId?: string; + readonly topicIdAbsent?: string; + readonly learningPhase?: "LEARN" | "DEEPEN" | "EXPAND"; + readonly stateUnchanged?: boolean; + readonly questionNotRepeat?: "exact-or-heuristic"; + }; +} + +export interface TutorQualityGitState { + readonly sha: string; + readonly trackedClean: boolean; + readonly relevantUntrackedClean: boolean; +} + +export interface TutorQualityEvaluationOptions { + readonly scenarios: readonly TutorQualityEvaluationScenario[]; + readonly provider: LLMProvider; + readonly providerId: string; + readonly endpointOrigin?: string; + readonly credential?: string; + readonly requestedModel: string; + readonly samples: number; + readonly maxCalls: number; + readonly timeoutMs?: number; + readonly temperature?: number; + readonly outputDir?: string; + readonly git: TutorQualityGitState; + readonly now?: () => Date; + readonly runSuffix?: () => string; + readonly artifactWriter?: ( + report: TutorQualityEvaluationReport, + transcripts: readonly TutorQualityTranscript[], + ) => Promise; +} + +export interface TutorQualityDeterministicCheck { + readonly name: string; + readonly passed: boolean; + readonly details?: string; +} + +export interface TutorQualityInvariantViolation { + readonly code: string; + readonly source: "raw-provider" | "final-tutor" | "state" | "scenario"; + readonly turnIndex?: number; + readonly details?: string; +} + +export interface TutorQualityTechnicalError { + readonly kind: string; + readonly name: string; +} + +export interface TutorQualityProviderRequestArtifact { + readonly model: string; + readonly systemPrompt: string; + readonly userPrompt: string; +} + +export interface TutorQualityTranscriptTurn { + readonly index: number; + readonly input: TutorQualityTurn; + readonly startedAt: string; + readonly durationMs: number; + readonly providerCalls: TutorQualityProviderCallCounts; + readonly providerRequest?: TutorQualityProviderRequestArtifact; + readonly rawProviderResult?: Record; + readonly finalTutorResult?: Record; + readonly returnedModel?: string; + readonly deterministicChecks: readonly TutorQualityDeterministicCheck[]; + readonly technicalError?: TutorQualityTechnicalError; +} + +export interface TutorQualityTranscript { + readonly schemaVersion: "tutor-quality-transcript-v1"; + readonly runId: string; + readonly evaluationIdentity: string; + readonly metadata: TutorQualityMetadata; + readonly scenario: { + readonly id: string; + readonly corpusId: string; + readonly corpusVersion: number; + readonly sketchRef: string; + readonly sketch: string; + readonly syntheticTurns: readonly TutorQualityTurn[]; + }; + readonly stateBefore?: TutorProgressionState; + readonly stateAfter?: TutorProgressionState; + readonly turns: readonly TutorQualityTranscriptTurn[]; + readonly deterministicChecks: readonly TutorQualityDeterministicCheck[]; + readonly executionStatus: TutorQualityExecutionStatus; + readonly invalidReason?: string; + readonly technicalError?: TutorQualityTechnicalError; + readonly invariantViolations: readonly TutorQualityInvariantViolation[]; +} + +export interface TutorQualityMetadata { + readonly providerId: string; + readonly endpointOrigin?: string; + readonly requestedModel: string; + readonly returnedModels: readonly string[]; + readonly promptRevision: { + readonly id: string; + readonly templateDigest: string; + }; + readonly courseContentRevision: string; + readonly corpusId: string; + readonly corpusVersion: number; + readonly gitSha: string; + readonly gitState: "clean"; + readonly sampleIndex: number; + readonly sampleCount: number; + readonly sampleStartedAt: string; + readonly sampleDurationMs: number; + readonly providerCalls: TutorQualityProviderCallCounts; + readonly timeoutMs?: number; + readonly temperature?: number; + readonly maxCalls: number; +} + +export interface TutorQualityProviderCallCounts { + readonly total: number; + readonly modelListCalls: number; + readonly generationCalls: number; +} + +export interface TutorQualityScenarioAggregate { + readonly samplesRequested: number; + readonly samplesObserved: number; + readonly completed: number; + readonly invalid: number; + readonly technicalFailures: number; + readonly notRun: number; + readonly invariantViolationSamples: number; + readonly budgetExhausted: number; + readonly providerCalls: TutorQualityProviderCallCounts; + readonly technicalErrorKinds: Readonly>; + readonly rates: TutorQualityRates; +} + +export interface TutorQualityRates { + readonly completedOfObserved: number | null; + readonly invalidOfObserved: number | null; + readonly technicalFailureOfObserved: number | null; + readonly notRunOfObserved: number | null; + readonly invariantViolationOfObserved: number | null; +} + +export interface TutorQualityEvaluationReport { + readonly schemaVersion: "tutor-quality-report-v1"; + readonly runId: string; + readonly evaluationIdentity: string; + readonly runStatus: TutorQualityExecutionStatus; + readonly reason?: string; + readonly providerId: string; + readonly requestedModel: string; + readonly credentialPresent: boolean; + readonly samplesRequested: number; + readonly samplesObserved: number; + readonly completed: number; + readonly invalid: number; + readonly technicalFailures: number; + readonly notRun: number; + readonly invariantViolationSamples: number; + readonly budgetExhausted: number; + readonly providerCalls: TutorQualityProviderCallCounts; + readonly technicalErrorKinds: Readonly>; + readonly byScenario: Readonly>; + readonly rates: TutorQualityRates; + readonly monetaryCost: "unavailable"; +} + +export interface TutorQualityEvaluationResult { + readonly report: TutorQualityEvaluationReport; + readonly transcripts: readonly TutorQualityTranscript[]; +} + +interface ProviderCapture { + readonly request: LLMProviderRequest; + readonly response?: ProviderQuestionResult; + readonly error?: unknown; +} + +class EvaluationBudgetExceeded extends Error { + constructor() { + super("call-budget-exhausted"); + this.name = "EvaluationBudgetExceeded"; + } +} + +class CountingProvider implements LLMProvider { + private totalCalls = 0; + private modelListCalls = 0; + private generationCalls = 0; + private readonly generations: ProviderCapture[] = []; + + constructor( + private readonly provider: LLMProvider, + private readonly maxCalls: number, + ) {} + + get counts(): TutorQualityProviderCallCounts { + return { + total: this.totalCalls, + modelListCalls: this.modelListCalls, + generationCalls: this.generationCalls, + }; + } + + get generationCaptures(): readonly ProviderCapture[] { + return this.generations; + } + + async listModels(credential: string): Promise { + this.reserveCall(); + this.modelListCalls += 1; + return this.provider.listModels(credential); + } + + async generateLearningQuestion(request: LLMProviderRequest, credential: string): Promise { + this.reserveCall(); + this.generationCalls += 1; + try { + const response = await this.provider.generateLearningQuestion(request, credential); + this.generations.push({ request, response }); + return response; + } catch (error) { + this.generations.push({ request, error }); + throw error; + } + } + + private reserveCall(): void { + if (this.totalCalls >= this.maxCalls) throw new EvaluationBudgetExceeded(); + this.totalCalls += 1; + } +} + +function stableJson(value: unknown): string { + if (Array.isArray(value)) return `[${value.map(stableJson).join(",")}]`; + if (value !== null && typeof value === "object") { + return `{${Object.entries(value as Record) + .filter(([, item]) => item !== undefined) + .sort(([left], [right]) => left.localeCompare(right)) + .map(([key, item]) => `${JSON.stringify(key)}:${stableJson(item)}`) + .join(",")}}`; + } + return JSON.stringify(value); +} + +function hashCanonical(value: unknown): string { + return createHash("sha256").update(stableJson(value)).digest("hex"); +} + +function clone(value: T): T { + return structuredClone(value); +} + +function courseRevision(scenario: TutorQualityEvaluationScenario): string { + return scenario.courseContent?.revision ?? "free-tutor"; +} + +function makeEvaluationIdentity(options: TutorQualityEvaluationOptions): string { + const first = options.scenarios[0]; + return hashCanonical({ + unosimGitSha: options.git.sha, + courseContentRevisions: [...new Set(options.scenarios.map(courseRevision))].sort((left, right) => left.localeCompare(right)), + corpusId: first?.corpusId, + corpusVersion: first?.corpusVersion, + providerId: options.providerId, + requestedModel: options.requestedModel, + promptRevision: { + id: TUTOR_PROMPT_REVISION.id, + templateDigest: TUTOR_PROMPT_REVISION.templateDigest, + }, + parameters: { + timeoutMs: options.timeoutMs, + temperature: options.temperature, + sampleCount: options.samples, + maxCalls: options.maxCalls, + difficulties: options.scenarios.flatMap((scenario) => scenario.turns.map((turn) => turn.difficulty ?? 30)), + }, + }); +} + +function makeRunId(options: TutorQualityEvaluationOptions): string { + const now = (options.now ?? (() => new Date()))().toISOString().replaceAll(/[^0-9TZ]/g, ""); + return `tq2a-${now}-${(options.runSuffix ?? randomUUID)()}`; +} + +function metadata( + options: TutorQualityEvaluationOptions, + scenario: TutorQualityEvaluationScenario, + sampleIndex: number, + returnedModels: readonly string[] = [], + sampleStartedAt = new Date(0).toISOString(), + sampleDurationMs = 0, + providerCalls: TutorQualityProviderCallCounts = EMPTY_PROVIDER_CALLS, +): TutorQualityMetadata { + return { + providerId: options.providerId, + ...(options.endpointOrigin ? { endpointOrigin: options.endpointOrigin } : {}), + requestedModel: options.requestedModel, + returnedModels, + promptRevision: { + id: TUTOR_PROMPT_REVISION.id, + templateDigest: TUTOR_PROMPT_REVISION.templateDigest, + }, + courseContentRevision: courseRevision(scenario), + corpusId: scenario.corpusId, + corpusVersion: scenario.corpusVersion, + gitSha: options.git.sha, + gitState: "clean", + sampleIndex, + sampleCount: options.samples, + sampleStartedAt, + sampleDurationMs, + providerCalls, + ...(options.timeoutMs === undefined ? {} : { timeoutMs: options.timeoutMs }), + ...(options.temperature === undefined ? {} : { temperature: options.temperature }), + maxCalls: options.maxCalls, + }; +} + +function technicalError(error: unknown): TutorQualityTechnicalError { + if (error instanceof EvaluationBudgetExceeded) return { kind: "call-budget-exhausted", name: error.name }; + if (error instanceof TutorProviderError) return { kind: error.kind, name: error.name }; + return { kind: "provider-error", name: "ProviderError" }; +} + +function redact(value: string, credential: string | undefined): string { + return credential && credential.length > 0 ? value.split(credential).join("[REDACTED]") : value; +} + +function safeResult(value: unknown, credential?: string): Record | undefined { + if (value === null || typeof value !== "object" || Array.isArray(value)) return undefined; + const allowed = new Set([ + "responseStyle", "feedback", "question", "topic", "difficulty", "answerRating", "mermaid", + "topicId", "conceptId", "questionId", "indicatorId", "questionKind", "strategyId", "strategySource", + "contentRevision", "learningPhase", "activeTopicId", "masteredTopicIds", "progressionBlockedReason", + "extensionTargetTopicId", + ]); + const result: Record = {}; + for (const [key, item] of Object.entries(value as Record)) { + if (!allowed.has(key)) continue; + if (typeof item === "string") { + result[key] = redact(item, credential); + } else if (typeof item === "number" || typeof item === "boolean" || item === undefined) { + result[key] = item; + } else if (Array.isArray(item) && item.every((entry) => typeof entry === "string")) { + result[key] = item.map((entry) => redact(entry, credential)); + } + } + return result; +} + +function requestArtifact(request: LLMProviderRequest, credential?: string): TutorQualityProviderRequestArtifact { + return { + model: request.model, + systemPrompt: redact(request.systemPrompt, credential), + userPrompt: redact(request.userPrompt, credential), + }; +} + +function safeTurn(turn: TutorQualityTurn, credential?: string): TutorQualityTurn { + if (turn.kind === "initial") return turn; + return { + ...turn, + question: redact(turn.question, credential), + answer: redact(turn.answer, credential), + bindsToQuestion: redact(turn.bindsToQuestion, credential), + ...(turn.history === undefined ? {} : { + history: turn.history.map((entry) => ({ + ...entry, + question: redact(entry.question, credential), + answer: redact(entry.answer, credential), + ...(entry.feedback === undefined ? {} : { feedback: redact(entry.feedback, credential) }), + })), + }), + }; +} + +function historyFromSource(history: readonly TutorDialogTurn[] | undefined): readonly TutorDialogTurn[] { + return history ?? []; +} + +function addCheck( + checks: TutorQualityDeterministicCheck[], + name: string, + passed: boolean, + details?: string, +): void { + checks.push({ name, passed, ...(details ? { details } : {}) }); +} + +function addViolation( + violations: TutorQualityInvariantViolation[], + code: string, + source: TutorQualityInvariantViolation["source"], + turnIndex: number | undefined, + details?: string, +): void { + violations.push({ code, source, ...(turnIndex === undefined ? {} : { turnIndex }), ...(details ? { details } : {}) }); +} + +function deterministicRawChecks( + rawResult: unknown, + turn: TutorQualityTurn, + turnIndex: number, + checks: TutorQualityDeterministicCheck[], + violations: TutorQualityInvariantViolation[], +): void { + const inspection = inspectLearningQuestion(rawResult, turn.difficulty); + const issueCodes = new Set(inspection.violations.map(({ code }) => code)); + const schemaValid = !issueCodes.has("schema-invalid"); + addCheck(checks, "raw-provider-schema-valid", schemaValid); + addCheck(checks, "raw-provider-one-primary-question", schemaValid && !issueCodes.has("multiple-primary-questions")); + addCheck(checks, "raw-provider-no-complete-solution", schemaValid && !issueCodes.has("complete-solution")); + for (const issue of inspection.violations) { + addViolation(violations, issue.code, "raw-provider", turnIndex); + } + if (typeof rawResult === "object" && rawResult !== null && !Array.isArray(rawResult)) { + const question = (rawResult as { question?: unknown }).question; + if (typeof question === "string" && turn.kind === "dialog") { + const previous = [turn.question, ...(turn.history ?? []).map(({ question: historyQuestion }) => historyQuestion)]; + const repeated = isSemanticallyRepeatedQuestion(question, previous); + addCheck(checks, "raw-provider-question-not-repeated", !repeated); + if (repeated) { + addViolation(violations, "question-repeat", "raw-provider", turnIndex, question === turn.question ? "exact" : "stage1-heuristic"); + } + } + } +} + +interface ExpectedCheckContext { + readonly scenario: TutorQualityEvaluationScenario; + readonly result: TutorContentResult; + readonly turn: TutorQualityTurn; + readonly stateBefore: TutorProgressionState | undefined; + readonly stateAfter: TutorProgressionState | undefined; + readonly checks: TutorQualityDeterministicCheck[]; + readonly violations: TutorQualityInvariantViolation[]; + readonly turnIndex: number; +} + +function expectedTopicCheck(context: ExpectedCheckContext): void { + const expectedTopic = context.scenario.expected?.topicId; + if (expectedTopic === undefined) return; + const passed = context.result.topicId === expectedTopic; + addCheck(context.checks, "expected-topic", passed, expectedTopic); + if (!passed) addViolation(context.violations, "topic-mismatch", "final-tutor", context.turnIndex, expectedTopic); +} + +function expectedTopicAbsentCheck(context: ExpectedCheckContext): void { + const forbiddenTopic = context.scenario.expected?.topicIdAbsent; + if (forbiddenTopic === undefined) return; + const passed = context.result.topicId !== forbiddenTopic && context.result.activeTopicId !== forbiddenTopic; + addCheck(context.checks, "expected-topic-absent", passed, forbiddenTopic); + if (!passed) addViolation(context.violations, "forbidden-topic-activation", "final-tutor", context.turnIndex, forbiddenTopic); +} + +function expectedPhaseCheck(context: ExpectedCheckContext): void { + const expectedPhase = context.scenario.expected?.learningPhase; + if (expectedPhase === undefined) return; + const passed = context.result.learningPhase === expectedPhase || context.stateAfter?.phase === expectedPhase; + addCheck(context.checks, "expected-learning-phase", passed, expectedPhase); + if (!passed) addViolation(context.violations, "phase-mismatch", "final-tutor", context.turnIndex, expectedPhase); +} + +function expectedStateCheck(context: ExpectedCheckContext): void { + if (!context.scenario.expected?.stateUnchanged || !context.stateBefore || !context.stateAfter) return; + const passed = stableJson(context.stateBefore) === stableJson(context.stateAfter); + addCheck(context.checks, "expected-state-unchanged", passed); + if (!passed) addViolation(context.violations, "state-changed-unexpectedly", "state", context.turnIndex); +} + +function expectedQuestionCheck(context: ExpectedCheckContext): void { + if (context.scenario.expected?.questionNotRepeat !== "exact-or-heuristic" || context.turn.kind !== "dialog") return; + const previousQuestions = [context.turn.question, ...(context.turn.history ?? []).map(({ question }) => question)]; + const passed = !isSemanticallyRepeatedQuestion(context.result.question, previousQuestions); + addCheck(context.checks, "final-question-not-repeated", passed); + if (!passed) addViolation(context.violations, "question-repeat", "final-tutor", context.turnIndex, "stage1-heuristic"); +} + +function expectedChecks(context: ExpectedCheckContext): void { + if (!context.scenario.expected) return; + expectedTopicCheck(context); + expectedTopicAbsentCheck(context); + expectedPhaseCheck(context); + expectedStateCheck(context); + expectedQuestionCheck(context); +} + +function applicationMetadataChecks( + result: TutorContentResult, + courseContent: TutorPlanningContentContext | undefined, + stateAfter: TutorProgressionState | undefined, + checks: TutorQualityDeterministicCheck[], + violations: TutorQualityInvariantViolation[], + turnIndex: number, +): void { + if (!courseContent) return; + const revisionPassed = result.contentRevision === undefined || result.contentRevision === courseContent.revision; + addCheck(checks, "content-revision-consistent", revisionPassed); + if (!revisionPassed) addViolation(violations, "content-revision-mismatch", "final-tutor", turnIndex); + if (!stateAfter) return; + const phasePassed = result.learningPhase === undefined + || stateAfter.phase === undefined + || result.learningPhase === stateAfter.phase + || (result.learningPhase === "LEARN" && stateAfter.phase === "DEEPEN"); + addCheck(checks, "phase-state-consistent", phasePassed); + if (!phasePassed) addViolation(violations, "state-phase-mismatch", "state", turnIndex); + const topicPassed = result.activeTopicId === undefined || result.activeTopicId === stateAfter.activeTopicId; + addCheck(checks, "active-topic-state-consistent", topicPassed); + if (!topicPassed) addViolation(violations, "state-topic-mismatch", "state", turnIndex); +} + +function questionIdReuseCheck( + result: TutorContentResult, + turn: TutorQualityTurn, + courseContent: TutorPlanningContentContext | undefined, + checks: TutorQualityDeterministicCheck[], + violations: TutorQualityInvariantViolation[], + turnIndex: number, +): void { + if (!courseContent?.tutor || turn.kind !== "dialog" || !result.questionId) return; + const usedIds = new Set((turn.history ?? []).map(({ questionId }) => questionId).filter((id): id is string => id !== undefined)); + if (result.questionId === undefined) return; + const passed = !usedIds.has(result.questionId); + addCheck(checks, "question-id-not-reused", passed); + if (!passed) addViolation(violations, "question-id-reused", "final-tutor", turnIndex, result.questionId); +} + +function sampleTemplate( + options: TutorQualityEvaluationOptions, + scenario: TutorQualityEvaluationScenario, + runId: string, + evaluationIdentity: string, + sampleIndex: number, +): TutorQualityTranscript { + return { + schemaVersion: "tutor-quality-transcript-v1", + runId, + evaluationIdentity, + metadata: metadata(options, scenario, sampleIndex), + scenario: { + id: scenario.id, + corpusId: scenario.corpusId, + corpusVersion: scenario.corpusVersion, + sketchRef: scenario.sketchRef, + sketch: redact(scenario.sketch, options.credential), + syntheticTurns: scenario.turns.map((turn) => safeTurn(turn, options.credential)), + }, + ...(scenario.courseContent?.progressionState ? { stateBefore: clone(scenario.courseContent.progressionState) } : {}), + turns: [], + deterministicChecks: [], + executionStatus: "completed", + invariantViolations: [], + }; +} + +interface SampleTurnContext { + readonly options: TutorQualityEvaluationOptions; + readonly provider: CountingProvider; + readonly service: TutorService; + readonly scenario: TutorQualityEvaluationScenario; + readonly content: TutorPlanningContentContext | undefined; + readonly stateBefore: TutorProgressionState | undefined; + readonly finalQuestions: Map; + readonly violations: TutorQualityInvariantViolation[]; + readonly turn: TutorQualityTurn; + readonly turnIndex: number; +} + +interface InvokedTutorTurn { + readonly finalResult?: TutorContentResult; + readonly error?: TutorQualityTechnicalError; + readonly invalidReason?: string; +} + +interface ProcessedCapture { + readonly returnedModel?: string; + readonly invalidReason?: string; +} + +interface SampleTurnOutcome { + readonly transcriptTurn?: TutorQualityTranscriptTurn; + readonly returnedModel?: string; + readonly executionStatus?: Extract; + readonly invalidReason?: string; + readonly terminalError?: TutorQualityTechnicalError; + readonly stop: boolean; +} + +function invalidTurnContext(context: SampleTurnContext): string | undefined { + const { turn, turnIndex, finalQuestions, violations } = context; + if (turn.kind === "initial") return undefined; + if (turn.bindsToQuestion !== turn.question) { + addViolation(violations, "unbound-question-context", "scenario", turnIndex); + return "unbound-question-context"; + } + if (turn.continuationOf !== undefined) { + const precedingQuestion = finalQuestions.get(turn.continuationOf); + if (precedingQuestion === undefined || precedingQuestion !== turn.bindsToQuestion) { + addViolation(violations, "preceding-question-mismatch", "scenario", turnIndex); + return "preceding-question-mismatch"; + } + } + return undefined; +} + +async function invokeTutorTurn(context: SampleTurnContext): Promise { + const { options, service, scenario, content, turn } = context; + const invalidReason = invalidTurnContext(context); + if (invalidReason) return { invalidReason }; + try { + if (turn.kind === "initial") { + const response = await service.generateQuestion(scenario.sketch, options.credential, options.requestedModel, turn.difficulty ?? 30, content); + return { finalResult: response.result }; + } + const response = await service.generateDialogResponse( + scenario.sketch, + historyFromSource(turn.history), + turn.question, + turn.answer, + options.credential, + options.requestedModel, + turn.difficulty ?? 30, + content, + ); + return { finalResult: response.result }; + } catch (error_) { + return { error: technicalError(error_) }; + } +} + +function latestCapture(provider: CountingProvider, beforeGenerationCount: number): ProviderCapture | undefined { + return provider.generationCaptures.length > beforeGenerationCount + ? provider.generationCaptures.at(-1) + : undefined; +} + +function processCapture( + capture: ProviderCapture | undefined, + context: SampleTurnContext, + checks: TutorQualityDeterministicCheck[], +): ProcessedCapture { + if (!capture?.response) return {}; + const returnedModel = capture.response.model; + if (typeof returnedModel === "string" && returnedModel.length > 0) { + if (returnedModel !== context.options.requestedModel) { + addCheck(checks, "returned-model-matches-request", false, returnedModel); + deterministicRawChecks(capture.response.result, context.turn, context.turnIndex, checks, context.violations); + return { returnedModel, invalidReason: "returned-model-mismatch" }; + } + addCheck(checks, "returned-model-matches-request", true); + deterministicRawChecks(capture.response.result, context.turn, context.turnIndex, checks, context.violations); + return { returnedModel }; + } + addCheck(checks, "returned-model-matches-request", false, "missing"); + deterministicRawChecks(capture.response.result, context.turn, context.turnIndex, checks, context.violations); + return { invalidReason: "returned-model-missing" }; +} + +function processFinalResult( + context: SampleTurnContext, + invocation: InvokedTutorTurn, + checks: TutorQualityDeterministicCheck[], +): void { + const { finalResult, error } = invocation; + if (!finalResult) { + if (error) addCheck(checks, "final-tutor-response-present", false, error.kind); + return; + } + context.finalQuestions.set(context.turnIndex, finalResult.question); + addCheck(checks, "final-tutor-response-present", true); + expectedChecks({ + scenario: context.scenario, + result: finalResult, + turn: context.turn, + stateBefore: context.stateBefore, + stateAfter: context.content?.progressionState, + checks, + violations: context.violations, + turnIndex: context.turnIndex, + }); + applicationMetadataChecks(finalResult, context.content, context.content?.progressionState, checks, context.violations, context.turnIndex); + questionIdReuseCheck(finalResult, context.turn, context.content, checks, context.violations, context.turnIndex); +} + +function buildTranscriptTurn( + context: SampleTurnContext, + startedAt: Date, + callsBefore: TutorQualityProviderCallCounts, + finishedAt: Date, + capture: ProviderCapture | undefined, + invocation: InvokedTutorTurn, + checks: readonly TutorQualityDeterministicCheck[], +): TutorQualityTranscriptTurn { + const { options, provider, turn, turnIndex } = context; + return { + index: turnIndex, + input: safeTurn(turn, options.credential), + startedAt: startedAt.toISOString(), + durationMs: Math.max(0, finishedAt.getTime() - startedAt.getTime()), + providerCalls: subtractCounts(provider.counts, callsBefore), + ...(capture?.request ? { providerRequest: requestArtifact(capture.request, options.credential) } : {}), + ...(capture?.response ? { + rawProviderResult: safeResult(capture.response.result, options.credential), + ...(typeof capture.response.model === "string" ? { returnedModel: redact(capture.response.model, options.credential) } : {}), + } : {}), + ...(invocation.finalResult ? { finalTutorResult: safeResult(invocation.finalResult, options.credential) } : {}), + deterministicChecks: checks, + ...(invocation.error ? { technicalError: invocation.error } : {}), + }; +} + +async function executeSampleTurn(context: SampleTurnContext): Promise { + const startedAt = (context.options.now ?? (() => new Date()))(); + const callsBefore = context.provider.counts; + const checks: TutorQualityDeterministicCheck[] = []; + const beforeGenerationCount = context.provider.generationCaptures.length; + const invocation = await invokeTutorTurn(context); + if (invocation.invalidReason) return { invalidReason: invocation.invalidReason, executionStatus: "invalid", stop: true }; + const capture = latestCapture(context.provider, beforeGenerationCount); + const captureResult = processCapture(capture, context, checks); + processFinalResult(context, invocation, checks); + const finishedAt = (context.options.now ?? (() => new Date()))(); + return { + transcriptTurn: buildTranscriptTurn(context, startedAt, callsBefore, finishedAt, capture, invocation, checks), + returnedModel: captureResult.returnedModel, + ...(invocation.error ? { executionStatus: "technical-failure" as const, terminalError: invocation.error } : {}), + ...(captureResult.invalidReason ? { executionStatus: "invalid" as const, invalidReason: captureResult.invalidReason } : {}), + stop: invocation.error !== undefined, + }; +} + +interface SampleExecutionState { + readonly violations: TutorQualityInvariantViolation[]; + readonly turns: TutorQualityTranscriptTurn[]; + readonly returnedModels: string[]; + readonly finalQuestions: Map; + executionStatus: TutorQualityExecutionStatus; + invalidReason: string | undefined; + terminalError: TutorQualityTechnicalError | undefined; +} + +function applySampleTurnOutcome(state: SampleExecutionState, outcome: SampleTurnOutcome): void { + if (outcome.executionStatus === "technical-failure") state.executionStatus = "technical-failure"; + if (outcome.executionStatus === "invalid") state.executionStatus = "invalid"; + if (outcome.invalidReason) state.invalidReason = state.invalidReason ?? outcome.invalidReason; + if (outcome.terminalError) state.terminalError = outcome.terminalError; + if (outcome.returnedModel) state.returnedModels.push(outcome.returnedModel); + if (outcome.transcriptTurn) state.turns.push(outcome.transcriptTurn); +} + +interface SampleTurnsContext { + readonly options: TutorQualityEvaluationOptions; + readonly provider: CountingProvider; + readonly service: TutorService; + readonly scenario: TutorQualityEvaluationScenario; + readonly content: TutorPlanningContentContext | undefined; + readonly stateBefore: TutorProgressionState | undefined; + readonly execution: SampleExecutionState; +} + +async function executeSampleTurns(context: SampleTurnsContext): Promise { + const { options, provider, service, scenario, content, stateBefore, execution } = context; + for (const [turnIndex, turn] of scenario.turns.entries()) { + const outcome = await executeSampleTurn({ + options, + provider, + service, + scenario, + content, + stateBefore, + finalQuestions: execution.finalQuestions, + violations: execution.violations, + turn, + turnIndex, + }); + applySampleTurnOutcome(execution, outcome); + if (outcome.stop) break; + } +} + +function recordFailureStateCheck( + terminalError: TutorQualityTechnicalError | undefined, + stateBefore: TutorProgressionState | undefined, + stateAfter: TutorProgressionState | undefined, + turns: TutorQualityTranscriptTurn[], + violations: TutorQualityInvariantViolation[], +): void { + if (!terminalError || !stateBefore || !stateAfter) return; + const unchanged = stableJson(stateBefore) === stableJson(stateAfter); + const stateChecks = turns.at(-1)?.deterministicChecks; + if (stateChecks) { + (stateChecks as TutorQualityDeterministicCheck[]).push({ name: "state-unchanged-after-failure", passed: unchanged }); + } + if (!unchanged) addViolation(violations, "state-mutated-after-failure", "state", undefined); +} + +async function runSample( + options: TutorQualityEvaluationOptions, + provider: CountingProvider, + scenario: TutorQualityEvaluationScenario, + runId: string, + evaluationIdentity: string, + sampleIndex: number, +): Promise { + const sampleStartedAt = (options.now ?? (() => new Date()))(); + const callsBeforeSample = provider.counts; + const content = scenario.courseContent ? clone(scenario.courseContent) : undefined; + const stateBefore = content?.progressionState ? clone(content.progressionState) : undefined; + const service = new TutorService(provider, content ? new CurriculumTutorAdapter() : undefined); + const execution: SampleExecutionState = { + violations: [], + turns: [], + returnedModels: [], + finalQuestions: new Map(), + executionStatus: "completed", + invalidReason: undefined, + terminalError: undefined, + }; + await executeSampleTurns({ options, provider, service, scenario, content, stateBefore, execution }); + + const stateAfter = content?.progressionState ? clone(content.progressionState) : undefined; + recordFailureStateCheck(execution.terminalError, stateBefore, stateAfter, execution.turns, execution.violations); + const sample = sampleTemplate(options, scenario, runId, evaluationIdentity, sampleIndex); + const sampleFinishedAt = (options.now ?? (() => new Date()))(); + const sampleCalls = subtractCounts(provider.counts, callsBeforeSample); + return { + ...sample, + metadata: metadata( + options, + scenario, + sampleIndex, + execution.returnedModels, + sampleStartedAt.toISOString(), + Math.max(0, sampleFinishedAt.getTime() - sampleStartedAt.getTime()), + sampleCalls, + ), + ...(stateBefore ? { stateBefore } : {}), + ...(stateAfter ? { stateAfter } : {}), + turns: execution.turns, + deterministicChecks: execution.turns.flatMap(({ deterministicChecks }) => deterministicChecks), + executionStatus: execution.executionStatus, + ...(execution.invalidReason ? { invalidReason: execution.invalidReason } : {}), + ...(execution.terminalError ? { technicalError: execution.terminalError } : {}), + invariantViolations: execution.violations, + }; +} + +function emptyAggregate(samples: number): TutorQualityScenarioAggregate { + return { + samplesRequested: samples, + samplesObserved: 0, + completed: 0, + invalid: 0, + technicalFailures: 0, + notRun: 0, + invariantViolationSamples: 0, + budgetExhausted: 0, + providerCalls: { total: 0, modelListCalls: 0, generationCalls: 0 }, + technicalErrorKinds: {}, + rates: ratesFor(0, 0, 0, 0, 0, 0), + }; +} + +function ratio(numerator: number, denominator: number): number | null { + return denominator === 0 ? null : numerator / denominator; +} + +function ratesFor( + completed: number, + invalid: number, + technicalFailures: number, + notRun: number, + invariantViolations: number, + observed: number, +): TutorQualityRates { + return { + completedOfObserved: ratio(completed, observed), + invalidOfObserved: ratio(invalid, observed), + technicalFailureOfObserved: ratio(technicalFailures, observed), + notRunOfObserved: ratio(notRun, observed), + invariantViolationOfObserved: ratio(invariantViolations, observed), + }; +} + +function addCounts(left: TutorQualityProviderCallCounts, right: TutorQualityProviderCallCounts): TutorQualityProviderCallCounts { + return { + total: left.total + right.total, + modelListCalls: left.modelListCalls + right.modelListCalls, + generationCalls: left.generationCalls + right.generationCalls, + }; +} + +function subtractCounts( + current: TutorQualityProviderCallCounts, + previous: TutorQualityProviderCallCounts, +): TutorQualityProviderCallCounts { + return { + total: current.total - previous.total, + modelListCalls: current.modelListCalls - previous.modelListCalls, + generationCalls: current.generationCalls - previous.generationCalls, + }; +} + +function addTranscriptToAggregate( + aggregate: TutorQualityScenarioAggregate, + transcript: TutorQualityTranscript, + calls: TutorQualityProviderCallCounts, +): TutorQualityScenarioAggregate { + const samplesObserved = aggregate.samplesObserved + 1; + const completed = aggregate.completed + Number(transcript.executionStatus === "completed"); + const invalid = aggregate.invalid + Number(transcript.executionStatus === "invalid"); + const technicalFailures = aggregate.technicalFailures + Number(transcript.executionStatus === "technical-failure"); + const notRun = aggregate.notRun + Number(transcript.executionStatus === "not-run"); + const invariantViolationSamples = aggregate.invariantViolationSamples + Number(transcript.invariantViolations.length > 0); + const budgetExhausted = aggregate.budgetExhausted + Number(transcript.technicalError?.kind === "call-budget-exhausted"); + const technicalErrorKinds = transcript.technicalError + ? { ...aggregate.technicalErrorKinds, [transcript.technicalError.kind]: (aggregate.technicalErrorKinds[transcript.technicalError.kind] ?? 0) + 1 } + : aggregate.technicalErrorKinds; + return { + ...aggregate, + samplesObserved, + completed, + invalid, + technicalFailures, + notRun, + invariantViolationSamples, + budgetExhausted, + providerCalls: addCounts(aggregate.providerCalls, calls), + technicalErrorKinds, + rates: ratesFor(completed, invalid, technicalFailures, notRun, invariantViolationSamples, samplesObserved), + }; +} + +interface BaseReportContext { + readonly options: TutorQualityEvaluationOptions; + readonly runId: string; + readonly evaluationIdentity: string; + readonly runStatus: TutorQualityExecutionStatus; + readonly reason?: string; + readonly calls: TutorQualityProviderCallCounts; + readonly byScenario: Readonly>; + readonly samplesObserved?: number; +} + +function baseReport(context: BaseReportContext): TutorQualityEvaluationReport { + const { options, runId, evaluationIdentity, runStatus, reason, calls, byScenario, samplesObserved = 0 } = context; + const aggregates = Object.values(byScenario); + const completed = aggregates.reduce((sum, item) => sum + item.completed, 0); + const invalid = aggregates.reduce((sum, item) => sum + item.invalid, 0); + const technicalFailures = aggregates.reduce((sum, item) => sum + item.technicalFailures, 0); + const notRun = aggregates.reduce((sum, item) => sum + item.notRun, 0); + const invariantViolationSamples = aggregates.reduce((sum, item) => sum + item.invariantViolationSamples, 0); + const budgetExhausted = aggregates.reduce((sum, item) => sum + item.budgetExhausted, 0); + const sampleTechnicalErrorKinds = Object.fromEntries( + aggregates.flatMap((item) => Object.entries(item.technicalErrorKinds)).reduce((entries, [kind, count]) => { + const current = entries.get(kind) ?? 0; + entries.set(kind, current + count); + return entries; + }, new Map()), + ); + const preflightTechnicalFailure = runStatus === "technical-failure" ? 1 : 0; + const technicalErrorKinds = reason && preflightTechnicalFailure > 0 + ? { ...sampleTechnicalErrorKinds, [reason]: (sampleTechnicalErrorKinds[reason] ?? 0) + 1 } + : sampleTechnicalErrorKinds; + return { + schemaVersion: "tutor-quality-report-v1", + runId, + evaluationIdentity, + runStatus, + ...(reason ? { reason } : {}), + providerId: options.providerId, + requestedModel: options.requestedModel, + credentialPresent: Boolean(options.credential), + samplesRequested: options.scenarios.length * options.samples, + samplesObserved, + completed, + invalid, + technicalFailures: technicalFailures + preflightTechnicalFailure, + notRun, + invariantViolationSamples, + budgetExhausted, + providerCalls: calls, + technicalErrorKinds, + byScenario, + rates: ratesFor(completed, invalid, technicalFailures + preflightTechnicalFailure, notRun, invariantViolationSamples, samplesObserved), + monetaryCost: "unavailable", + }; +} + +async function writeArtifactsToDirectory( + outputDir: string | undefined, + report: TutorQualityEvaluationReport, + transcripts: readonly TutorQualityTranscript[], +): Promise { + if (!outputDir) return; + await mkdir(outputDir, { recursive: true }); + await writeFile(`${outputDir}/report.json`, `${JSON.stringify(report, null, 2)}\n`, "utf8"); + await Promise.all(transcripts.map((transcript) => { + const safeId = transcript.scenario.id.replaceAll(/[^A-Za-z0-9._-]/g, "_"); + const index = transcript.metadata.sampleIndex; + return writeFile(`${outputDir}/transcript-${safeId}-${index}.json`, `${JSON.stringify(transcript, null, 2)}\n`, "utf8"); + })); +} + +async function writeArtifacts( + options: TutorQualityEvaluationOptions, + report: TutorQualityEvaluationReport, + transcripts: readonly TutorQualityTranscript[], +): Promise { + if (options.artifactWriter) { + await options.artifactWriter(report, transcripts); + return; + } + await writeArtifactsToDirectory(options.outputDir, report, transcripts); +} + +function invalidBasicPreflightReason(options: TutorQualityEvaluationOptions): string | undefined { + if (!options.requestedModel || options.requestedModel === "auto") return "fixed-model-required"; + if (!options.git.sha) return "git-sha-missing"; + if (!options.git.trackedClean || !options.git.relevantUntrackedClean) return "dirty-relevant-worktree"; + if (!Number.isInteger(options.samples) || options.samples < 1) return "invalid-sample-count"; + if (options.samples > MAX_TUTOR_QUALITY_SAMPLES) return "sample-count-exceeds-limit"; + if (!Number.isInteger(options.maxCalls) || options.maxCalls < 0) return "invalid-call-budget"; + if (options.maxCalls > MAX_TUTOR_QUALITY_CALLS) return "call-budget-exceeds-limit"; + return undefined; +} + +function invalidCorpusPreflightReason(options: TutorQualityEvaluationOptions): string | undefined { + if (options.scenarios.length === 0) return "empty-corpus"; + const first = options.scenarios[0]; + if (options.scenarios.some((scenario) => scenario.corpusId !== first?.corpusId || scenario.corpusVersion !== first?.corpusVersion)) { + return "mixed-corpus-versions"; + } + for (const scenario of options.scenarios) { + for (const turn of scenario.turns) { + if (turn.kind === "dialog" && turn.bindsToQuestion !== turn.question) return "unbound-question-context"; + } + } + return undefined; +} + +function invalidPreflightReason(options: TutorQualityEvaluationOptions): string | undefined { + return invalidBasicPreflightReason(options) ?? invalidCorpusPreflightReason(options); +} + +export async function runTutorQualityEvaluation(options: TutorQualityEvaluationOptions): Promise { + const runId = makeRunId(options); + const evaluationIdentity = makeEvaluationIdentity(options); + const invalidReason = invalidPreflightReason(options); + const emptyByScenario = Object.fromEntries(options.scenarios.map((scenario) => [scenario.id, emptyAggregate(options.samples)])); + + if (invalidReason) { + const report = baseReport({ options, runId, evaluationIdentity, runStatus: "invalid", reason: invalidReason, calls: EMPTY_PROVIDER_CALLS, byScenario: emptyByScenario }); + await writeArtifacts(options, report, []); + return { report, transcripts: [] }; + } + if (!options.credential) { + const report = baseReport({ options, runId, evaluationIdentity, runStatus: "not-run", reason: "missing-credential", calls: EMPTY_PROVIDER_CALLS, byScenario: emptyByScenario }); + await writeArtifacts(options, report, []); + return { report, transcripts: [] }; + } + if (options.maxCalls === 0) { + const report = baseReport({ options, runId, evaluationIdentity, runStatus: "not-run", reason: "call-budget-zero", calls: EMPTY_PROVIDER_CALLS, byScenario: emptyByScenario }); + await writeArtifacts(options, report, []); + return { report, transcripts: [] }; + } + + const provider = new CountingProvider(options.provider, options.maxCalls); + let availableModels: readonly string[]; + try { + availableModels = await provider.listModels(options.credential); + } catch (error) { + const report = baseReport({ options, runId, evaluationIdentity, runStatus: "technical-failure", reason: technicalError(error).kind, calls: provider.counts, byScenario: emptyByScenario }); + await writeArtifacts(options, report, []); + return { report, transcripts: [] }; + } + if (!availableModels.includes(options.requestedModel)) { + const report = baseReport({ options, runId, evaluationIdentity, runStatus: "invalid", reason: "model-unavailable", calls: provider.counts, byScenario: emptyByScenario }); + await writeArtifacts(options, report, []); + return { report, transcripts: [] }; + } + + const transcripts: TutorQualityTranscript[] = []; + const byScenario: Record = Object.fromEntries( + options.scenarios.map((scenario) => [scenario.id, emptyAggregate(options.samples)]), + ); + for (const scenario of options.scenarios) { + for (let sampleIndex = 0; sampleIndex < options.samples; sampleIndex += 1) { + const callsBeforeSample = provider.counts; + const transcript = await runSample(options, provider, scenario, runId, evaluationIdentity, sampleIndex); + transcripts.push(transcript); + const aggregate = byScenario[scenario.id]; + if (!aggregate) throw new Error(`Missing scenario aggregate for ${scenario.id}`); + byScenario[scenario.id] = addTranscriptToAggregate(aggregate, transcript, subtractCounts(provider.counts, callsBeforeSample)); + if (transcript.executionStatus === "invalid" && transcript.invalidReason === "returned-model-mismatch") { + // The mismatch belongs to this sample; subsequent samples remain observable. + } + } + } + const report = baseReport({ options, runId, evaluationIdentity, runStatus: "completed", calls: provider.counts, byScenario, samplesObserved: transcripts.length }); + await writeArtifacts(options, report, transcripts); + return { report, transcripts }; +} + +export { EvaluationBudgetExceeded }; diff --git a/server/services/tutor/tutor-service.ts b/server/services/tutor/tutor-service.ts index 87bd7f61..e6723950 100644 --- a/server/services/tutor/tutor-service.ts +++ b/server/services/tutor/tutor-service.ts @@ -1,3 +1,4 @@ +import { createHash } from "node:crypto"; import { analyzeStaticIO } from "@shared/io-registry-parser"; import { tutorContentResultSchema, @@ -439,25 +440,59 @@ function buildPhilosophicalFallback( }; } -function validateLearningQuestion(result: TutorContentResult, difficulty?: TutorDifficulty): TutorContentResult { - const { mermaid: rawMermaid, ...resultWithoutMermaid } = result; - const sanitizedMermaid = sanitizeMermaid(rawMermaid); +export type LearningQuestionViolationCode = + | "schema-invalid" + | "multiple-primary-questions" + | "complete-solution" + | "philosophical-answer-rating"; + +export interface LearningQuestionViolation { + readonly code: LearningQuestionViolationCode; +} + +export interface LearningQuestionInspection { + readonly normalizedResult?: TutorContentResult; + readonly violations: readonly LearningQuestionViolation[]; +} + +function inspectLearningQuestion(result: unknown, difficulty?: TutorDifficulty): LearningQuestionInspection { + const input = typeof result === "object" && result !== null && !Array.isArray(result) + ? result as Record + : {}; + const preSchemaViolations: LearningQuestionViolation[] = []; + if (input.responseStyle === "philosophical" && input.answerRating !== undefined) { + preSchemaViolations.push({ code: "philosophical-answer-rating" }); + } + const { mermaid: rawMermaid, ...resultWithoutMermaid } = input; + const sanitizedMermaid = typeof rawMermaid === "string" ? sanitizeMermaid(rawMermaid) : undefined; const parsed = tutorContentResultSchema.safeParse({ ...resultWithoutMermaid, ...(sanitizedMermaid ? { mermaid: sanitizedMermaid } : {}), }); - if ( - !parsed.success - || (parsed.data.question.match(/\?/g)?.length ?? 0) > 1 - || containsCompleteSolution(parsed.data.question) - || (parsed.data.feedback !== undefined && containsCompleteSolution(parsed.data.feedback)) - ) { - throw new TutorProviderError("invalid-response"); + if (!parsed.success) { + return { + violations: [...preSchemaViolations, { code: "schema-invalid" }], + }; } - if (parsed.data.responseStyle === "philosophical" && parsed.data.answerRating !== undefined) { + const violations: LearningQuestionViolation[] = [...preSchemaViolations]; + if ((parsed.data.question.match(/\?/g)?.length ?? 0) > 1) { + violations.push({ code: "multiple-primary-questions" }); + } + if (containsCompleteSolution(parsed.data.question) || (parsed.data.feedback !== undefined && containsCompleteSolution(parsed.data.feedback))) { + violations.push({ code: "complete-solution" }); + } + return { + normalizedResult: difficulty === undefined ? parsed.data : { ...parsed.data, difficulty }, + violations, + }; +} + +function validateLearningQuestion(result: TutorContentResult, difficulty?: TutorDifficulty): TutorContentResult { + const inspection = inspectLearningQuestion(result, difficulty); + if (inspection.violations.length > 0 || inspection.normalizedResult === undefined) { throw new TutorProviderError("invalid-response"); } - return difficulty === undefined ? parsed.data : { ...parsed.data, difficulty }; + return inspection.normalizedResult; } function stripProviderPlanningMetadata(result: TutorContentResult): TutorContentResult { @@ -710,5 +745,40 @@ export { isClearlyNonLearningAnswer, isSemanticallyRepeatedQuestion, sanitizeMermaid, + inspectLearningQuestion, validateLearningQuestion, }; + +export interface TutorPromptTemplateSources { + readonly system: string; + readonly initialUser: string; + readonly dialogUser: string; +} + +export function digestTutorPromptTemplates(sources: TutorPromptTemplateSources): string { + return createHash("sha256").update(JSON.stringify(sources)).digest("hex"); +} + +const TUTOR_PROMPT_TEMPLATE_SOURCES: TutorPromptTemplateSources = { + system: TUTOR_SYSTEM_PROMPT, + initialUser: [ + buildUserPrompt.toString(), + TUTOR_DIFFICULTY_GUIDANCE, + TUTOR_CONCRETE_REFERENCE_GUIDANCE, + buildTutorStrategyGuidance.toString(), + buildTutorLearningObjectivesGuidance.toString(), + ].join("\n"), + dialogUser: [ + buildDialogPrompt.toString(), + TUTOR_DIFFICULTY_GUIDANCE, + TUTOR_CONCRETE_REFERENCE_GUIDANCE, + buildTutorStrategyGuidance.toString(), + buildTutorLearningObjectivesGuidance.toString(), + ].join("\n"), +}; + +export const TUTOR_PROMPT_REVISION = { + id: "tutor-prompts-v1", + sources: TUTOR_PROMPT_TEMPLATE_SOURCES, + templateDigest: digestTutorPromptTemplates(TUTOR_PROMPT_TEMPLATE_SOURCES), +} as const; diff --git a/ssot/ssot_function_definition_TutorQualityStage2A.md b/ssot/ssot_function_definition_TutorQualityStage2A.md new file mode 100644 index 00000000..3cb360c1 --- /dev/null +++ b/ssot/ssot_function_definition_TutorQualityStage2A.md @@ -0,0 +1,256 @@ +# Tutor Quality – Stage 2A: Real-Provider Evaluation Foundation + +Status: normative for the Stage-2A evaluation runner and its artifacts. This +SSOT does not change the Stage-1 PR hard gate and does not claim to measure +learning effect. + +## 1. Purpose and boundary + +Stage 2A makes real Tutor responses observable, reproducible enough for +comparison, and available for later human or calibrated-judge review. Its +measurement target is **evaluation integrity**: + +- the same versioned scenarios can be run again; +- every sample identifies the exact code, Course Content, prompt, provider, + model, and relevant parameters used; +- the normal `TutorService` and provider trust boundary remains in force; +- deterministic Stage-1 invariants are applied to real responses; +- technical failures are separated from Tutor-quality invariant violations; +- complete, secret-free transcripts can be inspected later. + +Stage 2A does **not** measure whether a learner actually learned, assign a +semantic quality score, optimize a strategy, or make a Tutor decision. Those +activities are Stage 2B or later. + +## 2. Relationship to Stage 1 + +Stage 1 remains the deterministic PR hard-gate layer. Stage 2A is a separate, +explicitly invoked observation layer. Real-provider calls MUST NOT be added to +normal unit tests, pull-request gates, or other required CI checks. + +The existing trust boundaries remain authoritative: + +- `TutorService` owns validation, planning application, repair, and state + commit; +- `CurriculumTutorAdapter` owns Course Content matching and progression; +- `KiconnectProvider` owns the OpenAI-compatible provider transport and parser; +- the evaluation runner owns only scenario orchestration, metadata, artifact + writing, and aggregation; +- provider output remains untrusted content, never application-owned planning + metadata. + +The runner MUST call the normal `TutorService`/`LLMProvider` path. It MUST NOT +call a provider HTTP endpoint directly or duplicate TutorService validation. + +## 3. Versioned corpus and scenario contract + +The anchor corpus is repository-owned and versioned. A corpus manifest has a +stable `corpusId`, an integer `corpusVersion`, and stable scenario IDs. A +scenario contains only synthetic learner data and references repository-owned +sketch/Course Content fixtures. A scenario MAY contain multiple sequential +Tutor turns; each sample starts from a fresh clone of the declared initial +history and progression state. + +Minimum scenario fields: + +- `id` and corpus version; +- sketch fixture/reference and its digest; +- Course Content revision/reference or explicit free-Tutor mode; +- initial history and progression state, when applicable; +- deterministic difficulty and simulated learner answers for each turn; +- expected structural observations only, such as phase/topic/state policy; +- no credential, personal identifier, or live learner data. + +The initial Stage-2A anchor set is deliberately small: + +1. `TQ-REG-001` PWM: strong answer and no exact or Stage-1-heuristic question + repetition; +2. simple variable question; +3. Serial-output prediction; +4. incorrect answer and remediation; +5. partially correct answer and focused follow-up; +6. strong answer and progression; +7. unmatched Topic / free Tutor; +8. LEARN to DEEPEN; +9. EXPAND; +10. clearly off-topic learner answer. + +Adding, removing, or semantically changing an anchor scenario MUST increase +`corpusVersion`. A scenario edit is an evaluation change, not a hidden +implementation change. Non-semantic formatting changes may keep the version +only when the parsed scenario and its digest remain identical. + +The first implementation uses state-seeded turns for multi-step cases. Every +scripted learner answer is bound to a declared preceding question context. A +continuation step without that binding, or with a different preceding +question, is `invalid`; the runner MUST NOT silently pretend that the answer +was given to an arbitrary real-model question. Scenarios that do not need a +deterministic binding should remain single-step cases. + +## 4. Run identity and metadata + +Each run has both a unique `runId` and a stable `evaluationIdentity`. + +`evaluationIdentity` is the SHA-256 of canonical JSON containing at least: + +- UnoSim Git SHA; +- Course Content revision(s); +- corpus ID and version; +- provider ID; +- requested model ID; +- prompt revision identifier and effective-template digest; +- relevant provider parameters, including timeout, temperature when known, + difficulty, sample count, and call budget. + +The `runId` additionally identifies this concrete invocation and MUST include a +UTC start time and collision-resistant suffix. The identity and run ID are +stored in every transcript and the aggregate report. + +The runner MUST record, without secrets: + +- provider ID and endpoint origin when useful for diagnosis; +- requested model and returned model; +- prompt revision identifier and effective-template digest; +- Course Content revision; +- scenario/corpus version; +- Git SHA and dirty-state indicator; +- sample index, turn index, timestamps, duration, and call counts; +- timeout, temperature, difficulty, and configured call/sample limits. + +`promptRevision` is not an arbitrary label. It consists of a versioned +application-owned prompt identifier and a SHA-256 digest of the effective +system/user prompt templates (before scenario values are inserted). A change +to an application-owned prompt template MUST change this revision. Per-turn +prompt digests MAY additionally be stored in transcripts. + +Real-provider evaluation requires a clean relevant Git state. Any tracked or +indexed change makes the run preflight `invalid` and no provider call is +issued. Untracked files are invalidating only when they are under a +versioned evaluation/runtime input root and could affect the run; known +untracked editor files, protected local SSOT files, and ignored output +directories do not affect the preflight. The runner does not upload a diff as +a substitute for the exact Git SHA. + +Missing or inconsistent identity metadata, including an omitted or `auto` +model, makes a sample `invalid`. It MUST NOT be reported as a Tutor-quality +failure. If the provider returns a model different from the requested fixed +model, the sample is also `invalid` and no quality conclusion is drawn. + +## 5. Provider and credential rules + +The first implementation supports the existing Kiconnect/OpenAI-compatible +provider through `KiconnectProvider` and a fixed model ID supplied by the +scenario invocation. A runner preflight MUST verify that the requested model +is available. Any returned-model mismatch invalidates the sample rather than +silently accepting `auto` fallback. + +Credentials are read only from a configured invocation environment variable. +The CLI may accept the variable's name, but never its value. They MUST +never appear in a transcript, aggregate report, log line, exception message, +Git diff, or uploaded artifact. Authorization headers are provider-internal +and are never part of the evaluation artifact model. + +Missing credentials cause an explicit `not-run/missing-credential` result for +manual/workflow execution. The runner still writes a run-level report with +that reason and zero provider calls, but no sample transcripts. Missing +credentials do not fail normal CI because Stage 2A is not part of normal CI. + +## 6. Transcript artifact contract + +One JSON transcript is written per scenario sample. It contains: + +- schema version, run ID, evaluation identity, and complete metadata; +- scenario inputs and synthetic learner answers; +- ordered logical Tutor turns; +- the Tutor request context needed for later review, excluding credentials and + transport headers; +- parsed `LLMProvider` result as received by `TutorService` before + application-side repair/normalization, normalized final Tutor result, + returned model, or technical error; +- deterministic check records and state-before/state-after snapshots; +- a terminal sample status. + +Every sample has two independent status axes: + +`executionStatus` is one of: + +- `completed`: all requested turns executed and metadata is valid; +- `invalid`: identity/model/contract metadata is missing or inconsistent; +- `technical-failure`: a provider error, timeout, malformed provider response, + or mid-run call-budget exhaustion prevented a turn; +- `not-run`: preflight stopped execution, for example because credentials were + missing or the call budget was zero before the first call. + +`invariantViolations` is always an array. It is empty or contains deterministic +Stage-1 violation records. A repaired raw response may therefore be +`executionStatus: completed` with non-empty `invariantViolations`; the report +MUST preserve that distinction. Missing credentials and preflight budget +exhaustion are `not-run`, not Tutor-quality failures. + +The artifact writer uses an allow-list of fields. It MUST NOT serialize the +credential environment, `process.env`, Authorization headers, or arbitrary +provider response envelopes. + +## 7. Deterministic checks and aggregation + +Stage 2A may report only properties with a deterministic rule. The runner +reuses the existing TutorService validation and bounded repeat heuristic and +records at least. The validation exposes one shared pure diagnostic result for +schema, question-count, complete-solution, and related deterministic issues; +`TutorService` retains its existing throw/repair behavior by consuming that +result, and the evaluator consumes the same records rather than duplicating +rules. + +- provider response/schema validity; +- exactly one primary question; +- complete-solution rejection; +- raw exact or heuristic question repetition; +- application-owned State-/Topic-/Phase-/revision consistency; +- no forbidden Question-ID reuse when Course Content is active; +- state unchanged after provider/technical failure; +- requested/returned model consistency; +- provider error category and timeout category; +- successful scenario/turn execution; +- provider-call count and budget exhaustion. + +The aggregate report contains counts and rates with explicit denominators per +scenario and overall. It MUST keep these categories separate: + +1. `invalid` evaluation metadata; +2. `not-run` preflight outcomes; +3. technical provider failures; +4. deterministic Tutor-quality invariant violations; +5. completed observations. + +There are no semantic quality grades, learning-support scores, or pass/fail +claims about actual learning in Stage 2A. + +Every invocation has a hard maximum sample count and provider-call budget. The +budget covers every external provider call, including `listModels()` preflight +and model resolution as well as generation; reports additionally separate +model-list calls from generation calls. The runner stops before issuing a call +that would exceed the budget. Provider token/cost data is reported only when +the provider supplies it; otherwise the report states that monetary cost is +unavailable and still reports exact call counts. + +## 8. Execution and CI + +The evaluation is available through a local CLI with explicit model, sample, +output-directory, credential-environment-variable name, and call-budget +inputs. A credential value is never a CLI argument. A manual +`workflow_dispatch` job MAY invoke the same CLI with a repository secret and +upload transcripts/report artifacts. The workflow MUST have no `pull_request` +trigger and MUST not gate merges. A missing secret is a skipped/not-run +evaluation, not a normal CI failure. + +Evaluation output is disposable run data. It is not committed to the +repository by the runner and should be written to a caller-selected output +directory or uploaded as an artifact with bounded retention. + +## 9. Explicit Stage-2B boundary + +Stage 2B may add a semantic rubric, calibrated LLM-as-Judge, inter-rater +agreement with human review, and baseline-vs-candidate statistical analysis. +Those layers must consume Stage-2A transcripts and deterministic reports; they +must not change Stage-1 invariants or reinterpret Stage-2A technical failures +as learning outcomes. diff --git a/tests/server/services/tutor/evaluation/anchor-corpus.test.ts b/tests/server/services/tutor/evaluation/anchor-corpus.test.ts new file mode 100644 index 00000000..dc5e642a --- /dev/null +++ b/tests/server/services/tutor/evaluation/anchor-corpus.test.ts @@ -0,0 +1,139 @@ +import { readFileSync } from "node:fs"; +import { fileURLToPath } from "node:url"; +import { parse as parseYaml } from "yaml"; +import { describe, expect, it } from "vitest"; +import { + compareTutorQualityCorpusVersions, + parseTutorQualityCorpus, + type TutorQualityCorpusSource, +} from "../../../../../server/services/tutor/evaluation/anchor-corpus"; +import { + ANCHOR_COURSE_CONTENT_FIXTURE_IDS, + createAnchorCourseContent, +} from "../../../../../server/services/tutor/evaluation/anchor-course-content"; + +const references = { + sketches: new Set(["variable.ino"]), + courseContentFixtures: new Set(["variables"]), +}; + +function validSource(): TutorQualityCorpusSource { + return { + corpusId: "test-corpus", + corpusVersion: 1, + scenarios: [ + { + id: "variable", + sketch: "variable.ino", + courseContent: "variables", + turns: [ + { + kind: "dialog", + question: "Welchen Wert hat x?", + answer: "x ist drei.", + bindsToQuestion: "Welchen Wert hat x?", + difficulty: 20, + }, + ], + expected: { topicId: "variables-and-serial" }, + }, + ], + }; +} + +describe("Tutor Quality anchor corpus contract", () => { + it("contains the reviewed version-1 anchor set", () => { + const source = parseYaml(readFileSync(fileURLToPath(new URL("../../../../../evals/tutor-quality/anchor-corpus.yaml", import.meta.url)), "utf8")) as TutorQualityCorpusSource; + const corpus = parseTutorQualityCorpus(source, { + sketches: new Set(source.scenarios.map(({ sketch }) => sketch)), + courseContentFixtures: new Set(["variables", "progression-learn", "progression-expand"]), + }); + + expect(corpus.corpusVersion).toBe(1); + expect(corpus.scenarios.map(({ id }) => id)).toEqual([ + "TQ-REG-001", + "simple-variable", + "serial-output-prediction", + "incorrect-answer-remediation", + "partial-answer-follow-up", + "strong-answer-progression", + "unmatched-topic-free-tutor", + "learn-to-deepen", + "expand", + "off-topic-answer", + ]); + }); + + it("parses stable scenario references and explicit question bindings", () => { + const corpus = parseTutorQualityCorpus(validSource(), references); + + expect(corpus.corpusId).toBe("test-corpus"); + expect(corpus.scenarios[0]).toMatchObject({ + id: "variable", + sketch: "variable.ino", + courseContent: "variables", + turns: [{ bindsToQuestion: "Welchen Wert hat x?" }], + }); + }); + + it.each([ + ["duplicate scenario id", (source: TutorQualityCorpusSource) => ({ + ...source, + scenarios: [...source.scenarios, source.scenarios[0]!], + })], + ["missing sketch reference", (source: TutorQualityCorpusSource) => ({ + ...source, + scenarios: [{ ...source.scenarios[0]!, sketch: "missing.ino" }], + })], + ["auto model", (source: TutorQualityCorpusSource) => ({ + ...source, + scenarios: [{ ...source.scenarios[0]!, model: "auto" }], + })], + ["unbound continuation", (source: TutorQualityCorpusSource) => ({ + ...source, + scenarios: [{ + ...source.scenarios[0]!, + turns: [{ ...source.scenarios[0]!.turns[0]!, bindsToQuestion: undefined }], + }], + })], + ["self-referencing continuation", (source: TutorQualityCorpusSource) => ({ + ...source, + scenarios: [{ + ...source.scenarios[0]!, + turns: [{ ...source.scenarios[0]!.turns[0]!, continuationOf: 0 }], + }], + })], + ])("rejects %s", (_label, mutate) => { + expect(() => parseTutorQualityCorpus(mutate(validSource()), references)).toThrow(); + }); + + it("keeps historical version comparison outside current-corpus parsing", () => { + const previous = parseTutorQualityCorpus(validSource(), references); + const changed = parseTutorQualityCorpus({ + ...validSource(), + scenarios: [{ ...validSource().scenarios[0]!, turns: [{ + ...validSource().scenarios[0]!.turns[0]!, + answer: "x ist vier.", + bindsToQuestion: "Welchen Wert hat x?", + }] }], + }, references); + const bumped = { ...changed, corpusVersion: previous.corpusVersion + 1 }; + + expect(compareTutorQualityCorpusVersions(previous, changed).valid).toBe(false); + expect(compareTutorQualityCorpusVersions(previous, bumped).valid).toBe(true); + expect(compareTutorQualityCorpusVersions(previous, previous).valid).toBe(true); + }); + + it("provides valid typed Course Content fixtures for progression anchors", () => { + for (const fixtureId of ANCHOR_COURSE_CONTENT_FIXTURE_IDS) { + const content = createAnchorCourseContent(fixtureId); + expect(content.tutor?.status).toBe("valid"); + expect(content.revision).toHaveLength(40); + } + expect(createAnchorCourseContent("progression-expand").progressionState).toMatchObject({ + activeTopicId: "variables-and-serial", + phase: "EXPAND", + masteredTopicIds: ["variables-and-serial"], + }); + }); +}); diff --git a/tests/server/services/tutor/evaluation/real-provider-evaluation.test.ts b/tests/server/services/tutor/evaluation/real-provider-evaluation.test.ts new file mode 100644 index 00000000..bed1f879 --- /dev/null +++ b/tests/server/services/tutor/evaluation/real-provider-evaluation.test.ts @@ -0,0 +1,352 @@ +import { mkdtemp, readFile, readdir, rm } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, describe, expect, it } from "vitest"; +import { createAnchorCourseContent } from "../../../../../server/services/tutor/evaluation/anchor-course-content"; +import { + MAX_TUTOR_QUALITY_CALLS, + MAX_TUTOR_QUALITY_SAMPLES, + runTutorQualityEvaluation, + type TutorQualityEvaluationScenario, +} from "../../../../../server/services/tutor/evaluation/real-provider-evaluation"; +import { TutorProviderError, type LLMProvider } from "../../../../../server/services/tutor/llm-provider"; + +const temporaryDirectories: string[] = []; + +afterEach(async () => { + await Promise.all(temporaryDirectories.splice(0).map((directory) => rm(directory, { recursive: true, force: true }))); +}); + +function scenario(overrides: Partial = {}): TutorQualityEvaluationScenario { + return { + id: "repeat-case", + corpusId: "test-corpus", + corpusVersion: 1, + sketchRef: "inline.ino", + sketch: "int counter = 3; void setup() { Serial.begin(9600); } void loop() {}", + courseContent: undefined, + turns: [{ + kind: "dialog", + question: "Welche Rolle spielt counter im Sketch?", + answer: "counter speichert einen ganzzahligen Wert.", + bindsToQuestion: "Welche Rolle spielt counter im Sketch?", + difficulty: 30, + }], + ...overrides, + }; +} + +function providerFor(result: unknown, returnedModel = "fake-model"): LLMProvider { + return { + async listModels() { + return ["fake-model"]; + }, + async generateLearningQuestion() { + return { model: returnedModel, result: result as never }; + }, + }; +} + +function options(provider: LLMProvider, overrides: Partial[0]> = {}) { + return { + scenarios: [scenario()], + provider, + providerId: "fake-provider", + credential: "super-secret-value", + requestedModel: "fake-model", + samples: 1, + maxCalls: 10, + git: { sha: "a".repeat(40), trackedClean: true, relevantUntrackedClean: true }, + ...overrides, + }; +} + +describe("real-provider Tutor Quality evaluation runner", () => { + it("captures the raw repeated question and preserves the repaired final response", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ + responseStyle: "normal", + answerRating: 5, + question: "Welche Rolle spielt counter im Sketch?", + }))); + + const transcript = result.transcripts[0]!; + expect(transcript.executionStatus).toBe("completed"); + expect(transcript.invariantViolations.map(({ code }) => code)).toContain("question-repeat"); + expect(transcript.turns[0]?.rawProviderResult).toMatchObject({ question: "Welche Rolle spielt counter im Sketch?" }); + expect(transcript.turns[0]?.finalTutorResult?.question).not.toBe("Welche Rolle spielt counter im Sketch?"); + expect(result.report.providerCalls).toMatchObject({ total: 3, modelListCalls: 2, generationCalls: 1 }); + }); + + it("records the existing bounded heuristic for a near-repeat", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ + responseStyle: "normal", + answerRating: 3, + question: "Welche Rolle hat counter im Sketch?", + }))); + const repeat = result.transcripts[0]?.invariantViolations.find(({ code }) => code === "question-repeat"); + + expect(repeat?.details).toBe("stage1-heuristic"); + }); + + it("separates provider failure from state mutation and does not commit state", async () => { + const content = createAnchorCourseContent("progression-learn"); + const provider: LLMProvider = { + async listModels() { + return ["fake-model"]; + }, + async generateLearningQuestion() { + throw new TutorProviderError("provider-timeout"); + }, + }; + const result = await runTutorQualityEvaluation(options(provider, { + scenarios: [scenario({ id: "timeout", courseContent: content })], + })); + + const transcript = result.transcripts[0]!; + expect(transcript.executionStatus).toBe("technical-failure"); + expect(transcript.technicalError?.kind).toBe("provider-timeout"); + expect(transcript.deterministicChecks.find(({ name }) => name === "state-unchanged-after-failure")?.passed).toBe(true); + expect(transcript.stateBefore).toEqual(transcript.stateAfter); + }); + + it("marks a returned-model mismatch invalid without treating it as quality", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ question: "Welche Beobachtung ist belegt?" }, "other-model"))); + const transcript = result.transcripts[0]!; + + expect(transcript.executionStatus).toBe("invalid"); + expect(transcript.invalidReason).toBe("returned-model-mismatch"); + expect(transcript.invariantViolations).toHaveLength(0); + }); + + it("marks a missing returned model invalid without leaking or throwing", async () => { + const result = await runTutorQualityEvaluation(options({ + async listModels() { + return ["fake-model"]; + }, + async generateLearningQuestion() { + return { + model: undefined as never, + result: { question: "Welche Beobachtung ist belegt?" }, + }; + }, + })); + const transcript = result.transcripts[0]!; + + expect(transcript.executionStatus).toBe("invalid"); + expect(transcript.invalidReason).toBe("returned-model-missing"); + expect(transcript.turns[0]).not.toHaveProperty("returnedModel"); + }); + + it("counts every provider call against the budget and stops before generation", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ question: "Welche Beobachtung ist belegt?" }), { maxCalls: 2 })); + const transcript = result.transcripts[0]!; + + expect(transcript.executionStatus).toBe("technical-failure"); + expect(transcript.technicalError?.kind).toBe("call-budget-exhausted"); + expect(result.report.providerCalls).toMatchObject({ total: 2, modelListCalls: 2, generationCalls: 0 }); + }); + + it("writes a run-level missing-credential report without sample transcripts or secrets", async () => { + const outputDir = await mkdtemp(path.join(os.tmpdir(), "unosim-tq-stage2a-")); + temporaryDirectories.push(outputDir); + const result = await runTutorQualityEvaluation(options(providerFor({ question: "Welche Beobachtung ist belegt?" }), { + credential: undefined, + outputDir, + })); + const files = await readdir(outputDir); + const reportText = await readFile(path.join(outputDir, "report.json"), "utf8"); + + expect(result.report.runStatus).toBe("not-run"); + expect(result.report.reason).toBe("missing-credential"); + expect(result.report.providerCalls.total).toBe(0); + expect(files).toEqual(["report.json"]); + expect(reportText).not.toContain("super-secret-value"); + }); + + it("supports an injected artifact writer without touching the repository", async () => { + let writtenReport: string | undefined; + let writtenTranscriptCount = -1; + const result = await runTutorQualityEvaluation(options(providerFor({ + responseStyle: "normal", + answerRating: 4, + question: "Welche Beobachtung ist belegt?", + }), { + artifactWriter: async (report, transcripts) => { + writtenReport = report.runId; + writtenTranscriptCount = transcripts.length; + }, + })); + + expect(writtenReport).toBe(result.report.runId); + expect(writtenTranscriptCount).toBe(1); + }); + + it("redacts a credential echoed by a fake provider from every transcript field", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ + responseStyle: "normal", + answerRating: 4, + question: "super-secret-value?", + feedback: "super-secret-value", + }))); + const transcriptText = JSON.stringify(result.transcripts[0]); + + expect(transcriptText).not.toContain("super-secret-value"); + expect(transcriptText).toContain("[REDACTED]"); + }); + + it("redacts credentials from scripted learner input in every transcript turn", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ + responseStyle: "normal", + answerRating: 4, + question: "Welche Beobachtung ist belegt?", + }), { + scenarios: [scenario({ + turns: [{ + kind: "dialog", + question: "Welche Beobachtung ist belegt?", + answer: "super-secret-value", + bindsToQuestion: "Welche Beobachtung ist belegt?", + }], + })], + })); + const transcriptText = JSON.stringify(result.transcripts[0]); + + expect(transcriptText).not.toContain("super-secret-value"); + expect(transcriptText).toContain("[REDACTED]"); + }); + + it("records raw complete-solution and schema violations separately from technical failure", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ + question: "void setup() {} void loop() {}", + }))); + const transcript = result.transcripts[0]!; + + expect(transcript.executionStatus).toBe("technical-failure"); + expect(transcript.technicalError?.kind).toBe("invalid-response"); + expect(transcript.invariantViolations.map(({ code }) => code)).toContain("complete-solution"); + }); + + it("records malformed provider output as a schema violation and technical failure", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ + responseStyle: "normal", + answerRating: "not-a-rating", + }))); + const transcript = result.transcripts[0]!; + + expect(transcript.executionStatus).toBe("technical-failure"); + expect(transcript.technicalError?.kind).toBe("invalid-response"); + expect(transcript.invariantViolations.map(({ code }) => code)).toContain("schema-invalid"); + }); + + it("does not trust provider-supplied planning metadata", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ + responseStyle: "normal", + question: "Welche Beobachtung ist belegt?", + topicId: "forged-topic", + questionId: "forged-question", + learningPhase: "EXPAND", + contentRevision: "f".repeat(40), + }), { + scenarios: [scenario({ + id: "forged-metadata", + sketch: "int counter = 3; Serial.println(counter);", + courseContent: createAnchorCourseContent("variables"), + turns: [{ kind: "initial", difficulty: 20 }], + })], + })); + const transcript = result.transcripts[0]!; + + expect(transcript.executionStatus).toBe("completed"); + expect(transcript.turns[0]?.rawProviderResult).toMatchObject({ topicId: "forged-topic", learningPhase: "EXPAND" }); + expect(transcript.turns[0]?.finalTutorResult).toMatchObject({ + topicId: "variables-and-serial", + learningPhase: "LEARN", + }); + }); + + it("starts every sample from a fresh progression-state clone", async () => { + const content = createAnchorCourseContent("progression-learn"); + const provider = providerFor({ + responseStyle: "normal", + answerRating: 5, + question: "Welche Folgefrage ist als Nächstes sinnvoll?", + }); + const result = await runTutorQualityEvaluation(options(provider, { + samples: 2, + scenarios: [scenario({ + id: "fresh-state", + sketch: "int counter = 3; Serial.println(counter);", + courseContent: content, + turns: [{ + kind: "dialog", + question: "Welche Rolle spielt der Integer-Datentyp im aktuellen Sketch?", + answer: "int speichert den ganzzahligen Wert von counter.", + bindsToQuestion: "Welche Rolle spielt der Integer-Datentyp im aktuellen Sketch?", + difficulty: 30, + }], + })], + })); + + expect(result.transcripts).toHaveLength(2); + expect(result.transcripts.map(({ stateBefore }) => stateBefore?.phase)).toEqual([undefined, undefined]); + }); + + it("rejects dirty or unavailable-model preflight without generation calls", async () => { + const dirty = await runTutorQualityEvaluation(options(providerFor({ question: "Welche Beobachtung ist belegt?" }), { + git: { sha: "a".repeat(40), trackedClean: false, relevantUntrackedClean: true }, + })); + expect(dirty.report).toMatchObject({ runStatus: "invalid", reason: "dirty-relevant-worktree" }); + expect(dirty.report.providerCalls.total).toBe(0); + + const unavailable = await runTutorQualityEvaluation(options({ + async listModels() { + return ["different-model"]; + }, + async generateLearningQuestion() { + throw new Error("must not be called"); + }, + })); + expect(unavailable.report).toMatchObject({ runStatus: "invalid", reason: "model-unavailable" }); + expect(unavailable.report.providerCalls).toMatchObject({ total: 1, generationCalls: 0 }); + + const unboundedSamples = await runTutorQualityEvaluation(options(providerFor({ question: "Welche Beobachtung ist belegt?" }), { + samples: MAX_TUTOR_QUALITY_SAMPLES + 1, + })); + expect(unboundedSamples.report).toMatchObject({ runStatus: "invalid", reason: "sample-count-exceeds-limit" }); + expect(unboundedSamples.report.providerCalls.total).toBe(0); + + const unboundedCalls = await runTutorQualityEvaluation(options(providerFor({ question: "Welche Beobachtung ist belegt?" }), { + maxCalls: MAX_TUTOR_QUALITY_CALLS + 1, + })); + expect(unboundedCalls.report).toMatchObject({ runStatus: "invalid", reason: "call-budget-exceeds-limit" }); + expect(unboundedCalls.report.providerCalls.total).toBe(0); + }); + + it("does not apply a learner answer to an arbitrary preceding real-model question", async () => { + const result = await runTutorQualityEvaluation(options(providerFor({ + responseStyle: "normal", + question: "Welche neue Beobachtung ist belegt?", + answerRating: 4, + }), { + scenarios: [scenario({ + id: "bound-continuation", + turns: [ + { kind: "initial", difficulty: 20 }, + { + kind: "dialog", + question: "Welche deklarierte Frage soll gelten?", + answer: "Eine Antwort auf die deklarierte Frage.", + bindsToQuestion: "Welche deklarierte Frage soll gelten?", + continuationOf: 0, + difficulty: 20, + }, + ], + })], + })); + const transcript = result.transcripts[0]!; + + expect(transcript.executionStatus).toBe("invalid"); + expect(transcript.invalidReason).toBe("preceding-question-mismatch"); + expect(transcript.turns).toHaveLength(1); + }); +}); diff --git a/tests/server/services/tutor/evaluation/tutor-prompt-revision.test.ts b/tests/server/services/tutor/evaluation/tutor-prompt-revision.test.ts new file mode 100644 index 00000000..b1510b92 --- /dev/null +++ b/tests/server/services/tutor/evaluation/tutor-prompt-revision.test.ts @@ -0,0 +1,25 @@ +import { describe, expect, it } from "vitest"; +import { + TUTOR_PROMPT_REVISION, + digestTutorPromptTemplates, + type TutorPromptTemplateSources, +} from "../../../../../server/services/tutor/tutor-service"; + +describe("Tutor prompt revision metadata", () => { + it("exposes versioned system and user template sources", () => { + expect(TUTOR_PROMPT_REVISION.id).toMatch(/^tutor-prompts-v\d+$/); + expect(TUTOR_PROMPT_REVISION.templateDigest).toMatch(/^[a-f0-9]{64}$/); + expect(TUTOR_PROMPT_REVISION.sources.system).toContain("didaktischer Tutor"); + expect(TUTOR_PROMPT_REVISION.sources.initialUser).toContain("Sketch:"); + expect(TUTOR_PROMPT_REVISION.sources.dialogUser).toContain("Nutzerantwort"); + expect(TUTOR_PROMPT_REVISION.templateDigest).toBe(digestTutorPromptTemplates(TUTOR_PROMPT_REVISION.sources)); + }); + + it("changes the digest when an effective template source changes", () => { + const changed: TutorPromptTemplateSources = { + ...TUTOR_PROMPT_REVISION.sources, + system: `${TUTOR_PROMPT_REVISION.sources.system}\nchanged`, + }; + expect(digestTutorPromptTemplates(changed)).not.toBe(TUTOR_PROMPT_REVISION.templateDigest); + }); +}); diff --git a/tests/server/services/tutor/evaluation/tutor-quality-real-provider-cli.test.ts b/tests/server/services/tutor/evaluation/tutor-quality-real-provider-cli.test.ts new file mode 100644 index 00000000..39269203 --- /dev/null +++ b/tests/server/services/tutor/evaluation/tutor-quality-real-provider-cli.test.ts @@ -0,0 +1,103 @@ +import { mkdtemp, readdir, rm } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { describe, expect, it } from "vitest"; +import { + MAX_TUTOR_QUALITY_CALLS, + MAX_TUTOR_QUALITY_SAMPLES, + loadTutorQualityEvaluationScenarios, + isTutorQualityRelevantUntrackedPath, + parseTutorQualityCliArgs, + readTutorQualityGitState, + runTutorQualityCli, +} from "../../../../../scripts/tutor-quality-real-provider-eval"; +import type { LLMProvider } from "../../../../../server/services/tutor/llm-provider"; + +describe("Tutor Quality real-provider CLI contract", () => { + it("requires a fixed model and accepts only a credential environment-variable name", () => { + expect(() => parseTutorQualityCliArgs(["--output-dir", "/tmp/tq", "--model", "auto"])).toThrow(); + expect(() => parseTutorQualityCliArgs(["--output-dir", "/tmp/tq", "--model", "pilot", "--api-key", "secret"])).toThrow(); + + expect(parseTutorQualityCliArgs([ + "--output-dir", "/tmp/tq", + "--model", "pilot-model", + "--samples", "2", + "--max-calls", "12", + "--credential-env", "TEST_TUTOR_CREDENTIAL", + ])).toMatchObject({ + model: "pilot-model", + samples: 2, + maxCalls: 12, + credentialEnv: "TEST_TUTOR_CREDENTIAL", + outputDir: "/tmp/tq", + }); + }); + + it("rejects malformed numeric limits and credential values", () => { + expect(() => parseTutorQualityCliArgs(["--output-dir", "/tmp/tq", "--model", "pilot", "--samples", "0"])).toThrow(); + expect(() => parseTutorQualityCliArgs(["--output-dir", "/tmp/tq", "--model", "pilot", "--credential-env", "not-a-value"])).toThrow(); + expect(() => parseTutorQualityCliArgs(["--output-dir", "/tmp/tq", "--model", "pilot", "--samples", String(MAX_TUTOR_QUALITY_SAMPLES + 1)])).toThrow(); + expect(() => parseTutorQualityCliArgs(["--output-dir", "/tmp/tq", "--model", "pilot", "--max-calls", String(MAX_TUTOR_QUALITY_CALLS + 1)])).toThrow(); + }); + + it("does not treat protected editor SSOT files or the output directory as relevant inputs", () => { + expect(isTutorQualityRelevantUntrackedPath("ssot/ssot_function_tutor_model_registration.md", ".tutor-quality-output")).toBe(false); + expect(isTutorQualityRelevantUntrackedPath("evals/tutor-quality/anchor-corpus.yaml", ".tutor-quality-output")).toBe(true); + expect(isTutorQualityRelevantUntrackedPath(".tutor-quality-output/report.json", ".tutor-quality-output")).toBe(false); + }); + + it("reads the local Git state through a fixed executable location", () => { + const state = readTutorQualityGitState(process.cwd(), "test-results"); + expect(state.sha).toMatch(/^[0-9a-f]{40}$/); + }); + + it("materializes and runs the complete versioned anchor corpus with a fake provider", async () => { + const scenarios = await loadTutorQualityEvaluationScenarios(process.cwd(), "evals/tutor-quality/anchor-corpus.yaml"); + const outputDir = await mkdtemp(path.join(os.tmpdir(), "unosim-tq-cli-test-")); + const provider: LLMProvider = { + async listModels() { + return ["pilot-model"]; + }, + async generateLearningQuestion() { + return { + model: "pilot-model", + result: { + responseStyle: "normal" as const, + answerRating: 5 as const, + question: "Welche konkrete Beobachtung ist im aktuellen Sketch belegt?", + }, + }; + }, + }; + try { + expect(scenarios).toHaveLength(10); + const result = await runTutorQualityCli([ + "--model", "pilot-model", + "--samples", "1", + "--max-calls", "100", + "--credential-env", "TEST_TUTOR_CREDENTIAL", + "--output-dir", outputDir, + ], { + cwd: process.cwd(), + environment: { TEST_TUTOR_CREDENTIAL: "secret-value" }, + provider, + git: { sha: "a".repeat(40), trackedClean: true, relevantUntrackedClean: true }, + }); + expect(result.transcripts.find(({ scenario }) => scenario.id === "TQ-REG-001")?.stateAfter) + .toEqual(result.transcripts.find(({ scenario }) => scenario.id === "TQ-REG-001")?.stateBefore); + expect(result.transcripts.filter(({ invariantViolations }) => invariantViolations.length > 0).map(({ scenario, invariantViolations }) => ({ id: scenario.id, invariantViolations }))).toEqual([]); + expect(result.report).toMatchObject({ + runStatus: "completed", + samplesRequested: 10, + samplesObserved: 10, + invalid: 0, + technicalFailures: 0, + invariantViolationSamples: 0, + }); + expect(result.report.providerCalls.generationCalls).toBeGreaterThan(0); + expect(await readdir(outputDir)).toContain("report.json"); + } finally { + await rm(outputDir, { recursive: true, force: true }); + } + }); +}); diff --git a/tests/server/services/tutor/evaluation/tutor-question-diagnostics.test.ts b/tests/server/services/tutor/evaluation/tutor-question-diagnostics.test.ts new file mode 100644 index 00000000..57341298 --- /dev/null +++ b/tests/server/services/tutor/evaluation/tutor-question-diagnostics.test.ts @@ -0,0 +1,25 @@ +import { describe, expect, it } from "vitest"; +import { + inspectLearningQuestion, + validateLearningQuestion, +} from "../../../../../server/services/tutor/tutor-service"; + +describe("shared Tutor learning-question diagnostics", () => { + it("reports granular deterministic violations", () => { + expect(inspectLearningQuestion({ question: "Welche? Zweite?" }).violations.map(({ code }) => code)) + .toContain("multiple-primary-questions"); + expect(inspectLearningQuestion({ question: "void setup() {} void loop() {}" }).violations.map(({ code }) => code)) + .toContain("complete-solution"); + expect(inspectLearningQuestion({ responseStyle: "normal", question: "Welche?", answerRating: 4 }).violations).toHaveLength(0); + expect(inspectLearningQuestion({ responseStyle: "philosophical", question: "Welche?", answerRating: 4 }).violations.map(({ code }) => code)) + .toContain("philosophical-answer-rating"); + expect(inspectLearningQuestion({ responseStyle: "normal", answerRating: "4" }).violations.map(({ code }) => code)) + .toContain("schema-invalid"); + }); + + it("keeps the existing validation throw behavior backed by the same records", () => { + const invalid = { question: "void setup() {} void loop() {}" }; + expect(() => validateLearningQuestion(invalid as never)).toThrowError(expect.objectContaining({ kind: "invalid-response" })); + expect(inspectLearningQuestion(invalid).violations.map(({ code }) => code)).toContain("complete-solution"); + }); +});