From 3f5c51bdbaa9646e90f4909c191e074589829a74 Mon Sep 17 00:00:00 2001 From: wyuc Date: Thu, 6 Aug 2026 05:28:56 -0400 Subject: [PATCH] refactor(pbl-v2): split the planner core from the loop; inject the LLM call path (#1069) Implements #1062. Pure planner-core extracted from the loop's file; both planners receive their LLM entry as a parameter (call sites pass the real callLLM unchanged); operations/ split into kernel vs runtime with a lint-enforced boundary; planner prompts pinned byte-identical via fixtures generated from the pre-refactor implementation. --- app/api/pbl/v2/open-task/route.ts | 2 +- app/api/pbl/v2/task/update/route.ts | 4 +- components/scene-renderers/pbl-renderer.tsx | 4 +- .../pbl/v2/apply-instructor-event.ts | 8 +- components/scene-renderers/pbl/v2/chat.tsx | 11 +- .../scene-renderers/pbl/v2/completion.tsx | 4 +- .../pbl/v2/eval-cards/milestone-card.tsx | 4 +- components/scene-renderers/pbl/v2/hero.tsx | 8 +- .../pbl/v2/scene-stage/scenario-stage.tsx | 2 +- components/scene-renderers/pbl/v2/sidebar.tsx | 2 +- .../scene-renderers/pbl/v2/submission.tsx | 15 +- .../pbl/v2/use-instructor-stream.ts | 4 +- eslint.config.mjs | 43 ++ eval/pbl-v2-planner/runner.ts | 8 +- lib/generation/scene-generator.ts | 13 +- lib/pbl/v2/agents/evaluator.ts | 10 +- lib/pbl/v2/agents/instructor.ts | 16 +- lib/pbl/v2/agents/planner-core.ts | 479 ++++++++++++++++ lib/pbl/v2/agents/planner-single-call.ts | 16 +- lib/pbl/v2/agents/planner.ts | 518 +----------------- lib/pbl/v2/agents/simulator.ts | 4 +- lib/pbl/v2/agents/tier-guidance.ts | 2 +- .../v2/operations/{ => kernel}/engagement.ts | 2 +- .../v2/operations/{ => kernel}/proficiency.ts | 2 +- .../v2/operations/{ => kernel}/progress.ts | 2 +- .../operations/{ => kernel}/runtime-events.ts | 2 +- .../{ => kernel}/task-completion.ts | 2 +- .../operations/{ => runtime}/advance-patch.ts | 8 +- .../{ => runtime}/completion-stats.ts | 2 +- .../{ => runtime}/dynamic-signals.ts | 10 +- .../operations/{ => runtime}/eval-prompts.ts | 8 +- .../{ => runtime}/eval-tail-parser.ts | 0 .../v2/operations/{ => runtime}/evaluation.ts | 13 +- .../{ => runtime}/file-validation.ts | 0 .../operations/{ => runtime}/quiz-snapshot.ts | 6 +- .../v2/operations/{ => runtime}/schemas.ts | 2 +- .../v2/operations/{ => runtime}/submission.ts | 4 +- .../{ => runtime}/workspace-launch.ts | 4 +- lib/pbl/v2/runtime/fold.ts | 2 +- lib/pbl/v2/types.ts | 6 +- .../classroom/pbl-fallback-hydration.test.ts | 2 +- tests/pbl/v2/advance-patch.test.ts | 11 +- tests/pbl/v2/completion-stats.test.ts | 2 +- tests/pbl/v2/drain.test.ts | 6 +- tests/pbl/v2/dynamic-signals.test.ts | 4 +- tests/pbl/v2/eval-prompts.test.ts | 8 +- tests/pbl/v2/eval-tail-parser.test.ts | 2 +- tests/pbl/v2/evaluator.test.ts | 2 +- tests/pbl/v2/file-validation.test.ts | 2 +- .../v2/fixtures/planner-system-ordinary.txt | 149 +++++ .../v2/fixtures/planner-system-scenario.txt | 195 +++++++ tests/pbl/v2/fold.test.ts | 4 +- tests/pbl/v2/hydration.test.ts | 12 +- tests/pbl/v2/instructor.test.ts | 2 +- tests/pbl/v2/learner-state.test.ts | 2 +- .../pbl/v2/microtask-engagement-cache.test.ts | 4 +- .../pbl/v2/planner-core-prompt-golden.test.ts | 105 ++++ tests/pbl/v2/planner-single-call.test.ts | 53 +- tests/pbl/v2/planner.test.ts | 17 +- tests/pbl/v2/proficiency.test.ts | 2 +- tests/pbl/v2/progress.test.ts | 8 +- tests/pbl/v2/quiz-snapshot.test.ts | 4 +- tests/pbl/v2/runtime-events.test.ts | 18 +- tests/pbl/v2/runtime-llm-entry.test.ts | 2 +- tests/pbl/v2/runtime-model-pin.test.ts | 2 +- tests/pbl/v2/simulator.test.ts | 5 +- tests/pbl/v2/submission.test.ts | 2 +- tests/pbl/v2/workspace-launch.test.ts | 2 +- 68 files changed, 1229 insertions(+), 650 deletions(-) create mode 100644 lib/pbl/v2/agents/planner-core.ts rename lib/pbl/v2/operations/{ => kernel}/engagement.ts (99%) rename lib/pbl/v2/operations/{ => kernel}/proficiency.ts (99%) rename lib/pbl/v2/operations/{ => kernel}/progress.ts (99%) rename lib/pbl/v2/operations/{ => kernel}/runtime-events.ts (99%) rename lib/pbl/v2/operations/{ => kernel}/task-completion.ts (99%) rename lib/pbl/v2/operations/{ => runtime}/advance-patch.ts (97%) rename lib/pbl/v2/operations/{ => runtime}/completion-stats.ts (99%) rename lib/pbl/v2/operations/{ => runtime}/dynamic-signals.ts (96%) rename lib/pbl/v2/operations/{ => runtime}/eval-prompts.ts (99%) rename lib/pbl/v2/operations/{ => runtime}/eval-tail-parser.ts (100%) rename lib/pbl/v2/operations/{ => runtime}/evaluation.ts (90%) rename lib/pbl/v2/operations/{ => runtime}/file-validation.ts (100%) rename lib/pbl/v2/operations/{ => runtime}/quiz-snapshot.ts (95%) rename lib/pbl/v2/operations/{ => runtime}/schemas.ts (99%) rename lib/pbl/v2/operations/{ => runtime}/submission.ts (98%) rename lib/pbl/v2/operations/{ => runtime}/workspace-launch.ts (92%) create mode 100644 tests/pbl/v2/fixtures/planner-system-ordinary.txt create mode 100644 tests/pbl/v2/fixtures/planner-system-scenario.txt create mode 100644 tests/pbl/v2/planner-core-prompt-golden.test.ts diff --git a/app/api/pbl/v2/open-task/route.ts b/app/api/pbl/v2/open-task/route.ts index 9cf603abc..ae8c7ae4d 100644 --- a/app/api/pbl/v2/open-task/route.ts +++ b/app/api/pbl/v2/open-task/route.ts @@ -21,7 +21,7 @@ import { resolveModelFromRequest } from '@/lib/server/resolve-model'; import { createSSEResponse } from '@/lib/pbl/v2/api/sse'; import { applyRequestLocaleToProject } from '@/lib/pbl/v2/api/locale'; import { runInstructorTurn } from '@/lib/pbl/v2/agents/instructor'; -import { applyQuizSignalsToProject } from '@/lib/pbl/v2/operations/quiz-snapshot'; +import { applyQuizSignalsToProject } from '@/lib/pbl/v2/operations/runtime/quiz-snapshot'; import type { PBLProjectV2, PriorQuizResult } from '@/lib/pbl/v2/types'; export const maxDuration = 300; diff --git a/app/api/pbl/v2/task/update/route.ts b/app/api/pbl/v2/task/update/route.ts index 20336e146..24ca53288 100644 --- a/app/api/pbl/v2/task/update/route.ts +++ b/app/api/pbl/v2/task/update/route.ts @@ -30,9 +30,9 @@ import { advanceMicrotask, completeRoleplayAct, appendTaskDividerMessage, -} from '@/lib/pbl/v2/operations/progress'; +} from '@/lib/pbl/v2/operations/kernel/progress'; import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; -import { currentPendingTaskCompletion } from '@/lib/pbl/v2/operations/task-completion'; +import { currentPendingTaskCompletion } from '@/lib/pbl/v2/operations/kernel/task-completion'; interface UpdateRequest { project: PBLProjectV2; diff --git a/components/scene-renderers/pbl-renderer.tsx b/components/scene-renderers/pbl-renderer.tsx index 95f656a19..26ed7c7e5 100644 --- a/components/scene-renderers/pbl-renderer.tsx +++ b/components/scene-renderers/pbl-renderer.tsx @@ -12,8 +12,8 @@ import { projectV2ToLegacyProjectConfig, upgradeLegacyPBLConfigToProjectV2, } from '@/lib/pbl/v2/compat'; -import { normalizeProjectRuntime } from '@/lib/pbl/v2/operations/progress'; -import { transitionProjectUiPhase } from '@/lib/pbl/v2/operations/runtime-events'; +import { normalizeProjectRuntime } from '@/lib/pbl/v2/operations/kernel/progress'; +import { transitionProjectUiPhase } from '@/lib/pbl/v2/operations/kernel/runtime-events'; import { useStageStore } from '@/lib/store/stage'; import { cn } from '@/lib/utils/cn'; import { PBLRoleSelection } from './pbl/role-selection'; diff --git a/components/scene-renderers/pbl/v2/apply-instructor-event.ts b/components/scene-renderers/pbl/v2/apply-instructor-event.ts index 4eacebf42..eece6c8c9 100644 --- a/components/scene-renderers/pbl/v2/apply-instructor-event.ts +++ b/components/scene-renderers/pbl/v2/apply-instructor-event.ts @@ -11,15 +11,15 @@ import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; import type { PBLSSEEvent } from '@/lib/pbl/v2/api/sse'; -import { applyAdvanceProjectPatch } from '@/lib/pbl/v2/operations/advance-patch'; -import { capEngagementEvents } from '@/lib/pbl/v2/operations/engagement'; -import { PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/progress'; +import { applyAdvanceProjectPatch } from '@/lib/pbl/v2/operations/runtime/advance-patch'; +import { capEngagementEvents } from '@/lib/pbl/v2/operations/kernel/engagement'; +import { PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/kernel/progress'; import { appendProficiencyUpdatedRuntimeEvent, appendRuntimeEvent, milestoneIdForMicrotask, mintRuntimeEventId, -} from '@/lib/pbl/v2/operations/runtime-events'; +} from '@/lib/pbl/v2/operations/kernel/runtime-events'; import { isStandaloneDividerMessage, stripEmbeddedDividerMarkers } from './protocol-markers'; import type { PBLChatMessage, PBLRuntimeActorType } from '@/lib/pbl/v2/types'; diff --git a/components/scene-renderers/pbl/v2/chat.tsx b/components/scene-renderers/pbl/v2/chat.tsx index 79c8b2e4d..82bd2b807 100644 --- a/components/scene-renderers/pbl/v2/chat.tsx +++ b/components/scene-renderers/pbl/v2/chat.tsx @@ -33,15 +33,18 @@ import type { PBLProjectV2, PBLScenarioCharacter, } from '@/lib/pbl/v2/types'; -import { normalizeProjectRuntime, PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/progress'; +import { + normalizeProjectRuntime, + PBL_SIMULATOR_AGENT_ID, +} from '@/lib/pbl/v2/operations/kernel/progress'; import { appendRuntimeEvent, milestoneIdForMicrotask, mintRuntimeEventId, transitionProjectUiPhase, -} from '@/lib/pbl/v2/operations/runtime-events'; -import { stripEvaluationTail } from '@/lib/pbl/v2/operations/eval-tail-parser'; -import { isTaskCompletionReadyMessageContent } from '@/lib/pbl/v2/operations/task-completion'; +} from '@/lib/pbl/v2/operations/kernel/runtime-events'; +import { stripEvaluationTail } from '@/lib/pbl/v2/operations/runtime/eval-tail-parser'; +import { isTaskCompletionReadyMessageContent } from '@/lib/pbl/v2/operations/kernel/task-completion'; import { cn } from '@/lib/utils/cn'; import { useInstructorStream, type StreamDisplayState } from './use-instructor-stream'; import { instructorIntroText } from './instructor-intro'; diff --git a/components/scene-renderers/pbl/v2/completion.tsx b/components/scene-renderers/pbl/v2/completion.tsx index 22f1ad41e..4892c4dbd 100644 --- a/components/scene-renderers/pbl/v2/completion.tsx +++ b/components/scene-renderers/pbl/v2/completion.tsx @@ -37,7 +37,7 @@ import { motion } from 'motion/react'; import type { ReactNode } from 'react'; import type { PBLEvaluation, PBLProjectV2, PBLScenarioActGoals } from '@/lib/pbl/v2/types'; import { useI18n } from '@/lib/hooks/use-i18n'; -import { stripEvaluationTail } from '@/lib/pbl/v2/operations/eval-tail-parser'; +import { stripEvaluationTail } from '@/lib/pbl/v2/operations/runtime/eval-tail-parser'; import { computeCompletionStats, type CompletionStats, @@ -45,7 +45,7 @@ import { type ScenarioCompletionStats, type ScenarioActGoalScaffold, type StageDetail, -} from '@/lib/pbl/v2/operations/completion-stats'; +} from '@/lib/pbl/v2/operations/runtime/completion-stats'; interface Props { readonly project: PBLProjectV2; diff --git a/components/scene-renderers/pbl/v2/eval-cards/milestone-card.tsx b/components/scene-renderers/pbl/v2/eval-cards/milestone-card.tsx index ac39c1bfe..61bf79ccb 100644 --- a/components/scene-renderers/pbl/v2/eval-cards/milestone-card.tsx +++ b/components/scene-renderers/pbl/v2/eval-cards/milestone-card.tsx @@ -17,7 +17,7 @@ * The button is the ONLY way to cross into the next stage — * while `pendingHandover.consumed === false` the next milestone * stays LOCKED in the store (gate enforced by continueAfter - * Handover in operations/progress.ts). + * Handover in operations/kernel/progress.ts). * * Visual scope: * - full-width (not a bubble); breaks out of the chat's max-width @@ -39,7 +39,7 @@ import { useI18n } from '@/lib/hooks/use-i18n'; import { sanitizeMilestoneEvaluationFeedback, stripEvaluationTail, -} from '@/lib/pbl/v2/operations/eval-tail-parser'; +} from '@/lib/pbl/v2/operations/runtime/eval-tail-parser'; import { MarkdownText } from '../markdown-text'; import { StarRating } from './star-rating'; import type { PBLEvaluation, PBLHandover } from '@/lib/pbl/v2/types'; diff --git a/components/scene-renderers/pbl/v2/hero.tsx b/components/scene-renderers/pbl/v2/hero.tsx index b6779cc04..5ce326216 100644 --- a/components/scene-renderers/pbl/v2/hero.tsx +++ b/components/scene-renderers/pbl/v2/hero.tsx @@ -29,14 +29,14 @@ import { import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; import { useStageStore } from '@/lib/store/stage'; -import { buildQuizSnapshot } from '@/lib/pbl/v2/operations/quiz-snapshot'; -import { hasStartedProject, resetProjectProgress } from '@/lib/pbl/v2/operations/progress'; -import { transitionProjectUiPhase } from '@/lib/pbl/v2/operations/runtime-events'; +import { buildQuizSnapshot } from '@/lib/pbl/v2/operations/runtime/quiz-snapshot'; +import { hasStartedProject, resetProjectProgress } from '@/lib/pbl/v2/operations/kernel/progress'; +import { transitionProjectUiPhase } from '@/lib/pbl/v2/operations/kernel/runtime-events'; import { invalidatePendingWorkspaceLaunch, isCurrentWorkspaceLaunch, prepareCurrentWorkspaceLaunchProject, -} from '@/lib/pbl/v2/operations/workspace-launch'; +} from '@/lib/pbl/v2/operations/runtime/workspace-launch'; import { useI18n } from '@/lib/hooks/use-i18n'; import { AlertDialog, diff --git a/components/scene-renderers/pbl/v2/scene-stage/scenario-stage.tsx b/components/scene-renderers/pbl/v2/scene-stage/scenario-stage.tsx index 7930b30b5..f8c84ebdb 100644 --- a/components/scene-renderers/pbl/v2/scene-stage/scenario-stage.tsx +++ b/components/scene-renderers/pbl/v2/scene-stage/scenario-stage.tsx @@ -31,7 +31,7 @@ import { ChevronDown, ChevronUp, Hand } from 'lucide-react'; import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; import { useI18n } from '@/lib/hooks/use-i18n'; import { cn } from '@/lib/utils/cn'; -import { PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/progress'; +import { PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/kernel/progress'; import { SceneBackdrop } from './scene-backdrop'; import { sanitizeSceneVisual } from './scene-types'; diff --git a/components/scene-renderers/pbl/v2/sidebar.tsx b/components/scene-renderers/pbl/v2/sidebar.tsx index d210ef2ac..6c85810e6 100644 --- a/components/scene-renderers/pbl/v2/sidebar.tsx +++ b/components/scene-renderers/pbl/v2/sidebar.tsx @@ -32,7 +32,7 @@ import { Drama, } from 'lucide-react'; import type { PBLProjectV2, PBLMicrotask, PBLMilestone } from '@/lib/pbl/v2/types'; -import { PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/progress'; +import { PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/kernel/progress'; import { cn } from '@/lib/utils/cn'; import { useI18n } from '@/lib/hooks/use-i18n'; diff --git a/components/scene-renderers/pbl/v2/submission.tsx b/components/scene-renderers/pbl/v2/submission.tsx index c9312653a..b113a91e0 100644 --- a/components/scene-renderers/pbl/v2/submission.tsx +++ b/components/scene-renderers/pbl/v2/submission.tsx @@ -41,14 +41,17 @@ import { X, } from 'lucide-react'; -import { addSubmission, listSubmissionsForMicrotask } from '@/lib/pbl/v2/operations/submission'; +import { + addSubmission, + listSubmissionsForMicrotask, +} from '@/lib/pbl/v2/operations/runtime/submission'; import { findModelById } from '@/lib/ai/model-aliases'; import { TEXT_PDF_IMAGE_ACCEPT, isImageFile, isPdfFile, isValidTextFile, -} from '@/lib/pbl/v2/operations/file-validation'; +} from '@/lib/pbl/v2/operations/runtime/file-validation'; import { uploadBlobToStorage } from '@/lib/storage/client'; import type { PBLChatMessage, @@ -60,20 +63,20 @@ import type { PBLSSEEvent } from '@/lib/pbl/v2/api/sse'; import { applyInstructorEvent } from './apply-instructor-event'; import { getCurrentModelConfig } from '@/lib/utils/model-config'; import { useSettingsStore } from '@/lib/store/settings'; -import { normalizeProjectRuntime } from '@/lib/pbl/v2/operations/progress'; +import { normalizeProjectRuntime } from '@/lib/pbl/v2/operations/kernel/progress'; import { appendRuntimeEvent, milestoneIdForMicrotask, mintRuntimeEventId, -} from '@/lib/pbl/v2/operations/runtime-events'; -import { trackSubmissionScore } from '@/lib/pbl/v2/operations/dynamic-signals'; +} from '@/lib/pbl/v2/operations/kernel/runtime-events'; +import { trackSubmissionScore } from '@/lib/pbl/v2/operations/runtime/dynamic-signals'; import { appendTaskCompletionReadyMessage, recordPendingTaskCompletionEvidence, setPendingTaskCompletion, TASK_EVAL_PASS_SCORE, taskEvaluationCanComplete, -} from '@/lib/pbl/v2/operations/task-completion'; +} from '@/lib/pbl/v2/operations/kernel/task-completion'; import { useI18n } from '@/lib/hooks/use-i18n'; import i18n from '@/lib/i18n/config'; import { diff --git a/components/scene-renderers/pbl/v2/use-instructor-stream.ts b/components/scene-renderers/pbl/v2/use-instructor-stream.ts index 95d92b95c..9d7f9324d 100644 --- a/components/scene-renderers/pbl/v2/use-instructor-stream.ts +++ b/components/scene-renderers/pbl/v2/use-instructor-stream.ts @@ -45,8 +45,8 @@ import { useCallback, useRef, useState } from 'react'; import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; import type { PBLSSEEvent } from '@/lib/pbl/v2/api/sse'; -import { trackSubmissionScore } from '@/lib/pbl/v2/operations/dynamic-signals'; -import { normalizeProjectRuntime } from '@/lib/pbl/v2/operations/progress'; +import { trackSubmissionScore } from '@/lib/pbl/v2/operations/runtime/dynamic-signals'; +import { normalizeProjectRuntime } from '@/lib/pbl/v2/operations/kernel/progress'; import { getCurrentModelConfig } from '@/lib/utils/model-config'; import { createLogger } from '@/lib/logger'; import { applyInstructorEvent } from './apply-instructor-event'; diff --git a/eslint.config.mjs b/eslint.config.mjs index fef43c89f..19dcfdf37 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -487,6 +487,48 @@ const eslintConfig = defineConfig([ ], }, }, + // PBL v2 project-definition boundary: kernel operations encode the shared + // project invariants used by planners, runtime, and UI. Runtime-only + // operations may depend on the kernel, but the kernel must never reach back + // into operations/runtime. Match the module string in every literal form so + // static imports, re-exports, dynamic imports, and require-like calls cannot + // quietly invert the boundary. + { + files: ['lib/pbl/v2/operations/kernel/**/*.{ts,tsx,js,jsx,mjs,cjs}'], + rules: { + 'no-restricted-imports': [ + 'error', + { + patterns: [ + { + group: [ + '../runtime', + '../runtime/*', + '@/lib/pbl/v2/operations/runtime', + '@/lib/pbl/v2/operations/runtime/*', + ], + message: + 'PBL v2 kernel operations must not import operations/runtime. Runtime operations may depend on the project-definition kernel, never the reverse.', + }, + ], + }, + ], + 'no-restricted-syntax': [ + 'error', + ...AI_SDK_DYNAMIC_IMPORT_BAN, + { + selector: 'Literal[value=/\\/runtime(?:\\/|$)/]', + message: + 'PBL v2 kernel operations must not import operations/runtime. Runtime operations may depend on the project-definition kernel, never the reverse.', + }, + { + selector: 'TemplateElement[value.cooked=/\\/runtime(?:\\/|$)/]', + message: + 'PBL v2 kernel operations must not import operations/runtime. Runtime operations may depend on the project-definition kernel, never the reverse.', + }, + ], + }, + }, // Single LLM entry point (machine-enforced): server-side model calls go through // `callLLM` / `streamLLM` in lib/ai/llm.ts. That wrapper is where usage // accounting (`recordUsage`), the `LLM_THINKING_DISABLED` kill switch, and @@ -571,6 +613,7 @@ const eslintConfig = defineConfig([ // Blocks above that configure no-restricted-syntax for their own boundary. 'lib/choreography/**', 'lib/video-export/**', + 'lib/pbl/v2/operations/kernel/**', 'packages/@openmaic/renderer/**', 'packages/@openmaic/storage/**', 'packages/@openmaic/generation/**', diff --git a/eval/pbl-v2-planner/runner.ts b/eval/pbl-v2-planner/runner.ts index c554bab84..d8677b095 100644 --- a/eval/pbl-v2-planner/runner.ts +++ b/eval/pbl-v2-planner/runner.ts @@ -42,8 +42,10 @@ import { createAnthropic } from '@ai-sdk/anthropic'; import { createOpenAI } from '@ai-sdk/openai'; import { generateText, type LanguageModel } from 'ai'; -import { generatePBLV2Project, PlannerV2Error } from '@/lib/pbl/v2/agents/planner'; +import { callLLM } from '@/lib/ai/llm'; +import { generatePBLV2Project } from '@/lib/pbl/v2/agents/planner'; import { generatePBLV2ProjectSingleCall } from '@/lib/pbl/v2/agents/planner-single-call'; +import { PlannerV2Error } from '@/lib/pbl/v2/agents/planner-core'; import { parseJsonResponse } from '@/lib/generation/json-repair'; import { buildCompareHtml } from './compare-html'; import type { PBLPlannerV2Input, PBLProjectV2 } from '@/lib/pbl/v2/types'; @@ -455,8 +457,8 @@ function runVariant( thinkingConfig?: ThinkingConfig, ): Promise { return variant === 'single-call' - ? generatePBLV2ProjectSingleCall(input, model, undefined, thinkingConfig) - : generatePBLV2Project(input, model, undefined, thinkingConfig); + ? generatePBLV2ProjectSingleCall(input, model, callLLM, undefined, thinkingConfig) + : generatePBLV2Project(input, model, callLLM, undefined, thinkingConfig); } async function runOne( diff --git a/lib/generation/scene-generator.ts b/lib/generation/scene-generator.ts index 545812b83..68615792a 100644 --- a/lib/generation/scene-generator.ts +++ b/lib/generation/scene-generator.ts @@ -25,8 +25,10 @@ import type { PromptId } from '@/lib/prompts/types'; import type { LanguageModel } from 'ai'; import { createStageAPI } from '@/lib/api/stage-api'; import { generatePBLContent } from '@/lib/pbl/generate-pbl'; -import { generatePBLV2Project, PlannerV2Error } from '@/lib/pbl/v2/agents/planner'; +import { callLLM } from '@/lib/ai/llm'; +import { generatePBLV2Project } from '@/lib/pbl/v2/agents/planner'; import { generatePBLV2ProjectSingleCall } from '@/lib/pbl/v2/agents/planner-single-call'; +import { PlannerV2Error } from '@/lib/pbl/v2/agents/planner-core'; import { projectV2ToLegacyProjectConfig } from '@/lib/pbl/v2/compat'; import type { PBLPlannerV2Input, PBLProjectV2 } from '@/lib/pbl/v2/types'; import { buildPrompt, PROMPT_IDS } from '@/lib/prompts'; @@ -948,6 +950,7 @@ async function generatePBLSceneContent( generatePBLV2ProjectSingleCall( plannerInput, languageModel, + callLLM, { onProgress }, thinkingConfig, ), @@ -955,7 +958,13 @@ async function generatePBLSceneContent( { label: 'loop', run: () => - generatePBLV2Project(plannerInput, languageModel, { onProgress }, thinkingConfig), + generatePBLV2Project( + plannerInput, + languageModel, + callLLM, + { onProgress }, + thinkingConfig, + ), }, ]; diff --git a/lib/pbl/v2/agents/evaluator.ts b/lib/pbl/v2/agents/evaluator.ts index 4891f3f7c..3e052717b 100644 --- a/lib/pbl/v2/agents/evaluator.ts +++ b/lib/pbl/v2/agents/evaluator.ts @@ -56,13 +56,13 @@ import type { PBLProjectV2, } from '../types'; import type { PBLSSEEvent } from '../api/sse'; -import { addEvaluation } from '../operations/evaluation'; -import { latestSubmissionForMicrotask } from '../operations/submission'; +import { addEvaluation } from '../operations/runtime/evaluation'; +import { latestSubmissionForMicrotask } from '../operations/runtime/submission'; import { buildFinalEvalPrompt, buildMilestoneEvalPrompt, buildTaskEvalPrompt, -} from '../operations/eval-prompts'; +} from '../operations/runtime/eval-prompts'; import { normalizeOptionalString, normalizeScore, @@ -71,8 +71,8 @@ import { parseEvaluationTail, sanitizeMilestoneEvaluationFeedback, stripEvaluationTail, -} from '../operations/eval-tail-parser'; -import { normalizeActGoals } from '../operations/completion-stats'; +} from '../operations/runtime/eval-tail-parser'; +import { normalizeActGoals } from '../operations/runtime/completion-stats'; const log = createLogger('PBL v2 Evaluator'); diff --git a/lib/pbl/v2/agents/instructor.ts b/lib/pbl/v2/agents/instructor.ts index 4ad47286e..6c661b6b2 100644 --- a/lib/pbl/v2/agents/instructor.ts +++ b/lib/pbl/v2/agents/instructor.ts @@ -31,26 +31,26 @@ import type { PBLProficiency, } from '../types'; import type { PBLSSEEvent } from '../api/sse'; -import { RecordObservationArgs, AdjustDifficultyArgs } from '../operations/schemas'; +import { RecordObservationArgs, AdjustDifficultyArgs } from '../operations/runtime/schemas'; import { recordEvent, microtaskEngagement, milestoneSynthesisSatisfied, -} from '../operations/engagement'; +} from '../operations/kernel/engagement'; import { currentMicrotask, advanceMicrotask as advanceMicrotaskOp, normalizeProjectRuntime, -} from '../operations/progress'; -import { summarizeLatestSubmissionForMicrotask } from '../operations/submission'; +} from '../operations/kernel/progress'; +import { summarizeLatestSubmissionForMicrotask } from '../operations/runtime/submission'; import { applyProficiencyDirective, tickTurnOnProject, trackObservation, -} from '../operations/dynamic-signals'; -import { DEFAULT_TIER, proficiencyDirectiveFromTarget } from '../operations/proficiency'; -import { buildAdvanceProjectPatch } from '../operations/advance-patch'; -import { formatScenarioTranscript } from '../operations/eval-prompts'; +} from '../operations/runtime/dynamic-signals'; +import { DEFAULT_TIER, proficiencyDirectiveFromTarget } from '../operations/kernel/proficiency'; +import { buildAdvanceProjectPatch } from '../operations/runtime/advance-patch'; +import { formatScenarioTranscript } from '../operations/runtime/eval-prompts'; const log = createLogger('PBL v2 Instructor'); diff --git a/lib/pbl/v2/agents/planner-core.ts b/lib/pbl/v2/agents/planner-core.ts new file mode 100644 index 000000000..f38c46821 --- /dev/null +++ b/lib/pbl/v2/agents/planner-core.ts @@ -0,0 +1,479 @@ +/** + * PBL v2 — shared planner core. + * + * Pure project construction, prompt assembly, completion validation, and + * deterministic normalization shared by both planner call strategies. This + * module deliberately has no AI SDK, schema-library, or app LLM entry import. + */ + +import { createLogger } from '@/lib/logger'; +import { loadPBLV2Prompt } from '../prompts/loader'; +import { computeInitialAssessment, reseatAssessmentTier } from '../operations/kernel/proficiency'; + +import type { PBLProjectV2, PBLPlannerV2Input, PBLProficiency, PBLRole } from '../types'; + +const log = createLogger('PBL v2 Planner'); + +/** SCENARIO ONLY. Packaged-format version stamped on scenario projects + * (`project.schemaVersion`). Absent on ordinary projects (baseline). + * Bump when the packaged scenario format changes so loaders can + * migrate. */ +export const SCENARIO_SCHEMA_VERSION = 1; + +// --------------------------------------------------------------------------- +// Callbacks (so the caller can stream progress to a UI later) +// --------------------------------------------------------------------------- + +export interface PlannerV2Callbacks { + /** Fired on each successful tool call. Used by the future Generating + * page to show "Adding milestone: X" etc. */ + onProgress?: (event: PlannerV2ProgressEvent) => void; +} + +export type PlannerV2ProgressEvent = + | { kind: 'project_info'; title: string } + | { kind: 'role'; roleType: PBLRole['type']; name: string } + | { kind: 'milestone'; title: string; index: number } + | { kind: 'microtask'; milestoneTitle: string; title: string; index: number } + | { kind: 'complete'; milestoneCount: number; microtaskCount: number }; + +// --------------------------------------------------------------------------- +// Errors +// --------------------------------------------------------------------------- + +/** Thrown when the Planner finishes but the result is unusable (no + * Instructor, no milestones, etc.). The caller should fall back to v1 + * or to a slide so the student is never stranded on an empty PBL. */ +export class PlannerV2Error extends Error { + constructor( + message: string, + public readonly partial: PBLProjectV2, + ) { + super(message); + this.name = 'PlannerV2Error'; + } +} + +// --------------------------------------------------------------------------- +// Empty / starter shape +// --------------------------------------------------------------------------- + +export function emptyProject(input: PBLPlannerV2Input): PBLProjectV2 { + const now = new Date().toISOString(); + + // Compute the planner-time initial proficiency assessment from + // static signals (outline keywords + prior-scene difficulty + user + // bio). Quiz accuracy is not yet available — that's folded in at + // Hero entry by the pre-play recalibration path. + // + // See `lib/pbl/v2/operations/kernel/proficiency.ts` for the full algorithm + // and the calibration table. The Planner LLM does NOT decide this + // value: it consumes `assessment.tier` as a directive when + // dimensioning microtasks. + const assessment = computeInitialAssessment({ + outline: input.outline, + priorScenes: input.courseContext.allOutlines, + userBio: input.user?.bio, + userRequirement: input.user?.requirement, + priorQuizResults: input.priorQuizResults, + source: 'planner', + }); + const proficiency: PBLProficiency = assessment.tier; + // `languageDirective` is the SINGLE source of truth for content language — + // a free natural-language directive from the outline stage (e.g. "Reply in + // Simplified Chinese" / "中文为主,英文技术术语保留原文"). The Planner feeds it + // straight to the system prompt; there is no content-based locale guessing. + // + // `language` is only the BCP-47 locale the RUNTIME uses for deterministic + // platform strings (synthetic openers, divider labels). It is seeded from + // the authoritative UI locale (`targetLanguage`) when known and otherwise + // left blank for the Hero locale-sync (hero.tsx) to fill on entry — it is + // NEVER inferred from content here. + const languageDirective = input.courseContext.languageDirective?.trim(); + const language = input.targetLanguage?.trim() ?? ''; + + log.info( + `Planner v2 initial assessment: tier=${assessment.tier} score=${assessment.score.toFixed( + 2, + )} confidence=${assessment.confidence.toFixed(2)} signals=${assessment.signals.length} language=${language}`, + ); + + return { + uiPhase: 'hero', + title: '', + description: '', + learningObjective: '', + gains: [], + proficiency, + proficiencyAssessment: assessment, + language, + languageDirective: languageDirective || undefined, + tags: [], + status: 'designing', + roles: [], + milestones: [], + submissions: [], + evaluations: [], + threads: [], + engagementEvents: [], + createdAt: now, + updatedAt: now, + }; +} + +// --------------------------------------------------------------------------- +// Prompt assembly +// --------------------------------------------------------------------------- + +export function formatCourseContext(input: PBLPlannerV2Input): string { + const lines: string[] = []; + for (const o of input.courseContext.allOutlines) { + const marker = o.id === input.outline.id ? ' ← this PBL scene' : ''; + lines.push(`- [${o.order}] ${o.type.toUpperCase()}: ${o.title}${marker}`); + if (o.description) { + lines.push(` ${o.description}`); + } + } + return lines.join('\n'); +} + +export async function buildPlannerSystemPrompt( + input: PBLPlannerV2Input, + proficiency: PBLProficiency, + contentLanguage: string, + scenarioRoleplay: boolean, + promptName: string = 'planner-system', +): Promise { + const pblConfig = input.outline.pblConfig!; + + // The adaptive engine decided the tier in `emptyProject`; pass it + // through as a directive so the LLM dimensions microtasks + // accordingly. The Planner is expected to mirror this value when it + // calls `set_project_info`; if it picks a different tier the tool + // accepts the value (it's a hint, not a hard contract) but the + // engine logs the divergence. + // + // `contentLanguage` is resolved by the caller from + // `project.languageDirective || project.language` (single source). It may be + // a BCP-47 locale or a nuanced directive like "中文为主,英文技术术语保留原文". + return loadPBLV2Prompt(promptName, { + projectTopic: pblConfig.projectTopic, + projectDescription: pblConfig.projectDescription, + targetSkills: (pblConfig.targetSkills ?? []).join(', '), + milestoneCount: pblConfig.issueCount ?? 3, + proficiency: proficiency === '' ? 'intermediate' : proficiency, + language: contentLanguage, + courseContext: formatCourseContext(input), + languageDirective: input.courseContext.languageDirective, + // Optional free-form scenario brief from the outline stage. Empty for + // ordinary projects / non-scenario prompts (the slot collapses). + scenarioBrief: input.outline.pblConfig?.scenarioBrief ?? '', + // SCENARIO ONLY. Empty string for ordinary PBL projects → the + // `{{scenarioDesign}}` slot collapses to nothing and the prompt is + // byte-identical to before. The slot must always be provided (the + // interpolator leaves unknown placeholders literal). + scenarioDesign: buildScenarioDesignBlock(pblConfig, scenarioRoleplay), + }); +} + +/** SCENARIO ONLY. Build the role-play scenario-design instruction block + * injected into the Planner system prompt. Returns '' for ordinary PBL + * projects (so the prompt is byte-identical to before). When the + * outline opted into `scenarioRoleplay`, it instructs the Planner to + * fully author the scenario AND lay it out as the fixed three-stage + * skeleton: prep → roleplay(×1..N) → wrapup. */ +export function buildScenarioDesignBlock( + pblConfig: NonNullable, + scenarioRoleplay: boolean, +): string { + if (!scenarioRoleplay) return ''; + const brief = pblConfig.scenarioBrief?.trim(); + return [ + '## SCENARIO MODE — role-play scenario (this project only)', + '', + 'This PBL is a **role-play scenario**: the learner will step into a concrete situation and interact in-character with character(s) played by a separate Simulator agent. You author the WHOLE scenario now (it is frozen into the package); the runtime only produces the live dialogue. Two rules above all: (a) the premise is **given and concrete**, introduced to the learner by the Instructor — the learner must NEVER be asked to guess it; (b) every task must serve the real learning goal (how to do the thing well), not meta-guessing.', + '', + brief ? `Scenario brief from the platform: ${brief}\n` : '', + '### Step A — fully author the scenario with `set_scenario(...)`', + 'Call **exactly once, right after `set_project_info` and before any `add_milestone`**:', + '- `setting`: the concrete overall premise / what is going on (in the project language).', + '- `goal` (optional): what the learner is practising.', + "- `rules` (optional but REQUIRED whenever the scenario has any defined rule-set — games / interviews / debates / structured negotiations / etc.): write the CONCRETE rules a newcomer needs to actually take part, specific enough that the Instructor can teach them verbatim in prep. Not a vague label — include the real mechanics (e.g. a card game: hand ranking, betting rounds, blinds, what terms like Pot Odds / Fold / Call / Raise / Check mean; a debate: the motion, each side's stance, the speaking format; an interview: the rounds and what each assesses). Omit ONLY for free scenarios with no special rules (e.g. comforting a friend).", + '- `learnerRole` (optional): the learner\'s OWN role/position (e.g. "you are their close friend" / "you are the 5th player, on the button").', + '- `characters`: **EXACTLY ONE character** — this version plays a single counterpart throughout (the runtime only ever voices one). It needs `name`, `persona` (stable identity / relationship / personality / speaking style), **`situation`** (their CONCRETE current circumstance the learner faces — e.g. "just broke up last week, low mood, says they\'re fine but aren\'t"). `situation` is shown to the learner up front (prep intro + the always-visible scenario briefing), so it must hold ONLY what the learner can see/know at the start — keep any fact a later beat is meant to make them uncover OUT of it (see the No-spoilers rule below). Plus strongly-recommended `boundaries` (hard safety rails), and optional `openingLine`.', + '', + '### Step B — lay out the FIXED three-stage skeleton (milestones in this exact order)', + '1. **Prep stage** — `add_milestone({ ..., scenarioStage: "prep" })` as the FIRST milestone. Its `briefing` is the Instructor intro that **introduces the concrete premise to the learner**: the situation, each character\'s `situation`, what the learner is there to do, plus `rules` / `learnerRole` when present. The intro MUST match the roleplay stages you design next. Give prep **exactly ONE light microtask** (e.g. "了解背景,准备开始" / "Understand the setup, ready to begin") — NO assessment; the learner just confirms and advances. **Do NOT set `coreConcept`.**', + '2. **Roleplay stage(s)** — one or MORE `add_milestone({ ..., scenarioStage: "roleplay" })` in the middle (split a long scenario into several roleplay stages by round/phase to avoid one giant stage). Each roleplay milestone\'s `briefing` brings the learner into the scene. **Design the beats as a DRAMATIC ARC, not a flat checklist**: an opening hook → rising stakes/complication → a turning point or decision → a resolution. Each beat should be a MEANINGFUL decision/action unit (something the learner can actually DO), never empty filler. For **each microtask (beat)** under a roleplay milestone, provide:', + ' - `description`: the CONCRETE situation of this beat as the SYSTEM narrator states it to the learner — positions / cards / what just happened / whose turn (e.g. "你在 Button 位拿到 A♠ J♦;前面都 Fold,老周在 Cutoff 加注到 6 个筹码;轮到你决定 preflop"). The character NEVER states these facts — the system does; the character only reacts. Keep it factual scene-setting, not coaching.', + ' - `successWhen` (REQUIRED for every roleplay beat): the CONCRETE, OBSERVABLE in-scene action the learner must SAY or DO for this beat to count as done — the scenario\'s "deliverable" (e.g. "做出 preflop 决定:跟注、加注或弃牌" / "对对方说出的感受做出共情回应,并问一个跟进问题"). State it in plain SCENE terms (what they do in the fiction), NOT as a teaching goal. This is exactly what the advance detector watches, so a crisp `successWhen` is what stops off-topic / small-talk turns from advancing the scene. Make it a real decision/action, not "they chatted a bit".', + ' - `completionCriteria` (optional, legacy): a teaching-side note on what this beat is about; `successWhen` is preferred and takes precedence for advancing.', + ' - `characterObjective` (recommended): what the character PRIVATELY wants — and privately KNOWS — this beat: their in-scene drive (e.g. "试探你是否在虚张声势" / "想确认你是否真的在乎"), plus any fact the learner is meant to UNCOVER this beat (the hidden cause / secret / backstory the character only reveals when probed — e.g. "你昨天在空调房待了很久才着凉,但只有被仔细询问才说出来"). It makes the character pursue a goal and hold its secrets in character; it is private to the character — NEVER narrated, shown in the briefing, evaluated, or coached.', + ' - `skillFocus` (recommended): the single skill this beat practises (e.g. "底池赔率判断" / "积极倾听"). Surfaced to the learner (current-task panel + end-of-project per-act review); never spoken by the character.', + ' - The scene is FREE-FIRST: the learner always speaks/types their OWN response to the character, which is how a real interaction is practised. (Some beats may instead ask the learner to hand in a real artefact, e.g. "write them a letter".)', + ' - `narration` (optional): a short neutral scene-setting line the SYSTEM reads when this beat opens (e.g. "你们走进了一家安静的咖啡厅"). NEVER spoken by a character or the Instructor. All scene/state facts come from the system (narration + description), never from the character\'s mouth.', + ' - `hints` (recommended for roleplay beats): 1–2 SHORT, learner-facing coaching tips for THIS beat — what skill to focus on or how to handle it well (e.g. "先共情、再问问题,别急着给建议" / "注意你的位置和底池赔率,再决定下注"). They appear in the "hints" card of the learner\'s current-task side panel, are NEVER spoken by the character, and are the learner\'s in-the-moment guidance. Keep them concrete to this beat, not generic.', + ' - **Do NOT set `coreConcept`** on roleplay milestones.', + '3. **Wrapup stage** — `add_milestone({ ..., scenarioStage: "wrapup" })` as the LAST milestone. Its `debrief` holds the Instructor\'s light, encouraging feedback points (highlights / one thing to improve); the detailed report lives on the completion page. Give wrapup **exactly ONE light microtask** (e.g. "听取反馈,收尾" / "Hear the feedback, wrap up"). **Do NOT set `coreConcept`.**', + '', + '### Rules for scenario design', + '- The premise (situation / rules / positions) is GIVEN and introduced in prep — **never make a task that asks the learner to guess/invent it**.', + "- **No spoilers — never give away in learner-VISIBLE text what a beat is designed to make the learner discover.** The premise the learner can see up front — `setting`, each character's `situation` / `persona`, the prep `briefing`, and each beat's `description` / `narration` (and the always-visible scenario briefing built from these) — must contain ONLY what the learner already knows or can plainly observe at the outset. If any roleplay beat's `successWhen` requires the learner to UNCOVER something through the interaction (a hidden cause, a motive, a secret, the diagnosis, a backstory fact), that information MUST NOT appear in any learner-visible field. Put it ONLY in that beat's private `characterObjective`, where the character holds it and reveals it solely when the learner actually probes for it — never up front. E.g. a \"find out why\" beat: the real cause lives in `characterObjective`; `situation` states only the visible symptoms / where the character is right now.", + '- All scenario text (`setting`/`persona`/`situation`/`briefing`/narration/options…) follows the same content-language policy as Hard rule 1.', + '- Scene beats should feel like a real interaction unfolding, not a checklist; 2-4 beats per roleplay stage is plenty.', + '- **The roleplay character is a pure in-world participant, NEVER a coach.** When you write `persona` / `situation` / `openingLine` (and any character-facing text), the character must have its OWN motives and react like a real person in the scene. It must NEVER: ask the learner to explain/justify their reasoning ("说说你为什么这么选" / "一句话给我理由"), evaluate or grade the learner\'s moves ("这步打得对"), give strategy/meta hints ("想想我的范围里哪些牌会付钱"), or tell the learner it\'s their turn / what to decide. That is all out-of-scene/teaching content and it does NOT belong in the character\'s mouth.', + '- **Out-of-scene content has its own channels — never the character:** (a) the "this is a training table / I\'ll test you" framing and any rule teaching belong to the **prep Instructor** (`briefing`); (b) "it\'s your turn to act / a decision point has arrived" belongs to the **system `narration`** of that beat, stated neutrally; (c) strategy / what-to-watch-for hints belong to the microtask **`hints`** (side panel). Route each of these to its channel; the character only ever lives the scene.', + '', + '', + '### Step C — author ONE project-wide scene visual with `set_scene_visual(...)`', + 'After ALL roleplay milestones/beats exist, call `set_scene_visual` exactly once. Read back over EVERY roleplay stage/task you just wrote and distil the ONE shared place/atmosphere they all happen in, then describe it: a `caption` (a short phrase in the project language fitting all stages — derived from the real tasks, e.g. "深夜,各自房间隔着手机聊到天亮" / "决赛辩论赛场" / "牌桌现金局"), a 3-colour `palette` (`bg1`/`bg2`/`accent` hex matching the mood), and 2–4 `motifs` (emoji that evoke this exact scene). Make it specific to THIS project — never a generic placeholder.', + '', + '### Scenario tool workflow (supersedes the order above for this project)', + '1. `set_project_info(...)`', + '2. `set_scenario({ setting, goal?, rules?, learnerRole?, characters })`', + '3. `add_role({ type: "instructor", ... })`', + '4. `add_milestone({ scenarioStage: "prep" })` + its one light microtask', + '5. one or more `add_milestone({ scenarioStage: "roleplay" })` + their beats as a dramatic arc (successWhen [required] / characterObjective / skillFocus / narration?)', + '6. `add_milestone({ scenarioStage: "wrapup" })` + its one light microtask', + '7. `set_scene_visual({ caption, bg1, bg2, accent, motifs })` — based on all the roleplay stages above', + '8. `mark_design_complete()`', + ].join('\n'); +} + +export function ordinaryPBLTextOnlyGaps(project: PBLProjectV2): string[] { + const gaps: string[] = []; + for (const milestone of project.milestones) { + if ((milestone.documents ?? []).length > 0) { + gaps.push( + `ordinary PBL milestone "${milestone.title}" has hidden documents; inline any required primer, sample data, or starter content in visible milestone/microtask text instead`, + ); + } + } + return gaps; +} + +// --------------------------------------------------------------------------- +// Tools (Zod-validated, share the same mutable `project`) +// --------------------------------------------------------------------------- + +export function newId(prefix: string): string { + // Short, collision-resistant (12 hex chars). Avoids pulling in + // `nanoid` here so the planner stays dependency-free. + return ( + prefix + '_' + Math.random().toString(16).slice(2, 8) + Math.random().toString(16).slice(2, 8) + ); +} + +export function instructorProjectAnchor(project: PBLProjectV2): string { + // Internal meta-instruction appended to the Instructor's system prompt. It is + // written in English (the model follows it regardless of content language); + // the embedded title / description are already in the project's content + // language, and the Instructor answers the learner in that language per its + // own language rule. No locale branching here. + return [ + `You are the Instructor for THIS PBL project.`, + `Project title: ${project.title}`, + `Project description: ${project.description}`, + project.learningObjective ? `Learning objective: ${project.learningObjective}` : '', + 'If the learner asks what project they are doing, answer directly from this information, in the project content language. Never say you do not know the project, and never ask them what project they want to do unless the project title and description are empty.', + ] + .filter(Boolean) + .join('\n'); +} + +/** + * Apply the Planner's chosen proficiency tier onto the project, honoring + * the explicit-self-report lock and re-seating the adaptive assessment so + * score/counters stay consistent. Mirrors the decision logic inside the + * loop's `set_project_info` tool, factored out for the single-call planner. + * + * When the learner explicitly stated their level, that lock wins: the + * project is coerced to the locked tier regardless of the LLM's pick. + */ +export function applyPlannerProficiency( + project: PBLProjectV2, + proficiency: 'beginner' | 'intermediate' | 'advanced', +): void { + const assessment = project.proficiencyAssessment; + const explicitTierLocked = assessment?.signals[0]?.kind === 'user_level_explicit'; + const effectiveProficiency: PBLProficiency = + explicitTierLocked && assessment ? assessment.tier : proficiency; + + if (assessment && assessment.tier !== effectiveProficiency) { + log.info( + `Planner LLM overrode initial proficiency: engine=${assessment.tier} → llm=${effectiveProficiency}`, + ); + project.proficiencyAssessment = reseatAssessmentTier( + assessment, + effectiveProficiency, + 'planner', + ); + } + project.proficiency = effectiveProficiency; +} + +// --------------------------------------------------------------------------- +// Completion validation +// --------------------------------------------------------------------------- + +export function plannerCompletionGaps( + project: PBLProjectV2, + opts?: { scenarioRoleplay?: boolean }, +): string[] { + const errors: string[] = []; + if (!project.title) errors.push('title is empty'); + if (!project.description) errors.push('description is empty'); + if (!project.roles.some((r) => r.type === 'instructor')) { + errors.push('no Instructor role'); + } + if (project.milestones.length === 0) { + errors.push('no milestones'); + } + for (const m of project.milestones) { + if (m.microtasks.length === 0) { + errors.push(`milestone "${m.title}" has no microtasks`); + } + } + if (!opts?.scenarioRoleplay) { + errors.push(...ordinaryPBLTextOnlyGaps(project)); + } + // SCENARIO ONLY. When scenario mode was requested, the design must be + // a coherent role-play scenario: a full cast + the fixed three-stage + // skeleton (prep → roleplay(s) → wrapup). These checks never fire for + // ordinary projects (opts.scenarioRoleplay falsy). + if (opts?.scenarioRoleplay) { + if (!project.scenario) { + errors.push('scenario project but set_scenario was never called'); + } else { + const characters = project.scenario.characters ?? []; + if (characters.length === 0) { + errors.push('scenario has no characters (set_scenario needs at least one character)'); + } else { + characters.forEach((c, i) => { + if (!c?.name?.trim() || !c?.persona?.trim() || !c?.situation?.trim()) { + errors.push(`scenario character #${i + 1} is missing name, persona, or situation`); + } + }); + } + } + // Fixed three-stage skeleton: first 'prep', last 'wrapup', ≥1 'roleplay'. + const stages = project.milestones.map((m) => m.scenarioStage); + const roleplayCount = stages.filter((s) => s === 'roleplay').length; + if (project.milestones.length < 3) { + errors.push( + 'scenario project needs the three-stage skeleton: a prep stage, at least one roleplay stage, and a wrapup stage', + ); + } + if (stages[0] !== 'prep') { + errors.push('scenario project: the FIRST milestone must be scenarioStage:"prep"'); + } + if (stages[stages.length - 1] !== 'wrapup') { + errors.push('scenario project: the LAST milestone must be scenarioStage:"wrapup"'); + } + if (roleplayCount === 0) { + errors.push('scenario project: needs at least one scenarioStage:"roleplay" milestone'); + } + // The project-wide scene visual must be authored (caption + ≥1 emoji + // motif) so the entrance animation / banner fits this exact project. + const sv = project.scenario?.sceneVisual; + if (!sv?.caption?.trim() || (sv?.motifs?.length ?? 0) === 0) { + errors.push( + 'scenario project: call set_scene_visual once (a project-wide caption + 2–4 fitting emoji motifs + colours) AFTER authoring the roleplay stages', + ); + } + } + return errors; +} + +// --------------------------------------------------------------------------- +// Stage-synthesis normalization (deterministic "not too many / not too few") +// --------------------------------------------------------------------------- + +/** Hard cap on how many stages may carry a `synthesisCheck`. The + * integrative stage-end reverse-question is meant for the 1-2 stages + * that hold the project's core knowledge; more than that re-introduces + * the over-questioning failure mode. */ +export const MAX_SYNTHESIS_STAGES = 2; + +/** Tokenize text into latin words (len ≥ 2) + CJK bigrams for a cheap, + * language-agnostic relevance overlap. Deterministic. */ +function conceptTokens(text: string): Set { + const out = new Set(); + const lower = (text ?? '').toLowerCase(); + for (const w of lower.match(/[a-z0-9]{2,}/g) ?? []) out.add(w); + const cjk = lower.match(/[\u4e00-\u9fff]/g) ?? []; + for (let i = 0; i + 1 < cjk.length; i++) out.add(cjk[i] + cjk[i + 1]); + return out; +} + +/** Count how many of `refTokens` appear in `text`. */ +function overlapScore(text: string, refTokens: Set): number { + if (refTokens.size === 0) return 0; + const t = conceptTokens(text); + let score = 0; + for (const tok of refTokens) if (t.has(tok)) score++; + return score; +} + +/** + * Deterministically enforce "1-2 core stages get a synthesisCheck": + * - If the Planner over-flagged (> MAX), keep the MAX most relevant to + * the learning objective / project and drop `synthesisCheck` from + * the rest. + * - If the Planner flagged none, pick the single stage most aligned + * with the learning objective (avoiding the very first setup stage + * when there are ≥ 3 stages) and synthesise a `coreConcept` from the + * learning objective / that stage. This turns the "not too many / + * not too few" guarantee from a prompt hope into code. + * + * Exported for unit tests. + */ +export function normalizeSynthesisChecks(project: PBLProjectV2): void { + // SCENARIO ONLY exemption. Role-play scenario projects never carry a + // synthesisCheck on any stage (the integrative reflection is the light + // wrapup stage, not a mid-scenario reverse-question). Skip entirely so + // we never auto-attach one to a prep/roleplay/wrapup milestone. + if (project.scenario) return; + if (project.milestones.length === 0) return; + const refTokens = conceptTokens( + `${project.learningObjective ?? ''} ${project.title} ${project.description}`, + ); + const flagged = project.milestones.filter((m) => m.synthesisCheck); + + if (flagged.length > MAX_SYNTHESIS_STAGES) { + const ranked = flagged + .map((m) => ({ + m, + score: overlapScore( + `${m.title} ${m.description ?? ''} ${m.synthesisCheck?.coreConcept ?? ''}`, + refTokens, + ), + })) + .sort((a, b) => b.score - a.score || a.m.order - b.m.order); + for (const { m } of ranked.slice(MAX_SYNTHESIS_STAGES)) { + delete m.synthesisCheck; + } + return; + } + + if (flagged.length === 0) { + const ordered = project.milestones.slice().sort((a, b) => a.order - b.order); + const ranked = ordered + .map((m) => ({ m, score: overlapScore(`${m.title} ${m.description ?? ''}`, refTokens) })) + .sort((a, b) => b.score - a.score || a.m.order - b.m.order); + let pick = ranked[0]?.m; + // When nothing aligns (all-zero overlap), avoid the first stage + // (usually setup) and the last (usually polish): take the median. + if ((!pick || ranked[0].score === 0) && ordered.length >= 3) { + pick = ordered[Math.floor(ordered.length / 2)]; + } + if (pick) { + const coreConcept = ( + project.learningObjective?.trim() || + pick.description?.trim() || + pick.title + ).slice(0, 120); + pick.synthesisCheck = { coreConcept }; + } + } +} diff --git a/lib/pbl/v2/agents/planner-single-call.ts b/lib/pbl/v2/agents/planner-single-call.ts index 53f584e43..20de53021 100644 --- a/lib/pbl/v2/agents/planner-single-call.ts +++ b/lib/pbl/v2/agents/planner-single-call.ts @@ -16,15 +16,14 @@ * All the deterministic hydration (ids / status / order / assignee / * thread bootstrap / proficiency re-seat) and post-processing * (`normalizeProjectRuntime`, `normalizeSynthesisChecks`, completion - * gate) is shared with the loop via exported helpers in `./planner.ts`. + * gate) is shared with the loop via exported helpers in `./planner-core.ts`. */ import type { LanguageModel } from 'ai'; -import { callLLM } from '@/lib/ai/llm'; import { createLogger } from '@/lib/logger'; import { parseJsonResponse } from '@/lib/generation/json-repair'; -import { normalizeProjectRuntime, normalizeScenario } from '../operations/progress'; +import { normalizeProjectRuntime, normalizeScenario } from '../operations/kernel/progress'; import type { ThinkingConfig } from '@/lib/types/provider'; import { @@ -38,7 +37,7 @@ import { normalizeSynthesisChecks, plannerCompletionGaps, type PlannerV2Callbacks, -} from './planner'; +} from './planner-core'; import type { PBLProjectV2, @@ -56,6 +55,14 @@ const log = createLogger('PBL v2 Planner (single-call)'); const SINGLE_CALL_PROMPT = 'planner-single-call-system'; const SCENARIO_PROMPT = 'planner-scenario-single-call-system'; +/** Narrow call seam used by the untooled single-call planner. */ +export type PlannerSingleCallFn = ( + params: { model: LanguageModel; system: string; prompt: string }, + source: 'pbl-v2-planner-single', + retryOptions: undefined, + thinkingConfig?: ThinkingConfig, +) => Promise<{ text: string }>; + function buildSingleCallUserPrompt(scenarioRoleplay: boolean): string { const sharedChecklist = [ 'projectInfo has non-empty title, description, learningObjective, 3-5 gains, and the exact requested proficiency', @@ -477,6 +484,7 @@ function hydrateProject(project: PBLProjectV2, parsed: PlannerLLMOutput): void { export async function generatePBLV2ProjectSingleCall( input: PBLPlannerV2Input, model: LanguageModel, + callLLM: PlannerSingleCallFn, callbacks?: PlannerV2Callbacks, thinkingConfig?: ThinkingConfig, ): Promise { diff --git a/lib/pbl/v2/agents/planner.ts b/lib/pbl/v2/agents/planner.ts index 22f8a9016..66d87f172 100644 --- a/lib/pbl/v2/agents/planner.ts +++ b/lib/pbl/v2/agents/planner.ts @@ -22,17 +22,26 @@ import type { LanguageModel, StepResult, StopCondition, ToolSet } from 'ai'; import { tool, stepCountIs } from 'ai'; import { z } from 'zod'; -import { callLLM } from '@/lib/ai/llm'; import { createLogger } from '@/lib/logger'; -import { loadPBLV2Prompt } from '../prompts/loader'; -import { computeInitialAssessment, reseatAssessmentTier } from '../operations/proficiency'; -import { normalizeProjectRuntime, normalizeScenario } from '../operations/progress'; +import { normalizeProjectRuntime, normalizeScenario } from '../operations/kernel/progress'; import type { ThinkingConfig } from '@/lib/types/provider'; +import { + SCENARIO_SCHEMA_VERSION, + PlannerV2Error, + emptyProject, + buildPlannerSystemPrompt, + newId, + instructorProjectAnchor, + applyPlannerProficiency, + normalizeSynthesisChecks, + plannerCompletionGaps, + type PlannerV2Callbacks, +} from './planner-core'; + import type { PBLProjectV2, PBLPlannerV2Input, - PBLProficiency, PBLMilestone, PBLMicrotask, PBLRole, @@ -41,11 +50,20 @@ import type { const log = createLogger('PBL v2 Planner'); -/** SCENARIO ONLY. Packaged-format version stamped on scenario projects - * (`project.schemaVersion`). Absent on ordinary projects (baseline). - * Bump when the packaged scenario format changes so loaders can - * migrate. */ -export const SCENARIO_SCHEMA_VERSION = 1; +/** Narrow call seam used by the tool-calling planner. */ +export type PlannerCallFn = ( + params: { + model: LanguageModel; + system: string; + prompt: string; + tools: ToolSet; + stopWhen: StopCondition[]; + onStepFinish: (step: Pick, 'toolCalls'>) => void; + }, + source: 'pbl-v2-planner', + retryOptions: undefined, + thinkingConfig?: ThinkingConfig, +) => Promise; // --------------------------------------------------------------------------- // Loop budgets @@ -57,40 +75,6 @@ export const SCENARIO_SCHEMA_VERSION = 1; * errors. */ const MAX_PLANNER_STEPS = 80; -// --------------------------------------------------------------------------- -// Callbacks (so the caller can stream progress to a UI later) -// --------------------------------------------------------------------------- - -export interface PlannerV2Callbacks { - /** Fired on each successful tool call. Used by the future Generating - * page to show "Adding milestone: X" etc. */ - onProgress?: (event: PlannerV2ProgressEvent) => void; -} - -export type PlannerV2ProgressEvent = - | { kind: 'project_info'; title: string } - | { kind: 'role'; roleType: PBLRole['type']; name: string } - | { kind: 'milestone'; title: string; index: number } - | { kind: 'microtask'; milestoneTitle: string; title: string; index: number } - | { kind: 'complete'; milestoneCount: number; microtaskCount: number }; - -// --------------------------------------------------------------------------- -// Errors -// --------------------------------------------------------------------------- - -/** Thrown when the Planner finishes but the result is unusable (no - * Instructor, no milestones, etc.). The caller should fall back to v1 - * or to a slide so the student is never stranded on an empty PBL. */ -export class PlannerV2Error extends Error { - constructor( - message: string, - public readonly partial: PBLProjectV2, - ) { - super(message); - this.name = 'PlannerV2Error'; - } -} - // --------------------------------------------------------------------------- // Public entrypoint // --------------------------------------------------------------------------- @@ -111,6 +95,7 @@ export class PlannerV2Error extends Error { export async function generatePBLV2Project( input: PBLPlannerV2Input, model: LanguageModel, + callLLM: PlannerCallFn, callbacks?: PlannerV2Callbacks, thinkingConfig?: ThinkingConfig, ): Promise { @@ -214,264 +199,6 @@ export async function generatePBLV2Project( return project; } -// --------------------------------------------------------------------------- -// Empty / starter shape -// --------------------------------------------------------------------------- - -export function emptyProject(input: PBLPlannerV2Input): PBLProjectV2 { - const now = new Date().toISOString(); - - // Compute the planner-time initial proficiency assessment from - // static signals (outline keywords + prior-scene difficulty + user - // bio). Quiz accuracy is not yet available — that's folded in at - // Hero entry by the pre-play recalibration path. - // - // See `lib/pbl/v2/operations/proficiency.ts` for the full algorithm - // and the calibration table. The Planner LLM does NOT decide this - // value: it consumes `assessment.tier` as a directive when - // dimensioning microtasks. - const assessment = computeInitialAssessment({ - outline: input.outline, - priorScenes: input.courseContext.allOutlines, - userBio: input.user?.bio, - userRequirement: input.user?.requirement, - priorQuizResults: input.priorQuizResults, - source: 'planner', - }); - const proficiency: PBLProficiency = assessment.tier; - // `languageDirective` is the SINGLE source of truth for content language — - // a free natural-language directive from the outline stage (e.g. "Reply in - // Simplified Chinese" / "中文为主,英文技术术语保留原文"). The Planner feeds it - // straight to the system prompt; there is no content-based locale guessing. - // - // `language` is only the BCP-47 locale the RUNTIME uses for deterministic - // platform strings (synthetic openers, divider labels). It is seeded from - // the authoritative UI locale (`targetLanguage`) when known and otherwise - // left blank for the Hero locale-sync (hero.tsx) to fill on entry — it is - // NEVER inferred from content here. - const languageDirective = input.courseContext.languageDirective?.trim(); - const language = input.targetLanguage?.trim() ?? ''; - - log.info( - `Planner v2 initial assessment: tier=${assessment.tier} score=${assessment.score.toFixed( - 2, - )} confidence=${assessment.confidence.toFixed(2)} signals=${assessment.signals.length} language=${language}`, - ); - - return { - uiPhase: 'hero', - title: '', - description: '', - learningObjective: '', - gains: [], - proficiency, - proficiencyAssessment: assessment, - language, - languageDirective: languageDirective || undefined, - tags: [], - status: 'designing', - roles: [], - milestones: [], - submissions: [], - evaluations: [], - threads: [], - engagementEvents: [], - createdAt: now, - updatedAt: now, - }; -} - -// --------------------------------------------------------------------------- -// Prompt assembly -// --------------------------------------------------------------------------- - -export function formatCourseContext(input: PBLPlannerV2Input): string { - const lines: string[] = []; - for (const o of input.courseContext.allOutlines) { - const marker = o.id === input.outline.id ? ' ← this PBL scene' : ''; - lines.push(`- [${o.order}] ${o.type.toUpperCase()}: ${o.title}${marker}`); - if (o.description) { - lines.push(` ${o.description}`); - } - } - return lines.join('\n'); -} - -export async function buildPlannerSystemPrompt( - input: PBLPlannerV2Input, - proficiency: PBLProficiency, - contentLanguage: string, - scenarioRoleplay: boolean, - promptName: string = 'planner-system', -): Promise { - const pblConfig = input.outline.pblConfig!; - - // The adaptive engine decided the tier in `emptyProject`; pass it - // through as a directive so the LLM dimensions microtasks - // accordingly. The Planner is expected to mirror this value when it - // calls `set_project_info`; if it picks a different tier the tool - // accepts the value (it's a hint, not a hard contract) but the - // engine logs the divergence. - // - // `contentLanguage` is resolved by the caller from - // `project.languageDirective || project.language` (single source). It may be - // a BCP-47 locale or a nuanced directive like "中文为主,英文技术术语保留原文". - return loadPBLV2Prompt(promptName, { - projectTopic: pblConfig.projectTopic, - projectDescription: pblConfig.projectDescription, - targetSkills: (pblConfig.targetSkills ?? []).join(', '), - milestoneCount: pblConfig.issueCount ?? 3, - proficiency: proficiency === '' ? 'intermediate' : proficiency, - language: contentLanguage, - courseContext: formatCourseContext(input), - languageDirective: input.courseContext.languageDirective, - // Optional free-form scenario brief from the outline stage. Empty for - // ordinary projects / non-scenario prompts (the slot collapses). - scenarioBrief: input.outline.pblConfig?.scenarioBrief ?? '', - // SCENARIO ONLY. Empty string for ordinary PBL projects → the - // `{{scenarioDesign}}` slot collapses to nothing and the prompt is - // byte-identical to before. The slot must always be provided (the - // interpolator leaves unknown placeholders literal). - scenarioDesign: buildScenarioDesignBlock(pblConfig, scenarioRoleplay), - }); -} - -/** SCENARIO ONLY. Build the role-play scenario-design instruction block - * injected into the Planner system prompt. Returns '' for ordinary PBL - * projects (so the prompt is byte-identical to before). When the - * outline opted into `scenarioRoleplay`, it instructs the Planner to - * fully author the scenario AND lay it out as the fixed three-stage - * skeleton: prep → roleplay(×1..N) → wrapup. */ -export function buildScenarioDesignBlock( - pblConfig: NonNullable, - scenarioRoleplay: boolean, -): string { - if (!scenarioRoleplay) return ''; - const brief = pblConfig.scenarioBrief?.trim(); - return [ - '## SCENARIO MODE — role-play scenario (this project only)', - '', - 'This PBL is a **role-play scenario**: the learner will step into a concrete situation and interact in-character with character(s) played by a separate Simulator agent. You author the WHOLE scenario now (it is frozen into the package); the runtime only produces the live dialogue. Two rules above all: (a) the premise is **given and concrete**, introduced to the learner by the Instructor — the learner must NEVER be asked to guess it; (b) every task must serve the real learning goal (how to do the thing well), not meta-guessing.', - '', - brief ? `Scenario brief from the platform: ${brief}\n` : '', - '### Step A — fully author the scenario with `set_scenario(...)`', - 'Call **exactly once, right after `set_project_info` and before any `add_milestone`**:', - '- `setting`: the concrete overall premise / what is going on (in the project language).', - '- `goal` (optional): what the learner is practising.', - "- `rules` (optional but REQUIRED whenever the scenario has any defined rule-set — games / interviews / debates / structured negotiations / etc.): write the CONCRETE rules a newcomer needs to actually take part, specific enough that the Instructor can teach them verbatim in prep. Not a vague label — include the real mechanics (e.g. a card game: hand ranking, betting rounds, blinds, what terms like Pot Odds / Fold / Call / Raise / Check mean; a debate: the motion, each side's stance, the speaking format; an interview: the rounds and what each assesses). Omit ONLY for free scenarios with no special rules (e.g. comforting a friend).", - '- `learnerRole` (optional): the learner\'s OWN role/position (e.g. "you are their close friend" / "you are the 5th player, on the button").', - '- `characters`: **EXACTLY ONE character** — this version plays a single counterpart throughout (the runtime only ever voices one). It needs `name`, `persona` (stable identity / relationship / personality / speaking style), **`situation`** (their CONCRETE current circumstance the learner faces — e.g. "just broke up last week, low mood, says they\'re fine but aren\'t"). `situation` is shown to the learner up front (prep intro + the always-visible scenario briefing), so it must hold ONLY what the learner can see/know at the start — keep any fact a later beat is meant to make them uncover OUT of it (see the No-spoilers rule below). Plus strongly-recommended `boundaries` (hard safety rails), and optional `openingLine`.', - '', - '### Step B — lay out the FIXED three-stage skeleton (milestones in this exact order)', - '1. **Prep stage** — `add_milestone({ ..., scenarioStage: "prep" })` as the FIRST milestone. Its `briefing` is the Instructor intro that **introduces the concrete premise to the learner**: the situation, each character\'s `situation`, what the learner is there to do, plus `rules` / `learnerRole` when present. The intro MUST match the roleplay stages you design next. Give prep **exactly ONE light microtask** (e.g. "了解背景,准备开始" / "Understand the setup, ready to begin") — NO assessment; the learner just confirms and advances. **Do NOT set `coreConcept`.**', - '2. **Roleplay stage(s)** — one or MORE `add_milestone({ ..., scenarioStage: "roleplay" })` in the middle (split a long scenario into several roleplay stages by round/phase to avoid one giant stage). Each roleplay milestone\'s `briefing` brings the learner into the scene. **Design the beats as a DRAMATIC ARC, not a flat checklist**: an opening hook → rising stakes/complication → a turning point or decision → a resolution. Each beat should be a MEANINGFUL decision/action unit (something the learner can actually DO), never empty filler. For **each microtask (beat)** under a roleplay milestone, provide:', - ' - `description`: the CONCRETE situation of this beat as the SYSTEM narrator states it to the learner — positions / cards / what just happened / whose turn (e.g. "你在 Button 位拿到 A♠ J♦;前面都 Fold,老周在 Cutoff 加注到 6 个筹码;轮到你决定 preflop"). The character NEVER states these facts — the system does; the character only reacts. Keep it factual scene-setting, not coaching.', - ' - `successWhen` (REQUIRED for every roleplay beat): the CONCRETE, OBSERVABLE in-scene action the learner must SAY or DO for this beat to count as done — the scenario\'s "deliverable" (e.g. "做出 preflop 决定:跟注、加注或弃牌" / "对对方说出的感受做出共情回应,并问一个跟进问题"). State it in plain SCENE terms (what they do in the fiction), NOT as a teaching goal. This is exactly what the advance detector watches, so a crisp `successWhen` is what stops off-topic / small-talk turns from advancing the scene. Make it a real decision/action, not "they chatted a bit".', - ' - `completionCriteria` (optional, legacy): a teaching-side note on what this beat is about; `successWhen` is preferred and takes precedence for advancing.', - ' - `characterObjective` (recommended): what the character PRIVATELY wants — and privately KNOWS — this beat: their in-scene drive (e.g. "试探你是否在虚张声势" / "想确认你是否真的在乎"), plus any fact the learner is meant to UNCOVER this beat (the hidden cause / secret / backstory the character only reveals when probed — e.g. "你昨天在空调房待了很久才着凉,但只有被仔细询问才说出来"). It makes the character pursue a goal and hold its secrets in character; it is private to the character — NEVER narrated, shown in the briefing, evaluated, or coached.', - ' - `skillFocus` (recommended): the single skill this beat practises (e.g. "底池赔率判断" / "积极倾听"). Surfaced to the learner (current-task panel + end-of-project per-act review); never spoken by the character.', - ' - The scene is FREE-FIRST: the learner always speaks/types their OWN response to the character, which is how a real interaction is practised. (Some beats may instead ask the learner to hand in a real artefact, e.g. "write them a letter".)', - ' - `narration` (optional): a short neutral scene-setting line the SYSTEM reads when this beat opens (e.g. "你们走进了一家安静的咖啡厅"). NEVER spoken by a character or the Instructor. All scene/state facts come from the system (narration + description), never from the character\'s mouth.', - ' - `hints` (recommended for roleplay beats): 1–2 SHORT, learner-facing coaching tips for THIS beat — what skill to focus on or how to handle it well (e.g. "先共情、再问问题,别急着给建议" / "注意你的位置和底池赔率,再决定下注"). They appear in the "hints" card of the learner\'s current-task side panel, are NEVER spoken by the character, and are the learner\'s in-the-moment guidance. Keep them concrete to this beat, not generic.', - ' - **Do NOT set `coreConcept`** on roleplay milestones.', - '3. **Wrapup stage** — `add_milestone({ ..., scenarioStage: "wrapup" })` as the LAST milestone. Its `debrief` holds the Instructor\'s light, encouraging feedback points (highlights / one thing to improve); the detailed report lives on the completion page. Give wrapup **exactly ONE light microtask** (e.g. "听取反馈,收尾" / "Hear the feedback, wrap up"). **Do NOT set `coreConcept`.**', - '', - '### Rules for scenario design', - '- The premise (situation / rules / positions) is GIVEN and introduced in prep — **never make a task that asks the learner to guess/invent it**.', - "- **No spoilers — never give away in learner-VISIBLE text what a beat is designed to make the learner discover.** The premise the learner can see up front — `setting`, each character's `situation` / `persona`, the prep `briefing`, and each beat's `description` / `narration` (and the always-visible scenario briefing built from these) — must contain ONLY what the learner already knows or can plainly observe at the outset. If any roleplay beat's `successWhen` requires the learner to UNCOVER something through the interaction (a hidden cause, a motive, a secret, the diagnosis, a backstory fact), that information MUST NOT appear in any learner-visible field. Put it ONLY in that beat's private `characterObjective`, where the character holds it and reveals it solely when the learner actually probes for it — never up front. E.g. a \"find out why\" beat: the real cause lives in `characterObjective`; `situation` states only the visible symptoms / where the character is right now.", - '- All scenario text (`setting`/`persona`/`situation`/`briefing`/narration/options…) follows the same content-language policy as Hard rule 1.', - '- Scene beats should feel like a real interaction unfolding, not a checklist; 2-4 beats per roleplay stage is plenty.', - '- **The roleplay character is a pure in-world participant, NEVER a coach.** When you write `persona` / `situation` / `openingLine` (and any character-facing text), the character must have its OWN motives and react like a real person in the scene. It must NEVER: ask the learner to explain/justify their reasoning ("说说你为什么这么选" / "一句话给我理由"), evaluate or grade the learner\'s moves ("这步打得对"), give strategy/meta hints ("想想我的范围里哪些牌会付钱"), or tell the learner it\'s their turn / what to decide. That is all out-of-scene/teaching content and it does NOT belong in the character\'s mouth.', - '- **Out-of-scene content has its own channels — never the character:** (a) the "this is a training table / I\'ll test you" framing and any rule teaching belong to the **prep Instructor** (`briefing`); (b) "it\'s your turn to act / a decision point has arrived" belongs to the **system `narration`** of that beat, stated neutrally; (c) strategy / what-to-watch-for hints belong to the microtask **`hints`** (side panel). Route each of these to its channel; the character only ever lives the scene.', - '', - '', - '### Step C — author ONE project-wide scene visual with `set_scene_visual(...)`', - 'After ALL roleplay milestones/beats exist, call `set_scene_visual` exactly once. Read back over EVERY roleplay stage/task you just wrote and distil the ONE shared place/atmosphere they all happen in, then describe it: a `caption` (a short phrase in the project language fitting all stages — derived from the real tasks, e.g. "深夜,各自房间隔着手机聊到天亮" / "决赛辩论赛场" / "牌桌现金局"), a 3-colour `palette` (`bg1`/`bg2`/`accent` hex matching the mood), and 2–4 `motifs` (emoji that evoke this exact scene). Make it specific to THIS project — never a generic placeholder.', - '', - '### Scenario tool workflow (supersedes the order above for this project)', - '1. `set_project_info(...)`', - '2. `set_scenario({ setting, goal?, rules?, learnerRole?, characters })`', - '3. `add_role({ type: "instructor", ... })`', - '4. `add_milestone({ scenarioStage: "prep" })` + its one light microtask', - '5. one or more `add_milestone({ scenarioStage: "roleplay" })` + their beats as a dramatic arc (successWhen [required] / characterObjective / skillFocus / narration?)', - '6. `add_milestone({ scenarioStage: "wrapup" })` + its one light microtask', - '7. `set_scene_visual({ caption, bg1, bg2, accent, motifs })` — based on all the roleplay stages above', - '8. `mark_design_complete()`', - ].join('\n'); -} - -export function ordinaryPBLTextOnlyGaps(project: PBLProjectV2): string[] { - const gaps: string[] = []; - for (const milestone of project.milestones) { - if ((milestone.documents ?? []).length > 0) { - gaps.push( - `ordinary PBL milestone "${milestone.title}" has hidden documents; inline any required primer, sample data, or starter content in visible milestone/microtask text instead`, - ); - } - } - return gaps; -} - -// --------------------------------------------------------------------------- -// Tools (Zod-validated, share the same mutable `project`) -// --------------------------------------------------------------------------- - -export function newId(prefix: string): string { - // Short, collision-resistant (12 hex chars). Avoids pulling in - // `nanoid` here so the planner stays dependency-free. - return ( - prefix + '_' + Math.random().toString(16).slice(2, 8) + Math.random().toString(16).slice(2, 8) - ); -} - -export function instructorProjectAnchor(project: PBLProjectV2): string { - // Internal meta-instruction appended to the Instructor's system prompt. It is - // written in English (the model follows it regardless of content language); - // the embedded title / description are already in the project's content - // language, and the Instructor answers the learner in that language per its - // own language rule. No locale branching here. - return [ - `You are the Instructor for THIS PBL project.`, - `Project title: ${project.title}`, - `Project description: ${project.description}`, - project.learningObjective ? `Learning objective: ${project.learningObjective}` : '', - 'If the learner asks what project they are doing, answer directly from this information, in the project content language. Never say you do not know the project, and never ask them what project they want to do unless the project title and description are empty.', - ] - .filter(Boolean) - .join('\n'); -} - -/** - * Apply the Planner's chosen proficiency tier onto the project, honoring - * the explicit-self-report lock and re-seating the adaptive assessment so - * score/counters stay consistent. Mirrors the decision logic inside the - * loop's `set_project_info` tool, factored out for the single-call planner. - * - * When the learner explicitly stated their level, that lock wins: the - * project is coerced to the locked tier regardless of the LLM's pick. - */ -export function applyPlannerProficiency( - project: PBLProjectV2, - proficiency: 'beginner' | 'intermediate' | 'advanced', -): void { - const assessment = project.proficiencyAssessment; - const explicitTierLocked = assessment?.signals[0]?.kind === 'user_level_explicit'; - const effectiveProficiency: PBLProficiency = - explicitTierLocked && assessment ? assessment.tier : proficiency; - - if (assessment && assessment.tier !== effectiveProficiency) { - log.info( - `Planner LLM overrode initial proficiency: engine=${assessment.tier} → llm=${effectiveProficiency}`, - ); - project.proficiencyAssessment = reseatAssessmentTier( - assessment, - effectiveProficiency, - 'planner', - ); - } - project.proficiency = effectiveProficiency; -} - function buildTools( project: PBLProjectV2, input: PBLPlannerV2Input, @@ -596,30 +323,7 @@ function buildTools( error: `The learner explicitly stated their level as ${project.proficiencyAssessment!.tier}. Call set_project_info again with proficiency="${project.proficiencyAssessment!.tier}".`, }; } - const effectiveProficiency = explicitTierLocked - ? project.proficiencyAssessment!.tier - : proficiency; - - // Log divergence from the platform's adaptive engine but - // accept the LLM's choice when there was no explicit learner - // self-report. Re-seat the whole assessment onto the chosen tier - // (not just `tier`) so score/counters stay consistent — otherwise a - // stale score later rebounds the learner back toward the engine's - // estimate once the dynamic retier gates clear (see reseatAssessmentTier). - if ( - project.proficiencyAssessment && - project.proficiencyAssessment.tier !== effectiveProficiency - ) { - log.info( - `Planner LLM overrode initial proficiency: engine=${project.proficiencyAssessment.tier} → llm=${effectiveProficiency}`, - ); - project.proficiencyAssessment = reseatAssessmentTier( - project.proficiencyAssessment, - effectiveProficiency, - 'planner', - ); - } - project.proficiency = effectiveProficiency; + applyPlannerProficiency(project, proficiency); project.updatedAt = new Date().toISOString(); projectInfoSet = true; callbacks?.onProgress?.({ kind: 'project_info', title }); @@ -1025,75 +729,6 @@ function buildTools( type PlannerCompletionToolResult = { ok: true } | { ok: false; gaps: string[]; nextAction: string }; -export function plannerCompletionGaps( - project: PBLProjectV2, - opts?: { scenarioRoleplay?: boolean }, -): string[] { - const errors: string[] = []; - if (!project.title) errors.push('title is empty'); - if (!project.description) errors.push('description is empty'); - if (!project.roles.some((r) => r.type === 'instructor')) { - errors.push('no Instructor role'); - } - if (project.milestones.length === 0) { - errors.push('no milestones'); - } - for (const m of project.milestones) { - if (m.microtasks.length === 0) { - errors.push(`milestone "${m.title}" has no microtasks`); - } - } - if (!opts?.scenarioRoleplay) { - errors.push(...ordinaryPBLTextOnlyGaps(project)); - } - // SCENARIO ONLY. When scenario mode was requested, the design must be - // a coherent role-play scenario: a full cast + the fixed three-stage - // skeleton (prep → roleplay(s) → wrapup). These checks never fire for - // ordinary projects (opts.scenarioRoleplay falsy). - if (opts?.scenarioRoleplay) { - if (!project.scenario) { - errors.push('scenario project but set_scenario was never called'); - } else { - const characters = project.scenario.characters ?? []; - if (characters.length === 0) { - errors.push('scenario has no characters (set_scenario needs at least one character)'); - } else { - characters.forEach((c, i) => { - if (!c?.name?.trim() || !c?.persona?.trim() || !c?.situation?.trim()) { - errors.push(`scenario character #${i + 1} is missing name, persona, or situation`); - } - }); - } - } - // Fixed three-stage skeleton: first 'prep', last 'wrapup', ≥1 'roleplay'. - const stages = project.milestones.map((m) => m.scenarioStage); - const roleplayCount = stages.filter((s) => s === 'roleplay').length; - if (project.milestones.length < 3) { - errors.push( - 'scenario project needs the three-stage skeleton: a prep stage, at least one roleplay stage, and a wrapup stage', - ); - } - if (stages[0] !== 'prep') { - errors.push('scenario project: the FIRST milestone must be scenarioStage:"prep"'); - } - if (stages[stages.length - 1] !== 'wrapup') { - errors.push('scenario project: the LAST milestone must be scenarioStage:"wrapup"'); - } - if (roleplayCount === 0) { - errors.push('scenario project: needs at least one scenarioStage:"roleplay" milestone'); - } - // The project-wide scene visual must be authored (caption + ≥1 emoji - // motif) so the entrance animation / banner fits this exact project. - const sv = project.scenario?.sceneVisual; - if (!sv?.caption?.trim() || (sv?.motifs?.length ?? 0) === 0) { - errors.push( - 'scenario project: call set_scene_visual once (a project-wide caption + 2–4 fitting emoji motifs + colours) AFTER authoring the roleplay stages', - ); - } - } - return errors; -} - function plannerCompletionNextAction( project: PBLProjectV2, opts?: { scenarioRoleplay?: boolean }, @@ -1160,96 +795,3 @@ function validateProject(project: PBLProjectV2, scenarioRoleplay = false): void throw new PlannerV2Error(`Planner v2 output failed validation: ${errors.join('; ')}`, project); } } - -// --------------------------------------------------------------------------- -// Stage-synthesis normalization (deterministic "not too many / not too few") -// --------------------------------------------------------------------------- - -/** Hard cap on how many stages may carry a `synthesisCheck`. The - * integrative stage-end reverse-question is meant for the 1-2 stages - * that hold the project's core knowledge; more than that re-introduces - * the over-questioning failure mode. */ -export const MAX_SYNTHESIS_STAGES = 2; - -/** Tokenize text into latin words (len ≥ 2) + CJK bigrams for a cheap, - * language-agnostic relevance overlap. Deterministic. */ -function conceptTokens(text: string): Set { - const out = new Set(); - const lower = (text ?? '').toLowerCase(); - for (const w of lower.match(/[a-z0-9]{2,}/g) ?? []) out.add(w); - const cjk = lower.match(/[\u4e00-\u9fff]/g) ?? []; - for (let i = 0; i + 1 < cjk.length; i++) out.add(cjk[i] + cjk[i + 1]); - return out; -} - -/** Count how many of `refTokens` appear in `text`. */ -function overlapScore(text: string, refTokens: Set): number { - if (refTokens.size === 0) return 0; - const t = conceptTokens(text); - let score = 0; - for (const tok of refTokens) if (t.has(tok)) score++; - return score; -} - -/** - * Deterministically enforce "1-2 core stages get a synthesisCheck": - * - If the Planner over-flagged (> MAX), keep the MAX most relevant to - * the learning objective / project and drop `synthesisCheck` from - * the rest. - * - If the Planner flagged none, pick the single stage most aligned - * with the learning objective (avoiding the very first setup stage - * when there are ≥ 3 stages) and synthesise a `coreConcept` from the - * learning objective / that stage. This turns the "not too many / - * not too few" guarantee from a prompt hope into code. - * - * Exported for unit tests. - */ -export function normalizeSynthesisChecks(project: PBLProjectV2): void { - // SCENARIO ONLY exemption. Role-play scenario projects never carry a - // synthesisCheck on any stage (the integrative reflection is the light - // wrapup stage, not a mid-scenario reverse-question). Skip entirely so - // we never auto-attach one to a prep/roleplay/wrapup milestone. - if (project.scenario) return; - if (project.milestones.length === 0) return; - const refTokens = conceptTokens( - `${project.learningObjective ?? ''} ${project.title} ${project.description}`, - ); - const flagged = project.milestones.filter((m) => m.synthesisCheck); - - if (flagged.length > MAX_SYNTHESIS_STAGES) { - const ranked = flagged - .map((m) => ({ - m, - score: overlapScore( - `${m.title} ${m.description ?? ''} ${m.synthesisCheck?.coreConcept ?? ''}`, - refTokens, - ), - })) - .sort((a, b) => b.score - a.score || a.m.order - b.m.order); - for (const { m } of ranked.slice(MAX_SYNTHESIS_STAGES)) { - delete m.synthesisCheck; - } - return; - } - - if (flagged.length === 0) { - const ordered = project.milestones.slice().sort((a, b) => a.order - b.order); - const ranked = ordered - .map((m) => ({ m, score: overlapScore(`${m.title} ${m.description ?? ''}`, refTokens) })) - .sort((a, b) => b.score - a.score || a.m.order - b.m.order); - let pick = ranked[0]?.m; - // When nothing aligns (all-zero overlap), avoid the first stage - // (usually setup) and the last (usually polish): take the median. - if ((!pick || ranked[0].score === 0) && ordered.length >= 3) { - pick = ordered[Math.floor(ordered.length / 2)]; - } - if (pick) { - const coreConcept = ( - project.learningObjective?.trim() || - pick.description?.trim() || - pick.title - ).slice(0, 120); - pick.synthesisCheck = { coreConcept }; - } - } -} diff --git a/lib/pbl/v2/agents/simulator.ts b/lib/pbl/v2/agents/simulator.ts index f75b342db..c8bd2b830 100644 --- a/lib/pbl/v2/agents/simulator.ts +++ b/lib/pbl/v2/agents/simulator.ts @@ -39,12 +39,12 @@ import type { PBLAgentThread, } from '../types'; import type { PBLSSEEvent } from '../api/sse'; -import { recordEvent } from '../operations/engagement'; +import { recordEvent } from '../operations/kernel/engagement'; import { currentMicrotask, normalizeProjectRuntime, PBL_SIMULATOR_AGENT_ID, -} from '../operations/progress'; +} from '../operations/kernel/progress'; const log = createLogger('PBL v2 Simulator'); diff --git a/lib/pbl/v2/agents/tier-guidance.ts b/lib/pbl/v2/agents/tier-guidance.ts index c1a738960..8b3ce5e3d 100644 --- a/lib/pbl/v2/agents/tier-guidance.ts +++ b/lib/pbl/v2/agents/tier-guidance.ts @@ -34,7 +34,7 @@ * evidence-path (Path B default) rules in instructor-base-rules.md. */ -import { DEFAULT_TIER } from '../operations/proficiency'; +import { DEFAULT_TIER } from '../operations/kernel/proficiency'; import type { PBLProficiency } from '../types'; const COMMON_RULES = [ diff --git a/lib/pbl/v2/operations/engagement.ts b/lib/pbl/v2/operations/kernel/engagement.ts similarity index 99% rename from lib/pbl/v2/operations/engagement.ts rename to lib/pbl/v2/operations/kernel/engagement.ts index 675959bff..5fd8690f6 100644 --- a/lib/pbl/v2/operations/engagement.ts +++ b/lib/pbl/v2/operations/kernel/engagement.ts @@ -20,7 +20,7 @@ import type { PBLEngagementEventKind, PBLEngagementSummary, PBLProjectV2, -} from '../types'; +} from '../../types'; /** Soft cap on the engagement ledger size. Older events are dropped * first; per-microtask summaries cache what's lost. Conservatively diff --git a/lib/pbl/v2/operations/proficiency.ts b/lib/pbl/v2/operations/kernel/proficiency.ts similarity index 99% rename from lib/pbl/v2/operations/proficiency.ts rename to lib/pbl/v2/operations/kernel/proficiency.ts index 671662607..11c6a71a4 100644 --- a/lib/pbl/v2/operations/proficiency.ts +++ b/lib/pbl/v2/operations/kernel/proficiency.ts @@ -46,7 +46,7 @@ import type { ProficiencySignal, ProficiencySignalKind, ProficiencyTransition, -} from '../types'; +} from '../../types'; import type { SceneOutline } from '@/lib/types/generation'; import { appendProficiencyUpdatedRuntimeEvent } from './runtime-events'; diff --git a/lib/pbl/v2/operations/progress.ts b/lib/pbl/v2/operations/kernel/progress.ts similarity index 99% rename from lib/pbl/v2/operations/progress.ts rename to lib/pbl/v2/operations/kernel/progress.ts index fc111b07a..e095a0ccf 100644 --- a/lib/pbl/v2/operations/progress.ts +++ b/lib/pbl/v2/operations/kernel/progress.ts @@ -23,7 +23,7 @@ import type { PBLMicrotask, PBLInternalAssessment, PBLHandover, -} from '../types'; +} from '../../types'; import { microtaskEngagement, recordEvent } from './engagement'; import { clearPendingTaskCompletion } from './task-completion'; import { diff --git a/lib/pbl/v2/operations/runtime-events.ts b/lib/pbl/v2/operations/kernel/runtime-events.ts similarity index 99% rename from lib/pbl/v2/operations/runtime-events.ts rename to lib/pbl/v2/operations/kernel/runtime-events.ts index 76c3a45c8..c47bcea79 100644 --- a/lib/pbl/v2/operations/runtime-events.ts +++ b/lib/pbl/v2/operations/kernel/runtime-events.ts @@ -1,4 +1,4 @@ -import type { PBLProjectV2, PBLRuntimeActorType, PBLRuntimeEvent, PBLUiPhase } from '../types'; +import type { PBLProjectV2, PBLRuntimeActorType, PBLRuntimeEvent, PBLUiPhase } from '../../types'; export const MAX_RUNTIME_EVENTS = 500; diff --git a/lib/pbl/v2/operations/task-completion.ts b/lib/pbl/v2/operations/kernel/task-completion.ts similarity index 99% rename from lib/pbl/v2/operations/task-completion.ts rename to lib/pbl/v2/operations/kernel/task-completion.ts index 1d426ecfc..f3b96dacf 100644 --- a/lib/pbl/v2/operations/task-completion.ts +++ b/lib/pbl/v2/operations/kernel/task-completion.ts @@ -6,7 +6,7 @@ import type { PBLMicrotask, PBLPendingTaskCompletion, PBLProjectV2, -} from '../types'; +} from '../../types'; import { recordEvent } from './engagement'; import { appendRuntimeEvent, milestoneIdForMicrotask, mintRuntimeEventId } from './runtime-events'; diff --git a/lib/pbl/v2/operations/advance-patch.ts b/lib/pbl/v2/operations/runtime/advance-patch.ts similarity index 97% rename from lib/pbl/v2/operations/advance-patch.ts rename to lib/pbl/v2/operations/runtime/advance-patch.ts index f149f0bfb..3732411b5 100644 --- a/lib/pbl/v2/operations/advance-patch.ts +++ b/lib/pbl/v2/operations/runtime/advance-patch.ts @@ -8,14 +8,14 @@ * checkpoint across the server/client boundary as one contract. */ -import type { PBLAdvanceProjectPatch } from '../api/sse'; -import { capEngagementEvents } from './engagement'; -import type { PBLProjectV2, PBLRuntimeEvent } from '../types'; +import type { PBLAdvanceProjectPatch } from '../../api/sse'; +import { capEngagementEvents } from '../kernel/engagement'; +import type { PBLProjectV2, PBLRuntimeEvent } from '../../types'; import { appendRuntimeEvent, appendStatusChangedRuntimeEvent, patchStatusChangedRuntimeEventId, -} from './runtime-events'; +} from '../kernel/runtime-events'; export function buildAdvanceProjectPatch( project: PBLProjectV2, diff --git a/lib/pbl/v2/operations/completion-stats.ts b/lib/pbl/v2/operations/runtime/completion-stats.ts similarity index 99% rename from lib/pbl/v2/operations/completion-stats.ts rename to lib/pbl/v2/operations/runtime/completion-stats.ts index 642411e5c..c64f22cc1 100644 --- a/lib/pbl/v2/operations/completion-stats.ts +++ b/lib/pbl/v2/operations/runtime/completion-stats.ts @@ -7,7 +7,7 @@ * no network calls, no LLM invocation. */ -import type { PBLProjectV2, PBLScenarioActGoals } from '../types'; +import type { PBLProjectV2, PBLScenarioActGoals } from '../../types'; // --------------------------------------------------------------------------- // Types diff --git a/lib/pbl/v2/operations/dynamic-signals.ts b/lib/pbl/v2/operations/runtime/dynamic-signals.ts similarity index 96% rename from lib/pbl/v2/operations/dynamic-signals.ts rename to lib/pbl/v2/operations/runtime/dynamic-signals.ts index a9db5e6d3..ed3b604bf 100644 --- a/lib/pbl/v2/operations/dynamic-signals.ts +++ b/lib/pbl/v2/operations/runtime/dynamic-signals.ts @@ -27,7 +27,7 @@ * the rationale. */ -import { microtaskEngagement, recordEvent } from './engagement'; +import { microtaskEngagement, recordEvent } from '../kernel/engagement'; import { ensureAssessment, explicitAssessment, @@ -39,10 +39,10 @@ import { stepProficiency, updateProjectAssessment, type ProficiencyDirective, -} from './proficiency'; -import { appendProficiencyUpdatedRuntimeEvent } from './runtime-events'; -import type { PBLProjectV2, ProficiencyTransition } from '../types'; -import type { PBLSSEEvent } from '../api/sse'; +} from '../kernel/proficiency'; +import { appendProficiencyUpdatedRuntimeEvent } from '../kernel/runtime-events'; +import type { PBLProjectV2, ProficiencyTransition } from '../../types'; +import type { PBLSSEEvent } from '../../api/sse'; /** Did a transition fire on the last signal? Used by callers that * need to render the transition history outside the SSE channel diff --git a/lib/pbl/v2/operations/eval-prompts.ts b/lib/pbl/v2/operations/runtime/eval-prompts.ts similarity index 99% rename from lib/pbl/v2/operations/eval-prompts.ts rename to lib/pbl/v2/operations/runtime/eval-prompts.ts index b8b8c70d8..b992ec7fb 100644 --- a/lib/pbl/v2/operations/eval-prompts.ts +++ b/lib/pbl/v2/operations/runtime/eval-prompts.ts @@ -22,13 +22,13 @@ * end to end. */ -import { loadPBLV2Prompt } from '../prompts/loader'; -import { microtaskEngagement } from './engagement'; +import { loadPBLV2Prompt } from '../../prompts/loader'; +import { microtaskEngagement } from '../kernel/engagement'; import { scenarioActGoalsScaffold } from './completion-stats'; -import { PBL_SIMULATOR_AGENT_ID } from './progress'; +import { PBL_SIMULATOR_AGENT_ID } from '../kernel/progress'; import { listEvaluationsForMicrotask } from './evaluation'; import { summarizeLatestSubmissionForMicrotask } from './submission'; -import type { PBLMicrotask, PBLMilestone, PBLProjectV2, PBLScenarioConfig } from '../types'; +import type { PBLMicrotask, PBLMilestone, PBLProjectV2, PBLScenarioConfig } from '../../types'; export interface EvalPromptPair { system: string; diff --git a/lib/pbl/v2/operations/eval-tail-parser.ts b/lib/pbl/v2/operations/runtime/eval-tail-parser.ts similarity index 100% rename from lib/pbl/v2/operations/eval-tail-parser.ts rename to lib/pbl/v2/operations/runtime/eval-tail-parser.ts diff --git a/lib/pbl/v2/operations/evaluation.ts b/lib/pbl/v2/operations/runtime/evaluation.ts similarity index 90% rename from lib/pbl/v2/operations/evaluation.ts rename to lib/pbl/v2/operations/runtime/evaluation.ts index b3a277135..17b7df973 100644 --- a/lib/pbl/v2/operations/evaluation.ts +++ b/lib/pbl/v2/operations/runtime/evaluation.ts @@ -8,8 +8,17 @@ * later PRs. */ -import type { PBLEvaluation, PBLEvaluationKind, PBLProjectV2, PBLScenarioActGoals } from '../types'; -import { appendRuntimeEvent, milestoneIdForMicrotask, mintRuntimeEventId } from './runtime-events'; +import type { + PBLEvaluation, + PBLEvaluationKind, + PBLProjectV2, + PBLScenarioActGoals, +} from '../../types'; +import { + appendRuntimeEvent, + milestoneIdForMicrotask, + mintRuntimeEventId, +} from '../kernel/runtime-events'; function newId(prefix: string): string { return ( diff --git a/lib/pbl/v2/operations/file-validation.ts b/lib/pbl/v2/operations/runtime/file-validation.ts similarity index 100% rename from lib/pbl/v2/operations/file-validation.ts rename to lib/pbl/v2/operations/runtime/file-validation.ts diff --git a/lib/pbl/v2/operations/quiz-snapshot.ts b/lib/pbl/v2/operations/runtime/quiz-snapshot.ts similarity index 95% rename from lib/pbl/v2/operations/quiz-snapshot.ts rename to lib/pbl/v2/operations/runtime/quiz-snapshot.ts index 681f1f3ab..2dd6e9caf 100644 --- a/lib/pbl/v2/operations/quiz-snapshot.ts +++ b/lib/pbl/v2/operations/runtime/quiz-snapshot.ts @@ -24,9 +24,9 @@ import { loadQuizAttemptState, type QuizAttemptState } from '@/lib/quiz/runtime'; import type { Scene } from '@/lib/types/stage'; import { createLogger } from '@/lib/logger'; -import { applyQuizSnapshot, ensureAssessment } from './proficiency'; -import { appendProficiencyUpdatedRuntimeEvent } from './runtime-events'; -import type { PBLProjectV2, PriorQuizResult } from '../types'; +import { applyQuizSnapshot, ensureAssessment } from '../kernel/proficiency'; +import { appendProficiencyUpdatedRuntimeEvent } from '../kernel/runtime-events'; +import type { PBLProjectV2, PriorQuizResult } from '../../types'; const log = createLogger('PBLQuizSnapshot'); diff --git a/lib/pbl/v2/operations/schemas.ts b/lib/pbl/v2/operations/runtime/schemas.ts similarity index 99% rename from lib/pbl/v2/operations/schemas.ts rename to lib/pbl/v2/operations/runtime/schemas.ts index 9c1284106..e41cf5951 100644 --- a/lib/pbl/v2/operations/schemas.ts +++ b/lib/pbl/v2/operations/runtime/schemas.ts @@ -8,7 +8,7 @@ */ import { z } from 'zod'; -import type { PBLClosingQuality } from '../types'; +import type { PBLClosingQuality } from '../../types'; // --------------------------------------------------------------------------- // Argument schemas diff --git a/lib/pbl/v2/operations/submission.ts b/lib/pbl/v2/operations/runtime/submission.ts similarity index 98% rename from lib/pbl/v2/operations/submission.ts rename to lib/pbl/v2/operations/runtime/submission.ts index 25967d46e..2a54ad7a4 100644 --- a/lib/pbl/v2/operations/submission.ts +++ b/lib/pbl/v2/operations/runtime/submission.ts @@ -11,8 +11,8 @@ * paste-text handling lives in the workspace submission panel. */ -import type { PBLProjectV2, PBLSubmission, PBLSubmissionKind } from '../types'; -import { appendRuntimeEvent, mintRuntimeEventId } from './runtime-events'; +import type { PBLProjectV2, PBLSubmission, PBLSubmissionKind } from '../../types'; +import { appendRuntimeEvent, mintRuntimeEventId } from '../kernel/runtime-events'; function newId(prefix: string): string { return ( diff --git a/lib/pbl/v2/operations/workspace-launch.ts b/lib/pbl/v2/operations/runtime/workspace-launch.ts similarity index 92% rename from lib/pbl/v2/operations/workspace-launch.ts rename to lib/pbl/v2/operations/runtime/workspace-launch.ts index f8bfcada1..501fc108e 100644 --- a/lib/pbl/v2/operations/workspace-launch.ts +++ b/lib/pbl/v2/operations/runtime/workspace-launch.ts @@ -1,5 +1,5 @@ -import type { PBLProjectV2, PriorQuizResult } from '../types'; -import { transitionProjectUiPhase } from './runtime-events'; +import type { PBLProjectV2, PriorQuizResult } from '../../types'; +import { transitionProjectUiPhase } from '../kernel/runtime-events'; /** Invalidate an async Hero launch before a different scene can reuse it. */ export function invalidatePendingWorkspaceLaunch( diff --git a/lib/pbl/v2/runtime/fold.ts b/lib/pbl/v2/runtime/fold.ts index 332b338ea..597719ad1 100644 --- a/lib/pbl/v2/runtime/fold.ts +++ b/lib/pbl/v2/runtime/fold.ts @@ -6,7 +6,7 @@ import type { PBLProjectV2, PBLRuntimeEvent, } from '@/lib/pbl/v2/types'; -import { MAX_ENGAGEMENT_EVENTS } from '@/lib/pbl/v2/operations/engagement'; +import { MAX_ENGAGEMENT_EVENTS } from '@/lib/pbl/v2/operations/kernel/engagement'; import { applyLearnerState, extractLearnerState, diff --git a/lib/pbl/v2/types.ts b/lib/pbl/v2/types.ts index a8726e707..39af7eb12 100644 --- a/lib/pbl/v2/types.ts +++ b/lib/pbl/v2/types.ts @@ -621,7 +621,7 @@ export interface PBLProficiencyAssessment { * `score > +0.33` → bucket `advanced` * Hysteresis: once a tier is entered, the score must move past * the *opposite* boundary (±0.20) to leave. See - * `scoreToTier(score, currentTier)` in operations/proficiency. */ + * `scoreToTier(score, currentTier)` in operations/kernel/proficiency. */ score: number; /** `[0, 1]`. Accumulates as more signals arrive. Gates tier * switches: cannot cross a boundary while confidence < 0.4. */ @@ -644,7 +644,7 @@ export interface PBLProficiencyAssessment { } /** Snapshot of a single prior quiz scene the learner attempted. - * Aggregated by `lib/pbl/v2/operations/quiz-snapshot.ts` from + * Aggregated by `lib/pbl/v2/operations/runtime/quiz-snapshot.ts` from * `lib/quiz/persistence.ts` localStorage and piggybacked on the * `/api/pbl/v2/open-task` request when the learner first enters * the Hero. */ @@ -795,7 +795,7 @@ export interface PBLProjectV2 { proficiency: PBLProficiency; /** Adaptive proficiency state — pre-play initial assessment + the * in-PBL EWMA-updated score. Drives Instructor's tier guidance. - * See `lib/pbl/v2/operations/proficiency.ts` for the algorithm. */ + * See `lib/pbl/v2/operations/kernel/proficiency.ts` for the algorithm. */ proficiencyAssessment?: PBLProficiencyAssessment; /** ISO 639 language code from outline language inference. * BCP-47 fallback locale for deterministic platform text (e.g. diff --git a/tests/classroom/pbl-fallback-hydration.test.ts b/tests/classroom/pbl-fallback-hydration.test.ts index 9aefe8890..307612857 100644 --- a/tests/classroom/pbl-fallback-hydration.test.ts +++ b/tests/classroom/pbl-fallback-hydration.test.ts @@ -25,7 +25,7 @@ vi.mock('@/lib/utils/database', () => ({ })); import type { PBLProjectConfig } from '@/lib/pbl/types'; -import { transitionProjectUiPhase } from '@/lib/pbl/v2/operations/runtime-events'; +import { transitionProjectUiPhase } from '@/lib/pbl/v2/operations/kernel/runtime-events'; import { drainProjectRuntime } from '@/lib/pbl/v2/runtime/drain'; import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; import { diff --git a/tests/pbl/v2/advance-patch.test.ts b/tests/pbl/v2/advance-patch.test.ts index cf87f85c8..f027ee099 100644 --- a/tests/pbl/v2/advance-patch.test.ts +++ b/tests/pbl/v2/advance-patch.test.ts @@ -3,10 +3,13 @@ import { describe, expect, it } from 'vitest'; import { applyAdvanceProjectPatch, buildAdvanceProjectPatch, -} from '@/lib/pbl/v2/operations/advance-patch'; -import { advanceMicrotask, startMicrotask } from '@/lib/pbl/v2/operations/progress'; -import { recordEvent } from '@/lib/pbl/v2/operations/engagement'; -import { appendRuntimeEvent, MAX_RUNTIME_EVENTS } from '@/lib/pbl/v2/operations/runtime-events'; +} from '@/lib/pbl/v2/operations/runtime/advance-patch'; +import { advanceMicrotask, startMicrotask } from '@/lib/pbl/v2/operations/kernel/progress'; +import { recordEvent } from '@/lib/pbl/v2/operations/kernel/engagement'; +import { + appendRuntimeEvent, + MAX_RUNTIME_EVENTS, +} from '@/lib/pbl/v2/operations/kernel/runtime-events'; import type { PBLProjectV2, PBLRuntimeEvent } from '@/lib/pbl/v2/types'; function makeProject(): PBLProjectV2 { diff --git a/tests/pbl/v2/completion-stats.test.ts b/tests/pbl/v2/completion-stats.test.ts index 5ec635d41..d7e9a6f69 100644 --- a/tests/pbl/v2/completion-stats.test.ts +++ b/tests/pbl/v2/completion-stats.test.ts @@ -5,7 +5,7 @@ import { humanizeConceptSignature, normalizeActGoals, scenarioActGoalsScaffold, -} from '@/lib/pbl/v2/operations/completion-stats'; +} from '@/lib/pbl/v2/operations/runtime/completion-stats'; import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; function baseProject(): PBLProjectV2 { diff --git a/tests/pbl/v2/drain.test.ts b/tests/pbl/v2/drain.test.ts index 173c4dc2a..de44d7e82 100644 --- a/tests/pbl/v2/drain.test.ts +++ b/tests/pbl/v2/drain.test.ts @@ -19,9 +19,9 @@ import { applyInstructorEvent } from '@/components/scene-renderers/pbl/v2/apply- import { applyAdvanceProjectPatch, buildAdvanceProjectPatch, -} from '@/lib/pbl/v2/operations/advance-patch'; -import { advanceMicrotask, startMicrotask } from '@/lib/pbl/v2/operations/progress'; -import { addSubmission } from '@/lib/pbl/v2/operations/submission'; +} from '@/lib/pbl/v2/operations/runtime/advance-patch'; +import { advanceMicrotask, startMicrotask } from '@/lib/pbl/v2/operations/kernel/progress'; +import { addSubmission } from '@/lib/pbl/v2/operations/runtime/submission'; import { clearStageDrainWatermarks, drainProjectRuntime } from '@/lib/pbl/v2/runtime/drain'; import { withRuntimeStorageExclusiveLock } from '@/lib/utils/chat-storage-lock'; import { diff --git a/tests/pbl/v2/dynamic-signals.test.ts b/tests/pbl/v2/dynamic-signals.test.ts index 190e14255..4dcb741cb 100644 --- a/tests/pbl/v2/dynamic-signals.test.ts +++ b/tests/pbl/v2/dynamic-signals.test.ts @@ -17,8 +17,8 @@ import { trackMicrotaskCompletion, trackObservation, trackSubmissionScore, -} from '@/lib/pbl/v2/operations/dynamic-signals'; -import { emptyAssessment } from '@/lib/pbl/v2/operations/proficiency'; +} from '@/lib/pbl/v2/operations/runtime/dynamic-signals'; +import { emptyAssessment } from '@/lib/pbl/v2/operations/kernel/proficiency'; import type { PBLProficiencyAssessment } from '@/lib/pbl/v2/types'; import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; diff --git a/tests/pbl/v2/eval-prompts.test.ts b/tests/pbl/v2/eval-prompts.test.ts index db24af4ad..55f6fc059 100644 --- a/tests/pbl/v2/eval-prompts.test.ts +++ b/tests/pbl/v2/eval-prompts.test.ts @@ -15,10 +15,10 @@ import { buildTaskEvalPrompt, formatProjectEngagementRollup, formatProjectSynthesisChecks, -} from '@/lib/pbl/v2/operations/eval-prompts'; -import { addSubmission } from '@/lib/pbl/v2/operations/submission'; -import { addEvaluation } from '@/lib/pbl/v2/operations/evaluation'; -import { recordEvent } from '@/lib/pbl/v2/operations/engagement'; +} from '@/lib/pbl/v2/operations/runtime/eval-prompts'; +import { addSubmission } from '@/lib/pbl/v2/operations/runtime/submission'; +import { addEvaluation } from '@/lib/pbl/v2/operations/runtime/evaluation'; +import { recordEvent } from '@/lib/pbl/v2/operations/kernel/engagement'; import type { PBLEngagementSummary, PBLMicrotask, diff --git a/tests/pbl/v2/eval-tail-parser.test.ts b/tests/pbl/v2/eval-tail-parser.test.ts index 81e9861cf..99be1273e 100644 --- a/tests/pbl/v2/eval-tail-parser.test.ts +++ b/tests/pbl/v2/eval-tail-parser.test.ts @@ -16,7 +16,7 @@ import { sanitizeMilestoneEvaluationFeedback, stripEvaluationTail, stripTemplatePlaceholders, -} from '@/lib/pbl/v2/operations/eval-tail-parser'; +} from '@/lib/pbl/v2/operations/runtime/eval-tail-parser'; describe('parseEvaluationTail', () => { it('returns null on empty / blank input', () => { diff --git a/tests/pbl/v2/evaluator.test.ts b/tests/pbl/v2/evaluator.test.ts index 56b7b0a1b..fedd3f199 100644 --- a/tests/pbl/v2/evaluator.test.ts +++ b/tests/pbl/v2/evaluator.test.ts @@ -24,7 +24,7 @@ import { runTaskEvaluation, } from '@/lib/pbl/v2/agents/evaluator'; import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; -import { addSubmission } from '@/lib/pbl/v2/operations/submission'; +import { addSubmission } from '@/lib/pbl/v2/operations/runtime/submission'; import type { PBLSSEEvent } from '@/lib/pbl/v2/api/sse'; type DoStreamConfig = NonNullable< diff --git a/tests/pbl/v2/file-validation.test.ts b/tests/pbl/v2/file-validation.test.ts index bd170271f..1f9ddbb9b 100644 --- a/tests/pbl/v2/file-validation.test.ts +++ b/tests/pbl/v2/file-validation.test.ts @@ -4,7 +4,7 @@ import { TEXT_FILE_EXTENSIONS, TEXT_FILE_ACCEPT, isValidTextFile, -} from '@/lib/pbl/v2/operations/file-validation'; +} from '@/lib/pbl/v2/operations/runtime/file-validation'; function file(name: string, type = ''): File { return new File(['dummy content'], name, { type }); diff --git a/tests/pbl/v2/fixtures/planner-system-ordinary.txt b/tests/pbl/v2/fixtures/planner-system-ordinary.txt new file mode 100644 index 000000000..242ffc473 --- /dev/null +++ b/tests/pbl/v2/fixtures/planner-system-ordinary.txt @@ -0,0 +1,149 @@ +You are the Planner of a Project-Based Learning (PBL) course module on the OpenMAIC platform. + +Your job: from the outline information the platform has already produced, **autonomously** design a complete, ready-to-run learning project for the student. The student will not be consulted during this design phase — by the time they reach the PBL scene, the project must already exist as a coherent, scaffolded plan. + +You are a **project designer**, not a course-outline generator. Slides and quizzes teach; your PBL scene turns that learning into a coherent project with a beginning, a middle, and an end. + +## What the platform gives you + +For the current PBL scene, the platform has already inferred: + +- **Project topic**: Neighborhood Air Quality Dashboard +- **Project description (what students will build)**: Analyze sensor readings and communicate the most important pattern. +- **Target skills (what students will learn)**: data cleaning, chart design, evidence-based explanation +- **Suggested milestone count**: 4 +- **Student proficiency tier (decided by the platform's adaptive engine)**: intermediate + +It also gives you the **course context** — the titles and descriptions of every other scene in the same course, in playback order: + +- [1] SLIDE: Reading Environmental Data + Distinguish measurements, trends, and anomalies. +- [2] PBL: Neighborhood Air Quality Dashboard ← this PBL scene + Build a dashboard that explains local air-quality patterns. + +Read the course context as **source material**, not as a checklist to copy. The slides / quizzes / interactive scenes **before** this PBL teach the prerequisites; the scenes **after** this PBL (if any) build on the project's outcome. Your project should follow naturally from what was just taught and give the learner a concrete project outcome the rest of the course can refer back to. + +Do **not** turn the course outline into another mini-course. If the course context says "concept A → operation B → review C", your project is not "learn A, learn B, review C". Your project is a purposeful path where the learner investigates, decides, sets up, builds, tests, presents, or reflects as needed to complete one coherent outcome. That outcome might be code, a document, a plan, a research question, a configured environment, an analysis, a presentation, a decision, or another domain-appropriate product. + +## Actual ordinary PBL workspace — text-only contract + +The ordinary PBL workspace gives the learner: + +- left: milestone/task roadmap +- center: Instructor chat +- right: current task submission area where they can paste text or upload their own work + +It does **NOT** provide a right-side briefing tab, resource tab, reference drawer, preloaded image, attached PDF, starter file download, or built-in dataset. Therefore the project must be completable from the visible milestone/task/instructor text plus the learner's own external tools. If a task needs a tiny sample dataset, prompt template, constraints list, scenario facts, rubric, or starter content, include that material directly inside the milestone/task text. Never tell the learner to open/read/view/download/inspect a provided resource that is not written in the tool-call text. + +## What you must produce + +A complete PBL project consisting of: + +1. **Project info** — title, short description, the explicit `learningObjective` (the verb the student will master; distinct from "what they will build"), and `gains`. The description must name the project outcome the student is working toward. `gains` is a SHORT list of **3-5 learner-facing "what you'll gain" statements** shown on the project Hero. Each names ONE **ability, awareness, or piece of knowledge the learner BUILDS by working through the project** — what they take away and can do afterwards — written as a readable phrase or short sentence in the project language. Typically expand each terse `Target skills` entry above into plain competency language. **Critically: a gain is NOT the project's final deliverable/result** (that belongs in `description`), NOT a task title, and NOT a single terse keyword. For a game-theory project, good gains are "理解纳什均衡的含义并能在具体场景中求解" / "学会用收益矩阵刻画双方策略与收益" / "培养把现实冲突抽象成博弈模型的建模意识" — NOT "完成一份定价方案" (that is the deliverable). +2. **Milestones** — major phases of the project. Aim for the suggested milestone count. Each milestone must have: + - A clear, action-oriented title + - A 1-2 sentence description + - A `briefing` (what the Instructor will say at the start of this milestone) + - A `completionCriteria` (how the Instructor will know the student is done) + - A `debrief` (what the Instructor will say at the end) + - **Optional** `coreConcept` — set this on **only the 1-2 stages that carry the project's CORE knowledge point**. It is a short description (in the project language) of the central concept that stage teaches, e.g. "为什么循环能避免重复代码". When set, the Instructor runs ONE integrative reverse-question about that concept at the end of the stage. **Leave it unset for ordinary setup / build / polish stages** — over-using it makes the learner feel interrogated. Most projects mark just one stage. +3. **Microtasks** — within each milestone, 2-4 specific, actionable steps the student will do. Each microtask must have: + - A title and 1-2 sentence description + - 1-3 hints the Instructor can offer if the student gets stuck (see Hard rules 11-15: hints/descriptions must guide not solve, leave the learner real choices, stay right-sized, and the final milestone must end on a consolidation step) +4. **Roles** — exactly **one Instructor** (always required). Do **not** create any other role type. For the Instructor provide: + - `name` — the guide title the learner sees. Use a SHORT **descriptive guide title tied to THIS project's topic**, ending in a guide word in the project language (教练 / 导师 / mentor / coach / etc.) — e.g. "排序项目教练", "RAG 项目导师", "数据可视化教练". Do **NOT** use a generic "Instructor" / "AI", and do **NOT** invent a personal human name (e.g. "林岚", "Alex"). + - `description` — a SHORT, **learner-facing introduction** shown as a hover tooltip on the instructor's avatar. Write it **TO the learner**, in the project language, in **2-3 short sentences max**. Say who the guide is (use the name), that they accompany the learner through the whole project and each task, that the learner can ask them anything at any time, and that they give feedback and check understanding along the way. Keep it warm and concrete to THIS project's topic. Do **NOT** expose internal mechanics or capabilities (reading conversation history, tool calls, "stage assessment / evaluation", scoring, advancing tasks, etc.) — include only what is meaningful and reassuring to a learner. + - `systemPrompt` — the Instructor's internal persona / voice that drives its behaviour. This is **NOT shown to the learner**; richer role detail lives here. +5. **No hidden resources** — if the student needs a primer, cheat sheet, starter content, constraints, sample rows, or reference material, put the minimal material directly in the relevant milestone or microtask text so it is visible without any separate resource UI. + +## Hard rules + +1. **Content language — strict, applies to EVERY field of EVERY tool call.** + Follow this content-language policy: **`Reply in English and keep technical terms precise.`**. + - If the policy is a BCP-47 locale code (`zh-CN` = 简体中文, `zh-TW` = 繁體中文, `en-US` = English, `ja-JP` = 日本語, `ru-RU` = Русский, `ar-SA` = العربية), reply ONLY in that language. + - If the policy contains nuanced natural-language instruction (e.g. "中文为主,英文技术术语保留原文"), follow it literally — the specific guidance takes priority over any default locale assumption. + EVERY text field you produce — `title`, `description`, `learningObjective`, every item in `gains`, role `name` / `description` / `systemPrompt`, every milestone's `title` / `description` / `briefing` / `completionCriteria` / `debrief`, every microtask's `title` / `description` / `hints` — must follow this policy. + Code samples, API names, and well-known technical terms (e.g. `HashMap`, `pandas`, `React`) stay in their native form within the otherwise localised prose. + + Classroom language context (may be empty or duplicative of the policy above): `Reply in English and keep technical terms precise.`. + +2. **Stay on the actual project topic — no template substitution.** The `set_project_info(title, description, learningObjective, gains)` fields must be **strictly derived from the outline's project metadata above** (`Project topic`, `Project description`, `Target skills`). You may rephrase, tighten, or translate, but you must NOT replace the topic with a different "common teaching project" from your training data. Same for every milestone / microtask / hint: they must serve THIS topic. + +3. **Project, not lesson sequence.** The project must have a named outcome and the milestones should feel like stages of doing that project, not a second lecture outline. + - Good shape: clarify the goal / gather inputs / set up tools / make decisions / build or draft / test or critique / revise / present or reflect, depending on the domain. + - Valid project steps include understanding requirements, researching references, choosing tools, installing or configuring software, defining a research question, planning an approach, checking assumptions, reviewing progress, and reflecting with the Instructor. + - Bad shape: "understand the concept" → "learn the operation" → "review what you learned" with no coherent project outcome tying the steps together. + +4. **Never call `ask_user`**. There is no `ask_user` tool. The student is not in this loop. +5. **No "skeleton confirmation"**. You do not pause for any approval; you design end-to-end in one pass and finish by calling `mark_design_complete`. +6. **Use the proficiency tier the platform has already decided** — `intermediate`. The platform's adaptive engine combines outline difficulty cues, prior-scene difficulty, student bio, and (later) quiz accuracy + in-PBL behaviour signals to pick this tier. Trust it; pass the same value through when you call `set_project_info`. + + Adapt the project depth to that tier: + - `beginner` → break tasks into smaller, more concrete steps; provide more hints; assume no prior knowledge of the specific tools + - `intermediate` → assume basic familiarity with the topic; tasks can be slightly broader + - `advanced` → assume strong familiarity; tasks can be high-level, fewer hints +7. **Keep scope tight**. A learner should be able to finish the project in a sitting (typically 15-45 minutes of guided work). When in doubt, prefer fewer, deeper microtasks over many shallow ones. +8. **The Instructor's voice is "warm coach, not lecturer"**. When you write `briefing` / `completionCriteria` / `debrief`, write them in the Instructor's voice — directly addressing the student in second person. +9. **Microtasks must build on each other**. Earlier ones create context, decisions, setup, materials, attempts, or reflections that later ones use. No floating tasks. +10. **Reference the course context**. If a prior scene taught a specific concept, microtasks can rely on it without re-teaching. If a later scene depends on the project's output, end on something that connects to it. + +11. **Hints and descriptions GUIDE, never SOLVE — this is the #1 failure to avoid.** + A hint or microtask `description` must NEVER contain the literal token the learner is meant to type: no method/function name, no operator, no syntax template, no exact variable name, no ready-to-paste line of code. State the GOAL and point at the concept; make the learner recall or look it up. + Apply this test to EVERY hint and description before writing it: *"Could the learner copy this straight into their editor and pass the task?"* If yes, rewrite it as a question or a where-to-look pointer. + Bad → Good: + - ❌ `"试试 unique = set(orders)"` → ✅ `"哪种数据结构天然不允许重复?怎么把列表转换过去?"` + - ❌ `"用 len() 数一下"` → ✅ `"怎样得到去重后还剩多少个元素?"` + - ❌ `"格式类似 d['新键'] = 值"` → ✅ `"给字典一个还不存在的键赋值,会发生什么?"` + - ❌ `"Python 有个方法叫 capitalize"` → ✅ `"字符串有没有内置方法能把首字母变大写?查查文档。"` + - ❌ `"用 += 累加 total"` / `"用 f-string 输出"` → ✅ `"每轮循环怎样把当前值加到总和上?"` / `"怎样把变量值拼进一句话输出?"` + The ban covers COMPLETE expressions, statements, and control-flow scaffolding too — not just method names: + - ❌ `"试着直接写 print('关键词' in 变量名)"` → ✅ `"Python 有个关键字能判断一个词是否在字符串里(结果是布尔值),是哪个?"` + - ❌ `"用 for comment in comments: 开始循环"` → ✅ `"怎样让程序对列表里的每一条都重复同样的处理?"` + - ❌ `"先判断 if not comment.strip():,再 continue 跳过"` → ✅ `"清洗后怎样识别一条其实是空的评论并跳过它?"` + Naming a library to INSTALL or a concept to UNDERSTAND is fine; handing the exact line / method / operator / loop / conditional that completes the step is not. + +12. **Leave the learner real choices (agency).** Do NOT dictate every variable name, exact output wording, or specific data value. In each milestone give the learner at least one genuine decision: their own sample data, their own scenario, their own naming, or which of several valid approaches to try. Every-token-dictated = a fill-in-the-blank worksheet, not a project. + +13. **Right-sized microtasks — no trivial fragmentation, no mega-tasks.** Each microtask is ONE substantive step that produces or demonstrates something real. NEVER make `"打印结果"` / `"运行一下"` / `"print 出来"` its own microtask — fold display and a quick check into the step that produced the thing. Likewise do NOT split a chain of trivial one-liners into separate tasks (e.g. "定义字符串" / "调用 strip" / "转小写" as three microtasks → combine into one "准备并规整你的样本数据"). Do not bundle several unrelated goals into one task either. 2-4 meaningful microtasks per milestone — prefer fewer, deeper steps. + +14. **End with consolidation — every project needs a real "done".** The FINAL milestone MUST contain a closing microtask that consolidates the whole project: run the complete product end-to-end, test it against at least one input (include an obvious edge case when the domain has one — e.g. an empty list), and/or a short reflection tying the pieces together — converging on ONE nameable deliverable the learner SEES working. A congratulatory `debrief` is NOT closure on its own. + +15. **Build phases, not lecture chapters.** Milestones are stages of building the product, not a concept-by-concept syllabus. `"布尔基础 → 逻辑运算 → if/else"` is a textbook outline; `"设定规则输入 → 组合出准入规则 → 根据判断给出结果"` is a project. If milestone titles read like chapter headings, reshape them around what the learner DOES. + +16. **Every task has a concrete, judgeable "done" definition (the design→runtime contract).** For each microtask, the `description` must make clear WHAT the learner produces / demonstrates / decides and what "done well" looks like — this written done-definition IS the contract the runtime advance + feedback depend on; leave it implicit and scoring drifts. Do NOT read "done" literally — judge it on TWO axes: (A) the task's NATURE and (B) its DELIVERY FORM (see rule 17). Classify the nature and write the done-criteria to match: + - **Convergent** (one checkable right answer: code runs, calculation correct, fact right) → done = correct / works. + - **Gradable-open** (no single answer, but clear better/worse by domain standards — a decision + its rationale, an argument's strength, an analysis, a plan) → done = quality of reasoning + meeting the domain criteria; you MUST STATE the criteria that separate a strong response from a weak one. This is neither "one correct answer" NOR "any stance passes". Most skill / analysis / decision tasks live here. + - **Open-reflective** (genuinely no right/wrong: an ethical stance, interpretation, reflection) → done = depth / honesty of thinking + a clearly stated position; NEVER "matched the expected answer". + ✘ Forbidden: vague tasks ("了解X" / "探索Y") with no checkable done-state; a gradable-open task with no stated criteria; a description or hint that hands the full answer. + +17. **Never manufacture a fake deliverable for open work.** "Must be evaluable" does NOT mean forcing a tangible artifact onto open / reflective work (e.g. a mandatory 500-word report or a quiz tacked onto a discussion). Design gradable-open as "make a real decision / take a position + justify it" with the domain rubric; design open-reflective as a stance / decision+rationale / plan / refined question / structured reflection, judged on reasoning. A truly outcome-less chat topic is a poor PBL fit — if you must, give it a process destination (explore angles → weigh the tensions → land on a personal view and have the learner state it). Match the DELIVERY FORM to the work — artifact (checkable product) / argument (written reasoning trace) / performance (a graceful action in a situated interaction) — and label the task's nature correctly; do not force a convergent shell onto open work or vice versa. +18. **Text-only resource grounding.** Do NOT mention a right-side briefing, resource panel, reference tab, preloaded image, screenshot, PDF, attachment, downloadable starter file, or provided dataset. If the learner needs information, make it visible in `briefing`, `completionCriteria`, `debrief`, a microtask `description`, or a `hint`. If the learner needs data, either ask them to create a small sample themselves or give the sample inline as text. If you write "read the following/below/given brief/material/case/dataset" or "阅读下面/以下/给定/提供的简报/资料/材料/案例/数据", the actual brief/material must appear immediately in that same visible text — do not refer to an implied brief that is not written out. + +## Tool workflow + +You have these tools. Call them in this order. There is no "mode" you must switch to; the platform tracks state for you. + +1. `set_project_info(title, description, learningObjective, gains, proficiency)` — exactly once +2. `add_role({ type: 'instructor', name, description, systemPrompt })` — exactly once +3. For each milestone (in order): + 1. `add_milestone(title, description, briefing, completionCriteria, debrief, coreConcept?)` — returns the milestone ID (`coreConcept` only on the 1-2 core-knowledge stages) + 2. For each microtask in that milestone (in order): `add_microtask(milestoneId, title, description, hints, order)` +4. `mark_design_complete()` — exactly once, at the very end + +If you call a tool with invalid arguments, the platform will return an error; correct the arguments and retry. Do not write narrative text between tool calls — the agentic loop is silent design, not chat. + +## A worked example shape (for calibration only — do not echo it back) + +If the topic is "Build a Python CSV analyser" with `beginner` proficiency: + +- Milestone 1 "Read the data" with microtasks "Open the CSV in pandas" / "Inspect the columns" / "Spot the data quality issues" +- Milestone 2 "Clean and aggregate" with microtasks "Handle missing values" / "Group by month" / "Sum the revenue" +- Milestone 3 "Visualise and report" with microtasks "Plot the monthly trend" / "Write 3 sentences summarising the finding" + +The shape is small and sequential. Some steps produce artefacts; some steps prepare the learner, clarify choices, configure tools, or check the work. The whole project still has a coherent outcome. + +Counterexample to avoid: "Milestone 1: Learn what a CSV is; Milestone 2: Learn grouping; Milestone 3: Review charts." That is a course outline, not a project. + + + +Now design the project for the platform. \ No newline at end of file diff --git a/tests/pbl/v2/fixtures/planner-system-scenario.txt b/tests/pbl/v2/fixtures/planner-system-scenario.txt new file mode 100644 index 000000000..ba79c30ae --- /dev/null +++ b/tests/pbl/v2/fixtures/planner-system-scenario.txt @@ -0,0 +1,195 @@ +You are the Planner of a Project-Based Learning (PBL) course module on the OpenMAIC platform. + +Your job: from the outline information the platform has already produced, **autonomously** design a complete, ready-to-run learning project for the student. The student will not be consulted during this design phase — by the time they reach the PBL scene, the project must already exist as a coherent, scaffolded plan. + +You are a **project designer**, not a course-outline generator. Slides and quizzes teach; your PBL scene turns that learning into a coherent project with a beginning, a middle, and an end. + +## What the platform gives you + +For the current PBL scene, the platform has already inferred: + +- **Project topic**: Museum Donor Negotiation +- **Project description (what students will build)**: Negotiate exhibit support without compromising curatorial independence. +- **Target skills (what students will learn)**: active listening, reframing, principled negotiation +- **Suggested milestone count**: 3 +- **Student proficiency tier (decided by the platform's adaptive engine)**: advanced + +It also gives you the **course context** — the titles and descriptions of every other scene in the same course, in playback order: + +- [3] PBL: Museum Donor Negotiation ← this PBL scene + Practice a high-stakes conversation with a prospective donor. + +Read the course context as **source material**, not as a checklist to copy. The slides / quizzes / interactive scenes **before** this PBL teach the prerequisites; the scenes **after** this PBL (if any) build on the project's outcome. Your project should follow naturally from what was just taught and give the learner a concrete project outcome the rest of the course can refer back to. + +Do **not** turn the course outline into another mini-course. If the course context says "concept A → operation B → review C", your project is not "learn A, learn B, review C". Your project is a purposeful path where the learner investigates, decides, sets up, builds, tests, presents, or reflects as needed to complete one coherent outcome. That outcome might be code, a document, a plan, a research question, a configured environment, an analysis, a presentation, a decision, or another domain-appropriate product. + +## Actual ordinary PBL workspace — text-only contract + +The ordinary PBL workspace gives the learner: + +- left: milestone/task roadmap +- center: Instructor chat +- right: current task submission area where they can paste text or upload their own work + +It does **NOT** provide a right-side briefing tab, resource tab, reference drawer, preloaded image, attached PDF, starter file download, or built-in dataset. Therefore the project must be completable from the visible milestone/task/instructor text plus the learner's own external tools. If a task needs a tiny sample dataset, prompt template, constraints list, scenario facts, rubric, or starter content, include that material directly inside the milestone/task text. Never tell the learner to open/read/view/download/inspect a provided resource that is not written in the tool-call text. + +## What you must produce + +A complete PBL project consisting of: + +1. **Project info** — title, short description, the explicit `learningObjective` (the verb the student will master; distinct from "what they will build"), and `gains`. The description must name the project outcome the student is working toward. `gains` is a SHORT list of **3-5 learner-facing "what you'll gain" statements** shown on the project Hero. Each names ONE **ability, awareness, or piece of knowledge the learner BUILDS by working through the project** — what they take away and can do afterwards — written as a readable phrase or short sentence in the project language. Typically expand each terse `Target skills` entry above into plain competency language. **Critically: a gain is NOT the project's final deliverable/result** (that belongs in `description`), NOT a task title, and NOT a single terse keyword. For a game-theory project, good gains are "理解纳什均衡的含义并能在具体场景中求解" / "学会用收益矩阵刻画双方策略与收益" / "培养把现实冲突抽象成博弈模型的建模意识" — NOT "完成一份定价方案" (that is the deliverable). +2. **Milestones** — major phases of the project. Aim for the suggested milestone count. Each milestone must have: + - A clear, action-oriented title + - A 1-2 sentence description + - A `briefing` (what the Instructor will say at the start of this milestone) + - A `completionCriteria` (how the Instructor will know the student is done) + - A `debrief` (what the Instructor will say at the end) + - **Optional** `coreConcept` — set this on **only the 1-2 stages that carry the project's CORE knowledge point**. It is a short description (in the project language) of the central concept that stage teaches, e.g. "为什么循环能避免重复代码". When set, the Instructor runs ONE integrative reverse-question about that concept at the end of the stage. **Leave it unset for ordinary setup / build / polish stages** — over-using it makes the learner feel interrogated. Most projects mark just one stage. +3. **Microtasks** — within each milestone, 2-4 specific, actionable steps the student will do. Each microtask must have: + - A title and 1-2 sentence description + - 1-3 hints the Instructor can offer if the student gets stuck (see Hard rules 11-15: hints/descriptions must guide not solve, leave the learner real choices, stay right-sized, and the final milestone must end on a consolidation step) +4. **Roles** — exactly **one Instructor** (always required). Do **not** create any other role type. For the Instructor provide: + - `name` — the guide title the learner sees. Use a SHORT **descriptive guide title tied to THIS project's topic**, ending in a guide word in the project language (教练 / 导师 / mentor / coach / etc.) — e.g. "排序项目教练", "RAG 项目导师", "数据可视化教练". Do **NOT** use a generic "Instructor" / "AI", and do **NOT** invent a personal human name (e.g. "林岚", "Alex"). + - `description` — a SHORT, **learner-facing introduction** shown as a hover tooltip on the instructor's avatar. Write it **TO the learner**, in the project language, in **2-3 short sentences max**. Say who the guide is (use the name), that they accompany the learner through the whole project and each task, that the learner can ask them anything at any time, and that they give feedback and check understanding along the way. Keep it warm and concrete to THIS project's topic. Do **NOT** expose internal mechanics or capabilities (reading conversation history, tool calls, "stage assessment / evaluation", scoring, advancing tasks, etc.) — include only what is meaningful and reassuring to a learner. + - `systemPrompt` — the Instructor's internal persona / voice that drives its behaviour. This is **NOT shown to the learner**; richer role detail lives here. +5. **No hidden resources** — if the student needs a primer, cheat sheet, starter content, constraints, sample rows, or reference material, put the minimal material directly in the relevant milestone or microtask text so it is visible without any separate resource UI. + +## Hard rules + +1. **Content language — strict, applies to EVERY field of EVERY tool call.** + Follow this content-language policy: **`Use Simplified Chinese, preserving standard English negotiation terms.`**. + - If the policy is a BCP-47 locale code (`zh-CN` = 简体中文, `zh-TW` = 繁體中文, `en-US` = English, `ja-JP` = 日本語, `ru-RU` = Русский, `ar-SA` = العربية), reply ONLY in that language. + - If the policy contains nuanced natural-language instruction (e.g. "中文为主,英文技术术语保留原文"), follow it literally — the specific guidance takes priority over any default locale assumption. + EVERY text field you produce — `title`, `description`, `learningObjective`, every item in `gains`, role `name` / `description` / `systemPrompt`, every milestone's `title` / `description` / `briefing` / `completionCriteria` / `debrief`, every microtask's `title` / `description` / `hints` — must follow this policy. + Code samples, API names, and well-known technical terms (e.g. `HashMap`, `pandas`, `React`) stay in their native form within the otherwise localised prose. + + Classroom language context (may be empty or duplicative of the policy above): `Use Simplified Chinese, preserving standard English negotiation terms.`. + +2. **Stay on the actual project topic — no template substitution.** The `set_project_info(title, description, learningObjective, gains)` fields must be **strictly derived from the outline's project metadata above** (`Project topic`, `Project description`, `Target skills`). You may rephrase, tighten, or translate, but you must NOT replace the topic with a different "common teaching project" from your training data. Same for every milestone / microtask / hint: they must serve THIS topic. + +3. **Project, not lesson sequence.** The project must have a named outcome and the milestones should feel like stages of doing that project, not a second lecture outline. + - Good shape: clarify the goal / gather inputs / set up tools / make decisions / build or draft / test or critique / revise / present or reflect, depending on the domain. + - Valid project steps include understanding requirements, researching references, choosing tools, installing or configuring software, defining a research question, planning an approach, checking assumptions, reviewing progress, and reflecting with the Instructor. + - Bad shape: "understand the concept" → "learn the operation" → "review what you learned" with no coherent project outcome tying the steps together. + +4. **Never call `ask_user`**. There is no `ask_user` tool. The student is not in this loop. +5. **No "skeleton confirmation"**. You do not pause for any approval; you design end-to-end in one pass and finish by calling `mark_design_complete`. +6. **Use the proficiency tier the platform has already decided** — `advanced`. The platform's adaptive engine combines outline difficulty cues, prior-scene difficulty, student bio, and (later) quiz accuracy + in-PBL behaviour signals to pick this tier. Trust it; pass the same value through when you call `set_project_info`. + + Adapt the project depth to that tier: + - `beginner` → break tasks into smaller, more concrete steps; provide more hints; assume no prior knowledge of the specific tools + - `intermediate` → assume basic familiarity with the topic; tasks can be slightly broader + - `advanced` → assume strong familiarity; tasks can be high-level, fewer hints +7. **Keep scope tight**. A learner should be able to finish the project in a sitting (typically 15-45 minutes of guided work). When in doubt, prefer fewer, deeper microtasks over many shallow ones. +8. **The Instructor's voice is "warm coach, not lecturer"**. When you write `briefing` / `completionCriteria` / `debrief`, write them in the Instructor's voice — directly addressing the student in second person. +9. **Microtasks must build on each other**. Earlier ones create context, decisions, setup, materials, attempts, or reflections that later ones use. No floating tasks. +10. **Reference the course context**. If a prior scene taught a specific concept, microtasks can rely on it without re-teaching. If a later scene depends on the project's output, end on something that connects to it. + +11. **Hints and descriptions GUIDE, never SOLVE — this is the #1 failure to avoid.** + A hint or microtask `description` must NEVER contain the literal token the learner is meant to type: no method/function name, no operator, no syntax template, no exact variable name, no ready-to-paste line of code. State the GOAL and point at the concept; make the learner recall or look it up. + Apply this test to EVERY hint and description before writing it: *"Could the learner copy this straight into their editor and pass the task?"* If yes, rewrite it as a question or a where-to-look pointer. + Bad → Good: + - ❌ `"试试 unique = set(orders)"` → ✅ `"哪种数据结构天然不允许重复?怎么把列表转换过去?"` + - ❌ `"用 len() 数一下"` → ✅ `"怎样得到去重后还剩多少个元素?"` + - ❌ `"格式类似 d['新键'] = 值"` → ✅ `"给字典一个还不存在的键赋值,会发生什么?"` + - ❌ `"Python 有个方法叫 capitalize"` → ✅ `"字符串有没有内置方法能把首字母变大写?查查文档。"` + - ❌ `"用 += 累加 total"` / `"用 f-string 输出"` → ✅ `"每轮循环怎样把当前值加到总和上?"` / `"怎样把变量值拼进一句话输出?"` + The ban covers COMPLETE expressions, statements, and control-flow scaffolding too — not just method names: + - ❌ `"试着直接写 print('关键词' in 变量名)"` → ✅ `"Python 有个关键字能判断一个词是否在字符串里(结果是布尔值),是哪个?"` + - ❌ `"用 for comment in comments: 开始循环"` → ✅ `"怎样让程序对列表里的每一条都重复同样的处理?"` + - ❌ `"先判断 if not comment.strip():,再 continue 跳过"` → ✅ `"清洗后怎样识别一条其实是空的评论并跳过它?"` + Naming a library to INSTALL or a concept to UNDERSTAND is fine; handing the exact line / method / operator / loop / conditional that completes the step is not. + +12. **Leave the learner real choices (agency).** Do NOT dictate every variable name, exact output wording, or specific data value. In each milestone give the learner at least one genuine decision: their own sample data, their own scenario, their own naming, or which of several valid approaches to try. Every-token-dictated = a fill-in-the-blank worksheet, not a project. + +13. **Right-sized microtasks — no trivial fragmentation, no mega-tasks.** Each microtask is ONE substantive step that produces or demonstrates something real. NEVER make `"打印结果"` / `"运行一下"` / `"print 出来"` its own microtask — fold display and a quick check into the step that produced the thing. Likewise do NOT split a chain of trivial one-liners into separate tasks (e.g. "定义字符串" / "调用 strip" / "转小写" as three microtasks → combine into one "准备并规整你的样本数据"). Do not bundle several unrelated goals into one task either. 2-4 meaningful microtasks per milestone — prefer fewer, deeper steps. + +14. **End with consolidation — every project needs a real "done".** The FINAL milestone MUST contain a closing microtask that consolidates the whole project: run the complete product end-to-end, test it against at least one input (include an obvious edge case when the domain has one — e.g. an empty list), and/or a short reflection tying the pieces together — converging on ONE nameable deliverable the learner SEES working. A congratulatory `debrief` is NOT closure on its own. + +15. **Build phases, not lecture chapters.** Milestones are stages of building the product, not a concept-by-concept syllabus. `"布尔基础 → 逻辑运算 → if/else"` is a textbook outline; `"设定规则输入 → 组合出准入规则 → 根据判断给出结果"` is a project. If milestone titles read like chapter headings, reshape them around what the learner DOES. + +16. **Every task has a concrete, judgeable "done" definition (the design→runtime contract).** For each microtask, the `description` must make clear WHAT the learner produces / demonstrates / decides and what "done well" looks like — this written done-definition IS the contract the runtime advance + feedback depend on; leave it implicit and scoring drifts. Do NOT read "done" literally — judge it on TWO axes: (A) the task's NATURE and (B) its DELIVERY FORM (see rule 17). Classify the nature and write the done-criteria to match: + - **Convergent** (one checkable right answer: code runs, calculation correct, fact right) → done = correct / works. + - **Gradable-open** (no single answer, but clear better/worse by domain standards — a decision + its rationale, an argument's strength, an analysis, a plan) → done = quality of reasoning + meeting the domain criteria; you MUST STATE the criteria that separate a strong response from a weak one. This is neither "one correct answer" NOR "any stance passes". Most skill / analysis / decision tasks live here. + - **Open-reflective** (genuinely no right/wrong: an ethical stance, interpretation, reflection) → done = depth / honesty of thinking + a clearly stated position; NEVER "matched the expected answer". + ✘ Forbidden: vague tasks ("了解X" / "探索Y") with no checkable done-state; a gradable-open task with no stated criteria; a description or hint that hands the full answer. + +17. **Never manufacture a fake deliverable for open work.** "Must be evaluable" does NOT mean forcing a tangible artifact onto open / reflective work (e.g. a mandatory 500-word report or a quiz tacked onto a discussion). Design gradable-open as "make a real decision / take a position + justify it" with the domain rubric; design open-reflective as a stance / decision+rationale / plan / refined question / structured reflection, judged on reasoning. A truly outcome-less chat topic is a poor PBL fit — if you must, give it a process destination (explore angles → weigh the tensions → land on a personal view and have the learner state it). Match the DELIVERY FORM to the work — artifact (checkable product) / argument (written reasoning trace) / performance (a graceful action in a situated interaction) — and label the task's nature correctly; do not force a convergent shell onto open work or vice versa. +18. **Text-only resource grounding.** Do NOT mention a right-side briefing, resource panel, reference tab, preloaded image, screenshot, PDF, attachment, downloadable starter file, or provided dataset. If the learner needs information, make it visible in `briefing`, `completionCriteria`, `debrief`, a microtask `description`, or a `hint`. If the learner needs data, either ask them to create a small sample themselves or give the sample inline as text. If you write "read the following/below/given brief/material/case/dataset" or "阅读下面/以下/给定/提供的简报/资料/材料/案例/数据", the actual brief/material must appear immediately in that same visible text — do not refer to an implied brief that is not written out. + +## Tool workflow + +You have these tools. Call them in this order. There is no "mode" you must switch to; the platform tracks state for you. + +1. `set_project_info(title, description, learningObjective, gains, proficiency)` — exactly once +2. `add_role({ type: 'instructor', name, description, systemPrompt })` — exactly once +3. For each milestone (in order): + 1. `add_milestone(title, description, briefing, completionCriteria, debrief, coreConcept?)` — returns the milestone ID (`coreConcept` only on the 1-2 core-knowledge stages) + 2. For each microtask in that milestone (in order): `add_microtask(milestoneId, title, description, hints, order)` +4. `mark_design_complete()` — exactly once, at the very end + +If you call a tool with invalid arguments, the platform will return an error; correct the arguments and retry. Do not write narrative text between tool calls — the agentic loop is silent design, not chat. + +## A worked example shape (for calibration only — do not echo it back) + +If the topic is "Build a Python CSV analyser" with `beginner` proficiency: + +- Milestone 1 "Read the data" with microtasks "Open the CSV in pandas" / "Inspect the columns" / "Spot the data quality issues" +- Milestone 2 "Clean and aggregate" with microtasks "Handle missing values" / "Group by month" / "Sum the revenue" +- Milestone 3 "Visualise and report" with microtasks "Plot the monthly trend" / "Write 3 sentences summarising the finding" + +The shape is small and sequential. Some steps produce artefacts; some steps prepare the learner, clarify choices, configure tools, or check the work. The whole project still has a coherent outcome. + +Counterexample to avoid: "Milestone 1: Learn what a CSV is; Milestone 2: Learn grouping; Milestone 3: Review charts." That is a course outline, not a project. + +## SCENARIO MODE — role-play scenario (this project only) + +This PBL is a **role-play scenario**: the learner will step into a concrete situation and interact in-character with character(s) played by a separate Simulator agent. You author the WHOLE scenario now (it is frozen into the package); the runtime only produces the live dialogue. Two rules above all: (a) the premise is **given and concrete**, introduced to the learner by the Instructor — the learner must NEVER be asked to guess it; (b) every task must serve the real learning goal (how to do the thing well), not meta-guessing. + +Scenario brief from the platform: A long-time donor wants naming control over a new exhibit; preserve the relationship and the museum's independence. + +### Step A — fully author the scenario with `set_scenario(...)` +Call **exactly once, right after `set_project_info` and before any `add_milestone`**: +- `setting`: the concrete overall premise / what is going on (in the project language). +- `goal` (optional): what the learner is practising. +- `rules` (optional but REQUIRED whenever the scenario has any defined rule-set — games / interviews / debates / structured negotiations / etc.): write the CONCRETE rules a newcomer needs to actually take part, specific enough that the Instructor can teach them verbatim in prep. Not a vague label — include the real mechanics (e.g. a card game: hand ranking, betting rounds, blinds, what terms like Pot Odds / Fold / Call / Raise / Check mean; a debate: the motion, each side's stance, the speaking format; an interview: the rounds and what each assesses). Omit ONLY for free scenarios with no special rules (e.g. comforting a friend). +- `learnerRole` (optional): the learner's OWN role/position (e.g. "you are their close friend" / "you are the 5th player, on the button"). +- `characters`: **EXACTLY ONE character** — this version plays a single counterpart throughout (the runtime only ever voices one). It needs `name`, `persona` (stable identity / relationship / personality / speaking style), **`situation`** (their CONCRETE current circumstance the learner faces — e.g. "just broke up last week, low mood, says they're fine but aren't"). `situation` is shown to the learner up front (prep intro + the always-visible scenario briefing), so it must hold ONLY what the learner can see/know at the start — keep any fact a later beat is meant to make them uncover OUT of it (see the No-spoilers rule below). Plus strongly-recommended `boundaries` (hard safety rails), and optional `openingLine`. + +### Step B — lay out the FIXED three-stage skeleton (milestones in this exact order) +1. **Prep stage** — `add_milestone({ ..., scenarioStage: "prep" })` as the FIRST milestone. Its `briefing` is the Instructor intro that **introduces the concrete premise to the learner**: the situation, each character's `situation`, what the learner is there to do, plus `rules` / `learnerRole` when present. The intro MUST match the roleplay stages you design next. Give prep **exactly ONE light microtask** (e.g. "了解背景,准备开始" / "Understand the setup, ready to begin") — NO assessment; the learner just confirms and advances. **Do NOT set `coreConcept`.** +2. **Roleplay stage(s)** — one or MORE `add_milestone({ ..., scenarioStage: "roleplay" })` in the middle (split a long scenario into several roleplay stages by round/phase to avoid one giant stage). Each roleplay milestone's `briefing` brings the learner into the scene. **Design the beats as a DRAMATIC ARC, not a flat checklist**: an opening hook → rising stakes/complication → a turning point or decision → a resolution. Each beat should be a MEANINGFUL decision/action unit (something the learner can actually DO), never empty filler. For **each microtask (beat)** under a roleplay milestone, provide: + - `description`: the CONCRETE situation of this beat as the SYSTEM narrator states it to the learner — positions / cards / what just happened / whose turn (e.g. "你在 Button 位拿到 A♠ J♦;前面都 Fold,老周在 Cutoff 加注到 6 个筹码;轮到你决定 preflop"). The character NEVER states these facts — the system does; the character only reacts. Keep it factual scene-setting, not coaching. + - `successWhen` (REQUIRED for every roleplay beat): the CONCRETE, OBSERVABLE in-scene action the learner must SAY or DO for this beat to count as done — the scenario's "deliverable" (e.g. "做出 preflop 决定:跟注、加注或弃牌" / "对对方说出的感受做出共情回应,并问一个跟进问题"). State it in plain SCENE terms (what they do in the fiction), NOT as a teaching goal. This is exactly what the advance detector watches, so a crisp `successWhen` is what stops off-topic / small-talk turns from advancing the scene. Make it a real decision/action, not "they chatted a bit". + - `completionCriteria` (optional, legacy): a teaching-side note on what this beat is about; `successWhen` is preferred and takes precedence for advancing. + - `characterObjective` (recommended): what the character PRIVATELY wants — and privately KNOWS — this beat: their in-scene drive (e.g. "试探你是否在虚张声势" / "想确认你是否真的在乎"), plus any fact the learner is meant to UNCOVER this beat (the hidden cause / secret / backstory the character only reveals when probed — e.g. "你昨天在空调房待了很久才着凉,但只有被仔细询问才说出来"). It makes the character pursue a goal and hold its secrets in character; it is private to the character — NEVER narrated, shown in the briefing, evaluated, or coached. + - `skillFocus` (recommended): the single skill this beat practises (e.g. "底池赔率判断" / "积极倾听"). Surfaced to the learner (current-task panel + end-of-project per-act review); never spoken by the character. + - The scene is FREE-FIRST: the learner always speaks/types their OWN response to the character, which is how a real interaction is practised. (Some beats may instead ask the learner to hand in a real artefact, e.g. "write them a letter".) + - `narration` (optional): a short neutral scene-setting line the SYSTEM reads when this beat opens (e.g. "你们走进了一家安静的咖啡厅"). NEVER spoken by a character or the Instructor. All scene/state facts come from the system (narration + description), never from the character's mouth. + - `hints` (recommended for roleplay beats): 1–2 SHORT, learner-facing coaching tips for THIS beat — what skill to focus on or how to handle it well (e.g. "先共情、再问问题,别急着给建议" / "注意你的位置和底池赔率,再决定下注"). They appear in the "hints" card of the learner's current-task side panel, are NEVER spoken by the character, and are the learner's in-the-moment guidance. Keep them concrete to this beat, not generic. + - **Do NOT set `coreConcept`** on roleplay milestones. +3. **Wrapup stage** — `add_milestone({ ..., scenarioStage: "wrapup" })` as the LAST milestone. Its `debrief` holds the Instructor's light, encouraging feedback points (highlights / one thing to improve); the detailed report lives on the completion page. Give wrapup **exactly ONE light microtask** (e.g. "听取反馈,收尾" / "Hear the feedback, wrap up"). **Do NOT set `coreConcept`.** + +### Rules for scenario design +- The premise (situation / rules / positions) is GIVEN and introduced in prep — **never make a task that asks the learner to guess/invent it**. +- **No spoilers — never give away in learner-VISIBLE text what a beat is designed to make the learner discover.** The premise the learner can see up front — `setting`, each character's `situation` / `persona`, the prep `briefing`, and each beat's `description` / `narration` (and the always-visible scenario briefing built from these) — must contain ONLY what the learner already knows or can plainly observe at the outset. If any roleplay beat's `successWhen` requires the learner to UNCOVER something through the interaction (a hidden cause, a motive, a secret, the diagnosis, a backstory fact), that information MUST NOT appear in any learner-visible field. Put it ONLY in that beat's private `characterObjective`, where the character holds it and reveals it solely when the learner actually probes for it — never up front. E.g. a "find out why" beat: the real cause lives in `characterObjective`; `situation` states only the visible symptoms / where the character is right now. +- All scenario text (`setting`/`persona`/`situation`/`briefing`/narration/options…) follows the same content-language policy as Hard rule 1. +- Scene beats should feel like a real interaction unfolding, not a checklist; 2-4 beats per roleplay stage is plenty. +- **The roleplay character is a pure in-world participant, NEVER a coach.** When you write `persona` / `situation` / `openingLine` (and any character-facing text), the character must have its OWN motives and react like a real person in the scene. It must NEVER: ask the learner to explain/justify their reasoning ("说说你为什么这么选" / "一句话给我理由"), evaluate or grade the learner's moves ("这步打得对"), give strategy/meta hints ("想想我的范围里哪些牌会付钱"), or tell the learner it's their turn / what to decide. That is all out-of-scene/teaching content and it does NOT belong in the character's mouth. +- **Out-of-scene content has its own channels — never the character:** (a) the "this is a training table / I'll test you" framing and any rule teaching belong to the **prep Instructor** (`briefing`); (b) "it's your turn to act / a decision point has arrived" belongs to the **system `narration`** of that beat, stated neutrally; (c) strategy / what-to-watch-for hints belong to the microtask **`hints`** (side panel). Route each of these to its channel; the character only ever lives the scene. + + +### Step C — author ONE project-wide scene visual with `set_scene_visual(...)` +After ALL roleplay milestones/beats exist, call `set_scene_visual` exactly once. Read back over EVERY roleplay stage/task you just wrote and distil the ONE shared place/atmosphere they all happen in, then describe it: a `caption` (a short phrase in the project language fitting all stages — derived from the real tasks, e.g. "深夜,各自房间隔着手机聊到天亮" / "决赛辩论赛场" / "牌桌现金局"), a 3-colour `palette` (`bg1`/`bg2`/`accent` hex matching the mood), and 2–4 `motifs` (emoji that evoke this exact scene). Make it specific to THIS project — never a generic placeholder. + +### Scenario tool workflow (supersedes the order above for this project) +1. `set_project_info(...)` +2. `set_scenario({ setting, goal?, rules?, learnerRole?, characters })` +3. `add_role({ type: "instructor", ... })` +4. `add_milestone({ scenarioStage: "prep" })` + its one light microtask +5. one or more `add_milestone({ scenarioStage: "roleplay" })` + their beats as a dramatic arc (successWhen [required] / characterObjective / skillFocus / narration?) +6. `add_milestone({ scenarioStage: "wrapup" })` + its one light microtask +7. `set_scene_visual({ caption, bg1, bg2, accent, motifs })` — based on all the roleplay stages above +8. `mark_design_complete()` + +Now design the project for the platform. \ No newline at end of file diff --git a/tests/pbl/v2/fold.test.ts b/tests/pbl/v2/fold.test.ts index 71f3009c8..55147cacc 100644 --- a/tests/pbl/v2/fold.test.ts +++ b/tests/pbl/v2/fold.test.ts @@ -2,8 +2,8 @@ import { describe, expect, it } from 'vitest'; import type { RuntimeRecord } from '@openmaic/dsl'; import { foldPBLRuntime } from '@/lib/pbl/v2/runtime/fold'; -import { MAX_ENGAGEMENT_EVENTS } from '@/lib/pbl/v2/operations/engagement'; -import { emptyAssessment } from '@/lib/pbl/v2/operations/proficiency'; +import { MAX_ENGAGEMENT_EVENTS } from '@/lib/pbl/v2/operations/kernel/engagement'; +import { emptyAssessment } from '@/lib/pbl/v2/operations/kernel/proficiency'; import { PBL_RUNTIME_PAYLOAD_VERSION, type PBLRuntimeStorePayload, diff --git a/tests/pbl/v2/hydration.test.ts b/tests/pbl/v2/hydration.test.ts index f5d52bff7..05343a496 100644 --- a/tests/pbl/v2/hydration.test.ts +++ b/tests/pbl/v2/hydration.test.ts @@ -10,20 +10,20 @@ import { BrowserRuntimeStore, type RuntimeSessionInit, type RuntimeStore } from import { applyInstructorEvent } from '@/components/scene-renderers/pbl/v2/apply-instructor-event'; import type { PBLProjectConfig } from '@/lib/pbl/types'; -import { recordEvent } from '@/lib/pbl/v2/operations/engagement'; -import { addEvaluation } from '@/lib/pbl/v2/operations/evaluation'; +import { recordEvent } from '@/lib/pbl/v2/operations/kernel/engagement'; +import { addEvaluation } from '@/lib/pbl/v2/operations/runtime/evaluation'; import { advanceMicrotask, continueAfterHandover, resetProjectProgress, startMicrotask, -} from '@/lib/pbl/v2/operations/progress'; -import { transitionProjectUiPhase } from '@/lib/pbl/v2/operations/runtime-events'; -import { addSubmission } from '@/lib/pbl/v2/operations/submission'; +} from '@/lib/pbl/v2/operations/kernel/progress'; +import { transitionProjectUiPhase } from '@/lib/pbl/v2/operations/kernel/runtime-events'; +import { addSubmission } from '@/lib/pbl/v2/operations/runtime/submission'; import { clearPendingTaskCompletion, setPendingTaskCompletion, -} from '@/lib/pbl/v2/operations/task-completion'; +} from '@/lib/pbl/v2/operations/kernel/task-completion'; import { drainProjectRuntime } from '@/lib/pbl/v2/runtime/drain'; import { foldPBLRuntime } from '@/lib/pbl/v2/runtime/fold'; import { diff --git a/tests/pbl/v2/instructor.test.ts b/tests/pbl/v2/instructor.test.ts index 3e1b57d5d..6508961f3 100644 --- a/tests/pbl/v2/instructor.test.ts +++ b/tests/pbl/v2/instructor.test.ts @@ -20,7 +20,7 @@ import { microtaskEngagement, milestoneSynthesisSatisfied, recordEvent, -} from '@/lib/pbl/v2/operations/engagement'; +} from '@/lib/pbl/v2/operations/kernel/engagement'; import type { PBLMilestone, PBLProjectV2 } from '@/lib/pbl/v2/types'; const now = '2026-05-29T00:00:00.000Z'; diff --git a/tests/pbl/v2/learner-state.test.ts b/tests/pbl/v2/learner-state.test.ts index 771a0d105..2f53ef1b7 100644 --- a/tests/pbl/v2/learner-state.test.ts +++ b/tests/pbl/v2/learner-state.test.ts @@ -5,7 +5,7 @@ import { extractLearnerState, stripToDesignTemplate, } from '@/lib/pbl/v2/runtime/learner-state'; -import { emptyAssessment } from '@/lib/pbl/v2/operations/proficiency'; +import { emptyAssessment } from '@/lib/pbl/v2/operations/kernel/proficiency'; import type { PBLProjectV2, PBLRuntimeEvent } from '@/lib/pbl/v2/types'; type ProjectFieldBoundary = 'learner-state' | 'design-template' | 'transient'; diff --git a/tests/pbl/v2/microtask-engagement-cache.test.ts b/tests/pbl/v2/microtask-engagement-cache.test.ts index 06c2110a4..2f4fa3fff 100644 --- a/tests/pbl/v2/microtask-engagement-cache.test.ts +++ b/tests/pbl/v2/microtask-engagement-cache.test.ts @@ -12,8 +12,8 @@ * always has data. These tests pin the cache behaviour. */ import { describe, expect, it } from 'vitest'; -import { advanceMicrotask } from '@/lib/pbl/v2/operations/progress'; -import { recordEvent } from '@/lib/pbl/v2/operations/engagement'; +import { advanceMicrotask } from '@/lib/pbl/v2/operations/kernel/progress'; +import { recordEvent } from '@/lib/pbl/v2/operations/kernel/engagement'; import type { PBLMicrotask, PBLMilestone, PBLProjectV2 } from '@/lib/pbl/v2/types'; function mkTask(id: string, status: PBLMicrotask['status'] = 'in_progress'): PBLMicrotask { diff --git a/tests/pbl/v2/planner-core-prompt-golden.test.ts b/tests/pbl/v2/planner-core-prompt-golden.test.ts new file mode 100644 index 000000000..71c668e96 --- /dev/null +++ b/tests/pbl/v2/planner-core-prompt-golden.test.ts @@ -0,0 +1,105 @@ +/** + * Golden planner-prompt parity for the #1062 core extraction. + * + * The fixture files were produced with Vitest's `-u` snapshot mode from the + * ORIGINAL `buildPlannerSystemPrompt` implementation in `planner.ts` at base + * commit 9d268943, before that implementation moved to `planner-core.ts`. + * These assertions intentionally compare the complete strings byte-for-byte. + */ +import { describe, expect, it } from 'vitest'; + +import { buildPlannerSystemPrompt } from '@/lib/pbl/v2/agents/planner-core'; +import type { PBLPlannerV2Input } from '@/lib/pbl/v2/types'; +import type { SceneOutline } from '@/lib/types/generation'; + +function ordinaryInput(): PBLPlannerV2Input { + const outline: SceneOutline = { + id: 'outline-pbl-golden-ordinary', + type: 'pbl', + title: 'Neighborhood Air Quality Dashboard', + description: 'Build a dashboard that explains local air-quality patterns.', + keyPoints: ['data cleaning', 'visualization', 'public communication'], + teachingObjective: 'Turn environmental observations into a clear public explanation.', + order: 2, + pblConfig: { + projectTopic: 'Neighborhood Air Quality Dashboard', + projectDescription: 'Analyze sensor readings and communicate the most important pattern.', + targetSkills: ['data cleaning', 'chart design', 'evidence-based explanation'], + issueCount: 4, + }, + }; + + return { + outline, + courseContext: { + allOutlines: [ + { + id: 'outline-slide-golden', + type: 'slide', + title: 'Reading Environmental Data', + description: 'Distinguish measurements, trends, and anomalies.', + keyPoints: ['measurements', 'trends'], + teachingObjective: 'Read a small environmental dataset.', + order: 1, + }, + outline, + ], + languageDirective: 'Reply in English and keep technical terms precise.', + }, + targetLanguage: 'en-US', + }; +} + +function scenarioInput(): PBLPlannerV2Input { + const outline: SceneOutline = { + id: 'outline-pbl-golden-scenario', + type: 'pbl', + title: 'Museum Donor Negotiation', + description: 'Practice a high-stakes conversation with a prospective donor.', + keyPoints: ['active listening', 'framing', 'negotiation'], + teachingObjective: 'Use careful questions and framing to reach a principled agreement.', + order: 3, + pblConfig: { + projectTopic: 'Museum Donor Negotiation', + projectDescription: 'Negotiate exhibit support without compromising curatorial independence.', + targetSkills: ['active listening', 'reframing', 'principled negotiation'], + issueCount: 3, + scenarioRoleplay: true, + scenarioBrief: + "A long-time donor wants naming control over a new exhibit; preserve the relationship and the museum's independence.", + }, + }; + + return { + outline, + courseContext: { + allOutlines: [outline], + languageDirective: 'Use Simplified Chinese, preserving standard English negotiation terms.', + }, + targetLanguage: 'zh-CN', + }; +} + +describe('PBL v2 planner core — pre-refactor prompt goldens', () => { + it('keeps the ordinary planner system prompt byte-identical', async () => { + const prompt = await buildPlannerSystemPrompt( + ordinaryInput(), + 'intermediate', + 'Reply in English and keep technical terms precise.', + false, + ); + + await expect(prompt).toMatchFileSnapshot('./fixtures/planner-system-ordinary.txt'); + }); + + it('keeps the scenario planner system prompt byte-identical', async () => { + const prompt = await buildPlannerSystemPrompt( + scenarioInput(), + 'advanced', + 'Use Simplified Chinese, preserving standard English negotiation terms.', + true, + ); + + await expect(prompt).toMatchFileSnapshot('./fixtures/planner-system-scenario.txt'); + }); +}); diff --git a/tests/pbl/v2/planner-single-call.test.ts b/tests/pbl/v2/planner-single-call.test.ts index 1dc655c38..133413592 100644 --- a/tests/pbl/v2/planner-single-call.test.ts +++ b/tests/pbl/v2/planner-single-call.test.ts @@ -11,9 +11,10 @@ import { describe, it, expect } from 'vitest'; import { MockLanguageModelV3 } from 'ai/test'; +import { callLLM } from '@/lib/ai/llm'; import { generatePBLV2ProjectSingleCall } from '@/lib/pbl/v2/agents/planner-single-call'; -import { PlannerV2Error } from '@/lib/pbl/v2/agents/planner'; -import { PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/progress'; +import { PlannerV2Error } from '@/lib/pbl/v2/agents/planner-core'; +import { PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/kernel/progress'; import type { SceneOutline } from '@/lib/types/generation'; import type { PBLPlannerV2Input } from '@/lib/pbl/v2/types'; @@ -139,7 +140,11 @@ function validOutput(overrides?: { proficiency?: string; coreConcept?: string }) describe('PBL v2 single-call planner — happy path', () => { it('parses + hydrates a complete project from one JSON response', async () => { - const project = await generatePBLV2ProjectSingleCall(plannerInput(), textModel(validOutput())); + const project = await generatePBLV2ProjectSingleCall( + plannerInput(), + textModel(validOutput()), + callLLM, + ); expect(project.title).toBe('CSV Data Analyzer project'); expect(project.status).toBe('active'); @@ -191,6 +196,7 @@ describe('PBL v2 single-call planner — happy path', () => { const project = await generatePBLV2ProjectSingleCall( plannerInput(), textModel(validOutput({ coreConcept: 'why a DataFrame beats raw rows' })), + callLLM, ); expect(project.milestones[0].synthesisCheck?.coreConcept).toBe( 'why a DataFrame beats raw rows', @@ -201,13 +207,18 @@ describe('PBL v2 single-call planner — happy path', () => { const project = await generatePBLV2ProjectSingleCall( plannerInput(), textModel(validOutput({ proficiency: 'advanced' })), + callLLM, ); expect(project.proficiency).toBe('advanced'); }); it('parses output even when the model wraps it in ```json fences', async () => { const fenced = '```json\n' + validOutput() + '\n```'; - const project = await generatePBLV2ProjectSingleCall(plannerInput(), textModel(fenced)); + const project = await generatePBLV2ProjectSingleCall( + plannerInput(), + textModel(fenced), + callLLM, + ); expect(project.milestones).toHaveLength(2); expect(project.title).toBe('CSV Data Analyzer project'); }); @@ -227,7 +238,11 @@ describe('PBL v2 single-call planner — guards + retry', () => { }); await expect( - generatePBLV2ProjectSingleCall(plannerInput(), textModel(noMilestones, noMilestones)), + generatePBLV2ProjectSingleCall( + plannerInput(), + textModel(noMilestones, noMilestones), + callLLM, + ), ).rejects.toBeInstanceOf(PlannerV2Error); }); @@ -236,6 +251,7 @@ describe('PBL v2 single-call planner — guards + retry', () => { generatePBLV2ProjectSingleCall( plannerInput(), textModel('Sorry, I cannot help with that.', 'Still not JSON.'), + callLLM, ), ).rejects.toBeInstanceOf(PlannerV2Error); }); @@ -243,7 +259,7 @@ describe('PBL v2 single-call planner — guards + retry', () => { it('throws PlannerV2Error when outline.pblConfig is missing (before any LLM call)', async () => { const input = plannerInput({ outline: pblOutline({ pblConfig: undefined }) }); await expect( - generatePBLV2ProjectSingleCall(input, textModel(validOutput())), + generatePBLV2ProjectSingleCall(input, textModel(validOutput()), callLLM), ).rejects.toBeInstanceOf(PlannerV2Error); }); @@ -252,7 +268,7 @@ describe('PBL v2 single-call planner — guards + retry', () => { delete noGains.projectInfo.gains; const text = JSON.stringify(noGains); await expect( - generatePBLV2ProjectSingleCall(plannerInput(), textModel(text, text)), + generatePBLV2ProjectSingleCall(plannerInput(), textModel(text, text), callLLM), ).rejects.toBeInstanceOf(PlannerV2Error); }); @@ -261,7 +277,7 @@ describe('PBL v2 single-call planner — guards + retry', () => { fewGains.projectInfo.gains = ['Only one gain']; const text = JSON.stringify(fewGains); await expect( - generatePBLV2ProjectSingleCall(plannerInput(), textModel(text, text)), + generatePBLV2ProjectSingleCall(plannerInput(), textModel(text, text), callLLM), ).rejects.toBeInstanceOf(PlannerV2Error); }); @@ -272,7 +288,7 @@ describe('PBL v2 single-call planner — guards + retry', () => { // Must reject with the PlannerV2Error contract so the caller falls back // cleanly — a TypeError from `.forEach` would escape that contract. await expect( - generatePBLV2ProjectSingleCall(plannerInput(), textModel(text, text)), + generatePBLV2ProjectSingleCall(plannerInput(), textModel(text, text), callLLM), ).rejects.toBeInstanceOf(PlannerV2Error); }); @@ -281,7 +297,7 @@ describe('PBL v2 single-call planner — guards + retry', () => { badScalar.projectInfo.title = 123; // schema drift: number where a string is expected const text = JSON.stringify(badScalar); await expect( - generatePBLV2ProjectSingleCall(plannerInput(), textModel(text, text)), + generatePBLV2ProjectSingleCall(plannerInput(), textModel(text, text), callLLM), ).rejects.toBeInstanceOf(PlannerV2Error); }); @@ -291,6 +307,7 @@ describe('PBL v2 single-call planner — guards + retry', () => { const project = await generatePBLV2ProjectSingleCall( plannerInput(), textModel(JSON.stringify(drift)), + callLLM, ); // Malformed hints coerce to [] — no throw. expect(project.milestones[0].microtasks[0].hints).toEqual([]); @@ -301,13 +318,17 @@ describe('PBL v2 single-call planner — guards + retry', () => { // Model insists on `advanced` both times — never matches the beginner lock. const advanced = validOutput({ proficiency: 'advanced' }); await expect( - generatePBLV2ProjectSingleCall(lockedInput, textModel(advanced, advanced)), + generatePBLV2ProjectSingleCall(lockedInput, textModel(advanced, advanced), callLLM), ).rejects.toBeInstanceOf(PlannerV2Error); }); it('accepts a matching proficiency under an explicit learner-level lock', async () => { const lockedInput = plannerInput({ user: { requirement: '我是零基础' } }); - const project = await generatePBLV2ProjectSingleCall(lockedInput, textModel(validOutput())); + const project = await generatePBLV2ProjectSingleCall( + lockedInput, + textModel(validOutput()), + callLLM, + ); expect(project.proficiency).toBe('beginner'); }); @@ -319,6 +340,7 @@ describe('PBL v2 single-call planner — guards + retry', () => { const project = await generatePBLV2ProjectSingleCall( plannerInput(), textModel(JSON.stringify(drift)), + callLLM, ); expect(project.milestones[1].documents).toBeUndefined(); }); @@ -456,6 +478,7 @@ describe('PBL v2 single-call planner — scenario roleplay', () => { const project = await generatePBLV2ProjectSingleCall( scenarioInput(), textModel(validScenarioOutput()), + callLLM, ); // Scenario frozen onto the project + schema stamped. @@ -495,7 +518,7 @@ describe('PBL v2 single-call planner — scenario roleplay', () => { it('throws PlannerV2Error when the scenario has no characters', async () => { const text = validScenarioOutput({ dropCharacters: true }); await expect( - generatePBLV2ProjectSingleCall(scenarioInput(), textModel(text, text)), + generatePBLV2ProjectSingleCall(scenarioInput(), textModel(text, text), callLLM), ).rejects.toBeInstanceOf(PlannerV2Error); }); @@ -504,7 +527,7 @@ describe('PBL v2 single-call planner — scenario roleplay', () => { delete drift.milestones[1].microtasks[0].successWhen; const text = JSON.stringify(drift); await expect( - generatePBLV2ProjectSingleCall(scenarioInput(), textModel(text, text)), + generatePBLV2ProjectSingleCall(scenarioInput(), textModel(text, text), callLLM), ).rejects.toBeInstanceOf(PlannerV2Error); }); @@ -513,7 +536,7 @@ describe('PBL v2 single-call planner — scenario roleplay', () => { drift.milestones[2].scenarioStage = 'roleplay'; // last is no longer wrapup const text = JSON.stringify(drift); await expect( - generatePBLV2ProjectSingleCall(scenarioInput(), textModel(text, text)), + generatePBLV2ProjectSingleCall(scenarioInput(), textModel(text, text), callLLM), ).rejects.toBeInstanceOf(PlannerV2Error); }); }); diff --git a/tests/pbl/v2/planner.test.ts b/tests/pbl/v2/planner.test.ts index b923b03c6..11c4297d0 100644 --- a/tests/pbl/v2/planner.test.ts +++ b/tests/pbl/v2/planner.test.ts @@ -16,15 +16,17 @@ import { describe, it, expect } from 'vitest'; import { generatePBLV2Project, + plannerStepHasAcceptedCompletion, +} from '@/lib/pbl/v2/agents/planner'; +import { PlannerV2Error, plannerCompletionGaps, - plannerStepHasAcceptedCompletion, normalizeSynthesisChecks, buildScenarioDesignBlock, MAX_SYNTHESIS_STAGES, type PlannerV2Callbacks, type PlannerV2ProgressEvent, -} from '@/lib/pbl/v2/agents/planner'; +} from '@/lib/pbl/v2/agents/planner-core'; import type { StepResult, ToolSet } from 'ai'; describe('PBL v2 — scenario design block (free-first dialogue)', () => { @@ -220,9 +222,9 @@ describe('PBL v2 Planner — error paths (no LLM needed)', () => { // the first thing it does. Using `as never` keeps the test // payload honest (we are deliberately violating the contract to // observe the guard). - await expect(generatePBLV2Project(input, undefined as never)).rejects.toBeInstanceOf( - PlannerV2Error, - ); + await expect( + generatePBLV2Project(input, undefined as never, undefined as never), + ).rejects.toBeInstanceOf(PlannerV2Error); }); }); @@ -463,6 +465,7 @@ describe('PBL v2 Planner — targetLanguage overrides detection (UI locale path) await generatePBLV2Project( { ...input, outline: { ...outline, pblConfig: undefined } }, undefined as never, + undefined as never, ); } catch (err) { // `partial` was built via emptyProject(input), which now reads @@ -496,7 +499,7 @@ describe('PBL v2 Planner — targetLanguage overrides detection (UI locale path) // No targetLanguage → language stays '' (no content-based locale guessing). }; try { - await generatePBLV2Project(input, undefined as never); + await generatePBLV2Project(input, undefined as never, undefined as never); } catch (err) { const partial = (err as PlannerV2Error).partial; expect(partial.language).toBe(''); @@ -522,7 +525,7 @@ describe('PBL v2 Planner — targetLanguage overrides detection (UI locale path) targetLanguage: ' ', }; try { - await generatePBLV2Project(input, undefined as never); + await generatePBLV2Project(input, undefined as never, undefined as never); } catch (err) { const partial = (err as PlannerV2Error).partial; // whitespace targetLanguage → '' (content is NOT scanned for a locale) diff --git a/tests/pbl/v2/proficiency.test.ts b/tests/pbl/v2/proficiency.test.ts index 17a573ec5..1bd323e9c 100644 --- a/tests/pbl/v2/proficiency.test.ts +++ b/tests/pbl/v2/proficiency.test.ts @@ -36,7 +36,7 @@ import { signalFromTaskSpeed, tickTurn, updateProjectAssessment, -} from '@/lib/pbl/v2/operations/proficiency'; +} from '@/lib/pbl/v2/operations/kernel/proficiency'; import type { PBLProficiencyAssessment, PBLProjectV2, PriorQuizResult } from '@/lib/pbl/v2/types'; import type { SceneOutline } from '@/lib/types/generation'; diff --git a/tests/pbl/v2/progress.test.ts b/tests/pbl/v2/progress.test.ts index 4e02aac49..05b34da1e 100644 --- a/tests/pbl/v2/progress.test.ts +++ b/tests/pbl/v2/progress.test.ts @@ -18,7 +18,7 @@ import { hasStartedProject, resetProjectProgress, completeRoleplayAct, -} from '@/lib/pbl/v2/operations/progress'; +} from '@/lib/pbl/v2/operations/kernel/progress'; import { appendTaskCompletionReadyMessage, currentPendingTaskCompletion, @@ -27,9 +27,9 @@ import { setPendingTaskCompletion, taskCompletionReadyText, taskEvaluationCanComplete, -} from '@/lib/pbl/v2/operations/task-completion'; -import { recordEvent } from '@/lib/pbl/v2/operations/engagement'; -import { runtimeEventEpoch } from '@/lib/pbl/v2/operations/runtime-events'; +} from '@/lib/pbl/v2/operations/kernel/task-completion'; +import { recordEvent } from '@/lib/pbl/v2/operations/kernel/engagement'; +import { runtimeEventEpoch } from '@/lib/pbl/v2/operations/kernel/runtime-events'; import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; function makeProject(): PBLProjectV2 { diff --git a/tests/pbl/v2/quiz-snapshot.test.ts b/tests/pbl/v2/quiz-snapshot.test.ts index 92e8c3d0e..f0eb5dfaf 100644 --- a/tests/pbl/v2/quiz-snapshot.test.ts +++ b/tests/pbl/v2/quiz-snapshot.test.ts @@ -11,8 +11,8 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; import { applyQuizSignalsToProject, buildQuizSnapshot, -} from '@/lib/pbl/v2/operations/quiz-snapshot'; -import { emptyAssessment } from '@/lib/pbl/v2/operations/proficiency'; +} from '@/lib/pbl/v2/operations/runtime/quiz-snapshot'; +import { emptyAssessment } from '@/lib/pbl/v2/operations/kernel/proficiency'; import type { PBLProjectV2, PriorQuizResult } from '@/lib/pbl/v2/types'; import type { Scene } from '@/lib/types/stage'; diff --git a/tests/pbl/v2/runtime-events.test.ts b/tests/pbl/v2/runtime-events.test.ts index 27659161d..15e6f6962 100644 --- a/tests/pbl/v2/runtime-events.test.ts +++ b/tests/pbl/v2/runtime-events.test.ts @@ -1,23 +1,23 @@ import { describe, expect, it } from 'vitest'; import { applyInstructorEvent } from '@/components/scene-renderers/pbl/v2/apply-instructor-event'; -import { applyAdvanceProjectPatch } from '@/lib/pbl/v2/operations/advance-patch'; -import { trackSubmissionScore } from '@/lib/pbl/v2/operations/dynamic-signals'; -import { emptyAssessment } from '@/lib/pbl/v2/operations/proficiency'; +import { applyAdvanceProjectPatch } from '@/lib/pbl/v2/operations/runtime/advance-patch'; +import { trackSubmissionScore } from '@/lib/pbl/v2/operations/runtime/dynamic-signals'; +import { emptyAssessment } from '@/lib/pbl/v2/operations/kernel/proficiency'; import { advanceMicrotask, continueAfterHandover, startMicrotask, -} from '@/lib/pbl/v2/operations/progress'; -import { applyQuizSignalsToProject } from '@/lib/pbl/v2/operations/quiz-snapshot'; -import { appendRuntimeEvent } from '@/lib/pbl/v2/operations/runtime-events'; -import { addSubmission } from '@/lib/pbl/v2/operations/submission'; +} from '@/lib/pbl/v2/operations/kernel/progress'; +import { applyQuizSignalsToProject } from '@/lib/pbl/v2/operations/runtime/quiz-snapshot'; +import { appendRuntimeEvent } from '@/lib/pbl/v2/operations/kernel/runtime-events'; +import { addSubmission } from '@/lib/pbl/v2/operations/runtime/submission'; import { appendTaskCompletionReadyMessage, clearPendingTaskCompletion, setPendingTaskCompletion, -} from '@/lib/pbl/v2/operations/task-completion'; -import { prepareWorkspaceLaunchProject } from '@/lib/pbl/v2/operations/workspace-launch'; +} from '@/lib/pbl/v2/operations/kernel/task-completion'; +import { prepareWorkspaceLaunchProject } from '@/lib/pbl/v2/operations/runtime/workspace-launch'; import type { PBLProjectV2, PBLRuntimeEvent, PriorQuizResult } from '@/lib/pbl/v2/types'; function makeProject(overrides: Partial = {}): PBLProjectV2 { diff --git a/tests/pbl/v2/runtime-llm-entry.test.ts b/tests/pbl/v2/runtime-llm-entry.test.ts index d74205649..24857c77f 100644 --- a/tests/pbl/v2/runtime-llm-entry.test.ts +++ b/tests/pbl/v2/runtime-llm-entry.test.ts @@ -36,7 +36,7 @@ vi.mock('@/lib/server/usage-storage', () => ({ import { thinkingContext } from '@/lib/ai/thinking-context'; import { runTaskEvaluation } from '@/lib/pbl/v2/agents/evaluator'; import { runSimulatorTurn } from '@/lib/pbl/v2/agents/simulator'; -import { addSubmission } from '@/lib/pbl/v2/operations/submission'; +import { addSubmission } from '@/lib/pbl/v2/operations/runtime/submission'; import type { PBLProjectV2 } from '@/lib/pbl/v2/types'; import type { PBLSSEEvent } from '@/lib/pbl/v2/api/sse'; import type { ThinkingConfig } from '@/lib/types/provider'; diff --git a/tests/pbl/v2/runtime-model-pin.test.ts b/tests/pbl/v2/runtime-model-pin.test.ts index 8a9a3c78d..f75123038 100644 --- a/tests/pbl/v2/runtime-model-pin.test.ts +++ b/tests/pbl/v2/runtime-model-pin.test.ts @@ -31,7 +31,7 @@ vi.mock('@/lib/pbl/v2/agents/evaluator', () => ({ runMilestoneEvaluation: vi.fn(), runFinalEvaluation: vi.fn(), })); -vi.mock('@/lib/pbl/v2/operations/quiz-snapshot', () => ({ +vi.mock('@/lib/pbl/v2/operations/runtime/quiz-snapshot', () => ({ applyQuizSignalsToProject: vi.fn(() => ({ updated: false, tierChanged: false })), })); vi.mock('@/lib/logger', () => ({ diff --git a/tests/pbl/v2/simulator.test.ts b/tests/pbl/v2/simulator.test.ts index 75507baf6..a0d82f30e 100644 --- a/tests/pbl/v2/simulator.test.ts +++ b/tests/pbl/v2/simulator.test.ts @@ -7,7 +7,10 @@ import { isFirstSceneEntry, } from '@/lib/pbl/v2/agents/simulator'; import type { PBLChatMessage, PBLAgentThread } from '@/lib/pbl/v2/types'; -import { normalizeProjectRuntime, PBL_SIMULATOR_AGENT_ID } from '@/lib/pbl/v2/operations/progress'; +import { + normalizeProjectRuntime, + PBL_SIMULATOR_AGENT_ID, +} from '@/lib/pbl/v2/operations/kernel/progress'; import type { PBLProjectV2, PBLMilestone, PBLMicrotask } from '@/lib/pbl/v2/types'; function roleplayMilestone(): PBLMilestone { diff --git a/tests/pbl/v2/submission.test.ts b/tests/pbl/v2/submission.test.ts index ee3fc44fc..422e8fd5a 100644 --- a/tests/pbl/v2/submission.test.ts +++ b/tests/pbl/v2/submission.test.ts @@ -11,7 +11,7 @@ import { listSubmissionsForMicrotask, summarizeLatestSubmissionForMicrotask, summarizeSubmissionsForMicrotask, -} from '@/lib/pbl/v2/operations/submission'; +} from '@/lib/pbl/v2/operations/runtime/submission'; import { buildRevisionGuidanceMessage } from '@/components/scene-renderers/pbl/v2/submission'; import type { PBLEvaluation, PBLProjectV2 } from '@/lib/pbl/v2/types'; diff --git a/tests/pbl/v2/workspace-launch.test.ts b/tests/pbl/v2/workspace-launch.test.ts index c15117a2c..97528fd85 100644 --- a/tests/pbl/v2/workspace-launch.test.ts +++ b/tests/pbl/v2/workspace-launch.test.ts @@ -5,7 +5,7 @@ import { isCurrentWorkspaceLaunch, prepareCurrentWorkspaceLaunchProject, prepareWorkspaceLaunchProject, -} from '@/lib/pbl/v2/operations/workspace-launch'; +} from '@/lib/pbl/v2/operations/runtime/workspace-launch'; import type { PBLProjectV2, PriorQuizResult } from '@/lib/pbl/v2/types'; function mkProject(overrides: Partial = {}): PBLProjectV2 {