diff --git a/.codex/agents/accessibility-specialist.toml b/.codex/agents/accessibility-specialist.toml index 7957ebc..7b257dc 100644 --- a/.codex/agents/accessibility-specialist.toml +++ b/.codex/agents/accessibility-specialist.toml @@ -3,12 +3,12 @@ description = "Owns accessibility specialist work: Review accessibility for cont model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/accessibility-specialist.md" -source_hash = "82c8550000b2882a1083338fdb97f46759ca8d493845d697b7594d219f362529" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/accessibility-specialist.md" +# source_hash = "82c8550000b2882a1083338fdb97f46759ca8d493845d697b7594d219f362529" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Accessibility Specialist role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/ai-programmer.toml b/.codex/agents/ai-programmer.toml index 6f60572..0e30f49 100644 --- a/.codex/agents/ai-programmer.toml +++ b/.codex/agents/ai-programmer.toml @@ -3,12 +3,12 @@ description = "Owns ai programmer work: Implement game AI behavior, decision log model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/ai-programmer.md" -source_hash = "f9decba41651b4c225f1c4d4bf3f3bcf0eca088caf9cb3a0b3cb28f43e392b56" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/ai-programmer.md" +# source_hash = "f9decba41651b4c225f1c4d4bf3f3bcf0eca088caf9cb3a0b3cb28f43e392b56" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the AI Programmer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/audio-director.toml b/.codex/agents/audio-director.toml index 1879b58..5d84323 100644 --- a/.codex/agents/audio-director.toml +++ b/.codex/agents/audio-director.toml @@ -3,12 +3,12 @@ description = "Owns audio director work: Define audio pillars, music direction, model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/audio-director.md" -source_hash = "6f40a1c899e6f2d1f348eb91af54fa3d40a29c7236469fdc59d15f0f08c0c85a" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/audio-director.md" +# source_hash = "6f40a1c899e6f2d1f348eb91af54fa3d40a29c7236469fdc59d15f0f08c0c85a" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Audio Director role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/community-manager.toml b/.codex/agents/community-manager.toml index 42315fc..2ef939d 100644 --- a/.codex/agents/community-manager.toml +++ b/.codex/agents/community-manager.toml @@ -3,12 +3,12 @@ description = "Owns community manager work: Plan player communication, feedback model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/community-manager.md" -source_hash = "ba552c736a42cb83f8d04d79199f29c3b8dc81914c7530e86c266ab11c034496" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/community-manager.md" +# source_hash = "ba552c736a42cb83f8d04d79199f29c3b8dc81914c7530e86c266ab11c034496" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Community Manager role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/creative-director.toml b/.codex/agents/creative-director.toml index 352647f..61cd3fa 100644 --- a/.codex/agents/creative-director.toml +++ b/.codex/agents/creative-director.toml @@ -3,12 +3,12 @@ description = "Owns creative director work: Set creative pillars, protect player model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/creative-director.md" -source_hash = "00b9dc2b494b5feeefa5203fdefbd9bd9a4789a23ed2d192898b5ef84036b966" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/creative-director.md" +# source_hash = "00b9dc2b494b5feeefa5203fdefbd9bd9a4789a23ed2d192898b5ef84036b966" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Creative Director role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/data-scientist.toml b/.codex/agents/data-scientist.toml index 826a701..73adb70 100644 --- a/.codex/agents/data-scientist.toml +++ b/.codex/agents/data-scientist.toml @@ -3,12 +3,12 @@ description = "Owns data scientist work: Define analytics events, success metric model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".codex/agents/data-scientist.toml" -source_hash = "51f87821156c61a975630826f11e834291c841b07efd545f4f89aa24fe1c65b8" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".codex/agents/data-scientist.toml" +# source_hash = "51f87821156c61a975630826f11e834291c841b07efd545f4f89aa24fe1c65b8" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Data Scientist role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/devops-engineer.toml b/.codex/agents/devops-engineer.toml index a156b55..d9cbfc4 100644 --- a/.codex/agents/devops-engineer.toml +++ b/.codex/agents/devops-engineer.toml @@ -3,12 +3,12 @@ description = "Owns devops engineer work: Design build, packaging, CI, release a model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/devops-engineer.md" -source_hash = "36bf5a49a121bbb9a7a611408d1d208436e1c9b1395b3f2e9c2cb7d461e50171" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/devops-engineer.md" +# source_hash = "36bf5a49a121bbb9a7a611408d1d208436e1c9b1395b3f2e9c2cb7d461e50171" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the DevOps Engineer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/economy-designer.toml b/.codex/agents/economy-designer.toml index b69d0bf..2828fe2 100644 --- a/.codex/agents/economy-designer.toml +++ b/.codex/agents/economy-designer.toml @@ -3,12 +3,12 @@ description = "Owns economy designer work: Model currencies, rewards, sinks, pac model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/economy-designer.md" -source_hash = "523d9faf11dfe6581df5fa3248c4a3d43b8d284715502c868cb8e796fd55eae0" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/economy-designer.md" +# source_hash = "523d9faf11dfe6581df5fa3248c4a3d43b8d284715502c868cb8e796fd55eae0" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Economy Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/engine-programmer.toml b/.codex/agents/engine-programmer.toml index ea392ab..a0dfbe8 100644 --- a/.codex/agents/engine-programmer.toml +++ b/.codex/agents/engine-programmer.toml @@ -3,12 +3,12 @@ description = "Owns engine programmer work: Work on engine integration, performa model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/engine-programmer.md" -source_hash = "322b59a3167298297657c846272567ef0d6994144649da626a6a7cc651d95c46" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/engine-programmer.md" +# source_hash = "322b59a3167298297657c846272567ef0d6994144649da626a6a7cc651d95c46" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Engine Programmer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/game-designer.toml b/.codex/agents/game-designer.toml index bf6957e..b616b8c 100644 --- a/.codex/agents/game-designer.toml +++ b/.codex/agents/game-designer.toml @@ -3,12 +3,12 @@ description = "Owns game designer work: Design implementation-level mechanics, f model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/game-designer.md" -source_hash = "a1b4fb4f1466b69140a23f8e12d3a5eba624d2a6318da32e2b755e471d41584f" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/game-designer.md" +# source_hash = "a1b4fb4f1466b69140a23f8e12d3a5eba624d2a6318da32e2b755e471d41584f" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Game Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/game-feel-designer.toml b/.codex/agents/game-feel-designer.toml index 7e86aae..3994864 100644 --- a/.codex/agents/game-feel-designer.toml +++ b/.codex/agents/game-feel-designer.toml @@ -3,12 +3,12 @@ description = "Owns game feel designer work: Tune controls, feedback, pacing, an model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".codex/agents/game-feel-designer.toml" -source_hash = "9f3a3af018bc213d900e5c9a6e1d41ee328c5c2f8f25d893527f00b0f90f54dd" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".codex/agents/game-feel-designer.toml" +# source_hash = "9f3a3af018bc213d900e5c9a6e1d41ee328c5c2f8f25d893527f00b0f90f54dd" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Game Feel Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/gameplay-programmer.toml b/.codex/agents/gameplay-programmer.toml index 548b4eb..2095acc 100644 --- a/.codex/agents/gameplay-programmer.toml +++ b/.codex/agents/gameplay-programmer.toml @@ -3,12 +3,12 @@ description = "Owns gameplay programmer work: Implement gameplay systems with fo model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/gameplay-programmer.md" -source_hash = "287ac70fb839e40b0e392e47e6e23ddd90851103d41c0a82df7f56e0e168834e" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/gameplay-programmer.md" +# source_hash = "287ac70fb839e40b0e392e47e6e23ddd90851103d41c0a82df7f56e0e168834e" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Gameplay Programmer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/godot-specialist.toml b/.codex/agents/godot-specialist.toml index b5efcfb..6fcb43a 100644 --- a/.codex/agents/godot-specialist.toml +++ b/.codex/agents/godot-specialist.toml @@ -3,12 +3,12 @@ description = "Owns godot specialist work: Apply Godot-specific implementation a model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/godot-specialist.md" -source_hash = "3156133f9bff9463a12cb48e6337f49a3ab39359550c18eabac4ac70b84b6e37" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/godot-specialist.md" +# source_hash = "3156133f9bff9463a12cb48e6337f49a3ab39359550c18eabac4ac70b84b6e37" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Godot Specialist role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/level-designer.toml b/.codex/agents/level-designer.toml index 2640f76..b9a7703 100644 --- a/.codex/agents/level-designer.toml +++ b/.codex/agents/level-designer.toml @@ -3,12 +3,12 @@ description = "Owns level designer work: Design levels, encounter spaces, naviga model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/level-designer.md" -source_hash = "6121007acde96a808235c4290669bb5de11f1874cd1da6c9d4f0c8c7fad7ef61" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/level-designer.md" +# source_hash = "6121007acde96a808235c4290669bb5de11f1874cd1da6c9d4f0c8c7fad7ef61" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Level Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/live-ops-designer.toml b/.codex/agents/live-ops-designer.toml index fee9d76..1a97945 100644 --- a/.codex/agents/live-ops-designer.toml +++ b/.codex/agents/live-ops-designer.toml @@ -3,12 +3,12 @@ description = "Owns live ops designer work: Design live events, retention loops, model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/live-ops-designer.md" -source_hash = "19fead72a61b75adceaf03763416ab3f4f0c88d8545bc2a844ae2e03d0e62f04" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/live-ops-designer.md" +# source_hash = "19fead72a61b75adceaf03763416ab3f4f0c88d8545bc2a844ae2e03d0e62f04" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Live Ops Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/localization-lead.toml b/.codex/agents/localization-lead.toml index b9c6e87..6bc43d1 100644 --- a/.codex/agents/localization-lead.toml +++ b/.codex/agents/localization-lead.toml @@ -3,12 +3,12 @@ description = "Owns localization lead work: Plan localization scope, string read model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/localization-lead.md" -source_hash = "96991fd33e46cc60179d1abca967ac56f6e809908a8a5fdd71aafa1be3571704" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/localization-lead.md" +# source_hash = "96991fd33e46cc60179d1abca967ac56f6e809908a8a5fdd71aafa1be3571704" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Localization Lead role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/market-analyst.toml b/.codex/agents/market-analyst.toml index 1345a7c..a325624 100644 --- a/.codex/agents/market-analyst.toml +++ b/.codex/agents/market-analyst.toml @@ -3,12 +3,12 @@ description = "Owns market analyst work: Analyze audience, competitors, position model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".codex/agents/market-analyst.toml" -source_hash = "91f30005d8a94c5d1fdff7b8b6a1fd0d112cd3b2192224aff247e6df5887d58b" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".codex/agents/market-analyst.toml" +# source_hash = "91f30005d8a94c5d1fdff7b8b6a1fd0d112cd3b2192224aff247e6df5887d58b" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Market Analyst role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/narrative-designer.toml b/.codex/agents/narrative-designer.toml index 56f6823..284a864 100644 --- a/.codex/agents/narrative-designer.toml +++ b/.codex/agents/narrative-designer.toml @@ -3,12 +3,12 @@ description = "Owns narrative designer work: Shape story, tone, world rules, cha model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".codex/agents/narrative-designer.toml" -source_hash = "02eadc45737f8a1e5950b32fa1d7daa4382608a4c82136ad6e162eb76616c684" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".codex/agents/narrative-designer.toml" +# source_hash = "02eadc45737f8a1e5950b32fa1d7daa4382608a4c82136ad6e162eb76616c684" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Narrative Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/network-programmer.toml b/.codex/agents/network-programmer.toml index 4268661..4bfa512 100644 --- a/.codex/agents/network-programmer.toml +++ b/.codex/agents/network-programmer.toml @@ -3,12 +3,12 @@ description = "Owns network programmer work: Implement networked gameplay, repli model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/network-programmer.md" -source_hash = "b3f107bb01b24475f93aebb6af1735fd90dd73afe070bae667cd9475b2e616e1" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/network-programmer.md" +# source_hash = "b3f107bb01b24475f93aebb6af1735fd90dd73afe070bae667cd9475b2e616e1" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Network Programmer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/performance-analyst.toml b/.codex/agents/performance-analyst.toml index 69050b9..a6c5894 100644 --- a/.codex/agents/performance-analyst.toml +++ b/.codex/agents/performance-analyst.toml @@ -3,12 +3,12 @@ description = "Owns performance analyst work: Profile frame time, memory, loadin model = "gpt-5.4" model_reasoning_effort = "medium" model_verbosity = "medium" -source_reference = ".claude/agents/performance-analyst.md" -source_hash = "c155149e303e200a00074721cd3d9bf5b92d080871ab6298670954aec5705468" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/performance-analyst.md" +# source_hash = "c155149e303e200a00074721cd3d9bf5b92d080871ab6298670954aec5705468" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Performance Analyst role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/producer.toml b/.codex/agents/producer.toml index 17bdeb1..2c695d9 100644 --- a/.codex/agents/producer.toml +++ b/.codex/agents/producer.toml @@ -3,12 +3,12 @@ description = "Owns producer work: Convert goals into bounded production plans, model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/producer.md" -source_hash = "05f8bf426001cbd07ca9ca40dde0348139e208f97af7d87abcfecf8ca84c3257" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/producer.md" +# source_hash = "05f8bf426001cbd07ca9ca40dde0348139e208f97af7d87abcfecf8ca84c3257" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Producer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/qa-playtester.toml b/.codex/agents/qa-playtester.toml index 4136e7a..d3f2d07 100644 --- a/.codex/agents/qa-playtester.toml +++ b/.codex/agents/qa-playtester.toml @@ -3,12 +3,12 @@ description = "Owns qa playtester work: Find reproducible gameplay, usability, a model = "gpt-5.4" model_reasoning_effort = "medium" model_verbosity = "medium" -source_reference = ".codex/agents/qa-playtester.toml" -source_hash = "c2b8d15cece22b3f5d7afe9e86f630790121186b5363bd7b9aa6355a2211010b" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".codex/agents/qa-playtester.toml" +# source_hash = "c2b8d15cece22b3f5d7afe9e86f630790121186b5363bd7b9aa6355a2211010b" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the QA Playtester role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/release-manager.toml b/.codex/agents/release-manager.toml index 70c9dea..cdb854e 100644 --- a/.codex/agents/release-manager.toml +++ b/.codex/agents/release-manager.toml @@ -3,12 +3,12 @@ description = "Owns release manager work: Assess ship readiness, release risk, p model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/release-manager.md" -source_hash = "75df18172fcb905c544972edcfe43589093f02f6abf41d388f952c1eadbe11bd" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/release-manager.md" +# source_hash = "75df18172fcb905c544972edcfe43589093f02f6abf41d388f952c1eadbe11bd" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Release Manager role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/security-engineer.toml b/.codex/agents/security-engineer.toml index 627bff7..5a073e2 100644 --- a/.codex/agents/security-engineer.toml +++ b/.codex/agents/security-engineer.toml @@ -3,12 +3,12 @@ description = "Owns security engineer work: Review game client, tools, build scr model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/security-engineer.md" -source_hash = "73bf76067ec312fa592d9c81b26c4a7b64f8831a50404244a1ec398a77091c7e" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/security-engineer.md" +# source_hash = "73bf76067ec312fa592d9c81b26c4a7b64f8831a50404244a1ec398a77091c7e" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Security Engineer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/senior-game-artist.toml b/.codex/agents/senior-game-artist.toml index fee4f8b..11a7117 100644 --- a/.codex/agents/senior-game-artist.toml +++ b/.codex/agents/senior-game-artist.toml @@ -3,12 +3,12 @@ description = "Owns senior game artist work: Define art direction, asset style, model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".codex/agents/senior-game-artist.toml" -source_hash = "fb3b1ec6a173ddac5a5c4cbc9b840d470d983b29b424fc553ef622c141d01d3e" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".codex/agents/senior-game-artist.toml" +# source_hash = "fb3b1ec6a173ddac5a5c4cbc9b840d470d983b29b424fc553ef622c141d01d3e" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Senior Game Artist role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/senior-game-designer.toml b/.codex/agents/senior-game-designer.toml index 01a14ec..4884543 100644 --- a/.codex/agents/senior-game-designer.toml +++ b/.codex/agents/senior-game-designer.toml @@ -3,12 +3,12 @@ description = "Owns senior game designer work: Own high-level systems, progressi model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".codex/agents/senior-game-designer.toml" -source_hash = "89528fc515b449688c707389b75eb10d66ed8247dc78e9834ec7b33ab608579d" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".codex/agents/senior-game-designer.toml" +# source_hash = "89528fc515b449688c707389b75eb10d66ed8247dc78e9834ec7b33ab608579d" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Senior Game Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/sound-designer.toml b/.codex/agents/sound-designer.toml index bd4ffdb..405de81 100644 --- a/.codex/agents/sound-designer.toml +++ b/.codex/agents/sound-designer.toml @@ -3,12 +3,12 @@ description = "Owns sound designer work: Specify sound effects, feedback cues, a model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/sound-designer.md" -source_hash = "4ebaef7c7069a5dd5a68ed12ca251035f0f9b2852fd29b9eca3ea2c5a4fdf088" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/sound-designer.md" +# source_hash = "4ebaef7c7069a5dd5a68ed12ca251035f0f9b2852fd29b9eca3ea2c5a4fdf088" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Sound Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/studio-orchestrator.toml b/.codex/agents/studio-orchestrator.toml index fc5cd0e..3759443 100644 --- a/.codex/agents/studio-orchestrator.toml +++ b/.codex/agents/studio-orchestrator.toml @@ -3,12 +3,12 @@ description = "Owns studio orchestrator work: Route work between roles, maintain model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".codex/agents/studio-orchestrator.toml" -source_hash = "cd8a03e5cc3890c988d77144d10650616d3eed1a172746dd0bc0b9d3650eb087" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".codex/agents/studio-orchestrator.toml" +# source_hash = "cd8a03e5cc3890c988d77144d10650616d3eed1a172746dd0bc0b9d3650eb087" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Studio Orchestrator role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/systems-designer.toml b/.codex/agents/systems-designer.toml index 3434b3d..e419771 100644 --- a/.codex/agents/systems-designer.toml +++ b/.codex/agents/systems-designer.toml @@ -3,12 +3,12 @@ description = "Owns systems designer work: Design interconnected rules, progress model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/systems-designer.md" -source_hash = "b12973f4dce65c5add8e31fc570c4ece2b6f9855d2c89ce1645c0901a14adc17" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/systems-designer.md" +# source_hash = "b12973f4dce65c5add8e31fc570c4ece2b6f9855d2c89ce1645c0901a14adc17" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Systems Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/technical-artist.toml b/.codex/agents/technical-artist.toml index a53e58f..37cc40f 100644 --- a/.codex/agents/technical-artist.toml +++ b/.codex/agents/technical-artist.toml @@ -3,12 +3,12 @@ description = "Owns technical artist work: Bridge art direction and runtime cons model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/technical-artist.md" -source_hash = "30feb7cd99da83ef1b1350b109535085239eb5b2d210eeb22faa8e54633ba77d" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/technical-artist.md" +# source_hash = "30feb7cd99da83ef1b1350b109535085239eb5b2d210eeb22faa8e54633ba77d" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Technical Artist role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/technical-director.toml b/.codex/agents/technical-director.toml index 147da8e..0a37d4d 100644 --- a/.codex/agents/technical-director.toml +++ b/.codex/agents/technical-director.toml @@ -3,12 +3,12 @@ description = "Owns technical director work: Coordinate technical architecture, model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/technical-director.md" -source_hash = "fe98888f29067eb81ef4dd29d93c3498048628f4bcf6404a1c594f8a6343f9b2" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/technical-director.md" +# source_hash = "fe98888f29067eb81ef4dd29d93c3498048628f4bcf6404a1c594f8a6343f9b2" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Technical Director role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/tools-programmer.toml b/.codex/agents/tools-programmer.toml index fa9c0c1..6a343f3 100644 --- a/.codex/agents/tools-programmer.toml +++ b/.codex/agents/tools-programmer.toml @@ -3,12 +3,12 @@ description = "Owns tools programmer work: Build editor, pipeline, automation, a model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/tools-programmer.md" -source_hash = "1304ed30dc06506f043cbe6df33617a483988a1f73aedec1f9867bbc5de7d20b" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/tools-programmer.md" +# source_hash = "1304ed30dc06506f043cbe6df33617a483988a1f73aedec1f9867bbc5de7d20b" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Tools Programmer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/ui-programmer.toml b/.codex/agents/ui-programmer.toml index 300ecf2..96f3dce 100644 --- a/.codex/agents/ui-programmer.toml +++ b/.codex/agents/ui-programmer.toml @@ -3,12 +3,12 @@ description = "Owns ui programmer work: Implement HUD, menu, input, state bindin model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/ui-programmer.md" -source_hash = "6cd29c2088724918b2680d3cc35c379276b7af11d6aa10ac990172280b1d5f48" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/ui-programmer.md" +# source_hash = "6cd29c2088724918b2680d3cc35c379276b7af11d6aa10ac990172280b1d5f48" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the UI Programmer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/ui-ux-designer.toml b/.codex/agents/ui-ux-designer.toml index aca01e6..2248a57 100644 --- a/.codex/agents/ui-ux-designer.toml +++ b/.codex/agents/ui-ux-designer.toml @@ -3,12 +3,12 @@ description = "Owns ui ux designer work: Design interface flows, usability heuri model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".codex/agents/ui-ux-designer.toml" -source_hash = "f211da5110cd95d6a5a92d1835249eb7a4a398561cb3a23a71ed364cdb245ce9" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".codex/agents/ui-ux-designer.toml" +# source_hash = "f211da5110cd95d6a5a92d1835249eb7a4a398561cb3a23a71ed364cdb245ce9" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the UI UX Designer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/unity-specialist.toml b/.codex/agents/unity-specialist.toml index c1aeeea..41536cd 100644 --- a/.codex/agents/unity-specialist.toml +++ b/.codex/agents/unity-specialist.toml @@ -3,12 +3,12 @@ description = "Owns unity specialist work: Apply Unity-specific implementation a model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/unity-specialist.md" -source_hash = "81777b58e3a30f4a131025e01af6ac8fb42065214df6271e59df1aa11e5374e4" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/unity-specialist.md" +# source_hash = "81777b58e3a30f4a131025e01af6ac8fb42065214df6271e59df1aa11e5374e4" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Unity Specialist role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/unreal-specialist.toml b/.codex/agents/unreal-specialist.toml index 3950c3a..ef412cf 100644 --- a/.codex/agents/unreal-specialist.toml +++ b/.codex/agents/unreal-specialist.toml @@ -3,12 +3,12 @@ description = "Owns unreal specialist work: Apply Unreal-specific implementation model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/unreal-specialist.md" -source_hash = "03794a0b048f1df4cd882ae73a964cd5368c629e00320e36051bc3d025a65c15" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/unreal-specialist.md" +# source_hash = "03794a0b048f1df4cd882ae73a964cd5368c629e00320e36051bc3d025a65c15" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Unreal Specialist role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/world-builder.toml b/.codex/agents/world-builder.toml index e8d697f..f07325f 100644 --- a/.codex/agents/world-builder.toml +++ b/.codex/agents/world-builder.toml @@ -3,12 +3,12 @@ description = "Owns world builder work: Define factions, places, rules, environm model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/world-builder.md" -source_hash = "9e3399c5478ea2491bfc8e42745948b7aec4c2045ecd008a318d4f72a8fad2ea" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/world-builder.md" +# source_hash = "9e3399c5478ea2491bfc8e42745948b7aec4c2045ecd008a318d4f72a8fad2ea" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the World Builder role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/.codex/agents/writer.toml b/.codex/agents/writer.toml index a6a5229..985fff5 100644 --- a/.codex/agents/writer.toml +++ b/.codex/agents/writer.toml @@ -3,12 +3,12 @@ description = "Owns writer work: Draft shippable dialogue, item text, barks, tut model = "gpt-5.5" model_reasoning_effort = "high" model_verbosity = "medium" -source_reference = ".claude/agents/writer.md" -source_hash = "8eafe2f3cd7f2c12acc0d8f93ed93bf638a915490ed5ca5a11721f1bf60f49d7" -primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] -allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] -invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." -stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." +# source_reference = ".claude/agents/writer.md" +# source_hash = "8eafe2f3cd7f2c12acc0d8f93ed93bf638a915490ed5ca5a11721f1bf60f49d7" +# primary_skills = ["cgs-standards-gameplay", "cgs-vertical-slice", "cgs-bugfix"] +# allowed_tool_categories = ["read", "edit", "shell", "tests", "git"] +# invocation_guidance = "Use for bounded game-development work where this role owns the decision or handoff." +# stop_conditions = "Stop on missing project state, unsafe write scope, absent verification path, or cross-role ownership conflict." developer_instructions = """ You are the Writer role for this Codex Game Studio template repository. Use AGENTS.md, .codex/studio.json when present, selected workflows, selected skills, and task-relevant files. diff --git a/eval-framework/README.md b/eval-framework/README.md new file mode 100644 index 0000000..63d8139 --- /dev/null +++ b/eval-framework/README.md @@ -0,0 +1,81 @@ +# Skill And Prompt Performance Eval Framework + +This is Open Game Studio's maintainer-only framework for evaluating how skills, workflow prompts, and agent-facing surfaces perform in realistic runs. + +It follows the CCGS pattern of catalog → rubric → behavioral scenario, and the Truthmark pattern of manual workflow-quality runs with deterministic boundaries, semantic judging, human review, and token tracking. + +Normal game-project users do not need this folder. It is not a hidden runtime, daemon, hosted service, or downstream requirement. If a downstream game repository only wants to build a game and not maintain Open Game Studio's prompt surfaces, users may delete `eval-framework/` and the related maintainer-only OpenSpec change files from their copy. + +## What this evaluates + +- Whether a skill or workflow is triggered for the right task. +- Whether the agent selects bounded context instead of loading the whole repository. +- Whether output quality matches the rubric and scenario contract. +- Whether write boundaries are respected. +- Whether verification is run or explicitly blocked. +- Whether the final report gives a human reviewer a useful verdict, risks, changed files, and next owner. +- Raw token usage and selected evaluation model for comparing prompt and workflow-surface changes over time. + +## What this does not evaluate + +- A pass is never awarded merely because a skill file is present. +- A pass is never awarded merely because a prompt contains a literal phrase. +- The framework is not a package-install gate for normal users. +- The framework does not enforce token-budget thresholds by default. + +## Files + +```text +eval-framework/ +├── catalog.json # targets, rubrics, scenarios, runner hosts +├── rubrics/ # deterministic gates + semantic dimensions +├── scenarios/ # realistic task prompts and expected boundaries +├── failure-taxonomy.md # stable failure labels +├── improvement-loop.md # how to turn failures into small fixes +└── runs/ # repository-saved evaluation summaries and compact audit files +``` + +## First-pass coverage + +This pass covers 31 behavior scenarios: + +| Area | Count | Examples | +|---|---:|---| +| Workflow prompts | 12 | `vertical-slice`, `bugfix`, `playtest`, `ship-check`, `sprint-plan` | +| Skills | 12 | `cgs-gate-check`, `cgs-skill-test`, `cgs-skill-improve`, `cgs-code-review` | +| Role prompts | 7 | `producer`, `qa-playtester`, `gameplay-programmer`, `release-manager` | + +This is not CCGS full parity. It is the first practical coverage threshold for the highest-risk workflow, skill-maintenance, QA/review/gate, and role-cluster surfaces. + +## Evaluation model and token estimation + +- Default manual evaluation model: `gpt-5.3-codex-spark`. +- Allowed models are listed in `catalog.json` under `modelPolicy.allowedEvaluationModels`. +- A manual runner may override the model per scenario or batch with the Codex `--model` value, but the selected model must be recorded in the run report and audit JSON. +- Token estimation is part of normal evaluation planning: estimate token usage before each staged run from the selected model, scenario count, expected context, and expected output/reasoning size. +- Every run should record raw token usage when the host exposes it: input, cached input, output, reasoning output, and total tokens. +- The framework does not estimate or record money cost. + +## Gradual evaluation plan + +You do not need to evaluate every skill and prompt at once. Use the staged plan in `catalog.json`: + +1. `smoke-critical` — run only the smallest critical workflow/skill/role set first. +2. `workflow-high-risk` — expand to high-priority workflow prompts. +3. `skill-maintenance` — evaluate skill-maintenance, gate, QA, review, and evidence skills. +4. `role-boundary` — evaluate role ownership and handoff boundaries. +5. `full-regression` — run all scenarios only before broad prompt-surface releases or after large refactors. + +## Manual run shape + +A future runner should follow this contract: + +1. Materialize the scenario fixture or use the current repository when the scenario says so. +2. Run one agent attempt with the scenario prompt. +3. Collect trace evidence, changed files, final report, verification output, and raw token usage. +4. Apply deterministic gates first. +5. Run semantic judges only after deterministic boundary checks are available. +6. Write a repository-saved run directory under `eval-framework/runs/-/`. +7. Save `summary.md` for human review and `audit.json` for machine-readable scenario results, selected model, raw token usage, trace references, failure labels, changed files, verification evidence, and recommended follow-up fixes. + +Evaluation results are repository artifacts when intentionally preserved. Do not rely on chat transcript as the record; commit the selected `eval-framework/runs/...` summary and audit files when an evaluation result should survive. diff --git a/eval-framework/catalog.json b/eval-framework/catalog.json new file mode 100644 index 0000000..6fca774 --- /dev/null +++ b/eval-framework/catalog.json @@ -0,0 +1,572 @@ +{ + "version": 1, + "manualOnly": true, + "lastReviewed": "2026-07-01", + "targets": [ + { + "id": "workflow.vertical-slice", + "kind": "workflow", + "priority": "critical", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/vertical-slice.md", + ".agents/skills/cgs-vertical-slice/SKILL.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-vertical-slice/behavior/scenario.json" + ] + }, + { + "id": "workflow.bugfix", + "kind": "workflow", + "priority": "critical", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/bugfix.md", + ".agents/skills/cgs-bugfix/SKILL.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-bugfix/behavior/scenario.json" + ] + }, + { + "id": "workflow.playtest", + "kind": "workflow", + "priority": "critical", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/playtest.md", + ".agents/skills/cgs-playtest-report/SKILL.md", + "templates/playtest_report_template.md", + "templates/test_evidence_template.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-playtest/behavior/scenario.json" + ] + }, + { + "id": "workflow.market-analysis", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/market-analysis.md", + ".agents/skills/cgs-content-audit/SKILL.md", + "templates/market_analysis_template.md", + "templates/pitch_document_template.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-market-analysis/behavior/scenario.json" + ] + }, + { + "id": "workflow.release-checklist", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/release-checklist.md", + ".agents/skills/cgs-release-checklist/SKILL.md", + "templates/release_notes_template.md", + "templates/risk_register_template.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-release-checklist/behavior/scenario.json" + ] + }, + { + "id": "workflow.ship-check", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/ship-check.md", + "templates/ship_check_template.md", + "templates/risk_register_template.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-ship-check/behavior/scenario.json" + ] + }, + { + "id": "workflow.prototype", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/prototype.md", + ".agents/skills/cgs-prototype/SKILL.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-prototype/behavior/scenario.json" + ] + }, + { + "id": "workflow.design-spec", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/design-spec.md", + ".agents/skills/cgs-design-system/SKILL.md", + "templates/gdd_template.md", + "templates/feature_spec_template.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-design-spec/behavior/scenario.json" + ] + }, + { + "id": "workflow.game-feel-tuning", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/game-feel-tuning.md", + ".agents/skills/cgs-balance-check/SKILL.md", + "templates/game_feel_tuning_template.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-game-feel-tuning/behavior/scenario.json" + ] + }, + { + "id": "workflow.ui-ux-review", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/ui-ux-review.md", + ".agents/skills/cgs-ui-ux-review/SKILL.md", + "templates/ui_ux_review_template.md", + "templates/accessibility_requirements_template.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-ui-ux-review/behavior/scenario.json" + ] + }, + { + "id": "workflow.architecture-review", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/architecture-review.md", + ".agents/skills/cgs-architecture-review/SKILL.md", + "templates/technical_design_template.md", + "templates/architecture_traceability_template.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-architecture-review/behavior/scenario.json" + ] + }, + { + "id": "workflow.sprint-plan", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/workflows/sprint-plan.md", + ".agents/skills/cgs-sprint-plan/SKILL.md", + "templates/sprint_plan_template.md" + ], + "rubric": "eval-framework/rubrics/prompt-workflow-behavior.json", + "scenarios": [ + "eval-framework/scenarios/workflow-sprint-plan/behavior/scenario.json" + ] + }, + { + "id": "cgs-skill-test", + "kind": "skill", + "priority": "critical", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-skill-test/SKILL.md", + "eval-framework/rubrics/skill-behavior.json" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-skill-test/behavioral-spec/scenario.json" + ] + }, + { + "id": "cgs-skill-improve", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-skill-improve/SKILL.md", + "eval-framework/failure-taxonomy.md", + "eval-framework/improvement-loop.md" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-skill-improve/failure-loop/scenario.json" + ] + }, + { + "id": "cgs-gate-check", + "kind": "skill", + "priority": "critical", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-gate-check/SKILL.md", + "eval-framework/rubrics/skill-behavior.json" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-gate-check/mode-boundary/scenario.json" + ] + }, + { + "id": "cgs-design-review", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-design-review/SKILL.md", + "eval-framework/rubrics/skill-behavior.json" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-design-review/read-only-verdict/scenario.json" + ] + }, + { + "id": "cgs-architecture-review", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-architecture-review/SKILL.md", + "eval-framework/rubrics/skill-behavior.json" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-architecture-review/risk-verdict/scenario.json" + ] + }, + { + "id": "cgs-story-readiness", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-story-readiness/SKILL.md", + "eval-framework/rubrics/skill-behavior.json" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-story-readiness/readiness-verdict/scenario.json" + ] + }, + { + "id": "cgs-story-done", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-story-done/SKILL.md", + "eval-framework/rubrics/skill-behavior.json" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-story-done/done-verdict/scenario.json" + ] + }, + { + "id": "cgs-qa-plan", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-qa-plan/SKILL.md", + "templates/test_plan_template.md" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-qa-plan/coverage-plan/scenario.json" + ] + }, + { + "id": "cgs-regression-suite", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-regression-suite/SKILL.md", + "templates/test_evidence_template.md" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-regression-suite/repeatability/scenario.json" + ] + }, + { + "id": "cgs-test-evidence-review", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-test-evidence-review/SKILL.md", + "templates/test_evidence_template.md" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-test-evidence-review/evidence-review/scenario.json" + ] + }, + { + "id": "cgs-test-flakiness", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-test-flakiness/SKILL.md", + "eval-framework/rubrics/skill-behavior.json" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-test-flakiness/flakiness-diagnosis/scenario.json" + ] + }, + { + "id": "cgs-code-review", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".agents/skills/cgs-code-review/SKILL.md", + "eval-framework/rubrics/skill-behavior.json" + ], + "rubric": "eval-framework/rubrics/skill-behavior.json", + "scenarios": [ + "eval-framework/scenarios/cgs-code-review/review-findings/scenario.json" + ] + }, + { + "id": "role.producer", + "kind": "role", + "priority": "critical", + "manualOnly": true, + "surfacePaths": [ + ".codex/agents/producer.toml" + ], + "rubric": "eval-framework/rubrics/role-behavior.json", + "scenarios": [ + "eval-framework/scenarios/role-producer/domain-boundary/scenario.json" + ] + }, + { + "id": "role.qa-playtester", + "kind": "role", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/agents/qa-playtester.toml" + ], + "rubric": "eval-framework/rubrics/role-behavior.json", + "scenarios": [ + "eval-framework/scenarios/role-qa-playtester/domain-boundary/scenario.json" + ] + }, + { + "id": "role.gameplay-programmer", + "kind": "role", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/agents/gameplay-programmer.toml" + ], + "rubric": "eval-framework/rubrics/role-behavior.json", + "scenarios": [ + "eval-framework/scenarios/role-gameplay-programmer/domain-boundary/scenario.json" + ] + }, + { + "id": "role.game-designer", + "kind": "role", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/agents/game-designer.toml" + ], + "rubric": "eval-framework/rubrics/role-behavior.json", + "scenarios": [ + "eval-framework/scenarios/role-game-designer/domain-boundary/scenario.json" + ] + }, + { + "id": "role.market-analyst", + "kind": "role", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/agents/market-analyst.toml" + ], + "rubric": "eval-framework/rubrics/role-behavior.json", + "scenarios": [ + "eval-framework/scenarios/role-market-analyst/domain-boundary/scenario.json" + ] + }, + { + "id": "role.technical-artist", + "kind": "role", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/agents/technical-artist.toml" + ], + "rubric": "eval-framework/rubrics/role-behavior.json", + "scenarios": [ + "eval-framework/scenarios/role-technical-artist/domain-boundary/scenario.json" + ] + }, + { + "id": "role.release-manager", + "kind": "role", + "priority": "high", + "manualOnly": true, + "surfacePaths": [ + ".codex/agents/release-manager.toml" + ], + "rubric": "eval-framework/rubrics/role-behavior.json", + "scenarios": [ + "eval-framework/scenarios/role-release-manager/domain-boundary/scenario.json" + ] + } + ], + "runners": { + "harnessHosts": [ + "fake" + ], + "manualAgentHosts": [ + "codex" + ], + "runOutputPolicy": { + "repositoryTracked": true, + "root": "eval-framework/runs", + "perRunDirectory": "-", + "requiredFiles": [ + "summary.md", + "audit.json" + ], + "summaryContract": [ + "run id and evaluation stage", + "selected model and token usage summary", + "scenario verdict table", + "deterministic failures and semantic findings", + "changed files and verification evidence", + "recommended follow-up fixes" + ], + "auditContract": [ + "machine-readable scenario results", + "raw token usage fields", + "model used per scenario", + "trace evidence references", + "failure taxonomy labels" + ], + "notes": "Evaluation results are repository artifacts. Commit intentionally selected run summaries and compact audit files when preserving an eval result; do not rely on chat transcript as the record." + } + }, + "modelPolicy": { + "defaultEvaluationModel": "gpt-5.3-codex-spark", + "allowedEvaluationModels": [ + "gpt-5.3-codex-spark", + "gpt-5.5", + "gpt-5.4", + "gpt-5.4-mini" + ], + "overrideMechanism": "Manual runners may pass an explicit Codex --model value per scenario or batch; record the selected model in the run report and audit JSON.", + "tokenEstimation": { + "required": true, + "fields": [ + "inputTokens", + "cachedInputTokens", + "outputTokens", + "reasoningOutputTokens", + "totalTokens" + ], + "notes": "Estimate token usage as part of each staged evaluation run using the selected model, scenario count, expected context, and expected output size; record actual token usage when the host exposes it." + } + }, + "evaluationPlan": [ + { + "stage": "smoke-critical", + "purpose": "Run the smallest critical token-estimation set first before touching broad coverage.", + "targetPriorities": [ + "critical" + ], + "scenarioKinds": [ + "workflow", + "skill", + "role" + ], + "recommendedMaxScenarios": 5 + }, + { + "stage": "workflow-high-risk", + "purpose": "Evaluate high-priority workflow prompts that drive real game-production behavior.", + "targetPriorities": [ + "critical", + "high" + ], + "scenarioKinds": [ + "workflow" + ], + "recommendedMaxScenarios": 12 + }, + { + "stage": "skill-maintenance", + "purpose": "Evaluate skill-maintenance, gate, QA, review, and evidence skills after workflow smoke passes.", + "targetPriorities": [ + "critical", + "high" + ], + "scenarioKinds": [ + "skill" + ], + "recommendedMaxScenarios": 12 + }, + { + "stage": "role-boundary", + "purpose": "Evaluate role prompts for ownership boundaries and handoff quality.", + "targetPriorities": [ + "critical", + "high" + ], + "scenarioKinds": [ + "role" + ], + "recommendedMaxScenarios": 7 + }, + { + "stage": "full-regression", + "purpose": "Run all scenarios only before large prompt-surface releases or after broad refactors.", + "targetPriorities": [ + "critical", + "high", + "medium", + "low" + ], + "scenarioKinds": [ + "workflow", + "skill", + "role", + "prompt" + ], + "recommendedMaxScenarios": 31 + } + ] +} diff --git a/eval-framework/failure-taxonomy.md b/eval-framework/failure-taxonomy.md new file mode 100644 index 0000000..fcf3d97 --- /dev/null +++ b/eval-framework/failure-taxonomy.md @@ -0,0 +1,16 @@ +# Failure Taxonomy + +Use stable labels when reviewing skill and prompt performance eval failures. + +- `skill-not-triggered`: the expected skill or workflow prompt was not used. +- `wrong-workflow`: the agent selected a different workflow lane. +- `over-triggered-workflow`: the agent ran a write workflow for a read-only review task. +- `missing-required-context`: the agent skipped required skill, workflow, rubric, template, or project-state context. +- `wrong-role-routing`: the prompt routed work to the wrong studio role. +- `forbidden-surface-write`: the agent modified skill, workflow, template, or source surfaces during an eval-only task. +- `missing-required-artifact`: the agent did not produce the expected report, plan, or evidence artifact. +- `verification-skipped-without-rationale`: required checks were neither run nor explained. +- `report-invalid`: the report is absent, malformed, or missing required evidence fields. +- `weak-human-review`: the output lacks a verdict, risks, next owner, or decision point. +- `token-bloat`: the agent loaded or repeated unnecessary context for the scenario. +- `judge-not-evaluable`: semantic judge output was missing, malformed, or insufficient. diff --git a/eval-framework/improvement-loop.md b/eval-framework/improvement-loop.md new file mode 100644 index 0000000..861f3e5 --- /dev/null +++ b/eval-framework/improvement-loop.md @@ -0,0 +1,18 @@ +# Improvement Loop + +When a skill or prompt performance eval fails: + +1. Read the run report first, then use machine-readable audit data for scenario verdicts, changed-file summaries, deterministic failures, judge summaries, and token usage. +2. Record a human review as `accepted`, `rejected`, `needs-rerun`, or `not-evaluable`. +3. Classify failures with labels from `failure-taxonomy.md`. +4. Choose the smallest fix target: + - skill text when task procedure or handoff language misled the agent; + - workflow prompt text when role routing, context selection, or stop conditions were weak; + - rubric text when the expectation was underspecified; + - scenario fixture when the setup did not isolate the intended behavior; + - deterministic validator when an objective boundary violation was missed; + - judge prompt/schema when semantic scoring was malformed. +5. Promote real failures into minimal scenarios instead of checking in large run artifacts. +6. Rerun the scenario and compare reports plus raw token counts. + +LLM judge scores are advisory until calibrated by human review. Deterministic boundary failures remain failures even if a semantic judge likes the final prose. diff --git a/eval-framework/rubrics/prompt-workflow-behavior.json b/eval-framework/rubrics/prompt-workflow-behavior.json new file mode 100644 index 0000000..432d8d3 --- /dev/null +++ b/eval-framework/rubrics/prompt-workflow-behavior.json @@ -0,0 +1,22 @@ +{ + "id": "prompt-workflow-behavior", + "manualOnly": true, + "deterministicGates": [ + "workflow-triggering", + "context-boundary", + "write-boundary", + "template-selection", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] +} diff --git a/eval-framework/rubrics/role-behavior.json b/eval-framework/rubrics/role-behavior.json new file mode 100644 index 0000000..795f029 --- /dev/null +++ b/eval-framework/rubrics/role-behavior.json @@ -0,0 +1,21 @@ +{ + "id": "role-behavior", + "manualOnly": true, + "deterministicGates": [ + "required-read", + "domain-boundary", + "write-boundary", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "domain-boundary", + "delegation-quality", + "output-quality", + "verification-discipline", + "handoff-quality", + "token-discipline" + ] +} diff --git a/eval-framework/rubrics/skill-behavior.json b/eval-framework/rubrics/skill-behavior.json new file mode 100644 index 0000000..0de3ff0 --- /dev/null +++ b/eval-framework/rubrics/skill-behavior.json @@ -0,0 +1,21 @@ +{ + "id": "skill-behavior", + "manualOnly": true, + "deterministicGates": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] +} diff --git a/eval-framework/runs/.gitkeep b/eval-framework/runs/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/eval-framework/scenarios/cgs-architecture-review/risk-verdict/prompt.md b/eval-framework/scenarios/cgs-architecture-review/risk-verdict/prompt.md new file mode 100644 index 0000000..1460cee --- /dev/null +++ b/eval-framework/scenarios/cgs-architecture-review/risk-verdict/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-architecture-review risk-verdict + +Evaluate architecture review risk identification and director-gate boundary. + +## Required context + +- `.agents/skills/cgs-architecture-review/SKILL.md` +- `eval-framework/rubrics/skill-behavior.json` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/architecture-review-skill-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-architecture-review/risk-verdict/scenario.json b/eval-framework/scenarios/cgs-architecture-review/risk-verdict/scenario.json new file mode 100644 index 0000000..ebc9fa4 --- /dev/null +++ b/eval-framework/scenarios/cgs-architecture-review/risk-verdict/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-architecture-review.risk-verdict", + "target": "cgs-architecture-review", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-architecture-review/risk-verdict/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-architecture-review/SKILL.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-architecture-review/risk-verdict/prompt.md" + ], + "mustChange": [ + "production/session-state/architecture-review-skill-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-code-review/review-findings/prompt.md b/eval-framework/scenarios/cgs-code-review/review-findings/prompt.md new file mode 100644 index 0000000..1cc6d8e --- /dev/null +++ b/eval-framework/scenarios/cgs-code-review/review-findings/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-code-review review-findings + +Evaluate code-review severity, evidence, and no-auto-write behavior. + +## Required context + +- `.agents/skills/cgs-code-review/SKILL.md` +- `eval-framework/rubrics/skill-behavior.json` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/code-review-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-code-review/review-findings/scenario.json b/eval-framework/scenarios/cgs-code-review/review-findings/scenario.json new file mode 100644 index 0000000..02efc7a --- /dev/null +++ b/eval-framework/scenarios/cgs-code-review/review-findings/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-code-review.review-findings", + "target": "cgs-code-review", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-code-review/review-findings/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-code-review/SKILL.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-code-review/review-findings/prompt.md" + ], + "mustChange": [ + "production/session-state/code-review-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-design-review/read-only-verdict/prompt.md b/eval-framework/scenarios/cgs-design-review/read-only-verdict/prompt.md new file mode 100644 index 0000000..7e19588 --- /dev/null +++ b/eval-framework/scenarios/cgs-design-review/read-only-verdict/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-design-review read-only-verdict + +Evaluate read-only structured design review and verdict quality. + +## Required context + +- `.agents/skills/cgs-design-review/SKILL.md` +- `eval-framework/rubrics/skill-behavior.json` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/design-review-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-design-review/read-only-verdict/scenario.json b/eval-framework/scenarios/cgs-design-review/read-only-verdict/scenario.json new file mode 100644 index 0000000..0cc1770 --- /dev/null +++ b/eval-framework/scenarios/cgs-design-review/read-only-verdict/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-design-review.read-only-verdict", + "target": "cgs-design-review", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-design-review/read-only-verdict/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-design-review/SKILL.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-design-review/read-only-verdict/prompt.md" + ], + "mustChange": [ + "production/session-state/design-review-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-gate-check/mode-boundary/prompt.md b/eval-framework/scenarios/cgs-gate-check/mode-boundary/prompt.md new file mode 100644 index 0000000..31e5438 --- /dev/null +++ b/eval-framework/scenarios/cgs-gate-check/mode-boundary/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-gate-check mode-boundary + +Evaluate phase gate mode handling and no auto-advance behavior. + +## Required context + +- `.agents/skills/cgs-gate-check/SKILL.md` +- `eval-framework/rubrics/skill-behavior.json` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/gate-check-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-gate-check/mode-boundary/scenario.json b/eval-framework/scenarios/cgs-gate-check/mode-boundary/scenario.json new file mode 100644 index 0000000..5eded80 --- /dev/null +++ b/eval-framework/scenarios/cgs-gate-check/mode-boundary/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-gate-check.mode-boundary", + "target": "cgs-gate-check", + "kind": "skill", + "priority": "critical", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-gate-check/mode-boundary/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-gate-check/SKILL.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-gate-check/mode-boundary/prompt.md" + ], + "mustChange": [ + "production/session-state/gate-check-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-qa-plan/coverage-plan/prompt.md b/eval-framework/scenarios/cgs-qa-plan/coverage-plan/prompt.md new file mode 100644 index 0000000..baa6e6b --- /dev/null +++ b/eval-framework/scenarios/cgs-qa-plan/coverage-plan/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-qa-plan coverage-plan + +Evaluate QA plan coverage, test levels, and execution evidence. + +## Required context + +- `.agents/skills/cgs-qa-plan/SKILL.md` +- `templates/test_plan_template.md` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/qa-plan-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-qa-plan/coverage-plan/scenario.json b/eval-framework/scenarios/cgs-qa-plan/coverage-plan/scenario.json new file mode 100644 index 0000000..395e4c6 --- /dev/null +++ b/eval-framework/scenarios/cgs-qa-plan/coverage-plan/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-qa-plan.coverage-plan", + "target": "cgs-qa-plan", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-qa-plan/coverage-plan/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-qa-plan/SKILL.md", + "templates/test_plan_template.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-qa-plan/coverage-plan/prompt.md" + ], + "mustChange": [ + "production/session-state/qa-plan-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-regression-suite/repeatability/prompt.md b/eval-framework/scenarios/cgs-regression-suite/repeatability/prompt.md new file mode 100644 index 0000000..ec513af --- /dev/null +++ b/eval-framework/scenarios/cgs-regression-suite/repeatability/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-regression-suite repeatability + +Evaluate regression-suite selection, repeatability, and evidence requirements. + +## Required context + +- `.agents/skills/cgs-regression-suite/SKILL.md` +- `templates/test_evidence_template.md` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/regression-suite-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-regression-suite/repeatability/scenario.json b/eval-framework/scenarios/cgs-regression-suite/repeatability/scenario.json new file mode 100644 index 0000000..b9b1ca7 --- /dev/null +++ b/eval-framework/scenarios/cgs-regression-suite/repeatability/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-regression-suite.repeatability", + "target": "cgs-regression-suite", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-regression-suite/repeatability/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-regression-suite/SKILL.md", + "templates/test_evidence_template.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-regression-suite/repeatability/prompt.md" + ], + "mustChange": [ + "production/session-state/regression-suite-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-skill-improve/failure-loop/prompt.md b/eval-framework/scenarios/cgs-skill-improve/failure-loop/prompt.md new file mode 100644 index 0000000..d5ac4e1 --- /dev/null +++ b/eval-framework/scenarios/cgs-skill-improve/failure-loop/prompt.md @@ -0,0 +1,22 @@ +# Scenario: cgs-skill-improve failure-loop + +Evaluate whether skill-improve converts failures into smallest useful fixes. + +## Required context + +- `.agents/skills/cgs-skill-improve/SKILL.md` +- `eval-framework/failure-taxonomy.md` +- `eval-framework/improvement-loop.md` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/skill-improvement-plan.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-skill-improve/failure-loop/scenario.json b/eval-framework/scenarios/cgs-skill-improve/failure-loop/scenario.json new file mode 100644 index 0000000..318e277 --- /dev/null +++ b/eval-framework/scenarios/cgs-skill-improve/failure-loop/scenario.json @@ -0,0 +1,52 @@ +{ + "id": "skill.cgs-skill-improve.failure-loop", + "target": "cgs-skill-improve", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-skill-improve/failure-loop/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-skill-improve/SKILL.md", + "eval-framework/failure-taxonomy.md", + "eval-framework/improvement-loop.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-skill-improve/failure-loop/prompt.md" + ], + "mustChange": [ + "production/session-state/skill-improvement-plan.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-skill-test/behavioral-spec/prompt.md b/eval-framework/scenarios/cgs-skill-test/behavioral-spec/prompt.md new file mode 100644 index 0000000..612d964 --- /dev/null +++ b/eval-framework/scenarios/cgs-skill-test/behavioral-spec/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-skill-test behavioral-spec + +Evaluate whether skill-test tests behavior specs rather than skill existence. + +## Required context + +- `.agents/skills/cgs-skill-test/SKILL.md` +- `eval-framework/rubrics/skill-behavior.json` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-skill-test/behavioral-spec/scenario.json b/eval-framework/scenarios/cgs-skill-test/behavioral-spec/scenario.json new file mode 100644 index 0000000..6864a16 --- /dev/null +++ b/eval-framework/scenarios/cgs-skill-test/behavioral-spec/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-skill-test.behavioral-spec", + "target": "cgs-skill-test", + "kind": "skill", + "priority": "critical", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-skill-test/behavioral-spec/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-skill-test/SKILL.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-skill-test/behavioral-spec/prompt.md" + ], + "mustChange": [ + "production/session-state/eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-story-done/done-verdict/prompt.md b/eval-framework/scenarios/cgs-story-done/done-verdict/prompt.md new file mode 100644 index 0000000..06e87e9 --- /dev/null +++ b/eval-framework/scenarios/cgs-story-done/done-verdict/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-story-done done-verdict + +Evaluate completion evidence, acceptance criteria, and handoff behavior. + +## Required context + +- `.agents/skills/cgs-story-done/SKILL.md` +- `eval-framework/rubrics/skill-behavior.json` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/story-done-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-story-done/done-verdict/scenario.json b/eval-framework/scenarios/cgs-story-done/done-verdict/scenario.json new file mode 100644 index 0000000..7b3faff --- /dev/null +++ b/eval-framework/scenarios/cgs-story-done/done-verdict/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-story-done.done-verdict", + "target": "cgs-story-done", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-story-done/done-verdict/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-story-done/SKILL.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-story-done/done-verdict/prompt.md" + ], + "mustChange": [ + "production/session-state/story-done-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-story-readiness/readiness-verdict/prompt.md b/eval-framework/scenarios/cgs-story-readiness/readiness-verdict/prompt.md new file mode 100644 index 0000000..ba67e56 --- /dev/null +++ b/eval-framework/scenarios/cgs-story-readiness/readiness-verdict/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-story-readiness readiness-verdict + +Evaluate story readiness blockers, verdict levels, and next-owner clarity. + +## Required context + +- `.agents/skills/cgs-story-readiness/SKILL.md` +- `eval-framework/rubrics/skill-behavior.json` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/story-readiness-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-story-readiness/readiness-verdict/scenario.json b/eval-framework/scenarios/cgs-story-readiness/readiness-verdict/scenario.json new file mode 100644 index 0000000..a28d173 --- /dev/null +++ b/eval-framework/scenarios/cgs-story-readiness/readiness-verdict/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-story-readiness.readiness-verdict", + "target": "cgs-story-readiness", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-story-readiness/readiness-verdict/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-story-readiness/SKILL.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-story-readiness/readiness-verdict/prompt.md" + ], + "mustChange": [ + "production/session-state/story-readiness-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-test-evidence-review/evidence-review/prompt.md b/eval-framework/scenarios/cgs-test-evidence-review/evidence-review/prompt.md new file mode 100644 index 0000000..3ebd95b --- /dev/null +++ b/eval-framework/scenarios/cgs-test-evidence-review/evidence-review/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-test-evidence-review evidence-review + +Evaluate test evidence review findings, gaps, and verifier discipline. + +## Required context + +- `.agents/skills/cgs-test-evidence-review/SKILL.md` +- `templates/test_evidence_template.md` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/test-evidence-review-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-test-evidence-review/evidence-review/scenario.json b/eval-framework/scenarios/cgs-test-evidence-review/evidence-review/scenario.json new file mode 100644 index 0000000..93e36a8 --- /dev/null +++ b/eval-framework/scenarios/cgs-test-evidence-review/evidence-review/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-test-evidence-review.evidence-review", + "target": "cgs-test-evidence-review", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-test-evidence-review/evidence-review/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-test-evidence-review/SKILL.md", + "templates/test_evidence_template.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-test-evidence-review/evidence-review/prompt.md" + ], + "mustChange": [ + "production/session-state/test-evidence-review-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/cgs-test-flakiness/flakiness-diagnosis/prompt.md b/eval-framework/scenarios/cgs-test-flakiness/flakiness-diagnosis/prompt.md new file mode 100644 index 0000000..1759a5d --- /dev/null +++ b/eval-framework/scenarios/cgs-test-flakiness/flakiness-diagnosis/prompt.md @@ -0,0 +1,21 @@ +# Scenario: cgs-test-flakiness flakiness-diagnosis + +Evaluate flaky-test diagnosis, reproduction notes, and isolation boundaries. + +## Required context + +- `.agents/skills/cgs-test-flakiness/SKILL.md` +- `eval-framework/rubrics/skill-behavior.json` +- `eval-framework/rubrics/skill-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/test-flakiness-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/cgs-test-flakiness/flakiness-diagnosis/scenario.json b/eval-framework/scenarios/cgs-test-flakiness/flakiness-diagnosis/scenario.json new file mode 100644 index 0000000..58a818d --- /dev/null +++ b/eval-framework/scenarios/cgs-test-flakiness/flakiness-diagnosis/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "skill.cgs-test-flakiness.flakiness-diagnosis", + "target": "cgs-test-flakiness", + "kind": "skill", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/cgs-test-flakiness/flakiness-diagnosis/prompt.md", + "expected": { + "mustRead": [ + ".agents/skills/cgs-test-flakiness/SKILL.md", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/rubrics/skill-behavior.json", + "eval-framework/scenarios/cgs-test-flakiness/flakiness-diagnosis/prompt.md" + ], + "mustChange": [ + "production/session-state/test-flakiness-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "task-framing", + "output-quality", + "verification-discipline", + "failure-handling", + "human-review-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/role-game-designer/domain-boundary/prompt.md b/eval-framework/scenarios/role-game-designer/domain-boundary/prompt.md new file mode 100644 index 0000000..a257da4 --- /dev/null +++ b/eval-framework/scenarios/role-game-designer/domain-boundary/prompt.md @@ -0,0 +1,20 @@ +# Scenario: role.game-designer domain-boundary + +Evaluate game designer acceptance criteria and design-system alignment. + +## Required context + +- `.codex/agents/game-designer.toml` +- `eval-framework/rubrics/role-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/game-designer-role-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/role-game-designer/domain-boundary/scenario.json b/eval-framework/scenarios/role-game-designer/domain-boundary/scenario.json new file mode 100644 index 0000000..57e7665 --- /dev/null +++ b/eval-framework/scenarios/role-game-designer/domain-boundary/scenario.json @@ -0,0 +1,50 @@ +{ + "id": "role.game-designer.domain-boundary", + "target": "role.game-designer", + "kind": "role", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/role-game-designer/domain-boundary/prompt.md", + "expected": { + "mustRead": [ + ".codex/agents/game-designer.toml", + "eval-framework/rubrics/role-behavior.json", + "eval-framework/scenarios/role-game-designer/domain-boundary/prompt.md" + ], + "mustChange": [ + "production/session-state/game-designer-role-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "domain-boundary", + "delegation-quality", + "output-quality", + "verification-discipline", + "handoff-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/role-gameplay-programmer/domain-boundary/prompt.md b/eval-framework/scenarios/role-gameplay-programmer/domain-boundary/prompt.md new file mode 100644 index 0000000..3e3095a --- /dev/null +++ b/eval-framework/scenarios/role-gameplay-programmer/domain-boundary/prompt.md @@ -0,0 +1,20 @@ +# Scenario: role.gameplay-programmer domain-boundary + +Evaluate gameplay programmer bounded implementation plan and verification evidence. + +## Required context + +- `.codex/agents/gameplay-programmer.toml` +- `eval-framework/rubrics/role-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/gameplay-programmer-role-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/role-gameplay-programmer/domain-boundary/scenario.json b/eval-framework/scenarios/role-gameplay-programmer/domain-boundary/scenario.json new file mode 100644 index 0000000..3571846 --- /dev/null +++ b/eval-framework/scenarios/role-gameplay-programmer/domain-boundary/scenario.json @@ -0,0 +1,50 @@ +{ + "id": "role.gameplay-programmer.domain-boundary", + "target": "role.gameplay-programmer", + "kind": "role", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/role-gameplay-programmer/domain-boundary/prompt.md", + "expected": { + "mustRead": [ + ".codex/agents/gameplay-programmer.toml", + "eval-framework/rubrics/role-behavior.json", + "eval-framework/scenarios/role-gameplay-programmer/domain-boundary/prompt.md" + ], + "mustChange": [ + "production/session-state/gameplay-programmer-role-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "domain-boundary", + "delegation-quality", + "output-quality", + "verification-discipline", + "handoff-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/role-market-analyst/domain-boundary/prompt.md b/eval-framework/scenarios/role-market-analyst/domain-boundary/prompt.md new file mode 100644 index 0000000..041cb44 --- /dev/null +++ b/eval-framework/scenarios/role-market-analyst/domain-boundary/prompt.md @@ -0,0 +1,20 @@ +# Scenario: role.market-analyst domain-boundary + +Evaluate market analyst positioning, evidence separation, and competitor caveats. + +## Required context + +- `.codex/agents/market-analyst.toml` +- `eval-framework/rubrics/role-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/market-analyst-role-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/role-market-analyst/domain-boundary/scenario.json b/eval-framework/scenarios/role-market-analyst/domain-boundary/scenario.json new file mode 100644 index 0000000..fbcffdb --- /dev/null +++ b/eval-framework/scenarios/role-market-analyst/domain-boundary/scenario.json @@ -0,0 +1,50 @@ +{ + "id": "role.market-analyst.domain-boundary", + "target": "role.market-analyst", + "kind": "role", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/role-market-analyst/domain-boundary/prompt.md", + "expected": { + "mustRead": [ + ".codex/agents/market-analyst.toml", + "eval-framework/rubrics/role-behavior.json", + "eval-framework/scenarios/role-market-analyst/domain-boundary/prompt.md" + ], + "mustChange": [ + "production/session-state/market-analyst-role-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "domain-boundary", + "delegation-quality", + "output-quality", + "verification-discipline", + "handoff-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/role-producer/domain-boundary/prompt.md b/eval-framework/scenarios/role-producer/domain-boundary/prompt.md new file mode 100644 index 0000000..d87c23f --- /dev/null +++ b/eval-framework/scenarios/role-producer/domain-boundary/prompt.md @@ -0,0 +1,20 @@ +# Scenario: role.producer domain-boundary + +Evaluate producer role prioritization, milestone risk, and next-owner clarity. + +## Required context + +- `.codex/agents/producer.toml` +- `eval-framework/rubrics/role-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/producer-role-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/role-producer/domain-boundary/scenario.json b/eval-framework/scenarios/role-producer/domain-boundary/scenario.json new file mode 100644 index 0000000..264edd8 --- /dev/null +++ b/eval-framework/scenarios/role-producer/domain-boundary/scenario.json @@ -0,0 +1,50 @@ +{ + "id": "role.producer.domain-boundary", + "target": "role.producer", + "kind": "role", + "priority": "critical", + "manualOnly": true, + "prompt": "eval-framework/scenarios/role-producer/domain-boundary/prompt.md", + "expected": { + "mustRead": [ + ".codex/agents/producer.toml", + "eval-framework/rubrics/role-behavior.json", + "eval-framework/scenarios/role-producer/domain-boundary/prompt.md" + ], + "mustChange": [ + "production/session-state/producer-role-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "domain-boundary", + "delegation-quality", + "output-quality", + "verification-discipline", + "handoff-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/role-qa-playtester/domain-boundary/prompt.md b/eval-framework/scenarios/role-qa-playtester/domain-boundary/prompt.md new file mode 100644 index 0000000..b1f0226 --- /dev/null +++ b/eval-framework/scenarios/role-qa-playtester/domain-boundary/prompt.md @@ -0,0 +1,20 @@ +# Scenario: role.qa-playtester domain-boundary + +Evaluate QA playtester reproduction evidence and implementation avoidance. + +## Required context + +- `.codex/agents/qa-playtester.toml` +- `eval-framework/rubrics/role-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/qa-playtester-role-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/role-qa-playtester/domain-boundary/scenario.json b/eval-framework/scenarios/role-qa-playtester/domain-boundary/scenario.json new file mode 100644 index 0000000..ae6b97b --- /dev/null +++ b/eval-framework/scenarios/role-qa-playtester/domain-boundary/scenario.json @@ -0,0 +1,50 @@ +{ + "id": "role.qa-playtester.domain-boundary", + "target": "role.qa-playtester", + "kind": "role", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/role-qa-playtester/domain-boundary/prompt.md", + "expected": { + "mustRead": [ + ".codex/agents/qa-playtester.toml", + "eval-framework/rubrics/role-behavior.json", + "eval-framework/scenarios/role-qa-playtester/domain-boundary/prompt.md" + ], + "mustChange": [ + "production/session-state/qa-playtester-role-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "domain-boundary", + "delegation-quality", + "output-quality", + "verification-discipline", + "handoff-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/role-release-manager/domain-boundary/prompt.md b/eval-framework/scenarios/role-release-manager/domain-boundary/prompt.md new file mode 100644 index 0000000..36558e4 --- /dev/null +++ b/eval-framework/scenarios/role-release-manager/domain-boundary/prompt.md @@ -0,0 +1,20 @@ +# Scenario: role.release-manager domain-boundary + +Evaluate release manager blocker separation, rollback notes, and ship/no-ship verdict. + +## Required context + +- `.codex/agents/release-manager.toml` +- `eval-framework/rubrics/role-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/release-manager-role-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/role-release-manager/domain-boundary/scenario.json b/eval-framework/scenarios/role-release-manager/domain-boundary/scenario.json new file mode 100644 index 0000000..088b27d --- /dev/null +++ b/eval-framework/scenarios/role-release-manager/domain-boundary/scenario.json @@ -0,0 +1,50 @@ +{ + "id": "role.release-manager.domain-boundary", + "target": "role.release-manager", + "kind": "role", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/role-release-manager/domain-boundary/prompt.md", + "expected": { + "mustRead": [ + ".codex/agents/release-manager.toml", + "eval-framework/rubrics/role-behavior.json", + "eval-framework/scenarios/role-release-manager/domain-boundary/prompt.md" + ], + "mustChange": [ + "production/session-state/release-manager-role-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "domain-boundary", + "delegation-quality", + "output-quality", + "verification-discipline", + "handoff-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/role-technical-artist/domain-boundary/prompt.md b/eval-framework/scenarios/role-technical-artist/domain-boundary/prompt.md new file mode 100644 index 0000000..fcee463 --- /dev/null +++ b/eval-framework/scenarios/role-technical-artist/domain-boundary/prompt.md @@ -0,0 +1,20 @@ +# Scenario: role.technical-artist domain-boundary + +Evaluate technical artist asset pipeline boundary and engine-specific evidence. + +## Required context + +- `.codex/agents/technical-artist.toml` +- `eval-framework/rubrics/role-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/technical-artist-role-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/role-technical-artist/domain-boundary/scenario.json b/eval-framework/scenarios/role-technical-artist/domain-boundary/scenario.json new file mode 100644 index 0000000..6665134 --- /dev/null +++ b/eval-framework/scenarios/role-technical-artist/domain-boundary/scenario.json @@ -0,0 +1,50 @@ +{ + "id": "role.technical-artist.domain-boundary", + "target": "role.technical-artist", + "kind": "role", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/role-technical-artist/domain-boundary/prompt.md", + "expected": { + "mustRead": [ + ".codex/agents/technical-artist.toml", + "eval-framework/rubrics/role-behavior.json", + "eval-framework/scenarios/role-technical-artist/domain-boundary/prompt.md" + ], + "mustChange": [ + "production/session-state/technical-artist-role-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "domain-boundary", + "delegation-quality", + "output-quality", + "verification-discipline", + "handoff-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-architecture-review/behavior/prompt.md b/eval-framework/scenarios/workflow-architecture-review/behavior/prompt.md new file mode 100644 index 0000000..c4737d4 --- /dev/null +++ b/eval-framework/scenarios/workflow-architecture-review/behavior/prompt.md @@ -0,0 +1,23 @@ +# Scenario: workflow.architecture-review behavior + +Evaluate architecture review verdict, risks, and traceability evidence. + +## Required context + +- `.codex/workflows/architecture-review.md` +- `.agents/skills/cgs-architecture-review/SKILL.md` +- `templates/technical_design_template.md` +- `templates/architecture_traceability_template.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/architecture-review-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-architecture-review/behavior/scenario.json b/eval-framework/scenarios/workflow-architecture-review/behavior/scenario.json new file mode 100644 index 0000000..6d87f71 --- /dev/null +++ b/eval-framework/scenarios/workflow-architecture-review/behavior/scenario.json @@ -0,0 +1,53 @@ +{ + "id": "workflow.architecture-review.behavior", + "target": "workflow.architecture-review", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-architecture-review/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/architecture-review.md", + ".agents/skills/cgs-architecture-review/SKILL.md", + "templates/technical_design_template.md", + "templates/architecture_traceability_template.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-architecture-review/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/architecture-review-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-bugfix/behavior/prompt.md b/eval-framework/scenarios/workflow-bugfix/behavior/prompt.md new file mode 100644 index 0000000..df42033 --- /dev/null +++ b/eval-framework/scenarios/workflow-bugfix/behavior/prompt.md @@ -0,0 +1,21 @@ +# Scenario: workflow.bugfix behavior + +Evaluate bugfix workflow reproduction, bounded fix guidance, and verification evidence. + +## Required context + +- `.codex/workflows/bugfix.md` +- `.agents/skills/cgs-bugfix/SKILL.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/bugfix-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-bugfix/behavior/scenario.json b/eval-framework/scenarios/workflow-bugfix/behavior/scenario.json new file mode 100644 index 0000000..d68d3e8 --- /dev/null +++ b/eval-framework/scenarios/workflow-bugfix/behavior/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "workflow.bugfix.behavior", + "target": "workflow.bugfix", + "kind": "workflow", + "priority": "critical", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-bugfix/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/bugfix.md", + ".agents/skills/cgs-bugfix/SKILL.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-bugfix/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/bugfix-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-design-spec/behavior/prompt.md b/eval-framework/scenarios/workflow-design-spec/behavior/prompt.md new file mode 100644 index 0000000..b441874 --- /dev/null +++ b/eval-framework/scenarios/workflow-design-spec/behavior/prompt.md @@ -0,0 +1,23 @@ +# Scenario: workflow.design-spec behavior + +Evaluate design-spec prompt requirements, acceptance criteria, and handoff clarity. + +## Required context + +- `.codex/workflows/design-spec.md` +- `.agents/skills/cgs-design-system/SKILL.md` +- `templates/gdd_template.md` +- `templates/feature_spec_template.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/design-spec-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-design-spec/behavior/scenario.json b/eval-framework/scenarios/workflow-design-spec/behavior/scenario.json new file mode 100644 index 0000000..7f5f47b --- /dev/null +++ b/eval-framework/scenarios/workflow-design-spec/behavior/scenario.json @@ -0,0 +1,53 @@ +{ + "id": "workflow.design-spec.behavior", + "target": "workflow.design-spec", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-design-spec/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/design-spec.md", + ".agents/skills/cgs-design-system/SKILL.md", + "templates/gdd_template.md", + "templates/feature_spec_template.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-design-spec/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/design-spec-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-game-feel-tuning/behavior/prompt.md b/eval-framework/scenarios/workflow-game-feel-tuning/behavior/prompt.md new file mode 100644 index 0000000..fadd94d --- /dev/null +++ b/eval-framework/scenarios/workflow-game-feel-tuning/behavior/prompt.md @@ -0,0 +1,22 @@ +# Scenario: workflow.game-feel-tuning behavior + +Evaluate game-feel tuning evidence, parameter discipline, and comparison notes. + +## Required context + +- `.codex/workflows/game-feel-tuning.md` +- `.agents/skills/cgs-balance-check/SKILL.md` +- `templates/game_feel_tuning_template.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/game-feel-tuning-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-game-feel-tuning/behavior/scenario.json b/eval-framework/scenarios/workflow-game-feel-tuning/behavior/scenario.json new file mode 100644 index 0000000..c72b2bb --- /dev/null +++ b/eval-framework/scenarios/workflow-game-feel-tuning/behavior/scenario.json @@ -0,0 +1,52 @@ +{ + "id": "workflow.game-feel-tuning.behavior", + "target": "workflow.game-feel-tuning", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-game-feel-tuning/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/game-feel-tuning.md", + ".agents/skills/cgs-balance-check/SKILL.md", + "templates/game_feel_tuning_template.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-game-feel-tuning/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/game-feel-tuning-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-market-analysis/behavior/prompt.md b/eval-framework/scenarios/workflow-market-analysis/behavior/prompt.md new file mode 100644 index 0000000..962de95 --- /dev/null +++ b/eval-framework/scenarios/workflow-market-analysis/behavior/prompt.md @@ -0,0 +1,23 @@ +# Scenario: workflow.market-analysis behavior + +Evaluate market-analysis prompt context selection and positioning output quality. + +## Required context + +- `.codex/workflows/market-analysis.md` +- `.agents/skills/cgs-content-audit/SKILL.md` +- `templates/market_analysis_template.md` +- `templates/pitch_document_template.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/market-analysis-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-market-analysis/behavior/scenario.json b/eval-framework/scenarios/workflow-market-analysis/behavior/scenario.json new file mode 100644 index 0000000..e59991b --- /dev/null +++ b/eval-framework/scenarios/workflow-market-analysis/behavior/scenario.json @@ -0,0 +1,53 @@ +{ + "id": "workflow.market-analysis.behavior", + "target": "workflow.market-analysis", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-market-analysis/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/market-analysis.md", + ".agents/skills/cgs-content-audit/SKILL.md", + "templates/market_analysis_template.md", + "templates/pitch_document_template.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-market-analysis/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/market-analysis-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-playtest/behavior/prompt.md b/eval-framework/scenarios/workflow-playtest/behavior/prompt.md new file mode 100644 index 0000000..1092fc5 --- /dev/null +++ b/eval-framework/scenarios/workflow-playtest/behavior/prompt.md @@ -0,0 +1,23 @@ +# Scenario: workflow.playtest behavior + +Evaluate QA/playtest routing and evidence quality without implementation writes. + +## Required context + +- `.codex/workflows/playtest.md` +- `.agents/skills/cgs-playtest-report/SKILL.md` +- `templates/playtest_report_template.md` +- `templates/test_evidence_template.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/playtest-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-playtest/behavior/scenario.json b/eval-framework/scenarios/workflow-playtest/behavior/scenario.json new file mode 100644 index 0000000..a92a438 --- /dev/null +++ b/eval-framework/scenarios/workflow-playtest/behavior/scenario.json @@ -0,0 +1,53 @@ +{ + "id": "workflow.playtest.behavior", + "target": "workflow.playtest", + "kind": "workflow", + "priority": "critical", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-playtest/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/playtest.md", + ".agents/skills/cgs-playtest-report/SKILL.md", + "templates/playtest_report_template.md", + "templates/test_evidence_template.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-playtest/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/playtest-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-prototype/behavior/prompt.md b/eval-framework/scenarios/workflow-prototype/behavior/prompt.md new file mode 100644 index 0000000..51c5b9a --- /dev/null +++ b/eval-framework/scenarios/workflow-prototype/behavior/prompt.md @@ -0,0 +1,21 @@ +# Scenario: workflow.prototype behavior + +Evaluate prototype hypothesis framing, cleanup boundaries, and validation notes. + +## Required context + +- `.codex/workflows/prototype.md` +- `.agents/skills/cgs-prototype/SKILL.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/prototype-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-prototype/behavior/scenario.json b/eval-framework/scenarios/workflow-prototype/behavior/scenario.json new file mode 100644 index 0000000..72e22fd --- /dev/null +++ b/eval-framework/scenarios/workflow-prototype/behavior/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "workflow.prototype.behavior", + "target": "workflow.prototype", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-prototype/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/prototype.md", + ".agents/skills/cgs-prototype/SKILL.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-prototype/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/prototype-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-release-checklist/behavior/prompt.md b/eval-framework/scenarios/workflow-release-checklist/behavior/prompt.md new file mode 100644 index 0000000..13d9fbc --- /dev/null +++ b/eval-framework/scenarios/workflow-release-checklist/behavior/prompt.md @@ -0,0 +1,23 @@ +# Scenario: workflow.release-checklist behavior + +Evaluate release checklist blocker separation, risk handling, and verification evidence. + +## Required context + +- `.codex/workflows/release-checklist.md` +- `.agents/skills/cgs-release-checklist/SKILL.md` +- `templates/release_notes_template.md` +- `templates/risk_register_template.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/release-checklist-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-release-checklist/behavior/scenario.json b/eval-framework/scenarios/workflow-release-checklist/behavior/scenario.json new file mode 100644 index 0000000..7bedefc --- /dev/null +++ b/eval-framework/scenarios/workflow-release-checklist/behavior/scenario.json @@ -0,0 +1,53 @@ +{ + "id": "workflow.release-checklist.behavior", + "target": "workflow.release-checklist", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-release-checklist/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/release-checklist.md", + ".agents/skills/cgs-release-checklist/SKILL.md", + "templates/release_notes_template.md", + "templates/risk_register_template.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-release-checklist/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/release-checklist-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-ship-check/behavior/prompt.md b/eval-framework/scenarios/workflow-ship-check/behavior/prompt.md new file mode 100644 index 0000000..5872831 --- /dev/null +++ b/eval-framework/scenarios/workflow-ship-check/behavior/prompt.md @@ -0,0 +1,22 @@ +# Scenario: workflow.ship-check behavior + +Evaluate ship-check readiness judgment and release-manager handoff. + +## Required context + +- `.codex/workflows/ship-check.md` +- `templates/ship_check_template.md` +- `templates/risk_register_template.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/ship-check-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-ship-check/behavior/scenario.json b/eval-framework/scenarios/workflow-ship-check/behavior/scenario.json new file mode 100644 index 0000000..124b8a1 --- /dev/null +++ b/eval-framework/scenarios/workflow-ship-check/behavior/scenario.json @@ -0,0 +1,52 @@ +{ + "id": "workflow.ship-check.behavior", + "target": "workflow.ship-check", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-ship-check/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/ship-check.md", + "templates/ship_check_template.md", + "templates/risk_register_template.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-ship-check/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/ship-check-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-sprint-plan/behavior/prompt.md b/eval-framework/scenarios/workflow-sprint-plan/behavior/prompt.md new file mode 100644 index 0000000..2aecd0a --- /dev/null +++ b/eval-framework/scenarios/workflow-sprint-plan/behavior/prompt.md @@ -0,0 +1,22 @@ +# Scenario: workflow.sprint-plan behavior + +Evaluate sprint-plan prioritization, dependency handling, and risk visibility. + +## Required context + +- `.codex/workflows/sprint-plan.md` +- `.agents/skills/cgs-sprint-plan/SKILL.md` +- `templates/sprint_plan_template.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/sprint-plan-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-sprint-plan/behavior/scenario.json b/eval-framework/scenarios/workflow-sprint-plan/behavior/scenario.json new file mode 100644 index 0000000..c5d0738 --- /dev/null +++ b/eval-framework/scenarios/workflow-sprint-plan/behavior/scenario.json @@ -0,0 +1,52 @@ +{ + "id": "workflow.sprint-plan.behavior", + "target": "workflow.sprint-plan", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-sprint-plan/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/sprint-plan.md", + ".agents/skills/cgs-sprint-plan/SKILL.md", + "templates/sprint_plan_template.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-sprint-plan/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/sprint-plan-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-ui-ux-review/behavior/prompt.md b/eval-framework/scenarios/workflow-ui-ux-review/behavior/prompt.md new file mode 100644 index 0000000..2adfc0b --- /dev/null +++ b/eval-framework/scenarios/workflow-ui-ux-review/behavior/prompt.md @@ -0,0 +1,23 @@ +# Scenario: workflow.ui-ux-review behavior + +Evaluate UI/UX review output quality and accessibility coverage. + +## Required context + +- `.codex/workflows/ui-ux-review.md` +- `.agents/skills/cgs-ui-ux-review/SKILL.md` +- `templates/ui_ux_review_template.md` +- `templates/accessibility_requirements_template.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/ui-ux-review-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-ui-ux-review/behavior/scenario.json b/eval-framework/scenarios/workflow-ui-ux-review/behavior/scenario.json new file mode 100644 index 0000000..f8675fe --- /dev/null +++ b/eval-framework/scenarios/workflow-ui-ux-review/behavior/scenario.json @@ -0,0 +1,53 @@ +{ + "id": "workflow.ui-ux-review.behavior", + "target": "workflow.ui-ux-review", + "kind": "workflow", + "priority": "high", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-ui-ux-review/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/ui-ux-review.md", + ".agents/skills/cgs-ui-ux-review/SKILL.md", + "templates/ui_ux_review_template.md", + "templates/accessibility_requirements_template.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-ui-ux-review/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/ui-ux-review-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/eval-framework/scenarios/workflow-vertical-slice/behavior/prompt.md b/eval-framework/scenarios/workflow-vertical-slice/behavior/prompt.md new file mode 100644 index 0000000..04156cd --- /dev/null +++ b/eval-framework/scenarios/workflow-vertical-slice/behavior/prompt.md @@ -0,0 +1,21 @@ +# Scenario: workflow.vertical-slice behavior + +Evaluate vertical-slice workflow planning, stop conditions, and verification evidence. + +## Required context + +- `.codex/workflows/vertical-slice.md` +- `.agents/skills/cgs-vertical-slice/SKILL.md` +- `eval-framework/rubrics/prompt-workflow-behavior.json` + +## Required behavior + +- Use the target skill, workflow, or role prompt only when it fits this task. +- Select bounded context; do not load unrelated studio surfaces. +- Produce `production/session-state/vertical-slice-eval-report.md` with verdict, evidence, changed files or proposed files, risks, verification notes, and next owner. +- Keep source, templates, skills, workflows, and agent definitions unchanged during the evaluation. +- Run `npm run validate` or explain the concrete blocker. + +## Semantic review focus + +Judge triggering, context selection, output quality, verification discipline, human-review usefulness, and token discipline. diff --git a/eval-framework/scenarios/workflow-vertical-slice/behavior/scenario.json b/eval-framework/scenarios/workflow-vertical-slice/behavior/scenario.json new file mode 100644 index 0000000..e0cd253 --- /dev/null +++ b/eval-framework/scenarios/workflow-vertical-slice/behavior/scenario.json @@ -0,0 +1,51 @@ +{ + "id": "workflow.vertical-slice.behavior", + "target": "workflow.vertical-slice", + "kind": "workflow", + "priority": "critical", + "manualOnly": true, + "prompt": "eval-framework/scenarios/workflow-vertical-slice/behavior/prompt.md", + "expected": { + "mustRead": [ + ".codex/workflows/vertical-slice.md", + ".agents/skills/cgs-vertical-slice/SKILL.md", + "eval-framework/rubrics/prompt-workflow-behavior.json", + "eval-framework/scenarios/workflow-vertical-slice/behavior/prompt.md" + ], + "mustChange": [ + "production/session-state/vertical-slice-eval-report.md" + ], + "mustNotChange": [ + "src/**", + ".agents/skills/**", + ".codex/workflows/**", + ".codex/agents/**", + "templates/**" + ], + "mustRunOrExplain": [ + "npm run validate" + ], + "report": { + "required": true + } + }, + "grading": { + "deterministic": [ + "required-read", + "write-boundary", + "required-change", + "verification-evidence", + "report-presence" + ], + "semanticDimensions": [ + "triggering", + "context-selection", + "role-routing", + "template-selection", + "output-quality", + "verification-discipline", + "stop-condition-quality", + "token-discipline" + ] + } +} diff --git a/openspec/changes/expand-eval-framework-coverage/.openspec.yaml b/openspec/changes/expand-eval-framework-coverage/.openspec.yaml new file mode 100644 index 0000000..e7cc357 --- /dev/null +++ b/openspec/changes/expand-eval-framework-coverage/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-07-01 diff --git a/openspec/changes/expand-eval-framework-coverage/design.md b/openspec/changes/expand-eval-framework-coverage/design.md new file mode 100644 index 0000000..980b5be --- /dev/null +++ b/openspec/changes/expand-eval-framework-coverage/design.md @@ -0,0 +1,45 @@ +## Context + +The branch currently contains an eval framework modeled after CCGS and Truthmark, but it has only three performance scenarios. CCGS has one spec per cataloged skill/agent, and Truthmark uses realistic workflow scenarios with deterministic boundaries, semantic judging, human review, and token tracking. Open Game Studio should not attempt full parity in one pass, but it needs enough coverage to exercise the highest-risk surfaces. + +## Goals / Non-Goals + +**Goals:** +- Reach a first real coverage threshold of at least 30 performance scenarios. +- Cover at least 10 workflow prompts, 10 skills, and 6 role prompts. +- Preserve manual-only behavior and no default CI agent/judge calls. +- Validate behavior contracts: required reads, forbidden writes, required artifacts, verification evidence, report presence, semantic dimensions, and raw token accounting. +- Keep scenario files reviewable and easy to extend. + +**Non-Goals:** +- Do not implement the real agent runner or LLM judge in this pass. +- Do not claim CCGS full parity. +- Do not add package/runtime requirements for downstream game projects. +- Do not evaluate success from skill-file existence or literal prompt presence alone. + +## Decisions + +1. Store scenarios as JSON plus prompt markdown under `eval-framework/scenarios///`. + - Rationale: this matches the current implementation and keeps each scenario reviewable. + - Alternative considered: a single generated mega-catalog. Rejected because it hides scenario intent. + +2. Use catalog targets as the coverage unit and scenario files as the executable unit. + - Rationale: targets map surfaces to rubrics; scenarios capture behavior expectations. + - Alternative considered: direct filesystem inventory. Rejected because existence coverage is not useful. + +3. Add a third rubric for role behavior instead of overloading skill/workflow rubrics. + - Rationale: role prompts need domain-boundary and delegation dimensions that differ from skill-improvement workflows. + +4. Enforce thresholds in tests and validation helpers. + - Rationale: this prevents future regressions back to scaffold-only coverage. + +## Risks / Trade-offs + +- Risk: Scenario count grows but quality stays shallow. Mitigation: require semantic dimensions and concrete expected behavior for every scenario. +- Risk: JSON scenarios become repetitive. Mitigation: keep prompts concise and scenario-specific; defer runner implementation. +- Risk: Full test suite may delete untracked tests during template smoke. Mitigation: stage new test files before full-suite runs and verify no deleted tests remain. +- Risk: Coverage is still below CCGS parity. Mitigation: report this as first-pass coverage, not parity. + +## Migration Plan + +Add OpenSpec artifacts, write failing tests for coverage thresholds, generate scenario files, update validation helpers if needed, then run focused tests, typecheck, full test suite, `npm run validate`, OpenSpec validation, and Truthmark checks. diff --git a/openspec/changes/expand-eval-framework-coverage/proposal.md b/openspec/changes/expand-eval-framework-coverage/proposal.md new file mode 100644 index 0000000..f66ce7e --- /dev/null +++ b/openspec/changes/expand-eval-framework-coverage/proposal.md @@ -0,0 +1,27 @@ +## Why + +The initial eval framework proves the shape, but three scenarios are not enough to judge skill and prompt quality across the studio. The next pass must cover the highest-risk workflows, skill-maintenance and QA/review/gate skills, and representative role clusters with behavior-focused scenarios rather than surface-existence checks. + +## What Changes + +- Expand `eval-framework/catalog.json` from scaffold coverage to a first real coverage pass. +- Add behavior scenarios for critical/high workflow prompts. +- Add behavior scenarios for skill-maintenance, QA, review, readiness, and gate skills. +- Add representative role prompt scenarios across major role clusters. +- Add validation/tests that enforce minimum scenario counts, coverage categories, semantic dimensions, and the no-existence-only strategy. +- Keep the framework manual-only and token-aware; no real agent or judge calls run in default CI. + +## Capabilities + +### New Capabilities +- `performance-eval-coverage`: Maintainer-only behavior evaluation coverage for skills, workflows, and role prompts. + +### Modified Capabilities +- `performance-eval-framework`: The existing eval framework SHALL enforce first-pass coverage thresholds and category distribution instead of only validating the scaffold. + +## Impact + +- `eval-framework/**` catalog, rubrics, scenario prompts, and expected behavior files. +- `src/performance-evaluation.ts` coverage validation helpers. +- `tests/performance-evaluation-framework.test.ts` TDD coverage for scenario counts and prompt/skill/role coverage. +- `src/validation.ts` remains the integration point for repository validation. diff --git a/openspec/changes/expand-eval-framework-coverage/specs/performance-eval-coverage/spec.md b/openspec/changes/expand-eval-framework-coverage/specs/performance-eval-coverage/spec.md new file mode 100644 index 0000000..7eeebbb --- /dev/null +++ b/openspec/changes/expand-eval-framework-coverage/specs/performance-eval-coverage/spec.md @@ -0,0 +1,35 @@ +## ADDED Requirements + +### Requirement: First-pass performance eval coverage +The eval framework SHALL provide a first-pass set of behavior scenarios across workflow prompts, skills, and role prompts. + +#### Scenario: Coverage threshold is met +- **WHEN** the eval framework catalog is loaded +- **THEN** it contains at least 30 performance evaluation scenarios +- **THEN** it covers at least 10 workflow targets +- **THEN** it covers at least 10 skill targets +- **THEN** it covers at least 6 role targets + +### Requirement: Behavior scenarios do not use existence-only success criteria +Each performance scenario SHALL define expected behavior evidence rather than passing because a skill, prompt, or file exists. + +#### Scenario: Scenario expectations are behavioral +- **WHEN** a scenario is validated +- **THEN** it includes at least one required read, write boundary, required artifact, verification expectation, report expectation, or semantic dimension +- **THEN** no deterministic gate uses skill-exists, file-exists, presence-only, or equivalent success criteria + +### Requirement: Manual-only eval execution +The eval framework SHALL remain maintainer-only and SHALL NOT run real agents or LLM judges during default validation or CI test commands. + +#### Scenario: Default validation is safe +- **WHEN** `npm run validate` runs +- **THEN** it validates catalog, rubric, and scenario contracts +- **THEN** it does not launch a real agent runner or LLM judge + +### Requirement: Token-aware quality comparison +The eval framework SHALL preserve raw token usage as comparable run metadata without enforcing budget or cost gates. + +#### Scenario: Usage is recorded with a scenario result +- **WHEN** a scenario observation includes token usage +- **THEN** the grader returns the usage with the result +- **THEN** the result status is based on deterministic and semantic evaluation inputs, not budget thresholds diff --git a/openspec/changes/expand-eval-framework-coverage/specs/performance-eval-framework/spec.md b/openspec/changes/expand-eval-framework-coverage/specs/performance-eval-framework/spec.md new file mode 100644 index 0000000..f7ff663 --- /dev/null +++ b/openspec/changes/expand-eval-framework-coverage/specs/performance-eval-framework/spec.md @@ -0,0 +1,22 @@ +## MODIFIED Requirements + +### Requirement: Framework contract validation +The eval framework SHALL validate catalog targets, rubrics, scenarios, manual-only behavior, semantic dimensions, and the absence of existence-only success criteria. + +#### Scenario: Framework validation reports coverage and behavior contracts +- **WHEN** repository validation runs +- **THEN** validation includes checks for manual-only behavior, target mappings, behavioral expectations, semantic dimensions, no existence-only checks, and first-pass coverage thresholds +- **THEN** validation fails if workflow, skill, or role coverage drops below the first-pass threshold + +### Requirement: Scenario grading contract +The scenario grader SHALL evaluate required reads, forbidden writes, required changed artifacts, verification evidence, report presence, and optional token usage. + +#### Scenario: Forbidden write fails before semantic judging +- **WHEN** an observation changes a path matching a scenario `mustNotChange` pattern +- **THEN** the deterministic grading result is fail +- **THEN** the failure includes a forbidden-write label + +#### Scenario: Token usage is preserved +- **WHEN** an observation includes raw token counts +- **THEN** the result includes those counts for comparison +- **THEN** the result does not enforce a token budget diff --git a/openspec/changes/expand-eval-framework-coverage/tasks.md b/openspec/changes/expand-eval-framework-coverage/tasks.md new file mode 100644 index 0000000..1d2adb4 --- /dev/null +++ b/openspec/changes/expand-eval-framework-coverage/tasks.md @@ -0,0 +1,37 @@ +## 1. OpenSpec Planning + +- [x] 1.1 Create proposal for first-pass eval coverage expansion +- [x] 1.2 Create design documenting CCGS/Truthmark-inspired decisions +- [x] 1.3 Create specs for coverage and framework validation requirements +- [x] 1.4 Validate OpenSpec change with `openspec validate expand-eval-framework-coverage --strict` + +## 2. TDD Coverage + +- [x] 2.1 Restore/add `tests/performance-evaluation-framework.test.ts` +- [x] 2.2 Add failing threshold tests for at least 30 scenarios, 10 workflows, 10 skills, and 6 roles +- [x] 2.3 Add validation expectations for first-pass coverage checks +- [x] 2.4 Run focused test and confirm RED before implementation + +## 3. Eval Framework Data + +- [x] 3.1 Add role behavior rubric +- [x] 3.2 Expand catalog to cover workflow prompt targets +- [x] 3.3 Expand catalog to cover skill-maintenance, QA, review, readiness, and gate targets +- [x] 3.4 Expand catalog to cover representative role prompt targets +- [x] 3.5 Add scenario JSON and prompt markdown for each target + +## 4. Validation Integration + +- [x] 4.1 Add coverage summary helper to `src/performance-evaluation.ts` +- [x] 4.2 Add validation check for first-pass coverage thresholds +- [x] 4.3 Keep manual-only validation free of real agent/judge calls + +## 5. Verification + +- [x] 5.1 Run focused performance eval framework tests +- [x] 5.2 Run `npm run typecheck` +- [x] 5.3 Stage new tests before full suite +- [x] 5.4 Run `npm run test` +- [x] 5.5 Run `npm run validate` +- [x] 5.6 Run OpenSpec validation and status +- [x] 5.7 Run Truthmark check/index and verify no deleted tests diff --git a/scripts/audit-prompt-surfaces.ts b/scripts/audit-prompt-surfaces.ts index 7976b9f..579dea6 100644 --- a/scripts/audit-prompt-surfaces.ts +++ b/scripts/audit-prompt-surfaces.ts @@ -5,7 +5,8 @@ import { fileURLToPath } from "node:url"; import { defaultModelPolicyForId, parsePromptSurfaceFrontmatter, - parseTomlArrayField, + parseTomlCommentArrayField, + parseTomlCommentStringField, parseTomlStringField, validateAgentDescriptionQuality, validateSkillDescriptionQuality, @@ -109,10 +110,10 @@ function tomlMetadata(body: string): Record { return { model: !!parseTomlStringField(body, "model"), reasoning: !!parseTomlStringField(body, "model_reasoning_effort"), - sourceReference: !!parseTomlStringField(body, "source_reference"), - sourceHash: !!parseTomlStringField(body, "source_hash"), - primarySkills: parseTomlArrayField(body, "primary_skills").length > 0, - toolPolicy: parseTomlArrayField(body, "allowed_tool_categories").length > 0 + sourceReference: !!parseTomlCommentStringField(body, "source_reference"), + sourceHash: !!parseTomlCommentStringField(body, "source_hash"), + primarySkills: parseTomlCommentArrayField(body, "primary_skills").length > 0, + toolPolicy: parseTomlCommentArrayField(body, "allowed_tool_categories").length > 0 }; } diff --git a/src/performance-evaluation.ts b/src/performance-evaluation.ts new file mode 100644 index 0000000..f2e4699 --- /dev/null +++ b/src/performance-evaluation.ts @@ -0,0 +1,424 @@ +import { existsSync, readFileSync } from "node:fs"; +import path from "node:path"; + +export type PerformanceEvaluationPriority = "critical" | "high" | "medium" | "low"; + +export type PerformanceEvaluationUsage = { + status: "recorded" | "unavailable" | "invalid"; + model?: string; + inputTokens?: number; + cachedInputTokens?: number; + outputTokens?: number; + reasoningOutputTokens?: number; + totalTokens?: number; +}; + +export type PerformanceEvaluationExpected = { + mustRead: string[]; + mustChange: string[]; + mustNotChange: string[]; + mustRunOrExplain: string[]; + report: { required: boolean }; +}; + +export type PerformanceEvaluationScenario = { + id: string; + target: string; + kind?: "skill" | "workflow" | "role" | "prompt"; + priority: PerformanceEvaluationPriority; + manualOnly: boolean; + prompt: string; + expected: PerformanceEvaluationExpected; + grading: { + deterministic: string[]; + semanticDimensions: string[]; + }; +}; + +export type PerformanceEvaluationTarget = { + id: string; + kind: "skill" | "workflow" | "role" | "prompt"; + priority: PerformanceEvaluationPriority; + manualOnly: boolean; + surfacePaths: string[]; + rubric: string; + scenarios: string[]; +}; + +export type PerformanceEvaluationTokenEstimationPolicy = { + required: boolean; + fields: string[]; + notes?: string; +}; + +export type PerformanceEvaluationModelPolicy = { + defaultEvaluationModel: string; + allowedEvaluationModels: string[]; + overrideMechanism: string; + tokenEstimation?: PerformanceEvaluationTokenEstimationPolicy; +}; + +export type PerformanceEvaluationPlanStage = { + stage: string; + purpose: string; + targetPriorities: PerformanceEvaluationPriority[]; + scenarioKinds: Array<"skill" | "workflow" | "role" | "prompt">; + recommendedMaxScenarios: number; +}; + +export type PerformanceEvaluationRunOutputPolicy = { + repositoryTracked: boolean; + root: string; + perRunDirectory: string; + requiredFiles: string[]; + summaryContract: string[]; + auditContract: string[]; + notes?: string; +}; + +export type PerformanceEvaluationCatalog = { + version: number; + manualOnly: boolean; + lastReviewed: string; + targets: PerformanceEvaluationTarget[]; + runners: { + harnessHosts: string[]; + manualAgentHosts: string[]; + runOutputPolicy?: PerformanceEvaluationRunOutputPolicy; + }; + modelPolicy?: PerformanceEvaluationModelPolicy; + evaluationPlan?: PerformanceEvaluationPlanStage[]; +}; + +export type PerformanceEvaluationRubric = { + id: string; + manualOnly: boolean; + deterministicGates: string[]; + semanticDimensions: string[]; +}; + +export type PerformanceEvaluationFramework = { + catalog: PerformanceEvaluationCatalog; + scenarios: PerformanceEvaluationScenario[]; + rubrics: PerformanceEvaluationRubric[]; +}; + +export type PerformanceEvaluationObservation = { + traceEvidence: string; + changedFiles: string[]; + finalReport?: string; + usage?: PerformanceEvaluationUsage; +}; + +export type PerformanceEvaluationFailure = { id: string; message: string }; + +export type PerformanceEvaluationResult = { + scenarioId: string; + status: "pass" | "fail"; + failures: PerformanceEvaluationFailure[]; + usage?: PerformanceEvaluationUsage; +}; + +export type PerformanceEvaluationValidationCheck = { id: string; status: "pass" | "fail"; message: string; path?: string }; + +export type PerformanceEvaluationCoverageSummary = { + targets: number; + scenarios: number; + surfacePaths: number; + byKind: Record<"skill" | "workflow" | "role" | "prompt", number>; +}; + +const FRAMEWORK_DIR = "eval-framework"; +const EXISTENCE_ONLY_PATTERN = /\b(?:skill[-_ ]?exists|file[-_ ]?exists|exists|presence[-_ ]?only|static[-_ ]?presence)\b/iu; + +function readJson(file: string): T { + return JSON.parse(readFileSync(file, "utf8")) as T; +} + +function pass(id: string, message: string, file?: string): PerformanceEvaluationValidationCheck { + return { id, status: "pass", message, path: file }; +} + +function fail(id: string, message: string, file?: string): PerformanceEvaluationValidationCheck { + return { id, status: "fail", message, path: file }; +} + +function slashPath(file: string): string { + return file.split(path.sep).join("/"); +} + +function escapeRegExp(value: string): string { + return value.replace(/[.*+?^${}()|[\]\\]/gu, "\\$&"); +} + +function globToRegExp(glob: string): RegExp { + const normalized = slashPath(glob); + let source = "^"; + for (let index = 0; index < normalized.length; index += 1) { + const char = normalized[index]; + const next = normalized[index + 1]; + if (char === "*" && next === "*") { + source += ".*"; + index += 1; + } else if (char === "*") { + source += "[^/]*"; + } else { + source += escapeRegExp(char); + } + } + source += "$"; + return new RegExp(source, "u"); +} + +function matchesPattern(file: string, pattern: string): boolean { + return globToRegExp(pattern).test(slashPath(file)); +} + +function asArray(value: unknown): string[] { + return Array.isArray(value) ? value.filter((item): item is string => typeof item === "string") : []; +} + +function includesRequiredEvidence(evidence: string, required: string): boolean { + return evidence.includes(required); +} + +function commandWasRunOrExplained(evidence: string, command: string): boolean { + if (evidence.includes(command)) return true; + const escaped = escapeRegExp(command); + const explanationPattern = new RegExp( + `(?:skip|skipped|cannot|can't|unable|did not run|not run|not available|would fail|missing dependency)[\\s\\S]{0,200}${escaped}|${escaped}[\\s\\S]{0,200}(?:skip|skipped|cannot|can't|unable|did not run|not run|not available|would fail|missing dependency)`, + "iu" + ); + return explanationPattern.test(evidence); +} + +export function gradePerformanceEvaluationScenario( + scenario: PerformanceEvaluationScenario, + observation: PerformanceEvaluationObservation +): PerformanceEvaluationResult { + const failures: PerformanceEvaluationFailure[] = []; + const evidence = `${observation.traceEvidence}\n${observation.finalReport ?? ""}`; + + for (const file of scenario.expected.mustRead) { + if (!includesRequiredEvidence(evidence, file)) { + failures.push({ id: "required-read-not-recorded", message: `${file} was not recorded in trace evidence` }); + } + } + + for (const changedFile of observation.changedFiles) { + for (const pattern of scenario.expected.mustNotChange) { + if (matchesPattern(changedFile, pattern)) { + failures.push({ id: "forbidden-write", message: `${changedFile} matched forbidden write pattern ${pattern}` }); + } + } + } + + for (const pattern of scenario.expected.mustChange) { + if (!observation.changedFiles.some((changedFile) => matchesPattern(changedFile, pattern))) { + failures.push({ id: "missing-required-change", message: `${pattern} was not changed` }); + } + } + + for (const command of scenario.expected.mustRunOrExplain) { + if (!commandWasRunOrExplained(evidence, command)) { + failures.push({ id: "verification-not-recorded", message: `${command} was not run or specifically explained` }); + } + } + + if (scenario.expected.report.required && !(observation.finalReport ?? "").trim()) { + failures.push({ id: "missing-report", message: "expected evaluation report was not produced" }); + } + + return { + scenarioId: scenario.id, + status: failures.length === 0 ? "pass" : "fail", + failures, + usage: observation.usage + }; +} + +export function loadPerformanceEvaluationFramework(root: string): PerformanceEvaluationFramework { + const frameworkRoot = path.join(root, FRAMEWORK_DIR); + const catalogPath = path.join(frameworkRoot, "catalog.json"); + const catalog = readJson(catalogPath); + const scenarioPaths = [...new Set(catalog.targets.flatMap((target) => target.scenarios))]; + const rubricPaths = [...new Set(catalog.targets.map((target) => target.rubric))]; + + return { + catalog, + scenarios: scenarioPaths.map((scenarioPath) => readJson(path.join(root, scenarioPath))), + rubrics: rubricPaths.map((rubricPath) => readJson(path.join(root, rubricPath))) + }; +} + +export function summarizePerformanceEvaluationCoverage(framework: PerformanceEvaluationFramework): PerformanceEvaluationCoverageSummary { + const byKind: PerformanceEvaluationCoverageSummary["byKind"] = { skill: 0, workflow: 0, role: 0, prompt: 0 }; + for (const target of framework.catalog.targets) { + byKind[target.kind] += 1; + } + return { + targets: framework.catalog.targets.length, + scenarios: framework.catalog.targets.reduce((total, target) => total + target.scenarios.length, 0), + surfacePaths: framework.catalog.targets.reduce((total, target) => total + target.surfacePaths.length, 0), + byKind + }; +} + +export function validatePerformanceEvaluationFramework(root: string): PerformanceEvaluationValidationCheck[] { + const checks: PerformanceEvaluationValidationCheck[] = []; + const catalogPath = path.join(root, FRAMEWORK_DIR, "catalog.json"); + if (!existsSync(catalogPath)) { + return [fail("performance_eval.catalog.manual_only", "performance evaluation catalog is missing", catalogPath)]; + } + + let framework: PerformanceEvaluationFramework; + try { + framework = loadPerformanceEvaluationFramework(root); + } catch (error) { + return [fail("performance_eval.catalog.manual_only", `performance evaluation framework could not be loaded: ${(error as Error).message}`, catalogPath)]; + } + + checks.push( + framework.catalog.manualOnly === true + ? pass("performance_eval.catalog.manual_only", "performance evaluation catalog is manual and behavior-focused", catalogPath) + : fail("performance_eval.catalog.manual_only", "performance evaluation catalog must be manual-only", catalogPath) + ); + + const targetProblems = framework.catalog.targets.flatMap((target) => { + const problems: string[] = []; + if (target.manualOnly !== true) problems.push(`${target.id} is not manual-only`); + if (!target.rubric) problems.push(`${target.id} has no rubric`); + if (target.scenarios.length === 0) problems.push(`${target.id} has no behavioral scenarios`); + if (target.surfacePaths.length === 0) problems.push(`${target.id} has no prompt/skill surface path`); + return problems; + }); + checks.push( + targetProblems.length === 0 + ? pass("performance_eval.catalog.targets", "catalog targets map surfaces to rubrics and scenarios", catalogPath) + : fail("performance_eval.catalog.targets", targetProblems.join("; "), catalogPath) + ); + + const scenarioProblems = framework.scenarios.flatMap((scenario) => { + const problems: string[] = []; + const hasBehaviorExpectation = + scenario.expected.mustRead.length > 0 || + scenario.expected.mustChange.length > 0 || + scenario.expected.mustNotChange.length > 0 || + scenario.expected.mustRunOrExplain.length > 0 || + scenario.expected.report.required; + if (scenario.manualOnly !== true) problems.push(`${scenario.id} is not manual-only`); + if (!scenario.prompt) problems.push(`${scenario.id} has no scenario prompt`); + if (!hasBehaviorExpectation) problems.push(`${scenario.id} has no behavior expectation`); + if (scenario.grading.semanticDimensions.length < 4) problems.push(`${scenario.id} has too few semantic dimensions`); + if (scenario.grading.deterministic.some((gate) => EXISTENCE_ONLY_PATTERN.test(gate))) { + problems.push(`${scenario.id} uses a presence-only deterministic gate`); + } + return problems; + }); + checks.push( + scenarioProblems.length === 0 + ? pass("performance_eval.scenarios.behavioral_expectations", "scenarios require reads, write boundaries, reports, verification, or semantic grading") + : fail("performance_eval.scenarios.behavioral_expectations", scenarioProblems.join("; ")) + ); + + const rubricProblems = framework.rubrics.flatMap((rubric) => { + const problems: string[] = []; + if (rubric.manualOnly !== true) problems.push(`${rubric.id} is not manual-only`); + if (rubric.semanticDimensions.length < 4) problems.push(`${rubric.id} has too few semantic dimensions`); + if (rubric.deterministicGates.some((gate) => EXISTENCE_ONLY_PATTERN.test(gate))) { + problems.push(`${rubric.id} uses a presence-only gate`); + } + return problems; + }); + checks.push( + rubricProblems.length === 0 + ? pass("performance_eval.rubrics.semantic_dimensions", "rubrics define semantic performance dimensions and deterministic boundaries") + : fail("performance_eval.rubrics.semantic_dimensions", rubricProblems.join("; ")) + ); + + const frameworkText = [ + JSON.stringify(framework.catalog), + ...framework.scenarios.map((scenario) => JSON.stringify(scenario.grading.deterministic)), + ...framework.rubrics.map((rubric) => JSON.stringify(rubric.deterministicGates)) + ].join("\n"); + checks.push( + EXISTENCE_ONLY_PATTERN.test(frameworkText) + ? fail("performance_eval.strategy.no_existence_only_checks", "performance evaluation strategy must not use presence-only checks as success criteria") + : pass("performance_eval.strategy.no_existence_only_checks", "evaluation strategy grades behavior, boundaries, reports, verification, and semantic quality") + ); + + const modelPolicy = framework.catalog.modelPolicy; + const allowedModels = modelPolicy?.allowedEvaluationModels ?? []; + const uniqueAllowedModels = new Set(allowedModels); + checks.push( + !!modelPolicy?.defaultEvaluationModel && + allowedModels.length > 0 && + uniqueAllowedModels.size === allowedModels.length && + allowedModels.includes(modelPolicy.defaultEvaluationModel) && + !!modelPolicy.overrideMechanism + ? pass("performance_eval.model_policy", `default evaluation model is ${modelPolicy.defaultEvaluationModel} with per-run override guidance`, catalogPath) + : fail("performance_eval.model_policy", "catalog must define a default evaluation model that is included in the unique allowed model list plus an override mechanism", catalogPath) + ); + + const requiredTokenFields = ["inputTokens", "cachedInputTokens", "outputTokens", "reasoningOutputTokens", "totalTokens"]; + const tokenFields = modelPolicy?.tokenEstimation?.fields ?? []; + checks.push( + modelPolicy?.tokenEstimation?.required === true && requiredTokenFields.every((field) => tokenFields.includes(field)) + ? pass("performance_eval.token_estimation", "evaluation runs record selected model and raw token usage as normal run-estimation metadata", catalogPath) + : fail("performance_eval.token_estimation", "model policy must include required token estimation fields for normal run estimation", catalogPath) + ); + + const plan = framework.catalog.evaluationPlan ?? []; + const hasGradualPlan = + plan.length >= 3 && + plan[0]?.targetPriorities.includes("critical") && + plan.some((stage) => stage.stage === "full-regression") && + plan.every((stage) => stage.scenarioKinds.length > 0 && stage.recommendedMaxScenarios > 0); + checks.push( + hasGradualPlan + ? pass("performance_eval.plan.gradual", "catalog defines staged evaluation so maintainers can run critical smoke, focused subsets, or full regression", catalogPath) + : fail("performance_eval.plan.gradual", "catalog must define a staged evaluation plan before full regression", catalogPath) + ); + + const readmePath = path.join(root, FRAMEWORK_DIR, "README.md"); + const readme = existsSync(readmePath) ? readFileSync(readmePath, "utf8").toLowerCase() : ""; + checks.push( + readme.includes("normal game-project users") && readme.includes("delete") && readme.includes("eval-framework") + ? pass("performance_eval.template_user_optional", "README tells downstream game-project users the eval framework is optional and deletable", readmePath) + : fail("performance_eval.template_user_optional", "README must tell downstream game-project users they may delete eval-framework/", readmePath) + ); + + const runOutputPolicy = framework.catalog.runners.runOutputPolicy; + const outputRequiredFiles = runOutputPolicy?.requiredFiles ?? []; + const outputSummary = runOutputPolicy?.summaryContract ?? []; + const outputAudit = runOutputPolicy?.auditContract ?? []; + checks.push( + runOutputPolicy?.repositoryTracked === true && + runOutputPolicy.root === "eval-framework/runs" && + outputRequiredFiles.includes("summary.md") && + outputRequiredFiles.includes("audit.json") && + outputSummary.length >= 5 && + outputAudit.length >= 5 && + readme.includes("summary.md") && + readme.includes("audit.json") && + readme.includes("repository artifacts") + ? pass("performance_eval.results.repository_saved", "evaluation results are summarized and saved as repository artifacts under eval-framework/runs", catalogPath) + : fail("performance_eval.results.repository_saved", "catalog and README must require summary.md and audit.json under eval-framework/runs for saved evaluation results", catalogPath) + ); + + const coverage = summarizePerformanceEvaluationCoverage(framework); + const coverageProblems = [ + coverage.scenarios >= 30 ? undefined : `expected at least 30 scenarios, found ${coverage.scenarios}`, + coverage.targets >= 30 ? undefined : `expected at least 30 targets, found ${coverage.targets}`, + coverage.byKind.workflow >= 10 ? undefined : `expected at least 10 workflow targets, found ${coverage.byKind.workflow}`, + coverage.byKind.skill >= 10 ? undefined : `expected at least 10 skill targets, found ${coverage.byKind.skill}`, + coverage.byKind.role >= 6 ? undefined : `expected at least 6 role targets, found ${coverage.byKind.role}` + ].filter((problem): problem is string => Boolean(problem)); + checks.push( + coverageProblems.length === 0 + ? pass("performance_eval.coverage.first_pass", `first-pass coverage includes ${coverage.scenarios} scenarios across ${coverage.byKind.workflow} workflows, ${coverage.byKind.skill} skills, and ${coverage.byKind.role} roles`) + : fail("performance_eval.coverage.first_pass", coverageProblems.join("; ")) + ); + + return checks; +} diff --git a/src/prompt-surface-metadata.ts b/src/prompt-surface-metadata.ts index 8a30e09..19c31db 100644 --- a/src/prompt-surface-metadata.ts +++ b/src/prompt-surface-metadata.ts @@ -104,12 +104,23 @@ export function parseTomlStringField(body: string, key: string): string | undefi return match?.[1]; } +export function parseTomlCommentStringField(body: string, key: string): string | undefined { + const match = new RegExp(`^#\\s*${key}\\s*=\\s*\"([^\"]*)\"`, "m").exec(body); + return match?.[1]; +} + export function parseTomlArrayField(body: string, key: string): string[] { const match = new RegExp(`^${key}\\s*=\\s*\\[([^\\]]*)\\]`, "m").exec(body); if (!match) return []; return match[1].split(",").map((part) => part.trim().replace(/^['\"]|['\"]$/g, "")).filter(Boolean); } +export function parseTomlCommentArrayField(body: string, key: string): string[] { + const match = new RegExp(`^#\\s*${key}\\s*=\\s*\\[([^\\]]*)\\]`, "m").exec(body); + if (!match) return []; + return match[1].split(",").map((part) => part.trim().replace(/^['\"]|['\"]$/g, "")).filter(Boolean); +} + export function inferComplexityFromId(id: string): PromptSurfaceComplexity { if (/help|status|changelog|patch-notes|smoke-check|project-stage-detect|standards/.test(id)) return "simple"; if (/bugfix|hotfix|qa|test|regression|playtest|localize|perf|code-review|dev-story|task|story/.test(id)) return "moderate"; diff --git a/src/validation.ts b/src/validation.ts index 42e72ab..589d3cc 100644 --- a/src/validation.ts +++ b/src/validation.ts @@ -8,6 +8,7 @@ import { projectAgentsMdRequiredSections, projectRolePromptSourceInput, renderPr import { validateApprovalStore } from "./approvals.js"; import { runBehavioralEvaluations } from "./behavioral-evaluation.js"; import { checkCodexAvailability } from "./codex-runtime.js"; +import { validatePerformanceEvaluationFramework } from "./performance-evaluation.js"; import { contextManifestInput, createContextManifest, type ContextManifest, type ContextManifestMeta } from "./context-manifest.js"; import { validateProjectCustomization } from "./customization.js"; import { createCodexStudioSession } from "./codex-session.js"; @@ -26,6 +27,8 @@ import { isReasoningEffort, parsePromptSurfaceFrontmatter, parseTomlArrayField, + parseTomlCommentArrayField, + parseTomlCommentStringField, parseTomlStringField, validateAgentDescriptionQuality, validateModelPolicy, @@ -307,11 +310,13 @@ function agentPromptSurfaceChecks(root: string, file: string): ValidationCheck[] const policy = validateModelPolicy({ model: model ?? "", model_reasoning_effort: effort }); checks.push(discovery.valid ? pass(`prompt_surface.agent.${id}.discovery_metadata`, `${id} discovery metadata is selection-oriented`, full) : fail(`prompt_surface.agent.${id}.discovery_metadata`, `${id} weak discovery metadata: ${discovery.diagnostics.map((diagnostic) => diagnostic.id).join(", ")}`, full)); checks.push(policy.valid ? pass(`prompt_surface.agent.${id}.model`, `${id} uses exact Codex model policy`, full) : fail(`prompt_surface.agent.${id}.model`, `${id} invalid model policy: ${policy.issues.join(", ")}`, full)); - checks.push(hashLooksValid(parseTomlStringField(body, "source_hash")) && !!parseTomlStringField(body, "source_reference") ? pass(`prompt_surface.agent.${id}.traceability`, `${id} source traceability present`, full) : fail(`prompt_surface.agent.${id}.traceability`, `${id} missing source_reference/source_hash`, full)); - const skills = parseTomlArrayField(body, "primary_skills"); + const sourceReference = parseTomlCommentStringField(body, "source_reference"); + const sourceHash = parseTomlCommentStringField(body, "source_hash"); + checks.push(hashLooksValid(sourceHash) && !!sourceReference ? pass(`prompt_surface.agent.${id}.traceability`, `${id} source traceability present`, full) : fail(`prompt_surface.agent.${id}.traceability`, `${id} missing commented source_reference/source_hash`, full)); + const skills = parseTomlCommentArrayField(body, "primary_skills"); const brokenSkill = skills.find((skill) => !existsSync(path.join(root, ".agents", "skills", skill, "SKILL.md"))); checks.push(skills.length && !brokenSkill ? pass(`prompt_surface.agent.${id}.links`, `${id} linked skills resolve`, full) : fail(`prompt_surface.agent.${id}.links`, brokenSkill ? `${id} linked skill missing: ${brokenSkill}` : `${id} missing linked skills`, full)); - checks.push(parseTomlArrayField(body, "allowed_tool_categories").length ? pass(`prompt_surface.agent.${id}.tool_policy`, `${id} tool policy present`, full) : fail(`prompt_surface.agent.${id}.tool_policy`, `${id} missing tool policy`, full)); + checks.push(parseTomlCommentArrayField(body, "allowed_tool_categories").length ? pass(`prompt_surface.agent.${id}.tool_policy`, `${id} tool policy present`, full) : fail(`prompt_surface.agent.${id}.tool_policy`, `${id} missing tool policy`, full)); checks.push(promptSurfaceDepth(body, ["Use When", "Do Not Use When", "Procedure", "Handoff Contract", "Stop Conditions"]) >= 40 ? pass(`prompt_surface.agent.${id}.depth`, `${id} prompt depth sufficient`, full) : fail(`prompt_surface.agent.${id}.depth`, `${id} prompt surface too thin`, full)); return checks; } @@ -458,6 +463,8 @@ export async function validateRepo(root = process.cwd()): Promise fail("templates", message))); for (const [id, info] of Object.entries(templateRegistry)) { diff --git a/tests/performance-evaluation-framework.test.ts b/tests/performance-evaluation-framework.test.ts new file mode 100644 index 0000000..0ebf2a0 --- /dev/null +++ b/tests/performance-evaluation-framework.test.ts @@ -0,0 +1,119 @@ +import { describe, test } from "node:test"; +import { expect } from "expect"; +import { + gradePerformanceEvaluationScenario, + loadPerformanceEvaluationFramework, + summarizePerformanceEvaluationCoverage, + validatePerformanceEvaluationFramework, + type PerformanceEvaluationScenario +} from "../src/performance-evaluation.js"; + +const scenario: PerformanceEvaluationScenario = { + id: "skill.cgs-skill-test.behavioral-spec", + target: "cgs-skill-test", + kind: "skill", + priority: "critical", + manualOnly: true, + prompt: "eval-framework/scenarios/cgs-skill-test/behavioral-spec/prompt.md", + expected: { + mustRead: [".agents/skills/cgs-skill-test/SKILL.md", "eval-framework/rubrics/skill-behavior.json"], + mustChange: ["production/session-state/eval-report.md"], + mustNotChange: ["src/**", ".agents/skills/**"], + mustRunOrExplain: ["npm run validate"], + report: { required: true } + }, + grading: { + deterministic: ["required-read", "write-boundary", "verification-evidence", "report-presence"], + semanticDimensions: ["triggering", "context-selection", "output-quality", "verification-discipline", "token-discipline"] + } +}; + +describe("performance evaluation framework", () => { + test("grades skill and prompt behavior from scenario expectations instead of skill existence", () => { + const passResult = gradePerformanceEvaluationScenario(scenario, { + traceEvidence: "Read .agents/skills/cgs-skill-test/SKILL.md and eval-framework/rubrics/skill-behavior.json; ran npm run validate", + changedFiles: ["production/session-state/eval-report.md"], + finalReport: "## Eval Report\nPASS", + usage: { status: "recorded", model: "gpt-5.3-codex-spark", inputTokens: 1200, cachedInputTokens: 200, outputTokens: 300, reasoningOutputTokens: 100, totalTokens: 1500 } + }); + + expect(passResult.status).toBe("pass"); + expect(passResult.usage?.model).toBe("gpt-5.3-codex-spark"); + expect(passResult.usage?.totalTokens).toBe(1500); + expect(passResult.failures).toEqual([]); + + const failResult = gradePerformanceEvaluationScenario(scenario, { + traceEvidence: "Read only .agents/skills/cgs-skill-test/SKILL.md", + changedFiles: [".agents/skills/cgs-skill-test/SKILL.md", "src/behavioral-evaluation.ts"], + finalReport: "" + }); + + expect(failResult.status).toBe("fail"); + expect(failResult.failures.map((failure) => failure.id)).toEqual(expect.arrayContaining([ + "required-read-not-recorded", + "forbidden-write", + "missing-required-change", + "verification-not-recorded", + "missing-report" + ])); + expect(failResult.failures.map((failure) => failure.id)).not.toContain("skill-missing"); + }); + + test("loads first-pass coverage across workflow prompts, skills, and role prompts", () => { + const framework = loadPerformanceEvaluationFramework(process.cwd()); + const summary = summarizePerformanceEvaluationCoverage(framework); + + expect(framework.catalog.manualOnly).toBe(true); + expect(summary.scenarios).toBeGreaterThanOrEqual(30); + expect(summary.targets).toBeGreaterThanOrEqual(30); + expect(summary.byKind.workflow).toBeGreaterThanOrEqual(10); + expect(summary.byKind.skill).toBeGreaterThanOrEqual(10); + expect(summary.byKind.role).toBeGreaterThanOrEqual(6); + expect(summary.surfacePaths).toBeGreaterThanOrEqual(30); + expect(framework.catalog.targets.every((target) => target.scenarios.length > 0)).toBe(true); + expect(framework.catalog.modelPolicy?.tokenEstimation?.required).toBe(true); + expect(framework.catalog.modelPolicy?.tokenEstimation?.fields).toEqual(expect.arrayContaining([ + "inputTokens", + "cachedInputTokens", + "outputTokens", + "reasoningOutputTokens", + "totalTokens" + ])); + expect(framework.catalog.modelPolicy?.defaultEvaluationModel).toBe("gpt-5.3-codex-spark"); + expect(framework.catalog.modelPolicy?.allowedEvaluationModels).toContain("gpt-5.5"); + expect(framework.catalog.runners.runOutputPolicy?.repositoryTracked).toBe(true); + expect(framework.catalog.runners.runOutputPolicy?.root).toBe("eval-framework/runs"); + expect(framework.catalog.runners.runOutputPolicy?.requiredFiles).toEqual(expect.arrayContaining(["summary.md", "audit.json"])); + expect(framework.catalog.evaluationPlan?.map((stage) => stage.stage)).toEqual([ + "smoke-critical", + "workflow-high-risk", + "skill-maintenance", + "role-boundary", + "full-regression" + ]); + expect(framework.scenarios.every((loaded) => loaded.grading.semanticDimensions.length >= 4)).toBe(true); + expect(framework.scenarios.some((loaded) => loaded.expected.mustNotChange.some((pattern) => pattern.includes(".agents/skills")))).toBe(true); + }); + + test("validation rejects existence-only checks and enforces first-pass coverage", () => { + const checks = validatePerformanceEvaluationFramework(process.cwd()); + const ids = checks.map((check) => check.id); + const messages = checks.map((check) => check.message.toLowerCase()); + + expect(checks.every((check) => check.status === "pass")).toBe(true); + expect(ids.some((id) => /skill.*exists|exists.*skill|presence-only/i.test(id))).toBe(false); + expect(messages.some((message) => message.includes("skill exists") || message.includes("existence-only"))).toBe(false); + expect(ids).toEqual(expect.arrayContaining([ + "performance_eval.catalog.manual_only", + "performance_eval.scenarios.behavioral_expectations", + "performance_eval.rubrics.semantic_dimensions", + "performance_eval.strategy.no_existence_only_checks", + "performance_eval.token_estimation", + "performance_eval.model_policy", + "performance_eval.plan.gradual", + "performance_eval.template_user_optional", + "performance_eval.results.repository_saved", + "performance_eval.coverage.first_pass" + ])); + }); +}); diff --git a/tests/prompt-surface-validation.test.ts b/tests/prompt-surface-validation.test.ts index b5274b9..716f789 100644 --- a/tests/prompt-surface-validation.test.ts +++ b/tests/prompt-surface-validation.test.ts @@ -13,7 +13,7 @@ function makeRoot(): string { mkdirSync(path.join(root, ".agents", "skills", "cgs-bugfix"), { recursive: true }); mkdirSync(path.join(root, ".agents", "skills", "cgs-standards-gameplay"), { recursive: true }); writeFileSync(path.join(root, "AGENTS.md"), "# Template\n\n.codex/agents game template guidance.\n"); - writeFileSync(path.join(root, ".codex", "agents", "producer.toml"), `name = "producer"\ndescription = "Producer"\nmodel = "gpt-5.5"\nmodel_reasoning_effort = "high"\nsource_reference = ".claude/agents/producer.md"\nsource_hash = "${"a".repeat(64)}"\nprimary_skills = ["cgs-bugfix"]\nallowed_tool_categories = ["read", "edit", "shell"]\ndeveloper_instructions = """\n## Stop Conditions\n\nStop.\n## Use When\n\nUse.\n## Do Not Use When\n\nDo not.\n## Procedure\n\n1. Do work.\n## Handoff Contract\n\nReport evidence.\n"""\n`); + writeFileSync(path.join(root, ".codex", "agents", "producer.toml"), `name = "producer"\ndescription = "Producer"\nmodel = "gpt-5.5"\nmodel_reasoning_effort = "high"\n# source_reference = ".claude/agents/producer.md"\n# source_hash = "${"a".repeat(64)}"\n# primary_skills = ["cgs-bugfix"]\n# allowed_tool_categories = ["read", "edit", "shell"]\ndeveloper_instructions = """\n## Stop Conditions\n\nStop.\n## Use When\n\nUse.\n## Do Not Use When\n\nDo not.\n## Procedure\n\n1. Do work.\n## Handoff Contract\n\nReport evidence.\n"""\n`); const skillBody = (name: string, model: string, effort: string, hash: string) => `---\nname: ${name}\ndescription: ${name}\nmodel: ${model}\nmodel_reasoning_effort: ${effort}\nargument-hint: describe target\nprimary-agent: producer\ntool-policy: read/edit/shell\nisolation: repository-root\nsource-reference: local\nsource-hash: ${hash}\n---\n\n# ${name}\n\n## Purpose\n\nPurpose.\n## Prerequisites\n\nPrereq.\n## Arguments\n\nArgs.\n## Phased Procedure\n\nProcedure.\n## Decision Gates\n\nGates.\n## Output Contract\n\nOutput.\n## Quality Gates\n\nQuality.\n## Failure Modes\n\nFailures.\n## Verification\n\nVerify.\n## Handoff\n\nHandoff.\n`; writeFileSync(path.join(root, ".agents", "skills", "cgs-bugfix", "SKILL.md"), skillBody("cgs-bugfix", "gpt-5.4", "medium", "b".repeat(64))); writeFileSync(path.join(root, ".agents", "skills", "cgs-standards-gameplay", "SKILL.md"), skillBody("cgs-standards-gameplay", "gpt-5.4-mini", "low", "c".repeat(64))); @@ -24,7 +24,7 @@ function makeRoot(): string { describe("prompt surface validation", () => { test("fails exact metadata regressions with stable diagnostics", () => { const root = makeRoot(); - writeFileSync(path.join(root, ".codex", "agents", "bad.toml"), `name = "bad"\ndescription = "Bad"\nmodel = "sonnet"\nmodel_reasoning_effort = "medium"\nsource_reference = "local"\nprimary_skills = ["missing-skill"]\nallowed_tool_categories = ["read"]\ndeveloper_instructions = """thin"""\n`); + writeFileSync(path.join(root, ".codex", "agents", "bad.toml"), `name = "bad"\ndescription = "Bad"\nmodel = "sonnet"\nmodel_reasoning_effort = "medium"\n# source_reference = "local"\n# primary_skills = ["missing-skill"]\n# allowed_tool_categories = ["read"]\ndeveloper_instructions = """thin"""\n`); const failures = validateTemplateSurfaces(root).filter((check) => check.status === "fail").map((check) => check.id); expect(failures).toEqual(expect.arrayContaining(["prompt_surface.agent.bad.model", "prompt_surface.agent.bad.traceability", "prompt_surface.agent.bad.links", "prompt_surface.agent.bad.depth"])); });