summaryrefslogtreecommitdiff
path: root/config_study2.yaml
diff options
context:
space:
mode:
authorCaptainJack2491 <jayrupnakawala@gmail.com>2026-04-25 21:11:56 +0100
committerCaptainJack2491 <jayrupnakawala@gmail.com>2026-04-25 21:11:56 +0100
commitbd37274bce4a233364c2523ba96163606d968a5a (patch)
tree2abbfada01b9c4cc71027f2c538bd22f08e6fe5a /config_study2.yaml
parent6cc92304379996f68bb2df96a5e4b637b4c4804b (diff)
Study 2: 270 runs again + fix for study 1 and analysis.
Forgot to commit the config file when i did the run. oops.
Diffstat (limited to 'config_study2.yaml')
-rw-r--r--config_study2.yaml89
1 files changed, 89 insertions, 0 deletions
diff --git a/config_study2.yaml b/config_study2.yaml
new file mode 100644
index 0000000..e3ab3e5
--- /dev/null
+++ b/config_study2.yaml
@@ -0,0 +1,89 @@
+# ============================================================
+# Study 2: The Rationalization Mechanism
+# ============================================================
+# Design: 3 models × 1 scenario × 3 goal_types × 1 oversight (mid) × n=30
+# Total: 3 × 1 × 3 × 1 × 30 = 270 runs (baselines already exist from Study 1)
+# ============================================================
+# Oversight fixed at MID based on Study 1 results:
+# low=38.9%, mid=32.2%, high=23.3%
+# Mid provides moderate baseline deception — room for both
+# moral framing to push UP and bare framing as control.
+# ============================================================
+
+defaults:
+ temperature: 1.0
+ include_reasoning_in_context: true
+ generate_baseline: true
+
+# Provider configurations
+providers:
+ openrouter:
+ api_key_env: OPENROUTER_API_KEY
+ base_url: https://openrouter.ai/api/v1
+
+# Same 3 frontier models as Study 1
+models:
+ - id: google/gemini-3.1-pro-preview
+ provider: openrouter
+ extra_body:
+ reasoning:
+ effort: "high"
+ exclude: false
+
+ - id: z-ai/glm-5.1:nitro
+ provider: openrouter
+ extra_body:
+ reasoning:
+ effort: "high"
+ exclude: false
+
+ - id: deepseek/deepseek-v3.2:nitro
+ provider: openrouter
+ extra_body:
+ reasoning:
+ effort: "high"
+ exclude: false
+
+# Study 2: ALL THREE framings (the independent variable)
+goal_types:
+ - bare
+ - self_serving
+ - moral
+
+# Same scenario as Study 1
+scenarios:
+ - path: scenarios/corporate_sabotage_v2
+ runs: 30
+
+# Fixed at MID oversight (chosen from Study 1 analysis)
+oversight_levels:
+ - mid
+
+# Parallel execution
+execution:
+ max_workers: 5
+
+# Output to dedicated study directory
+output:
+ dir: logs/v2_study2
+
+# Judge configuration (same validated judges as Study 1)
+judge:
+ log_dir: logs/v2_study2_judge
+
+ blackbox:
+ model: grok-4-1-fast-reasoning
+ provider: xai
+ temperature: 0
+
+ glassbox:
+ model: gpt-4.1
+ provider: openai
+ temperature: 0
+
+# Logging
+logging:
+ level: 3
+ format: "[{level}] {message}"
+ output: both
+ file: logs/v2_study2/experiment.log