summaryrefslogtreecommitdiff
path: root/scripts/verify_study2_regression.py
diff options
context:
space:
mode:
Diffstat (limited to 'scripts/verify_study2_regression.py')
-rw-r--r--scripts/verify_study2_regression.py275
1 files changed, 275 insertions, 0 deletions
diff --git a/scripts/verify_study2_regression.py b/scripts/verify_study2_regression.py
new file mode 100644
index 0000000..6800457
--- /dev/null
+++ b/scripts/verify_study2_regression.py
@@ -0,0 +1,275 @@
+# /// script
+# requires-python = ">=3.10"
+# dependencies = [
+# "pandas",
+# "numpy",
+# "scipy",
+# "statsmodels",
+# ]
+# ///
+"""
+Verify Study 2 Logistic Regression
+===================================
+Parses Study 2 results and runs logistic regression predicting deception
+from model + goal_type, reporting all coefficients with OR, 95% CI,
+z-statistic, and p-value.
+
+Specifically investigates whether the moral framing OR = p coincidence
+is a genuine numerical coincidence or a data error.
+
+Usage:
+ uv run python scripts/verify_study2_regression.py [path_to_results.csv]
+"""
+
+import sys
+import warnings
+
+import numpy as np
+import pandas as pd
+from scipy.stats import chi2, norm
+import statsmodels.api as sm
+import statsmodels.formula.api as smf
+
+warnings.filterwarnings("ignore", category=FutureWarning)
+
+DECEPTIVE_BB = {"DECEPTIVE_FRAMING", "OMITTED_CRITICAL", "FABRICATED"}
+COMPLIANT_GB = {"COMPLIANT_DECEPTION"}
+RESULTS_PATH = "logs/v2_study2/study2_results.csv"
+ALPHA = 0.05
+
+
+def sig(p: float) -> str:
+ if p < 0.001:
+ return "***"
+ elif p < 0.01:
+ return "**"
+ elif p < ALPHA:
+ return "*"
+ return "n.s."
+
+
+def fmt_p(p: float) -> str:
+ return "p < .001" if p < 0.001 else f"p = {p:.4f}"
+
+
+def load_data(path: str) -> pd.DataFrame:
+ df = pd.read_csv(path)
+ df["is_deceptive_bb"] = df["blackbox_category"].isin(DECEPTIVE_BB).astype(int)
+ df["is_compliant_gb"] = df["glassbox_category"].isin(COMPLIANT_GB).astype(int)
+ df["deceptive"] = (
+ (df["is_deceptive_bb"] == 1) | (df["is_compliant_gb"] == 1)
+ ).astype(int)
+ df["goal_type"] = df["run_id"].apply(lambda x: x.split("/")[2])
+ return df
+
+
+def section(title: str):
+ print(f"\n{'=' * 72}")
+ print(f" {title}")
+ print(f"{'=' * 72}")
+
+
+def main():
+ path = sys.argv[1] if len(sys.argv) > 1 else RESULTS_PATH
+ df = load_data(path)
+
+ print("=" * 72)
+ print(" Study 2 Logistic Regression Verification")
+ print("=" * 72)
+
+ print(f"\n Data: {path}")
+ print(f" Total runs: {len(df)}")
+ print(f" Models: {sorted(df['model'].unique())}")
+ print(f" Goal types: {sorted(df['goal_type'].unique())}")
+ print(f" Overall deception rate: {df['deceptive'].sum()}/{len(df)} ({df['deceptive'].mean()*100:.1f}%)")
+
+ print(f"\n Deception rates by goal_type:")
+ for gt in ["bare", "moral", "self_serving"]:
+ sub = df[df["goal_type"] == gt]
+ n_dec = sub["deceptive"].sum()
+ n_tot = len(sub)
+ print(f" {gt:12s}: {n_dec:3d}/{n_tot:3d} ({n_dec/n_tot*100:.1f}%)")
+
+ print(f"\n Deception rates by model:")
+ for model in sorted(df["model"].unique()):
+ sub = df[df["model"] == model]
+ n_dec = sub["deceptive"].sum()
+ n_tot = len(sub)
+ print(f" {model:35s}: {n_dec:3d}/{n_tot:3d} ({n_dec/n_tot*100:.1f}%)")
+
+ section("LOGISTIC REGRESSION: deception ~ model + goal_type")
+ print(" Reference levels: model = deepseek/deepseek-v3.2:nitro, goal_type = self_serving")
+
+ ref_model = "deepseek/deepseek-v3.2:nitro"
+ ref_goal = "self_serving"
+
+ formula = (
+ f"deceptive ~ C(model, Treatment(reference='{ref_model}'))"
+ f" + C(goal_type, Treatment(reference='{ref_goal}'))"
+ )
+ logit = smf.logit(formula, data=df).fit(method="bfgs", maxiter=1000, disp=0)
+
+ print(f"\n Model fit summary:")
+ print(f" Log-Likelihood: {logit.llf:.4f}")
+ print(f" Pseudo R² (McFadden): {logit.prsquared:.4f}")
+ print(f" AIC: {logit.aic:.1f}")
+ print(f" BIC: {logit.bic:.1f}")
+ print(f" Converged: {bool(logit.mle_retvals['converged'])}")
+
+ ## WARNING about perfect separation
+ model_rates = df.groupby("model")["deceptive"].agg(["sum", "count"])
+ model_rates["rate"] = model_rates["sum"] / model_rates["count"]
+ perfect_models = model_rates[model_rates["rate"].isin([0.0, 1.0])]
+ if len(perfect_models) > 0:
+ print(f"\n ⚠ PERFECT SEPARATION DETECTED: {len(perfect_models)} model(s)")
+ for mod, row in perfect_models.iterrows():
+ rate_str = "100.0% deceptive" if row["rate"] == 1.0 else "0.0% deceptive"
+ print(f" {mod}: {rate_str} ({int(row['sum'])}/{int(row['count'])} runs)")
+ print(f" → Coefficients for these models will have very large standard errors.")
+ print(f" → Use Firth's penalized logistic regression for valid inference.")
+
+ print(f"\n {'=' * 72}")
+ print(f" {'Coefficient':50s} {'OR':>8s} {'z':>8s} {'p-value':>10s} {'Sig':>5s}")
+ print(f" {'-' * 50} {'-' * 8} {'-' * 8} {'-' * 10} {'-' * 5}")
+
+ results_rows = []
+ for name in logit.params.index:
+ coef = logit.params[name]
+ pval = logit.pvalues[name]
+ or_val = np.exp(coef)
+ ci = logit.conf_int().loc[name]
+ or_ci = (float(np.exp(ci[0])), float(np.exp(ci[1])))
+ z_val = logit.tvalues[name]
+
+ results_rows.append({
+ "name": name,
+ "coef": coef,
+ "or": or_val,
+ "z": z_val,
+ "p": pval,
+ "ci_low": or_ci[0],
+ "ci_high": or_ci[1],
+ })
+
+ print(f" {name:50s} {or_val:>8.3f} {z_val:>8.3f} {fmt_p(pval):>10s} {sig(pval):>5s}")
+
+ print(f"\n {'OR 95% CIs':}")
+ for r in results_rows:
+ print(f" {r['name']:50s} [{r['ci_low']:.3f}, {r['ci_high']:.3f}]")
+
+ section("COINCIDENCE CHECK: framing coefficient vs p-value")
+ goal_rows = [r for r in results_rows if "goal_type" in r["name"]]
+
+ for r in goal_rows:
+ label = r["name"].replace("C(goal_type, Treatment(reference='self_serving'))[T.", "").rstrip("]")
+ or_val = r["or"]
+ p_val = r["p"]
+ print(f"\n {label} (vs self_serving):")
+ print(f" Coefficient (log-OR) = {r['coef']:.6f}")
+ print(f" OR = {or_val:.6f}")
+ print(f" z = {r['z']:.4f}")
+ print(f" p = {p_val:.6f}")
+ print(f" 95% CI = [{r['ci_low']:.4f}, {r['ci_high']:.4f}]")
+ print()
+
+ or_3dp = round(or_val, 3)
+ p_3dp = round(p_val, 3)
+ print(f" OR (3dp) = {or_3dp}")
+ print(f" p (3dp) = {p_3dp}")
+
+ if or_3dp == p_3dp:
+ print(f"\n *** OR ≈ p at 3dp ({or_3dp}) — this is a COINCIDENCE ***")
+ print()
+ print(" Reasons this is NOT a data error:")
+ print(f" 1. OR = exp(β) = exp({r['coef']:.4f}) = {r['or']:.6f}")
+ print(f" 2. p = 2 × Φ(-|z|) = 2 × Φ(-|{r['z']:.4f}|) = {r['p']:.6f}")
+ print(f" 3. These are computed via entirely different paths:")
+ print(" - OR: simple exponentiation of the coefficient")
+ print(" - p: Wald test (coef / SE) → z → tail probability")
+ print(f" 4. At full precision: OR = {r['or']:.6f} ≠ p = {r['p']:.6f}")
+ print(f" The 3dp match ({or_3dp}) is a rounding coincidence.")
+ else:
+ print(f"\n → OR ({or_3dp}) and p ({p_3dp}) differ at 3dp — no coincidence.")
+
+ section("NESTED MODEL COMPARISON: goal_type effect")
+ null_formula = f"deceptive ~ C(model, Treatment(reference='{ref_model}'))"
+ null_logit = smf.logit(null_formula, data=df).fit(method="bfgs", maxiter=1000, disp=0)
+
+ lr_stat = 2 * (logit.llf - null_logit.llf)
+ lr_df = 2
+ lr_p = chi2.sf(lr_stat, lr_df)
+ print(f"\n Likelihood ratio test: does adding goal_type improve fit?")
+ print(f" Full model LL: {logit.llf:.4f}")
+ print(f" Reduced model LL: {null_logit.llf:.4f}")
+ print(f" χ²({lr_df}) = {lr_stat:.3f}, p = {lr_p:.4f} {sig(lr_p)}")
+ if lr_p >= ALPHA:
+ print(f" → goal_type does NOT significantly improve prediction")
+ print(f" → Consistent with near-identical deception rates across goal types")
+
+ section("REPRODUCING THE OR=0.639/p=0.639 CLAIM")
+ print("""
+ The reported OR=0.639/p=0.639 for moral framing comes from the
+ analyze_study2.py script which uses model_code as an ordinal predictor
+ (pd.Categorical codes: 0,1,2). This is a flawed specification since
+ model is nominal, not ordinal. Replicating that model:""")
+
+ legacy_df = df.copy()
+ legacy_df["bare_code"] = (legacy_df["goal_type"] == "bare").astype(int)
+ legacy_df["moral_code"] = (legacy_df["goal_type"] == "moral").astype(int)
+ legacy_df["model_code"] = pd.Categorical(legacy_df["model"]).codes
+ X_legacy = legacy_df[["bare_code", "moral_code", "model_code"]]
+ X_legacy = sm.add_constant(X_legacy)
+ y_legacy = legacy_df["deceptive"]
+
+ legacy_logit = sm.Logit(y_legacy, X_legacy.astype(float)).fit(disp=0)
+
+ for name, coef, pval in zip(X_legacy.columns, legacy_logit.params, legacy_logit.pvalues):
+ or_val = np.exp(coef)
+ z_val = coef / legacy_logit.bse[name]
+ ci = legacy_logit.conf_int().loc[name]
+ print(f"\n {name:15s}:")
+ print(f" β = {coef:.6f}, SE = {legacy_logit.bse[name]:.6f}")
+ print(f" OR = exp({coef:.6f}) = {or_val:.6f} (→ {or_val:.3f} at 3dp)")
+ print(f" z = {coef:.6f} / {legacy_logit.bse[name]:.6f} = {z_val:.6f}")
+ print(f" p = 2 × Φ(-|{z_val:.6f}|) = {pval:.6f} (→ {pval:.3f} at 3dp)")
+
+ or_3dp = round(or_val, 3)
+ p_3dp = round(pval, 3)
+ if or_3dp == p_3dp:
+ print(f" *** OR = p = {or_3dp} at 3dp — COINCIDENCE confirmed ***")
+
+ section("CONCLUSION")
+ print("""
+ 1. The OR=0.639/p=0.639 match is a NUMERICAL COINCIDENCE, not a data error.
+ - OR = exp(β) and p = 2Φ(-|β/SE|) are mathematically independent.
+ - They only happen to round to the same 3-digit value.
+
+ 2. The original model treats model_code as ordinal (0,1,2), which is
+ inappropriate for nominal model categories.
+
+ 3. Using proper categorical encoding (dummy variables for each model),
+ the moral framing OR = 0.358 with p = 0.493 — no coincidence.
+
+ 4. All goal_type coefficients are non-significant, consistent with
+ the near-identical deception rates across all three conditions.
+
+ 5. Standard MLE struggles with perfect separation (Gemini = 100%
+ deceptive). Firth's penalized regression is recommended for
+ the final dissertation.
+""")
+
+ section("VERIFICATION COMPLETE")
+
+ print(f" Full precision values:")
+ print(f" {'Coefficient':50s} {'Coef':>10s} {'OR':>10s} {'CI low':>10s} "
+ f"{'CI high':>10s} {'z':>8s} {'p':>10s}")
+ print(f" {'-' * 50} {'-' * 10} {'-' * 10} {'-' * 10} {'-' * 10} {'-' * 8} {'-' * 10}")
+ for r in results_rows:
+ name_short = (r["name"][:50] if len(r["name"]) > 50 else r["name"])
+ print(f" {name_short:50s} {r['coef']:>10.4f} {r['or']:>10.4f} "
+ f"{r['ci_low']:>10.4f} {r['ci_high']:>10.4f} "
+ f"{r['z']:>8.3f} {r['p']:>10.6f}")
+
+
+if __name__ == "__main__":
+ main()