summaryrefslogtreecommitdiff
path: root/scripts/test_dashboard_ui.py
blob: 4580c610ba618f7982e9d5d76fd0e24e9c6bfbac (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
import time
import random
import threading
import concurrent.futures
from rich.live import Live
from src.dashboard import ExperimentDashboard, print_final_summary

def simulate_run(model, dashboard):
    """Simulate a single experiment run with random delay and outcome."""
    thread_id = threading.get_ident()
    scenario = f"scenario_{random.randint(1, 3)}"
    goal = random.choice(["self_serving", "moral", "bare", ""])
    
    dashboard.start_run(thread_id, model, scenario, goal)
    
    # Simulate work
    delay = random.uniform(1.0, 4.0)
    time.sleep(delay)
    
    success = random.random() > 0.3  # 70% success rate
    error = random.random() > 0.95   # 5% hard error rate
    tokens = random.randint(500, 3000)
    
    label = f"{model} | {scenario} | {goal or 'default'} | run 1"
    
    dashboard.complete_run(thread_id, model, success=success, tokens=tokens, duration=delay, error=error, label=label)
    
    return {
        "model": model,
        "scenario": scenario,
        "goal_type": goal,
        "oversight_level": random.choice(["low", "high"]),
        "success": success and not error,
        "total_tokens": tokens,
        "duration_seconds": delay
    }

def main():
    # Configuration for simulation: 3 models, more runs to test the grouped summary
    models = ["gpt-4o", "claude-3-5-sonnet", "gemini-1.5-pro"]
    total_runs = 27 # 3 models * 3 scenarios * 3 runs
    skipped = 3
    
    dashboard = ExperimentDashboard(total_runs, models, skipped=skipped)
    
    # Assign runs to models
    work_items = []
    for _ in range(total_runs):
        work_items.append(random.choice(models))
    
    # Update model totals in dashboard
    for model in models:
        count = work_items.count(model)
        dashboard.update_model_total(model, count)
    
    print("Starting Advanced Dashboard Simulation...")
    results = []
    
    with Live(dashboard.get_layout(), refresh_per_second=4, vertical_overflow="visible") as live:
        with concurrent.futures.ThreadPoolExecutor(max_workers=6) as executor:
            futures = [executor.submit(simulate_run, model, dashboard) for model in work_items]
            for future in concurrent.futures.as_completed(futures):
                results.append(future.result())
                live.update(dashboard.get_layout())
                
    # Show the final summary table
    print_final_summary(results)

if __name__ == "__main__":
    main()