From 2ed454c66d9865a8abcbecc0081f7023cf4b50f9 Mon Sep 17 00:00:00 2001 From: CaptainJack2491 Date: Mon, 9 Mar 2026 18:18:48 +0000 Subject: feat: Add web GUI for experiment framework - Add FastAPI backend (api/server.py) with endpoints for: - Config read/write - Scenario/model discovery - Experiment run management (start/cancel/status) - Real-time log streaming via SSE - Results fetching (CSV, images, judge results) - Add vanilla JS frontend (api/static/): - Clean dark-themed dashboard - Configuration panel with dropdowns - Live log terminal - Results viewer with tabs - Add documentation (docs/web_gui_plan.md) Dependencies added: fastapi, uvicorn, sse-starlette --- api/server.py | 491 ++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 491 insertions(+) create mode 100644 api/server.py (limited to 'api/server.py') diff --git a/api/server.py b/api/server.py new file mode 100644 index 0000000..ee5fd2b --- /dev/null +++ b/api/server.py @@ -0,0 +1,491 @@ +""" +FastAPI server for the Web GUI. +Wraps core project functionality without modifying it. +""" +import os +import sys +import asyncio +import subprocess +from pathlib import Path +from typing import Optional, Dict, Any, List +from datetime import datetime +from contextlib import asynccontextmanager + +from fastapi import FastAPI, HTTPException, BackgroundTasks, Request +from fastapi.responses import HTMLResponse, FileResponse, JSONResponse +from fastapi.staticfiles import StaticFiles +from sse_starlette.sse import EventSourceResponse +import yaml + +# Add project root to path to import core modules +PROJECT_ROOT = Path(__file__).parent.parent +sys.path.insert(0, str(PROJECT_ROOT)) + +from src.config_loader import ConfigLoader + + +# Global state for run management +class RunManager: + """Manages experiment runs.""" + + def __init__(self): + self.current_process: Optional[subprocess.Popen] = None + self.status: str = "idle" # idle, running, complete, error + self.start_time: Optional[datetime] = None + self.log_file_path: Optional[str] = None + self.config: Optional[ConfigLoader] = None + + def load_config(self, config_path: str = "config.yaml"): + """Load configuration and update log file path.""" + self.config = ConfigLoader(config_path) + self.config.load() + self.log_file_path = self.config.logging_config.get('file') + + # If relative path, make it absolute from project root + if self.log_file_path and not os.path.isabs(self.log_file_path): + self.log_file_path = os.path.join(PROJECT_ROOT, self.log_file_path) + + return self.config + + async def start_run(self, config_path: str = "config.yaml"): + """Start an experiment run in background.""" + if self.status == "running": + raise HTTPException(status_code=409, detail="A run is already in progress") + + # Load config to get log file path + self.load_config(config_path) + + self.status = "running" + self.start_time = datetime.now() + + # Start the run in background + cmd = [sys.executable, "-m", "uv", "run", "src/main.py"] + self.current_process = subprocess.Popen( + cmd, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + bufsize=1, + cwd=PROJECT_ROOT + ) + + return {"status": "started", "message": "Experiment run started"} + + def cancel_run(self): + """Cancel the current run.""" + if self.current_process: + self.current_process.terminate() + self.current_process = None + self.status = "cancelled" + return {"status": "cancelled"} + return {"status": "idle", "message": "No run to cancel"} + + def get_status(self): + """Get current run status.""" + if self.current_process and self.current_process.poll() is None: + self.status = "running" + elif self.status == "running": + self.status = "complete" + + return { + "status": self.status, + "start_time": self.start_time.isoformat() if self.start_time else None, + } + + +# Global instance +run_manager = RunManager() + + +@asynccontextmanager +async def lifespan(app: FastAPI): + """Application lifespan handler.""" + # Startup: Load config + try: + run_manager.load_config() + except Exception as e: + print(f"Warning: Could not load config: {e}") + + yield + + # Shutdown: Cancel any running process + if run_manager.current_process: + run_manager.cancel_run() + + +# Create FastAPI app +app = FastAPI( + title="AI Agent Reasoning Experiment Framework", + description="Web GUI for running experiments and viewing results", + version="0.1.0", + lifespan=lifespan +) + +# Mount static files +static_dir = Path(__file__).parent / "static" +if static_dir.exists(): + app.mount("/static", StaticFiles(directory=str(static_dir)), name="static") + + +# ============================================================================ +# Root Endpoint - Serve HTML +# ============================================================================ + +@app.get("/", response_class=HTMLResponse) +async def root(): + """Serve the main HTML page.""" + index_path = static_dir / "index.html" + if index_path.exists(): + return FileResponse(index_path) + return HTMLResponse(content="

index.html not found

", status_code=404) + + +# ============================================================================ +# Config Endpoints +# ============================================================================ + +@app.get("/api/config") +async def get_config(): + """Read current config.yaml.""" + config_path = PROJECT_ROOT / "config.yaml" + if not config_path.exists(): + raise HTTPException(status_code=404, detail="config.yaml not found") + + with open(config_path) as f: + config_data = yaml.safe_load(f) + + return config_data + + +@app.put("/api/config") +async def update_config(config_data: Dict[str, Any]): + """Update config.yaml.""" + config_path = PROJECT_ROOT / "config.yaml" + + with open(config_path, 'w') as f: + yaml.dump(config_data, f, default_flow_style=False) + + # Reload config in run manager + run_manager.load_config() + + return {"status": "saved", "message": "Configuration updated"} + + +@app.get("/api/logging") +async def get_logging_config(): + """Get logging configuration including file path.""" + config = run_manager.config + if not config: + raise HTTPException(status_code=500, detail="Config not loaded") + + return { + "level": config.logging_config.get('level'), + "format": config.logging_config.get('format'), + "output": config.logging_config.get('output'), + "file": config.logging_config.get('file'), + "file_absolute": run_manager.log_file_path + } + + +# ============================================================================ +# Discovery Endpoints +# ============================================================================ + +@app.get("/api/scenarios") +async def list_scenarios(): + """List all available scenarios from scenarios/ directory.""" + scenarios_dir = PROJECT_ROOT / "scenarios" + if not scenarios_dir.exists(): + return [] + + scenarios = [] + for item in scenarios_dir.iterdir(): + if item.is_dir() and not item.name.startswith('.'): + # Check for oversight levels + oversight_dir = item / "oversight" + oversight_levels = [] + if oversight_dir.exists(): + oversight_levels = [f.stem for f in oversight_dir.glob("*.md")] + + scenarios.append({ + "name": item.name, + "path": str(item.relative_to(PROJECT_ROOT)), + "oversight_levels": oversight_levels + }) + + return scenarios + + +@app.get("/api/scenarios/{scenario_name}") +async def get_scenario(scenario_name: str): + """Get details for a specific scenario.""" + scenario_path = PROJECT_ROOT / "scenarios" / scenario_name + if not scenario_path.exists(): + raise HTTPException(status_code=404, detail="Scenario not found") + + # Read scenario files + files = {} + for md_file in scenario_path.glob("*.md"): + if md_file.name != "regex_rules.yaml": + with open(md_file) as f: + files[md_file.stem] = f.read() + + # Check oversight levels + oversight_dir = scenario_path / "oversight" + oversight_levels = {} + if oversight_dir.exists(): + for md_file in oversight_dir.glob("*.md"): + with open(md_file) as f: + oversight_levels[md_file.stem] = f.read() + + return { + "name": scenario_name, + "files": files, + "oversight_levels": oversight_levels + } + + +@app.get("/api/models") +async def list_models(): + """List models from config.""" + config = run_manager.config + if not config: + raise HTTPException(status_code=500, detail="Config not loaded") + + return [ + { + "id": model.id, + "provider": model.provider, + "temperature": model.temperature, + "max_tokens": model.max_tokens + } + for model in config.models + ] + + +@app.get("/api/providers") +async def list_providers(): + """List providers from config.""" + config = run_manager.config + if not config: + raise HTTPException(status_code=500, detail="Config not loaded") + + return { + name: { + "base_url": provider.base_url, + "api_key_env": provider.api_key_env + } + for name, provider in config.providers.items() + } + + +# ============================================================================ +# Execution Endpoints +# ============================================================================ + +@app.post("/api/run") +async def start_run(background_tasks: BackgroundTasks): + """Start an experiment run.""" + try: + result = await run_manager.start_run() + return result + except HTTPException: + raise + except Exception as e: + raise HTTPException(status_code=500, detail=str(e)) + + +@app.get("/api/run/status") +async def get_run_status(): + """Get current run status.""" + return run_manager.get_status() + + +@app.delete("/api/run") +async def cancel_run(): + """Cancel the current run.""" + return run_manager.cancel_run() + + +@app.get("/api/logs/stream") +async def log_stream(): + """Stream logs in real-time using SSE.""" + async def event_generator(): + log_file = run_manager.log_file_path + + if not log_file or not os.path.exists(log_file): + yield {"event": "error", "data": "Log file not found"} + return + + # Track file position for tailing + file_pos = 0 + + while True: + # Check if process is still running + status = run_manager.get_status() + if status["status"] == "idle" and not run_manager.current_process: + break + + try: + if os.path.exists(log_file): + with open(log_file, 'r') as f: + f.seek(file_pos) + new_lines = f.readlines() + file_pos = f.tell() + + for line in new_lines: + yield {"event": "log", "data": line.rstrip()} + + # Check if process ended + if run_manager.current_process and run_manager.current_process.poll() is not None: + # Process finished, yield remaining logs + if os.path.exists(log_file): + with open(log_file, 'r') as f: + f.seek(file_pos) + remaining = f.read() + if remaining: + yield {"event": "log", "data": remaining} + break + + except Exception as e: + yield {"event": "error", "data": str(e)} + break + + await asyncio.sleep(0.5) + + yield {"event": "done", "data": "Run completed"} + + return EventSourceResponse(event_generator()) + + +# ============================================================================ +# Results Endpoints +# ============================================================================ + +@app.get("/api/results") +async def get_results(): + """Get experiment results as JSON.""" + config = run_manager.config + if not config: + raise HTTPException(status_code=500, detail="Config not loaded") + + output_dir = PROJECT_ROOT / config.output_dir + + # Look for CSV files + csv_files = list(output_dir.glob("*.csv")) if output_dir.exists() else [] + + results = {} + for csv_file in csv_files: + import pandas as pd + import math + try: + df = pd.read_csv(csv_file) + + # Convert NaN values to None for JSON serialization + def clean_value(val): + if isinstance(val, float) and (math.isnan(val) or math.isinf(val)): + return None + return val + + # Clean each row + cleaned_data = [] + for record in df.to_dict(orient="records"): + cleaned_record = {k: clean_value(v) for k, v in record.items()} + cleaned_data.append(cleaned_record) + + results[csv_file.stem] = { + "columns": df.columns.tolist(), + "data": cleaned_data + } + except Exception as e: + results[csv_file.stem] = {"error": str(e)} + + return results + + +@app.get("/api/results/images") +async def list_result_images(): + """List generated visualization images.""" + config = run_manager.config + if not config: + return [] + + output_dir = PROJECT_ROOT / config.output_dir + viz_dir = output_dir / "viz" + + if not viz_dir.exists(): + return [] + + images = [] + for img in viz_dir.glob("*"): + if img.suffix.lower() in ['.png', '.jpg', '.jpeg', '.gif', '.svg']: + images.append({ + "name": img.name, + "path": str(img.relative_to(PROJECT_ROOT)) + }) + + return images + + +@app.get("/api/results/images/{image_name}") +async def get_result_image(image_name: str): + """Serve a specific image.""" + config = run_manager.config + if not config: + raise HTTPException(status_code=500, detail="Config not loaded") + + output_dir = PROJECT_ROOT / config.output_dir + viz_dir = output_dir / "viz" + image_path = viz_dir / image_name + + if not image_path.exists(): + raise HTTPException(status_code=404, detail="Image not found") + + return FileResponse(image_path) + + +# ============================================================================ +# Judge Results Endpoints +# ============================================================================ + +@app.get("/api/judge/results") +async def get_judge_results(): + """Get judge results if available.""" + config = run_manager.config + if not config: + raise HTTPException(status_code=500, detail="Config not loaded") + + judge_log_dir = config.logging_config.get('file') + if judge_log_dir: + judge_dir = Path(judge_log_dir).parent / "judge" + else: + judge_dir = PROJECT_ROOT / "logs" / "judge" + + if not judge_dir.exists(): + return {"message": "No judge results found"} + + # Look for judge result files + import glob + result_files = list(judge_dir.glob("*.csv")) + list(judge_dir.glob("*.json")) + + results = {} + for rf in result_files: + if rf.suffix == '.csv': + import pandas as pd + df = pd.read_csv(rf) + results[rf.stem] = { + "type": "csv", + "columns": df.columns.tolist(), + "data": df.to_dict(orient="records") + } + elif rf.suffix == '.json': + import json + with open(rf) as f: + results[rf.stem] = {"type": "json", "data": json.load(f)} + + return results + + +if __name__ == "__main__": + import uvicorn + uvicorn.run(app, host="0.0.0.0", port=8000) -- cgit v1.2.3 From d2b1fbd5ac63fd1eb17e68b5364b7dee5cc86e6e Mon Sep 17 00:00:00 2001 From: CaptainJack2491 Date: Mon, 9 Mar 2026 19:11:45 +0000 Subject: feat: upgrade web GUI to modular ES6 and add Chart.js This commit completely overhauls the initial GUI: - Splits monolithic JS/CSS into an ES6 module structure - Implements an industrial 'Control Room' dark theme - Adds dynamic file selection endpoints for CSV and JSON judge logs - Integrates Chart.js to automatically render visualizations from CSV data - Updates web_gui_plan.md to reflect architectural changes --- api/server.py | 95 +++++-- api/static/app.js | 516 ----------------------------------- api/static/css/base.css | 66 +++++ api/static/css/components.css | 161 +++++++++++ api/static/css/layout.css | 87 ++++++ api/static/css/results.css | 113 ++++++++ api/static/css/terminal.css | 77 ++++++ api/static/index.html | 191 +++++++------ api/static/js/api.js | 46 ++++ api/static/js/components/config.js | 107 ++++++++ api/static/js/components/results.js | 296 ++++++++++++++++++++ api/static/js/components/terminal.js | 49 ++++ api/static/js/main.js | 66 +++++ api/static/js/ui.js | 62 +++++ api/static/style.css | 461 ------------------------------- docs/web_gui_plan.md | 102 +++---- 16 files changed, 1352 insertions(+), 1143 deletions(-) delete mode 100644 api/static/app.js create mode 100644 api/static/css/base.css create mode 100644 api/static/css/components.css create mode 100644 api/static/css/layout.css create mode 100644 api/static/css/results.css create mode 100644 api/static/css/terminal.css create mode 100644 api/static/js/api.js create mode 100644 api/static/js/components/config.js create mode 100644 api/static/js/components/results.js create mode 100644 api/static/js/components/terminal.js create mode 100644 api/static/js/main.js create mode 100644 api/static/js/ui.js delete mode 100644 api/static/style.css (limited to 'api/server.py') diff --git a/api/server.py b/api/server.py index ee5fd2b..8bb1e24 100644 --- a/api/server.py +++ b/api/server.py @@ -362,17 +362,40 @@ async def log_stream(): # Results Endpoints # ============================================================================ +@app.get("/api/results/files") +async def get_results_files(): + """List all CSV files in the output directory.""" + config = run_manager.config + if not config: + raise HTTPException(status_code=500, detail="Config not loaded") + + output_dir = PROJECT_ROOT / config.output_dir + if not output_dir.exists(): + return [] + + files = [] + for p in output_dir.rglob("*.csv"): + files.append(str(p.relative_to(output_dir))) + return sorted(files) + @app.get("/api/results") -async def get_results(): - """Get experiment results as JSON.""" +async def get_results(file: str = None): + """Get experiment results as JSON, optionally for a specific file.""" config = run_manager.config if not config: raise HTTPException(status_code=500, detail="Config not loaded") output_dir = PROJECT_ROOT / config.output_dir - # Look for CSV files - csv_files = list(output_dir.glob("*.csv")) if output_dir.exists() else [] + csv_files = [] + if file: + file_path = output_dir / file + if not file_path.exists() or ".." in file: + raise HTTPException(status_code=404, detail="File not found") + csv_files.append(file_path) + else: + # Backward compatibility or default empty + csv_files = list(output_dir.glob("*.csv")) if output_dir.exists() else [] results = {} for csv_file in csv_files: @@ -448,40 +471,70 @@ async def get_result_image(image_name: str): # Judge Results Endpoints # ============================================================================ +@app.get("/api/judge/files") +async def get_judge_files(): + """Get list of judge result files.""" + config = run_manager.config + if not config: + raise HTTPException(status_code=500, detail="Config not loaded") + + judge_log_dir = config._config.get('judge', {}).get('log_dir', 'judge_logs') + judge_dir = PROJECT_ROOT / judge_log_dir + + if not judge_dir.exists(): + return [] + + files = [] + for p in judge_dir.rglob("*"): + if p.suffix in ['.csv', '.json']: + files.append(str(p.relative_to(judge_dir))) + return sorted(files) + + @app.get("/api/judge/results") -async def get_judge_results(): +async def get_judge_results(file: str = None): """Get judge results if available.""" config = run_manager.config if not config: raise HTTPException(status_code=500, detail="Config not loaded") - judge_log_dir = config.logging_config.get('file') - if judge_log_dir: - judge_dir = Path(judge_log_dir).parent / "judge" - else: - judge_dir = PROJECT_ROOT / "logs" / "judge" + judge_log_dir = config._config.get('judge', {}).get('log_dir', 'judge_logs') + judge_dir = PROJECT_ROOT / judge_log_dir if not judge_dir.exists(): return {"message": "No judge results found"} - # Look for judge result files - import glob - result_files = list(judge_dir.glob("*.csv")) + list(judge_dir.glob("*.json")) + result_files = [] + if file: + file_path = judge_dir / file + if not file_path.exists() or ".." in file: + raise HTTPException(status_code=404, detail="File not found") + result_files.append(file_path) + else: + # Default empty or backward compatibility + import glob + result_files = list(judge_dir.glob("*.csv")) + list(judge_dir.glob("*.json")) results = {} for rf in result_files: if rf.suffix == '.csv': import pandas as pd - df = pd.read_csv(rf) - results[rf.stem] = { - "type": "csv", - "columns": df.columns.tolist(), - "data": df.to_dict(orient="records") - } + try: + df = pd.read_csv(rf) + results[rf.stem] = { + "type": "csv", + "columns": df.columns.tolist(), + "data": df.to_dict(orient="records") + } + except Exception as e: + results[rf.stem] = {"error": str(e)} elif rf.suffix == '.json': import json - with open(rf) as f: - results[rf.stem] = {"type": "json", "data": json.load(f)} + try: + with open(rf) as f: + results[rf.stem] = {"type": "json", "data": json.load(f)} + except Exception as e: + results[rf.stem] = {"error": str(e)} return results diff --git a/api/static/app.js b/api/static/app.js deleted file mode 100644 index 0dcbd01..0000000 --- a/api/static/app.js +++ /dev/null @@ -1,516 +0,0 @@ -/* ========================================================================== - AI Agent Reasoning Experiment Framework - Web GUI App - ========================================================================== */ - -const API_BASE = ''; - -// State -let eventSource = null; -let isRunning = false; - -// ========================================================================== -// DOM Elements -// ========================================================================== - -const elements = { - // Configuration - scenarioSelect: document.getElementById('scenario-select'), - modelSelect: document.getElementById('model-select'), - oversightSelect: document.getElementById('oversight-select'), - runsInput: document.getElementById('runs-input'), - - // Buttons - runBtn: document.getElementById('run-btn'), - cancelBtn: document.getElementById('cancel-btn'), - clearLogsBtn: document.getElementById('clear-logs-btn'), - - // Status - statusBadge: document.getElementById('status-badge'), - statusTime: document.getElementById('status-time'), - - // Logs - logsContainer: document.getElementById('logs-container'), - - // Results - csvResults: document.getElementById('csv-results'), - imageResults: document.getElementById('image-results'), - judgeResults: document.getElementById('judge-results'), - - // Tabs - tabBtns: document.querySelectorAll('.tab-btn'), - tabContents: document.querySelectorAll('.tab-content'), -}; - -// ========================================================================== -// Initialization -// ========================================================================== - -document.addEventListener('DOMContentLoaded', async () => { - await loadConfig(); - await loadScenarios(); - await loadModels(); - setupEventListeners(); - startStatusPolling(); -}); - -// ========================================================================== -// API Functions -// ========================================================================== - -async function loadConfig() { - try { - const response = await fetch(`${API_BASE}/api/config`); - const config = await response.json(); - - // Set default values from config - if (config.defaults?.oversight) { - elements.oversightSelect.value = config.defaults.oversight; - } - } catch (error) { - console.error('Failed to load config:', error); - } -} - -async function loadScenarios() { - try { - const response = await fetch(`${API_BASE}/api/scenarios`); - const scenarios = await response.json(); - - elements.scenarioSelect.innerHTML = scenarios.map(s => - `` - ).join(''); - - // If only one scenario, auto-select it and load oversight levels - if (scenarios.length === 1) { - elements.scenarioSelect.value = scenarios[0].name; - updateOversightLevels(scenarios[0]); - } - } catch (error) { - console.error('Failed to load scenarios:', error); - elements.scenarioSelect.innerHTML = ''; - } -} - -async function loadModels() { - try { - const response = await fetch(`${API_BASE}/api/models`); - const models = await response.json(); - - elements.modelSelect.innerHTML = models.map(m => - `` - ).join(''); - } catch (error) { - console.error('Failed to load models:', error); - elements.modelSelect.innerHTML = ''; - } -} - -async function startRun() { - const scenario = elements.scenarioSelect.value; - const model = elements.modelSelect.value; - const oversight = elements.oversightSelect.value; - const runs = elements.runsInput.value; - - if (!scenario || !model) { - alert('Please select a scenario and model'); - return; - } - - // Show loading state - elements.runBtn.classList.add('loading'); - elements.runBtn.disabled = true; - elements.cancelBtn.disabled = false; - isRunning = true; - - // Clear logs - elements.logsContainer.innerHTML = ''; - appendLog('info', `Starting experiment: ${scenario} with ${model} (oversight: ${oversight}, runs: ${runs})`); - - try { - // TODO: For now, we just trigger the run with current config - // In the future, we could modify config via API - const response = await fetch(`${API_BASE}/api/run`, { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ scenario, model, oversight, runs }) - }); - - if (!response.ok) { - throw new Error(`HTTP ${response.status}`); - } - - // Start log streaming - startLogStream(); - updateStatus('running'); - - } catch (error) { - console.error('Failed to start run:', error); - appendLog('error', `Failed to start: ${error.message}`); - resetRunState(); - } -} - -async function cancelRun() { - try { - const response = await fetch(`${API_BASE}/api/run`, { - method: 'DELETE' - }); - - const result = await response.json(); - appendLog('warn', 'Run cancelled by user'); - updateStatus('cancelled'); - - } catch (error) { - console.error('Failed to cancel run:', error); - } finally { - stopLogStream(); - resetRunState(); - } -} - -// ========================================================================== -// Log Streaming -// ========================================================================== - -function startLogStream() { - stopLogStream(); // Close any existing connection - - eventSource = new EventSource(`${API_BASE}/api/logs/stream`); - - eventSource.onmessage = (event) => { - const data = event.data; - if (data) { - appendLog('info', data); - } - }; - - eventSource.addEventListener('log', (event) => { - const data = event.data; - if (data) { - appendLog('info', data); - } - }); - - eventSource.addEventListener('done', (event) => { - appendLog('info', '=== Run Complete ==='); - stopLogStream(); - isRunning = false; - updateStatus('complete'); - resetRunState(); - loadResults(); - }); - - eventSource.addEventListener('error', (event) => { - console.error('SSE Error:', event); - }); -} - -function stopLogStream() { - if (eventSource) { - eventSource.close(); - eventSource = null; - } -} - -function appendLog(level, message) { - // Remove empty state if present - const emptyState = elements.logsContainer.querySelector('.logs-empty'); - if (emptyState) { - emptyState.remove(); - } - - const line = document.createElement('div'); - line.className = 'log-line'; - - // Color based on content - let logLevel = 'log-level-3'; - if (message.includes('[ERROR]') || message.includes('error') || message.includes('Error')) { - logLevel = 'log-level-1'; - } else if (message.includes('[WARN]') || message.includes('warning') || message.includes('Warning')) { - logLevel = 'log-level-2'; - } else if (message.includes('[DEBUG]') || message.includes('[DEBUG+]')) { - logLevel = 'log-level-4'; - } - - line.classList.add(logLevel); - line.textContent = message; - - elements.logsContainer.appendChild(line); - - // Auto-scroll to bottom - elements.logsContainer.scrollTop = elements.logsContainer.scrollHeight; -} - -function clearLogs() { - elements.logsContainer.innerHTML = '
Run an experiment to see logs...
'; -} - -// ========================================================================== -// Status Polling -// ========================================================================== - -let statusPollingInterval = null; - -function startStatusPolling() { - statusPollingInterval = setInterval(async () => { - try { - const response = await fetch(`${API_BASE}/api/run/status`); - const status = await response.json(); - - if (status.status !== 'idle' && !isRunning) { - // Something is running but we don't know about it - isRunning = true; - elements.runBtn.classList.add('loading'); - elements.runBtn.disabled = true; - elements.cancelBtn.disabled = false; - startLogStream(); - } - - updateStatusDisplay(status.status, status.start_time); - - } catch (error) { - console.error('Status poll error:', error); - } - }, 2000); -} - -function updateStatus(status) { - updateStatusDisplay(status); - - if (status === 'running') { - elements.runBtn.classList.add('loading'); - elements.runBtn.disabled = true; - elements.cancelBtn.disabled = false; - } -} - -function updateStatusDisplay(status, startTime = null) { - elements.statusBadge.className = `badge badge-${status}`; - - const statusText = { - 'idle': 'Idle', - 'running': 'Running...', - 'complete': 'Complete', - 'error': 'Error', - 'cancelled': 'Cancelled' - }; - - elements.statusBadge.textContent = statusText[status] || status; - - if (startTime) { - const date = new Date(startTime); - elements.statusTime.textContent = `Started: ${date.toLocaleTimeString()}`; - } else if (status === 'idle') { - elements.statusTime.textContent = ''; - } -} - -function resetRunState() { - elements.runBtn.classList.remove('loading'); - elements.runBtn.disabled = false; - elements.cancelBtn.disabled = true; - isRunning = false; -} - -// ========================================================================== -// Results Loading -// ========================================================================== - -async function loadResults() { - await Promise.all([ - loadCSVResults(), - loadImageResults(), - loadJudgeResults() - ]); -} - -async function loadCSVResults() { - try { - const response = await fetch(`${API_BASE}/api/results`); - const results = await response.json(); - - if (Object.keys(results).length === 0) { - elements.csvResults.innerHTML = '
No results yet
'; - return; - } - - let html = ''; - - for (const [filename, data] of Object.entries(results)) { - if (data.error) { - html += `

${filename}

Error: ${data.error}

`; - continue; - } - - html += `

${filename}.csv

`; - html += '
'; - - // Header - html += ''; - for (const col of data.columns) { - html += ``; - } - html += ''; - - // Body - html += ''; - for (const row of data.data) { - html += ''; - for (const col of data.columns) { - html += ``; - } - html += ''; - } - html += '
${col}
${row[col] ?? ''}
'; - } - - elements.csvResults.innerHTML = html || '
No results yet
'; - - } catch (error) { - console.error('Failed to load CSV results:', error); - } -} - -async function loadImageResults() { - try { - const response = await fetch(`${API_BASE}/api/results/images`); - const images = await response.json(); - - if (images.length === 0) { - elements.imageResults.innerHTML = '
No visualizations yet
'; - return; - } - - elements.imageResults.innerHTML = images.map(img => ` -
- ${img.name} -
${img.name}
-
- `).join(''); - - } catch (error) { - console.error('Failed to load images:', error); - } -} - -async function loadJudgeResults() { - try { - const response = await fetch(`${API_BASE}/api/judge/results`); - const results = await response.json(); - - if (results.message || Object.keys(results).length === 0) { - elements.judgeResults.innerHTML = '
No judge results yet
'; - return; - } - - let html = ''; - - for (const [filename, data] of Object.entries(results)) { - if (data.type === 'csv') { - html += `

${filename}

`; - html += '
'; - - html += ''; - for (const col of data.columns) { - html += ``; - } - html += ''; - - for (const row of data.data) { - html += ''; - for (const col of data.columns) { - html += ``; - } - html += ''; - } - html += '
${col}
${row[col] ?? ''}
'; - } else { - html += `

${filename}

${JSON.stringify(data.data, null, 2)}
`; - } - } - - elements.judgeResults.innerHTML = html || '
No judge results
'; - - } catch (error) { - console.error('Failed to load judge results:', error); - } -} - -// ========================================================================== -// Event Listeners -// ========================================================================== - -function setupEventListeners() { - // Run button - elements.runBtn.addEventListener('click', startRun); - - // Cancel button - elements.cancelBtn.addEventListener('click', cancelRun); - - // Clear logs button - elements.clearLogsBtn.addEventListener('click', clearLogs); - - // Scenario selection - update oversight levels - elements.scenarioSelect.addEventListener('change', async (e) => { - const scenarioName = e.target.value; - if (!scenarioName) return; - - try { - const response = await fetch(`${API_BASE}/api/scenarios/${scenarioName}`); - const scenario = await response.json(); - updateOversightLevels(scenario); - } catch (error) { - console.error('Failed to load scenario details:', error); - } - }); - - // Tab switching - elements.tabBtns.forEach(btn => { - btn.addEventListener('click', () => { - const tabId = btn.dataset.tab; - - // Update active tab button - elements.tabBtns.forEach(b => b.classList.remove('active')); - btn.classList.add('active'); - - // Update active tab content - elements.tabContents.forEach(content => { - content.classList.remove('active'); - if (content.id === `tab-${tabId}`) { - content.classList.add('active'); - } - }); - - // Load results if switching to results tab - if (tabId === 'csv' || tabId === 'images' || tabId === 'judge') { - loadResults(); - } - }); - }); - - // Keyboard shortcuts - document.addEventListener('keydown', (e) => { - // Ctrl/Cmd + Enter to run - if ((e.ctrlKey || e.metaKey) && e.key === 'Enter' && !isRunning) { - startRun(); - } - }); -} - -function updateOversightLevels(scenario) { - const oversightSelect = elements.oversightSelect; - const levels = scenario.oversight_levels || []; - - if (levels.length === 0) { - // No scenario-specific oversight, use global levels - oversightSelect.innerHTML = ` - - - - `; - } else { - oversightSelect.innerHTML = levels.map(level => - `` - ).join(''); - } -} diff --git a/api/static/css/base.css b/api/static/css/base.css new file mode 100644 index 0000000..aa28ba5 --- /dev/null +++ b/api/static/css/base.css @@ -0,0 +1,66 @@ +:root { + /* Colors - Deep Industrial Dark */ + --color-bg-0: #0a0a0b; /* Deepest background */ + --color-bg-1: #121214; /* Sidebar / Cards */ + --color-bg-2: #1c1c1f; /* Input fields */ + --color-bg-3: #2a2a2e; /* Borders / Hover */ + + /* Text */ + --color-text-primary: #f0f0f2; + --color-text-secondary: #9ea0a6; + --color-text-muted: #62646c; + + /* Accent - Pure Purple */ + --color-accent: #9d4edd; + --color-accent-hover: #7b2cbf; + --color-accent-text: #ffffff; + + /* Semantic */ + --color-success: #10b981; + --color-warning: #f59e0b; + --color-danger: #ef4444; + --color-info: #3b82f6; + + /* Spacing & Borders */ + --border-width: 1px; + --border-radius: 4px; /* Crisp corners */ + --space-xs: 4px; + --space-sm: 8px; + --space-md: 16px; + --space-lg: 24px; + --space-xl: 32px; + + /* Transitions */ + --transition-fast: 0.15s ease; +} + +* { + margin: 0; + padding: 0; + box-sizing: border-box; +} + +body { + font-family: 'Inter', -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif; + background-color: var(--color-bg-0); + color: var(--color-text-primary); + line-height: 1.5; + height: 100vh; + overflow: hidden; /* Main scroll happens in panes */ +} + +/* Scrollbar Styling */ +::-webkit-scrollbar { + width: 6px; + height: 6px; +} +::-webkit-scrollbar-track { + background: var(--color-bg-0); +} +::-webkit-scrollbar-thumb { + background: var(--color-bg-3); + border-radius: 10px; +} +::-webkit-scrollbar-thumb:hover { + background: var(--color-accent); +} diff --git a/api/static/css/components.css b/api/static/css/components.css new file mode 100644 index 0000000..9b9a823 --- /dev/null +++ b/api/static/css/components.css @@ -0,0 +1,161 @@ +/* Buttons */ +.btn { + appearance: none; + border: none; + font-family: inherit; + font-size: 0.875rem; + font-weight: 600; + padding: 10px 16px; + border-radius: var(--border-radius); + cursor: pointer; + transition: var(--transition-fast); + display: inline-flex; + align-items: center; + justify-content: center; + gap: var(--space-sm); + text-transform: uppercase; + letter-spacing: 0.02em; +} + +.btn-primary { + background-color: var(--color-accent); + color: var(--color-accent-text); +} +.btn-primary:hover:not(:disabled) { + background-color: var(--color-accent-hover); +} + +.btn-danger { + background-color: transparent; + border: var(--border-width) solid var(--color-danger); + color: var(--color-danger); +} +.btn-danger:hover:not(:disabled) { + background-color: var(--color-danger); + color: white; +} + +.btn-ghost { + background-color: transparent; + color: var(--color-text-secondary); + border: var(--border-width) solid var(--color-bg-3); +} +.btn-ghost:hover:not(:disabled) { + background-color: var(--color-bg-3); + color: var(--color-text-primary); +} + +.btn:disabled { + opacity: 0.4; + cursor: not-allowed; + filter: grayscale(1); +} + +/* Forms */ +.form-group { + display: flex; + flex-direction: column; + gap: var(--space-xs); + margin-bottom: var(--space-md); +} + +.form-label { + font-size: 0.75rem; + font-weight: 700; + color: var(--color-text-muted); + text-transform: uppercase; +} + +.input-field { + background-color: var(--color-bg-2); + border: var(--border-width) solid var(--color-bg-3); + border-radius: var(--border-radius); + padding: 10px 12px; + color: var(--color-text-primary); + font-size: 0.9375rem; + width: 100%; + transition: border-color var(--transition-fast); +} + +.input-field:focus { + outline: none; + border-color: var(--color-accent); +} + +/* Tabs */ +.tab-nav { + display: flex; + border-bottom: var(--border-width) solid var(--color-bg-3); + padding: 0 var(--space-lg); + background-color: var(--color-bg-1); +} + +.tab-trigger { + background: none; + border: none; + padding: 14px 20px; + color: var(--color-text-secondary); + font-size: 0.875rem; + font-weight: 600; + cursor: pointer; + border-bottom: 2px solid transparent; + transition: var(--transition-fast); +} + +.tab-trigger:hover { + color: var(--color-text-primary); +} + +.tab-trigger.active { + color: var(--color-accent); + border-bottom-color: var(--color-accent); +} + +.tab-panel { + display: none; + padding: var(--space-lg); +} + +.tab-panel.active { + display: block; +} + +/* Cards */ +.card { + background-color: var(--color-bg-1); + border: var(--border-width) solid var(--color-bg-3); + border-radius: var(--border-radius); + overflow: hidden; +} + +.card-header { + padding: var(--space-md); + border-bottom: var(--border-width) solid var(--color-bg-3); + font-size: 0.75rem; + font-weight: 700; + text-transform: uppercase; + color: var(--color-text-muted); +} + +.card-body { + padding: var(--space-md); +} + +/* Badges */ +.badge { + font-size: 0.7rem; + font-weight: 800; + padding: 2px 8px; + border-radius: 2px; + text-transform: uppercase; +} + +.badge-idle { background: var(--color-bg-3); color: var(--color-text-secondary); } +.badge-running { background: var(--color-accent); color: white; animation: blink 1.5s infinite; } +.badge-complete { background: var(--color-success); color: white; } + +@keyframes blink { + 0% { opacity: 1; } + 50% { opacity: 0.6; } + 100% { opacity: 1; } +} diff --git a/api/static/css/layout.css b/api/static/css/layout.css new file mode 100644 index 0000000..ef0fc2d --- /dev/null +++ b/api/static/css/layout.css @@ -0,0 +1,87 @@ +.dashboard-root { + display: grid; + grid-template-columns: 320px 1fr; + grid-template-rows: 1fr 300px; + height: 100vh; + width: 100vw; +} + +/* Sidebar - Left Column */ +.sidebar { + grid-row: 1 / -1; + background-color: var(--color-bg-1); + border-right: var(--border-width) solid var(--color-bg-3); + padding: var(--space-lg); + display: flex; + flex-direction: column; + gap: var(--space-xl); + z-index: 10; +} + +/* Main Area - Top Right */ +.main-viewport { + grid-column: 2; + grid-row: 1; + overflow-y: auto; + background-color: var(--color-bg-0); + display: flex; + flex-direction: column; +} + +/* Terminal - Bottom Right */ +.terminal-dock { + grid-column: 2; + grid-row: 2; + background-color: var(--color-bg-1); + border-top: var(--border-width) solid var(--color-bg-3); + display: flex; + flex-direction: column; + overflow: hidden; +} + +/* Header */ +.app-header { + padding: var(--space-md) var(--space-lg); + border-bottom: var(--border-width) solid var(--color-bg-3); + display: flex; + justify-content: space-between; + align-items: center; +} + +.logo-area { + display: flex; + align-items: center; + gap: var(--space-sm); +} + +.logo-area h1 { + font-size: 1rem; + font-weight: 700; + letter-spacing: 0.05em; + text-transform: uppercase; + color: var(--color-accent); +} + +/* Mobile notice */ +@media (max-width: 900px) { + .dashboard-root { + grid-template-columns: 1fr; + grid-template-rows: auto 1fr auto; + overflow-y: auto; + } + .sidebar { + grid-row: 1; + border-right: none; + border-bottom: var(--border-width) solid var(--color-bg-3); + } + .main-viewport { + grid-column: 1; + grid-row: 2; + height: 500px; + } + .terminal-dock { + grid-column: 1; + grid-row: 3; + height: 300px; + } +} diff --git a/api/static/css/results.css b/api/static/css/results.css new file mode 100644 index 0000000..9a532b8 --- /dev/null +++ b/api/static/css/results.css @@ -0,0 +1,113 @@ +.results-container { + padding: var(--space-sm); +} + +.results-controls { + padding: var(--space-md) var(--space-lg); + background-color: var(--color-bg-1); + border-bottom: var(--border-width) solid var(--color-bg-3); + display: flex; + align-items: center; +} + +.data-table-wrapper { + margin-bottom: var(--space-xl); + border: var(--border-width) solid var(--color-bg-3); + border-radius: var(--border-radius); + overflow: hidden; +} + +.data-table-title { + background-color: var(--color-bg-1); + padding: var(--space-sm) var(--space-md); + font-size: 0.75rem; + font-weight: 800; + color: var(--color-accent); + text-transform: uppercase; + border-bottom: var(--border-width) solid var(--color-bg-3); +} + +.data-table { + width: 100%; + border-collapse: collapse; + font-size: 0.8125rem; + background-color: var(--color-bg-1); + table-layout: auto; +} + +.data-table th { + text-align: left; + padding: 12px 16px; + background-color: var(--color-bg-2); + color: var(--color-text-secondary); + font-weight: 700; + text-transform: uppercase; + font-size: 0.65rem; + letter-spacing: 0.05em; + border-bottom: var(--border-width) solid var(--color-bg-3); + white-space: nowrap; +} + +.data-table td { + padding: 10px 16px; + border-bottom: var(--border-width) solid var(--color-bg-3); + color: var(--color-text-primary); + max-width: 400px; + white-space: normal; + word-wrap: break-word; + vertical-align: top; +} + +.data-table tr:last-child td { + border-bottom: none; +} + +.data-table tr:hover td { + background-color: var(--color-bg-2); +} + +/* Image Grid */ +.image-grid { + display: grid; + grid-template-columns: repeat(auto-fill, minmax(400px, 1fr)); + gap: var(--space-lg); + padding: var(--space-lg); +} + +.image-card { + background-color: var(--color-bg-1); + border: var(--border-width) solid var(--color-bg-3); + border-radius: var(--border-radius); + overflow: hidden; + transition: var(--transition-fast); +} + +.image-card:hover { + border-color: var(--color-accent); +} + +.image-card img { + width: 100%; + height: auto; + display: block; + background-color: white; /* Charts usually have white background */ +} + +.image-card-footer { + padding: var(--space-sm) var(--space-md); + font-size: 0.75rem; + font-weight: 600; + color: var(--color-text-secondary); + text-align: center; + border-top: var(--border-width) solid var(--color-bg-3); +} + +/* Chart Canvas Container */ +.chart-container { + background-color: var(--color-bg-1); + border: var(--border-width) solid var(--color-bg-3); + border-radius: var(--border-radius); + padding: var(--space-md); + position: relative; + height: 300px; +} diff --git a/api/static/css/terminal.css b/api/static/css/terminal.css new file mode 100644 index 0000000..7cf79ba --- /dev/null +++ b/api/static/css/terminal.css @@ -0,0 +1,77 @@ +.terminal-header { + background-color: var(--color-bg-2); + padding: var(--space-xs) var(--space-md); + border-bottom: var(--border-width) solid var(--color-bg-3); + display: flex; + justify-content: space-between; + align-items: center; +} + +.terminal-title { + font-size: 0.7rem; + font-weight: 800; + text-transform: uppercase; + color: var(--color-text-muted); + letter-spacing: 0.1em; +} + +.terminal-body { + flex: 1; + overflow-y: auto; + padding: var(--space-md); + font-family: 'JetBrains Mono', 'Fira Code', monospace; + font-size: 0.8125rem; + line-height: 1.6; + background-color: #0d0d0f; /* Slightly darker for terminal */ +} + +.terminal-line { + white-space: pre-wrap; + word-break: break-all; + margin-bottom: 2px; +} + +.terminal-line.info { color: var(--color-text-primary); } +.terminal-line.error { color: var(--color-danger); font-weight: 600; } +.terminal-line.warning { color: var(--color-warning); } +.terminal-line.muted { color: var(--color-text-muted); } +.terminal-line.success { color: var(--color-success); } +.terminal-line.reasoning { color: var(--color-accent); border-left: 2px solid var(--color-accent); padding-left: var(--space-sm); margin: var(--space-sm) 0; } + +.btn-terminal { + background: none; + border: none; + color: var(--color-text-muted); + font-size: 0.65rem; + font-weight: 700; + text-transform: uppercase; + cursor: pointer; + padding: 4px 8px; + border-radius: 2px; + transition: var(--transition-fast); +} + +.btn-terminal:hover { + color: var(--color-text-primary); + background-color: var(--color-bg-3); +} + +.btn-terminal.active { + color: var(--color-accent); +} + +.empty-state { + display: flex; + flex-direction: column; + align-items: center; + justify-content: center; + height: 300px; + color: var(--color-text-muted); + font-size: 0.875rem; + text-transform: uppercase; + letter-spacing: 0.05em; + font-weight: 600; + border: 2px dashed var(--color-bg-3); + margin: var(--space-lg); + border-radius: var(--border-radius); +} diff --git a/api/static/index.html b/api/static/index.html index cdf8d2f..564618a 100644 --- a/api/static/index.html +++ b/api/static/index.html @@ -3,109 +3,142 @@ - AI Agent Reasoning Experiment Framework - + AI Reasoning Framework + + + + + + + + + + + + -
-
-

πŸ€– AI Agent Reasoning Experiment Framework

-

Run experiments, analyze deception, and visualize results

-
+
+ + + + +
+
+
+ + + +
+
+ +
+
+
+ + +
+
+
No active data.
-
+
+
+ + +
+
+
Select a CSV to generate visuals.
+
+ +

Saved Visualizations

-
No visualizations yet. Run an experiment first.
+
No visualizations generated.
-
-
-
No judge results yet.
+
+
+ + +
+
+
No judge reports.
- +
-
-

AI Agent Reasoning Experiment Framework β€’ Dissertation Project

-
+ +
+
+ Event Log +
+ + +
+
+
+
System ready. Waiting for run...
+
+
- + + diff --git a/api/static/js/api.js b/api/static/js/api.js new file mode 100644 index 0000000..ecfe30d --- /dev/null +++ b/api/static/js/api.js @@ -0,0 +1,46 @@ +/** + * api.js - Core API communication + */ + +const API_BASE = ''; + +export const api = { + async get(endpoint) { + const response = await fetch(`${API_BASE}${endpoint}`); + if (!response.ok) throw new Error(`API Error: ${response.status}`); + return await response.json(); + }, + + async post(endpoint, data) { + const response = await fetch(`${API_BASE}${endpoint}`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify(data) + }); + if (!response.ok) throw new Error(`API Error: ${response.status}`); + return await response.json(); + }, + + async delete(endpoint) { + const response = await fetch(`${API_BASE}${endpoint}`, { method: 'DELETE' }); + if (!response.ok) throw new Error(`API Error: ${response.status}`); + return await response.json(); + }, + + streamLogs(onLog, onDone, onError) { + const eventSource = new EventSource(`${API_BASE}/api/logs/stream`); + + eventSource.onmessage = (e) => onLog(e.data); + eventSource.addEventListener('log', (e) => onLog(e.data)); + eventSource.addEventListener('done', () => { + eventSource.close(); + onDone(); + }); + eventSource.onerror = (e) => { + console.error('SSE Error:', e); + if (onError) onError(e); + }; + + return eventSource; + } +}; diff --git a/api/static/js/components/config.js b/api/static/js/components/config.js new file mode 100644 index 0000000..d87c874 --- /dev/null +++ b/api/static/js/components/config.js @@ -0,0 +1,107 @@ +/** + * config.js - Experiment configuration and execution + */ +import { api } from '../api.js'; +import { ui } from '../ui.js'; +import { terminal } from './terminal.js'; +import { results } from './results.js'; + +let isRunning = false; + +export const config = { + async init() { + await Promise.all([ + this.loadScenarios(), + this.loadModels(), + this.loadDefaults() + ]); + + ui.elements.runBtn.addEventListener('click', () => this.startRun()); + ui.elements.cancelBtn.addEventListener('click', () => this.stopRun()); + + // Scenario detail updates + ui.elements.scenarioSelect.addEventListener('change', (e) => this.updateOversight(e.target.value)); + }, + + async loadScenarios() { + const scenarios = await api.get('/api/scenarios'); + ui.elements.scenarioSelect.innerHTML = scenarios.map(s => + `` + ).join(''); + if (scenarios[0]) this.updateOversight(scenarios[0].name); + }, + + async loadModels() { + const models = await api.get('/api/models'); + ui.elements.modelSelect.innerHTML = models.map(m => + `` + ).join(''); + }, + + async loadDefaults() { + const cfg = await api.get('/api/config'); + if (cfg.defaults?.oversight) { + ui.elements.oversightSelect.value = cfg.defaults.oversight; + } + }, + + async updateOversight(scenarioName) { + if (!scenarioName) return; + const details = await api.get(`/api/scenarios/${scenarioName}`); + const levels = details.oversight_levels || ['low', 'mid', 'high']; + ui.elements.oversightSelect.innerHTML = levels.map(l => + `` + ).join(''); + }, + + async startRun() { + const data = { + scenario: ui.elements.scenarioSelect.value, + model: ui.elements.modelSelect.value, + oversight: ui.elements.oversightSelect.value, + runs: ui.elements.runsInput.value + }; + + this.setRunningState(true); + terminal.clear(); + terminal.append(`[SYSTEM] Starting experiment: ${data.scenario} | ${data.model}`); + + try { + await api.post('/api/run', data); + api.streamLogs( + (log) => terminal.append(log), + () => { + this.setRunningState(false); + ui.updateStatus('complete'); + terminal.append('[SYSTEM] Run completed successfully.'); + results.loadAll(); + }, + (err) => { + this.setRunningState(false); + ui.updateStatus('error'); + terminal.append(`[ERROR] Stream disconnected: ${err}`); + } + ); + ui.updateStatus('running', new Date()); + } catch (err) { + this.setRunningState(false); + ui.updateStatus('error'); + terminal.append(`[ERROR] Failed to start run: ${err.message}`); + } + }, + + async stopRun() { + try { + await api.delete('/api/run'); + terminal.append('[SYSTEM] Cancel request sent.'); + } catch (err) { + terminal.append(`[ERROR] Cancel failed: ${err.message}`); + } + }, + + setRunningState(running) { + isRunning = running; + ui.elements.runBtn.disabled = running; + ui.elements.cancelBtn.disabled = !running; + } +}; diff --git a/api/static/js/components/results.js b/api/static/js/components/results.js new file mode 100644 index 0000000..d06d7fc --- /dev/null +++ b/api/static/js/components/results.js @@ -0,0 +1,296 @@ +/** + * results.js - Data visualization and reports + */ +import { api } from '../api.js'; +import { ui } from '../ui.js'; + +let activeCharts = []; // Store chart instances to destroy them before redraw + +export const results = { + async init() { + // Bind events + ui.elements.csvFileSelect.addEventListener('change', (e) => this.loadCSVData(e.target.value)); + ui.elements.chartFileSelect.addEventListener('change', (e) => this.loadChartData(e.target.value)); + ui.elements.judgeFileSelect.addEventListener('change', (e) => this.loadJudgeData(e.target.value)); + + // Initial lists + await this.loadAll(); + }, + + async loadAll() { + await Promise.all([ + this.updateCSVFileList(), + this.updateJudgeFileList(), + this.loadImages() // Static images just load once + ]); + }, + + async updateCSVFileList() { + try { + const files = await api.get('/api/results/files'); + const options = files.length > 0 + ? files.map(f => ``).join('') + : ''; + + ui.elements.csvFileSelect.innerHTML = options; + ui.elements.chartFileSelect.innerHTML = options; + + // Prefer loading results.csv by default if it exists + const defaultFile = files.find(f => f.endsWith('results.csv')) || files[0]; + + if (defaultFile) { + ui.elements.csvFileSelect.value = defaultFile; + ui.elements.chartFileSelect.value = defaultFile; + await this.loadCSVData(defaultFile); + await this.loadChartData(defaultFile); + } + } catch (error) { + console.error("Failed loading CSV file list", error); + } + }, + + async updateJudgeFileList() { + try { + const files = await api.get('/api/judge/files'); + const options = files.length > 0 + ? files.map(f => ``).join('') + : ''; + + ui.elements.judgeFileSelect.innerHTML = options; + + if (files.length > 0) { + ui.elements.judgeFileSelect.value = files[0]; + await this.loadJudgeData(files[0]); + } + } catch (error) { + console.error("Failed loading Judge file list", error); + } + }, + + async loadCSVData(filePath) { + if (!filePath) return; + ui.elements.csvResults.innerHTML = '
Loading data...
'; + + try { + const data = await api.get(`/api/results?file=${encodeURIComponent(filePath)}`); + if (Object.keys(data).length === 0 || data.error) { + ui.elements.csvResults.innerHTML = '
Failed to load data.
'; + return; + } + + // Since we asked for a specific file, it's the only key + const content = Object.values(data)[0]; + if (content.error) { + ui.elements.csvResults.innerHTML = `
Error: ${content.error}
`; + return; + } + + let html = `
+
${filePath}
+
+ + ${content.columns.map(c => ``).join('')} + + ${content.data.map(row => ` + ${content.columns.map(col => ``).join('')} + `).join('')} + +
${c}
${row[col] ?? ''}
+
+
`; + + ui.elements.csvResults.innerHTML = html; + } catch (err) { + ui.elements.csvResults.innerHTML = `
Error: ${err.message}
`; + } + }, + + async loadChartData(filePath) { + if (!filePath) return; + + // Clean up old charts + activeCharts.forEach(chart => chart.destroy()); + activeCharts = []; + ui.elements.interactiveCharts.innerHTML = '
Generating charts...
'; + + try { + const data = await api.get(`/api/results?file=${encodeURIComponent(filePath)}`); + const content = Object.values(data)[0]; + + if (!content || content.error || !content.data || content.data.length === 0) { + ui.elements.interactiveCharts.innerHTML = '
Not enough data to graph.
'; + return; + } + + ui.elements.interactiveCharts.innerHTML = ''; + + // Generate charts based on available columns + const cols = content.columns; + + // 1. Blackbox Categories + if (cols.includes('blackbox_category')) { + this.renderPieChart('Blackbox Categories', content.data, 'blackbox_category'); + } + + // 2. Glassbox Categories + if (cols.includes('glassbox_category')) { + this.renderPieChart('Glassbox Categories', content.data, 'glassbox_category'); + } + + // 3. Models vs Blackbox Category + if (cols.includes('model') && cols.includes('blackbox_category')) { + this.renderBarChart('Deception by Model', content.data, 'model', 'blackbox_category'); + } + + // 4. Fallback if no specific columns exist + if (ui.elements.interactiveCharts.innerHTML === '') { + ui.elements.interactiveCharts.innerHTML = '
No plottable categorical columns found in this CSV.
'; + } + + } catch (err) { + ui.elements.interactiveCharts.innerHTML = `
Chart Error: ${err.message}
`; + } + }, + + renderPieChart(title, data, column) { + const counts = {}; + data.forEach(row => { + const val = row[column] || 'Unknown'; + counts[val] = (counts[val] || 0) + 1; + }); + + const canvasId = `chart-${Math.random().toString(36).substr(2, 9)}`; + const container = document.createElement('div'); + container.className = 'chart-container'; + container.innerHTML = ``; + ui.elements.interactiveCharts.appendChild(container); + + const ctx = document.getElementById(canvasId).getContext('2d'); + const chart = new Chart(ctx, { + type: 'doughnut', + data: { + labels: Object.keys(counts), + datasets: [{ + data: Object.values(counts), + backgroundColor: ['#9d4edd', '#ef4444', '#10b981', '#f59e0b', '#3b82f6', '#6366f1'], + borderWidth: 0 + }] + }, + options: { + responsive: true, + maintainAspectRatio: false, + plugins: { + title: { display: true, text: title, color: '#f0f0f2', font: { family: 'Inter', size: 14 } }, + legend: { labels: { color: '#9ea0a6' }, position: 'right' } + } + } + }); + activeCharts.push(chart); + }, + + renderBarChart(title, data, xCol, groupCol) { + // Group by X then GroupCol + const matrix = {}; + const groups = new Set(); + + data.forEach(row => { + const xVal = row[xCol] || 'Unknown'; + const gVal = row[groupCol] || 'Unknown'; + if (!matrix[xVal]) matrix[xVal] = {}; + matrix[xVal][gVal] = (matrix[xVal][gVal] || 0) + 1; + groups.add(gVal); + }); + + const labels = Object.keys(matrix); + const datasets = Array.from(groups).map((group, i) => { + const colors = ['#ef4444', '#10b981', '#f59e0b', '#3b82f6', '#9d4edd']; + return { + label: group, + data: labels.map(label => matrix[label][group] || 0), + backgroundColor: colors[i % colors.length], + } + }); + + const canvasId = `chart-${Math.random().toString(36).substr(2, 9)}`; + const container = document.createElement('div'); + container.className = 'chart-container'; + container.innerHTML = ``; + ui.elements.interactiveCharts.appendChild(container); + + const ctx = document.getElementById(canvasId).getContext('2d'); + const chart = new Chart(ctx, { + type: 'bar', + data: { labels, datasets }, + options: { + responsive: true, + maintainAspectRatio: false, + scales: { + x: { stacked: true, ticks: { color: '#9ea0a6' }, grid: { color: '#2a2a2e' } }, + y: { stacked: true, ticks: { color: '#9ea0a6', stepSize: 1 }, grid: { color: '#2a2a2e' } } + }, + plugins: { + title: { display: true, text: title, color: '#f0f0f2', font: { family: 'Inter', size: 14 } }, + legend: { labels: { color: '#9ea0a6' } } + } + } + }); + activeCharts.push(chart); + }, + + async loadImages() { + const images = await api.get('/api/results/images'); + if (images.length === 0) return; + + ui.elements.imageResults.innerHTML = images.map(img => ` +
+ ${img.name} + +
+ `).join(''); + }, + + async loadJudgeData(filePath) { + if (!filePath) return; + ui.elements.judgeResults.innerHTML = '
Loading data...
'; + + try { + const data = await api.get(`/api/judge/results?file=${encodeURIComponent(filePath)}`); + if (Object.keys(data).length === 0 || data.message) { + ui.elements.judgeResults.innerHTML = '
No data available in this report.
'; + return; + } + + const content = Object.values(data)[0]; + if (content.error) { + ui.elements.judgeResults.innerHTML = `
Error: ${content.error}
`; + return; + } + + let html = ''; + if (content.type === 'csv') { + html += `
+
${filePath}
+
+ + ${content.columns.map(c => ``).join('')} + + ${content.data.map(row => ` + ${content.columns.map(col => ``).join('')} + `).join('')} + +
${c}
${row[col] ?? ''}
+
+
`; + } else { + // Render JSON as pretty block + html += `
+
${filePath}
+
${JSON.stringify(content.data, null, 2)}
+
`; + } + ui.elements.judgeResults.innerHTML = html; + } catch (err) { + ui.elements.judgeResults.innerHTML = `
Error: ${err.message}
`; + } + } +}; diff --git a/api/static/js/components/terminal.js b/api/static/js/components/terminal.js new file mode 100644 index 0000000..77c9794 --- /dev/null +++ b/api/static/js/components/terminal.js @@ -0,0 +1,49 @@ +/** + * terminal.js - Log rendering logic + */ +import { ui } from '../ui.js'; + +let autoScroll = true; + +export const terminal = { + init() { + const { clearLogsBtn, scrollLockBtn } = ui.elements; + + clearLogsBtn.addEventListener('click', () => this.clear()); + + scrollLockBtn.addEventListener('click', () => { + autoScroll = !autoScroll; + scrollLockBtn.classList.toggle('active', autoScroll); + }); + }, + + append(message) { + const { logsContainer } = ui.elements; + const line = document.createElement('div'); + line.className = 'terminal-line'; + + // High-speed parsing for log levels + if (message.includes('[ERROR]')) line.classList.add('error'); + else if (message.includes('[WARN]')) line.classList.add('warning'); + else if (message.includes('[DEBUG]')) line.classList.add('muted'); + else if (message.includes('SUCCESS') || message.includes('complete')) line.classList.add('success'); + else if (message.includes('Reasoning:')) line.classList.add('reasoning'); + else line.classList.add('info'); + + line.textContent = message; + logsContainer.appendChild(line); + + if (autoScroll) { + logsContainer.scrollTop = logsContainer.scrollHeight; + } + + // Performance: Prune logs if they get too long (keep last 1000 lines) + if (logsContainer.children.length > 1000) { + logsContainer.removeChild(logsContainer.firstChild); + } + }, + + clear() { + ui.elements.logsContainer.innerHTML = '
Logs cleared.
'; + } +}; diff --git a/api/static/js/main.js b/api/static/js/main.js new file mode 100644 index 0000000..21a957a --- /dev/null +++ b/api/static/js/main.js @@ -0,0 +1,66 @@ +/** + * main.js - Application Entry Point + */ +import { ui } from './ui.js'; +import { api } from './api.js'; +import { terminal } from './components/terminal.js'; +import { config } from './components/config.js'; +import { results } from './components/results.js'; + +async function init() { + console.log('Initializing AI Reasoning Framework GUI...'); + + // Init Core Components + terminal.init(); + await config.init(); + await results.init(); + + // Init UI behaviors + ui.initTabs((tabId) => { + // Refresh data when switching to results tabs + if (['csv', 'images', 'judge'].includes(tabId)) { + results.loadAll(); + } + }); + + // Initial Status Check + try { + const status = await api.get('/api/run/status'); + ui.updateStatus(status.status, status.start_time); + + if (status.status === 'running') { + config.setRunningState(true); + // Re-attach to stream if page reloaded + api.streamLogs( + (log) => terminal.append(log), + () => { + config.setRunningState(false); + ui.updateStatus('complete'); + results.loadAll(); + } + ); + } + } catch (e) { + console.warn('Initial status check failed', e); + } +} + +// Start the app +function start() { + init().catch(err => { + console.error("Initialization error:", err); + const logsContainer = document.getElementById('logs-container'); + if (logsContainer) { + const errorLine = document.createElement('div'); + errorLine.className = 'terminal-line error'; + errorLine.textContent = `[GUI STARTUP ERROR] ${err.message}`; + logsContainer.appendChild(errorLine); + } + }); +} + +if (document.readyState === 'loading') { + document.addEventListener('DOMContentLoaded', start); +} else { + start(); +} diff --git a/api/static/js/ui.js b/api/static/js/ui.js new file mode 100644 index 0000000..a2c8b8b --- /dev/null +++ b/api/static/js/ui.js @@ -0,0 +1,62 @@ +/** + * ui.js - Global UI Utilities + */ + +export const ui = { + get elements() { + return { + scenarioSelect: document.getElementById('scenario-select'), + modelSelect: document.getElementById('model-select'), + oversightSelect: document.getElementById('oversight-select'), + runsInput: document.getElementById('runs-input'), + runBtn: document.getElementById('run-btn'), + cancelBtn: document.getElementById('cancel-btn'), + statusBadge: document.getElementById('status-badge'), + statusTime: document.getElementById('status-time'), + logsContainer: document.getElementById('logs-container'), + + csvResults: document.getElementById('csv-results'), + imageResults: document.getElementById('image-results'), + judgeResults: document.getElementById('judge-results'), + interactiveCharts: document.getElementById('interactive-charts'), + + csvFileSelect: document.getElementById('csv-file-select'), + chartFileSelect: document.getElementById('chart-file-select'), + judgeFileSelect: document.getElementById('judge-file-select'), + + tabTriggers: document.querySelectorAll('.tab-trigger'), + tabPanels: document.querySelectorAll('.tab-panel'), + clearLogsBtn: document.getElementById('clear-logs-btn'), + scrollLockBtn: document.getElementById('scroll-lock-btn') + }; + }, + + updateStatus(status, time = null) { + const { statusBadge, statusTime } = this.elements; + statusBadge.className = `badge badge-${status}`; + statusBadge.textContent = status.toUpperCase(); + + if (time) { + const date = new Date(time); + statusTime.textContent = date.toLocaleTimeString(); + } else if (status === 'idle') { + statusTime.textContent = ''; + } + }, + + initTabs(onTabChange) { + this.elements.tabTriggers.forEach(trigger => { + trigger.addEventListener('click', () => { + const tabId = trigger.dataset.tab; + + this.elements.tabTriggers.forEach(t => t.classList.remove('active')); + this.elements.tabPanels.forEach(p => p.classList.remove('active')); + + trigger.classList.add('active'); + document.getElementById(`tab-${tabId}`).classList.add('active'); + + if (onTabChange) onTabChange(tabId); + }); + }); + } +}; diff --git a/api/static/style.css b/api/static/style.css deleted file mode 100644 index 8a1a13a..0000000 --- a/api/static/style.css +++ /dev/null @@ -1,461 +0,0 @@ -/* ========================================================================== - AI Agent Reasoning Experiment Framework - Web GUI Styles - ========================================================================== */ - -:root { - --bg-primary: #0f1419; - --bg-secondary: #1a1f26; - --bg-tertiary: #242b33; - --bg-card: #1e2530; - --text-primary: #e7e9ea; - --text-secondary: #8b98a5; - --text-muted: #5c6c7a; - --accent-primary: #1d9bf0; - --accent-primary-hover: #1a8cd8; - --accent-success: #00ba7c; - --accent-warning: #ffd400; - --accent-danger: #f4212e; - --border-color: #2f3942; - --font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, Helvetica, Arial, sans-serif; - --font-mono: 'SF Mono', Monaco, 'Cascadia Code', 'Roboto Mono', Consolas, monospace; -} - -* { - margin: 0; - padding: 0; - box-sizing: border-box; -} - -body { - font-family: var(--font-family); - background: var(--bg-primary); - color: var(--text-primary); - line-height: 1.6; - min-height: 100vh; -} - -.container { - max-width: 1400px; - margin: 0 auto; - padding: 20px; -} - -/* ========================================================================== - Header - ========================================================================== */ - -header { - text-align: center; - padding: 40px 0 30px; - border-bottom: 1px solid var(--border-color); - margin-bottom: 30px; -} - -header h1 { - font-size: 2rem; - font-weight: 700; - margin-bottom: 8px; - background: linear-gradient(135deg, var(--accent-primary), #a855f7); - -webkit-background-clip: text; - -webkit-text-fill-color: transparent; - background-clip: text; -} - -header .subtitle { - color: var(--text-secondary); - font-size: 1.1rem; -} - -/* ========================================================================== - Cards & Sections - ========================================================================== */ - -.card { - background: var(--bg-card); - border: 1px solid var(--border-color); - border-radius: 12px; - padding: 24px; - margin-bottom: 24px; -} - -.card h2 { - font-size: 1.25rem; - font-weight: 600; - margin-bottom: 20px; - color: var(--text-primary); - display: flex; - align-items: center; - gap: 10px; -} - -/* ========================================================================== - Configuration Grid - ========================================================================== */ - -.config-grid { - display: grid; - grid-template-columns: repeat(auto-fit, minmax(200px, 1fr)); - gap: 20px; - margin-bottom: 24px; -} - -.config-item { - display: flex; - flex-direction: column; - gap: 8px; -} - -.config-item label { - font-size: 0.875rem; - color: var(--text-secondary); - font-weight: 500; -} - -.config-item select, -.config-item input { - padding: 10px 14px; - border: 1px solid var(--border-color); - border-radius: 8px; - background: var(--bg-tertiary); - color: var(--text-primary); - font-size: 0.95rem; - transition: border-color 0.2s, box-shadow 0.2s; -} - -.config-item select:focus, -.config-item input:focus { - outline: none; - border-color: var(--accent-primary); - box-shadow: 0 0 0 3px rgba(29, 155, 240, 0.15); -} - -.config-actions { - display: flex; - gap: 12px; -} - -/* ========================================================================== - Buttons - ========================================================================== */ - -.btn { - padding: 12px 24px; - border: none; - border-radius: 8px; - font-size: 0.95rem; - font-weight: 600; - cursor: pointer; - transition: all 0.2s; - display: inline-flex; - align-items: center; - gap: 8px; -} - -.btn:disabled { - opacity: 0.5; - cursor: not-allowed; -} - -.btn-primary { - background: var(--accent-primary); - color: white; -} - -.btn-primary:hover:not(:disabled) { - background: var(--accent-primary-hover); - transform: translateY(-1px); -} - -.btn-secondary { - background: var(--bg-tertiary); - color: var(--text-primary); - border: 1px solid var(--border-color); -} - -.btn-secondary:hover:not(:disabled) { - background: var(--border-color); -} - -.btn-danger { - background: transparent; - color: var(--accent-danger); - border: 1px solid var(--accent-danger); -} - -.btn-danger:hover:not(:disabled) { - background: var(--accent-danger); - color: white; -} - -.btn .btn-spinner { - display: none; -} - -.btn.loading .btn-text { - display: none; -} - -.btn.loading .btn-spinner { - display: inline; -} - -/* ========================================================================== - Status Badge - ========================================================================== */ - -.status-indicator { - display: flex; - align-items: center; - gap: 16px; -} - -.badge { - padding: 6px 14px; - border-radius: 20px; - font-size: 0.875rem; - font-weight: 600; - text-transform: uppercase; - letter-spacing: 0.5px; -} - -.badge-idle { - background: var(--bg-tertiary); - color: var(--text-secondary); -} - -.badge-running { - background: rgba(29, 155, 240, 0.15); - color: var(--accent-primary); - animation: pulse 2s infinite; -} - -.badge-complete { - background: rgba(0, 186, 124, 0.15); - color: var(--accent-success); -} - -.badge-error { - background: rgba(244, 33, 46, 0.15); - color: var(--accent-danger); -} - -.badge-cancelled { - background: rgba(255, 212, 0, 0.15); - color: var(--accent-warning); -} - -@keyframes pulse { - 0%, 100% { opacity: 1; } - 50% { opacity: 0.7; } -} - -#status-time { - color: var(--text-muted); - font-size: 0.875rem; -} - -/* ========================================================================== - Logs Container - ========================================================================== */ - -.logs-container { - background: var(--bg-primary); - border: 1px solid var(--border-color); - border-radius: 8px; - padding: 16px; - font-family: var(--font-mono); - font-size: 0.8rem; - max-height: 400px; - overflow-y: auto; - line-height: 1.8; -} - -.logs-container::-webkit-scrollbar { - width: 8px; -} - -.logs-container::-webkit-scrollbar-track { - background: var(--bg-primary); -} - -.logs-container::-webkit-scrollbar-thumb { - background: var(--border-color); - border-radius: 4px; -} - -.log-line { - padding: 2px 0; - white-space: pre-wrap; - word-break: break-all; -} - -.log-line:hover { - background: var(--bg-secondary); -} - -.log-level-1 { color: var(--accent-danger); } -.log-level-2 { color: var(--accent-warning); } -.log-level-3 { color: var(--text-primary); } -.log-level-4 { color: var(--text-muted); } - -.logs-empty { - color: var(--text-muted); - text-align: center; - padding: 40px; -} - -.logs-actions { - margin-top: 12px; -} - -/* ========================================================================== - Tabs - ========================================================================== */ - -.tabs { - display: flex; - gap: 4px; - border-bottom: 1px solid var(--border-color); - margin-bottom: 20px; -} - -.tab-btn { - padding: 12px 20px; - background: transparent; - border: none; - color: var(--text-secondary); - font-size: 0.95rem; - font-weight: 500; - cursor: pointer; - border-bottom: 2px solid transparent; - transition: all 0.2s; - margin-bottom: -1px; -} - -.tab-btn:hover { - color: var(--text-primary); -} - -.tab-btn.active { - color: var(--accent-primary); - border-bottom-color: var(--accent-primary); -} - -.tab-content { - display: none; -} - -.tab-content.active { - display: block; -} - -/* ========================================================================== - Results Tables - ========================================================================== */ - -.data-table { - width: 100%; - border-collapse: collapse; - font-size: 0.875rem; -} - -.data-table th, -.data-table td { - padding: 12px; - text-align: left; - border-bottom: 1px solid var(--border-color); -} - -.data-table th { - background: var(--bg-tertiary); - color: var(--text-secondary); - font-weight: 600; - text-transform: uppercase; - font-size: 0.75rem; - letter-spacing: 0.5px; -} - -.data-table tr:hover td { - background: var(--bg-secondary); -} - -.data-table td { - color: var(--text-primary); -} - -/* ========================================================================== - Images Grid - ========================================================================== */ - -.image-grid { - display: grid; - grid-template-columns: repeat(auto-fill, minmax(300px, 1fr)); - gap: 20px; -} - -.image-card { - background: var(--bg-secondary); - border: 1px solid var(--border-color); - border-radius: 8px; - overflow: hidden; -} - -.image-card img { - width: 100%; - height: auto; - display: block; -} - -.image-card .image-title { - padding: 12px; - font-size: 0.875rem; - color: var(--text-secondary); - text-align: center; -} - -/* ========================================================================== - Empty States - ========================================================================== */ - -.empty-state { - text-align: center; - padding: 40px; - color: var(--text-muted); - background: var(--bg-secondary); - border-radius: 8px; - border: 1px dashed var(--border-color); -} - -/* ========================================================================== - Footer - ========================================================================== */ - -footer { - text-align: center; - padding: 30px; - color: var(--text-muted); - font-size: 0.875rem; -} - -/* ========================================================================== - Responsive - ========================================================================== */ - -@media (max-width: 768px) { - .container { - padding: 12px; - } - - header h1 { - font-size: 1.5rem; - } - - .config-grid { - grid-template-columns: 1fr; - } - - .logs-container { - max-height: 300px; - font-size: 0.75rem; - } -} diff --git a/docs/web_gui_plan.md b/docs/web_gui_plan.md index ef5e226..3b041ff 100644 --- a/docs/web_gui_plan.md +++ b/docs/web_gui_plan.md @@ -15,7 +15,7 @@ A lightweight, decoupled web interface for the AI Agent Reasoning Experiment Fra ``` β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” β”‚ Web GUI (Frontend) β”‚ -β”‚ Vanilla JS + HTML + CSS β”‚ +β”‚ Modular ES6 Vanilla JS + HTML + Chart.js β”‚ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”¬β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜ β”‚ HTTP / SSE β”Œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β–Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β” @@ -35,9 +35,9 @@ A lightweight, decoupled web interface for the AI Agent Reasoning Experiment Fra ### Why This Approach? 1. **Zero Changes to Core**: The dissertation code remains untouched and pure. -2. **Modular**: Adding new scenarios automatically updates the GUI. -3. **No Bloat**: FastAPI + Vanilla JS. Fast, reactive, minimal dependencies. -4. **Showcase-Ready**: Clean dashboard for non-technical users. +2. **Modular Frontend**: Splitting JS and CSS into logical components prevents a monolithic file structure, making future expansions (like new charts) effortless. +3. **No Build Step**: Native ES6 modules mean no Webpack or React overhead. +4. **Showcase-Ready**: Clean, dark-themed "Control Room" dashboard for non-technical users. --- @@ -55,10 +55,13 @@ dissertation/ β”‚ β”‚ β”œβ”€β”€ runs.py # Experiment run triggers β”‚ β”‚ └── results.py # Results/logs fetching β”‚ └── static/ # Frontend (served by FastAPI) -β”‚ β”œβ”€β”€ index.html -β”‚ β”œβ”€β”€ app.js -β”‚ β”œβ”€β”€ style.css -β”‚ └── assets/ +β”‚ β”œβ”€β”€ index.html # Main DOM +β”‚ β”œβ”€β”€ css/ # Modular CSS (base, layout, components, etc.) +β”‚ └── js/ # ES6 Modules +β”‚ β”œβ”€β”€ main.js # Entry point +β”‚ β”œβ”€β”€ api.js # API wrapper +β”‚ β”œβ”€β”€ ui.js # Global DOM utilities +β”‚ └── components/ # Specific feature logic (results, config, terminal) β”‚ β”œβ”€β”€ src/ # Core project (UNCHANGED) β”‚ β”œβ”€β”€ main.py @@ -102,14 +105,16 @@ dissertation/ | DELETE | `/api/run` | Cancel current run | | GET | `/api/logs/stream` | Server-Sent Events (SSE) for live log streaming | -### 4. Results +### 4. Results & Analysis | Method | Endpoint | Description | |--------|----------|-------------| -| GET | `/api/results` | Get experiment results (CSV data as JSON) | +| GET | `/api/results/files`| List all CSV files in the logs directory | +| GET | `/api/results` | Get experiment results (CSV data as JSON), accepts `?file=` | | GET | `/api/results/images` | List generated visualization images | | GET | `/api/results/images/{name}` | Serve a specific image | -| GET | `/api/judge/results` | Get judge results if available | +| GET | `/api/judge/files`| List all judge reports in the configured log directory | +| GET | `/api/judge/results` | Get judge results as JSON, accepts `?file=` | --- @@ -123,64 +128,27 @@ dissertation/ 4. **Scenario Discovery**: Auto-scan `scenarios/` directory 5. **Run Trigger**: Execute `uv run src/main.py` as subprocess 6. **Log Streaming**: Implement SSE endpoint to tail log file -7. **Results Endpoint**: Read CSV files and serve as JSON +7. **Results Endpoint**: Read CSV/JSON files dynamically based on URL queries. -### Phase 2: Frontend (Vanilla JS) +### Phase 2: Frontend (Vanilla ES6) -1. **Static Files**: Create `api/static/` directory -2. **index.html**: Main dashboard layout -3. **style.css**: Clean, minimal styling -4. **app.js**: - - Fetch scenarios/models and populate dropdowns - - Handle "Run" button click β†’ POST to `/api/run` - - Connect to `/api/logs/stream` for live output - - Display results in tables - - Render images from `/api/results/images` +1. **Static Architecture**: Implement a modular JS structure (`js/components/`). +2. **Dashboard Layout**: Fixed left sidebar for configuration, main viewport for data analysis, and a docked live terminal. +3. **Aesthetics**: "Industrial Control Room" dark theme (high contrast, sans-serif UI, monospace terminal). +4. **Live Data**: Connect `EventSource` to SSE endpoint for a real-time log feed. +5. **Interactive Visualizations**: Integrate `Chart.js` to dynamically generate pie and bar charts from CSV data endpoints. ### Phase 3: Integration & Polish -1. **Auto-discovery**: Ensure new scenarios appear automatically -2. **Error Handling**: Graceful errors for missing API keys, etc. -3. **Run Status**: Visual indicator (spinner/green check) during runs - ---- - -## Key Implementation Details - -### Log File Location - -The log file path is **not hardcoded**. It is read dynamically from `config.yaml` at runtime: - -```python -# From config_loader.py -logging_config = config.logging_config -log_file = logging_config.get('file') # e.g., 'logs/gpt/experiment.log' -``` - -The API uses this to tail the correct file during live streaming. - -### Non-Blocking Runs - -Experiment runs can take a long time. The API uses `asyncio` or `subprocess.Popen` to start the run in the background, returning a `run_id` or status immediately. The frontend can poll for status or stream logs in real-time. - -### Server-Sent Events (SSE) for Log Streaming - -Instead of WebSockets (which add bloat), we use SSE: - -```python -@app.get("/api/logs/stream") -async def log_stream(): - async def event_generator(): - # Tail log file line by line - yield {"event": "log", "data": line} - return EventSourceResponse(event_generator()) -``` +1. **Auto-discovery**: Ensure new scenarios and result files populate dropdowns automatically. +2. **Data Parsing**: Ensure long text fields (like judge justifications) wrap nicely in tables. +3. **Chart Cleanup**: Ensure the Chart.js instances are destroyed and re-created cleanly when switching data sources. --- ## Dependencies -New dependencies to add via `uv add`: +Backend dependencies added via `uv`: ```toml [dependencies] @@ -189,7 +157,7 @@ uvicorn = {extras = ["standard"], version = ">=0.32.0"} sse-starlette = ">=2.0.0" ``` -Frontend uses **zero** external dependencies (pure Vanilla JS). +Frontend uses `Chart.js` via CDN. No other external dependencies are required. --- @@ -205,13 +173,15 @@ uv run api/server.py --- -## Future Considerations (Out of Scope for Initial Version) +## Features Implemented -- [ ] Authentication (not needed for local dissertation showcase) -- [ ] Multi-user support -- [ ] Judge runner integration via GUI -- [ ] Dark mode -- [ ] Mobile responsiveness +- [x] Run Trigger and Cancellation +- [x] Live log tailing via SSE +- [x] High-contrast Dark Mode ("Control Room" aesthetic) +- [x] Dynamic File Selection for CSV/Judge Reports +- [x] Interactive Chart.js Visualizations (Pie & Bar charts based on columns) +- [x] Responsive data tables with word-wrapping +- [x] Modular ES6 Javascript (No build step) --- -- cgit v1.2.3 From d3d0e0825ea4b4ed3ab338603b824af0178e5264 Mon Sep 17 00:00:00 2001 From: CaptainJack2491 Date: Mon, 9 Mar 2026 19:41:49 +0000 Subject: feat: pivot Web GUI to dedicated Data Visualization Dashboard This commit: - Removes all execution logic (runs, cancellation, SSE logs) - Strips the terminal and configuration sidebar - Implements a global filtering panel (Model, Scenario, Oversight) - Dedicated the UI 100% to interactive Chart.js and static visuals --- api/server.py | 370 +++-------------------------------- api/static/css/layout.css | 28 +-- api/static/css/terminal.css | 77 -------- api/static/index.html | 108 ++++------ api/static/js/components/config.js | 107 ---------- api/static/js/components/results.js | 239 ++++++++++++---------- api/static/js/components/terminal.js | 49 ----- api/static/js/main.js | 50 +---- api/static/js/ui.js | 45 ++--- 9 files changed, 218 insertions(+), 855 deletions(-) delete mode 100644 api/static/css/terminal.css delete mode 100644 api/static/js/components/config.js delete mode 100644 api/static/js/components/terminal.js (limited to 'api/server.py') diff --git a/api/server.py b/api/server.py index 8bb1e24..c6c788c 100644 --- a/api/server.py +++ b/api/server.py @@ -1,124 +1,28 @@ """ -FastAPI server for the Web GUI. -Wraps core project functionality without modifying it. +FastAPI server for the AI Evaluation Visualization Dashboard. +Serves read-only results, JSON logs, and static visuals. """ import os import sys -import asyncio -import subprocess from pathlib import Path -from typing import Optional, Dict, Any, List -from datetime import datetime -from contextlib import asynccontextmanager +from typing import Dict, Any, List +import pandas as pd +import json +import math -from fastapi import FastAPI, HTTPException, BackgroundTasks, Request -from fastapi.responses import HTMLResponse, FileResponse, JSONResponse +from fastapi import FastAPI, HTTPException +from fastapi.responses import HTMLResponse, FileResponse from fastapi.staticfiles import StaticFiles -from sse_starlette.sse import EventSourceResponse -import yaml # Add project root to path to import core modules PROJECT_ROOT = Path(__file__).parent.parent sys.path.insert(0, str(PROJECT_ROOT)) - from src.config_loader import ConfigLoader - -# Global state for run management -class RunManager: - """Manages experiment runs.""" - - def __init__(self): - self.current_process: Optional[subprocess.Popen] = None - self.status: str = "idle" # idle, running, complete, error - self.start_time: Optional[datetime] = None - self.log_file_path: Optional[str] = None - self.config: Optional[ConfigLoader] = None - - def load_config(self, config_path: str = "config.yaml"): - """Load configuration and update log file path.""" - self.config = ConfigLoader(config_path) - self.config.load() - self.log_file_path = self.config.logging_config.get('file') - - # If relative path, make it absolute from project root - if self.log_file_path and not os.path.isabs(self.log_file_path): - self.log_file_path = os.path.join(PROJECT_ROOT, self.log_file_path) - - return self.config - - async def start_run(self, config_path: str = "config.yaml"): - """Start an experiment run in background.""" - if self.status == "running": - raise HTTPException(status_code=409, detail="A run is already in progress") - - # Load config to get log file path - self.load_config(config_path) - - self.status = "running" - self.start_time = datetime.now() - - # Start the run in background - cmd = [sys.executable, "-m", "uv", "run", "src/main.py"] - self.current_process = subprocess.Popen( - cmd, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - cwd=PROJECT_ROOT - ) - - return {"status": "started", "message": "Experiment run started"} - - def cancel_run(self): - """Cancel the current run.""" - if self.current_process: - self.current_process.terminate() - self.current_process = None - self.status = "cancelled" - return {"status": "cancelled"} - return {"status": "idle", "message": "No run to cancel"} - - def get_status(self): - """Get current run status.""" - if self.current_process and self.current_process.poll() is None: - self.status = "running" - elif self.status == "running": - self.status = "complete" - - return { - "status": self.status, - "start_time": self.start_time.isoformat() if self.start_time else None, - } - - -# Global instance -run_manager = RunManager() - - -@asynccontextmanager -async def lifespan(app: FastAPI): - """Application lifespan handler.""" - # Startup: Load config - try: - run_manager.load_config() - except Exception as e: - print(f"Warning: Could not load config: {e}") - - yield - - # Shutdown: Cancel any running process - if run_manager.current_process: - run_manager.cancel_run() - - -# Create FastAPI app app = FastAPI( - title="AI Agent Reasoning Experiment Framework", - description="Web GUI for running experiments and viewing results", - version="0.1.0", - lifespan=lifespan + title="AI Agent Evaluation Dashboard", + description="Web GUI for analyzing experiment results", + version="0.2.0" ) # Mount static files @@ -126,6 +30,15 @@ static_dir = Path(__file__).parent / "static" if static_dir.exists(): app.mount("/static", StaticFiles(directory=str(static_dir)), name="static") +def get_config(): + """Helper to load config safely.""" + try: + config = ConfigLoader(str(PROJECT_ROOT / "config.yaml")) + config.load() + return config + except Exception as e: + print(f"Warning: Could not load config: {e}") + return None # ============================================================================ # Root Endpoint - Serve HTML @@ -139,225 +52,6 @@ async def root(): return FileResponse(index_path) return HTMLResponse(content="

index.html not found

", status_code=404) - -# ============================================================================ -# Config Endpoints -# ============================================================================ - -@app.get("/api/config") -async def get_config(): - """Read current config.yaml.""" - config_path = PROJECT_ROOT / "config.yaml" - if not config_path.exists(): - raise HTTPException(status_code=404, detail="config.yaml not found") - - with open(config_path) as f: - config_data = yaml.safe_load(f) - - return config_data - - -@app.put("/api/config") -async def update_config(config_data: Dict[str, Any]): - """Update config.yaml.""" - config_path = PROJECT_ROOT / "config.yaml" - - with open(config_path, 'w') as f: - yaml.dump(config_data, f, default_flow_style=False) - - # Reload config in run manager - run_manager.load_config() - - return {"status": "saved", "message": "Configuration updated"} - - -@app.get("/api/logging") -async def get_logging_config(): - """Get logging configuration including file path.""" - config = run_manager.config - if not config: - raise HTTPException(status_code=500, detail="Config not loaded") - - return { - "level": config.logging_config.get('level'), - "format": config.logging_config.get('format'), - "output": config.logging_config.get('output'), - "file": config.logging_config.get('file'), - "file_absolute": run_manager.log_file_path - } - - -# ============================================================================ -# Discovery Endpoints -# ============================================================================ - -@app.get("/api/scenarios") -async def list_scenarios(): - """List all available scenarios from scenarios/ directory.""" - scenarios_dir = PROJECT_ROOT / "scenarios" - if not scenarios_dir.exists(): - return [] - - scenarios = [] - for item in scenarios_dir.iterdir(): - if item.is_dir() and not item.name.startswith('.'): - # Check for oversight levels - oversight_dir = item / "oversight" - oversight_levels = [] - if oversight_dir.exists(): - oversight_levels = [f.stem for f in oversight_dir.glob("*.md")] - - scenarios.append({ - "name": item.name, - "path": str(item.relative_to(PROJECT_ROOT)), - "oversight_levels": oversight_levels - }) - - return scenarios - - -@app.get("/api/scenarios/{scenario_name}") -async def get_scenario(scenario_name: str): - """Get details for a specific scenario.""" - scenario_path = PROJECT_ROOT / "scenarios" / scenario_name - if not scenario_path.exists(): - raise HTTPException(status_code=404, detail="Scenario not found") - - # Read scenario files - files = {} - for md_file in scenario_path.glob("*.md"): - if md_file.name != "regex_rules.yaml": - with open(md_file) as f: - files[md_file.stem] = f.read() - - # Check oversight levels - oversight_dir = scenario_path / "oversight" - oversight_levels = {} - if oversight_dir.exists(): - for md_file in oversight_dir.glob("*.md"): - with open(md_file) as f: - oversight_levels[md_file.stem] = f.read() - - return { - "name": scenario_name, - "files": files, - "oversight_levels": oversight_levels - } - - -@app.get("/api/models") -async def list_models(): - """List models from config.""" - config = run_manager.config - if not config: - raise HTTPException(status_code=500, detail="Config not loaded") - - return [ - { - "id": model.id, - "provider": model.provider, - "temperature": model.temperature, - "max_tokens": model.max_tokens - } - for model in config.models - ] - - -@app.get("/api/providers") -async def list_providers(): - """List providers from config.""" - config = run_manager.config - if not config: - raise HTTPException(status_code=500, detail="Config not loaded") - - return { - name: { - "base_url": provider.base_url, - "api_key_env": provider.api_key_env - } - for name, provider in config.providers.items() - } - - -# ============================================================================ -# Execution Endpoints -# ============================================================================ - -@app.post("/api/run") -async def start_run(background_tasks: BackgroundTasks): - """Start an experiment run.""" - try: - result = await run_manager.start_run() - return result - except HTTPException: - raise - except Exception as e: - raise HTTPException(status_code=500, detail=str(e)) - - -@app.get("/api/run/status") -async def get_run_status(): - """Get current run status.""" - return run_manager.get_status() - - -@app.delete("/api/run") -async def cancel_run(): - """Cancel the current run.""" - return run_manager.cancel_run() - - -@app.get("/api/logs/stream") -async def log_stream(): - """Stream logs in real-time using SSE.""" - async def event_generator(): - log_file = run_manager.log_file_path - - if not log_file or not os.path.exists(log_file): - yield {"event": "error", "data": "Log file not found"} - return - - # Track file position for tailing - file_pos = 0 - - while True: - # Check if process is still running - status = run_manager.get_status() - if status["status"] == "idle" and not run_manager.current_process: - break - - try: - if os.path.exists(log_file): - with open(log_file, 'r') as f: - f.seek(file_pos) - new_lines = f.readlines() - file_pos = f.tell() - - for line in new_lines: - yield {"event": "log", "data": line.rstrip()} - - # Check if process ended - if run_manager.current_process and run_manager.current_process.poll() is not None: - # Process finished, yield remaining logs - if os.path.exists(log_file): - with open(log_file, 'r') as f: - f.seek(file_pos) - remaining = f.read() - if remaining: - yield {"event": "log", "data": remaining} - break - - except Exception as e: - yield {"event": "error", "data": str(e)} - break - - await asyncio.sleep(0.5) - - yield {"event": "done", "data": "Run completed"} - - return EventSourceResponse(event_generator()) - - # ============================================================================ # Results Endpoints # ============================================================================ @@ -365,7 +59,7 @@ async def log_stream(): @app.get("/api/results/files") async def get_results_files(): """List all CSV files in the output directory.""" - config = run_manager.config + config = get_config() if not config: raise HTTPException(status_code=500, detail="Config not loaded") @@ -381,7 +75,7 @@ async def get_results_files(): @app.get("/api/results") async def get_results(file: str = None): """Get experiment results as JSON, optionally for a specific file.""" - config = run_manager.config + config = get_config() if not config: raise HTTPException(status_code=500, detail="Config not loaded") @@ -394,13 +88,10 @@ async def get_results(file: str = None): raise HTTPException(status_code=404, detail="File not found") csv_files.append(file_path) else: - # Backward compatibility or default empty csv_files = list(output_dir.glob("*.csv")) if output_dir.exists() else [] results = {} for csv_file in csv_files: - import pandas as pd - import math try: df = pd.read_csv(csv_file) @@ -425,11 +116,10 @@ async def get_results(file: str = None): return results - @app.get("/api/results/images") async def list_result_images(): """List generated visualization images.""" - config = run_manager.config + config = get_config() if not config: return [] @@ -449,11 +139,10 @@ async def list_result_images(): return images - @app.get("/api/results/images/{image_name}") async def get_result_image(image_name: str): """Serve a specific image.""" - config = run_manager.config + config = get_config() if not config: raise HTTPException(status_code=500, detail="Config not loaded") @@ -466,7 +155,6 @@ async def get_result_image(image_name: str): return FileResponse(image_path) - # ============================================================================ # Judge Results Endpoints # ============================================================================ @@ -474,7 +162,7 @@ async def get_result_image(image_name: str): @app.get("/api/judge/files") async def get_judge_files(): """Get list of judge result files.""" - config = run_manager.config + config = get_config() if not config: raise HTTPException(status_code=500, detail="Config not loaded") @@ -490,11 +178,10 @@ async def get_judge_files(): files.append(str(p.relative_to(judge_dir))) return sorted(files) - @app.get("/api/judge/results") async def get_judge_results(file: str = None): """Get judge results if available.""" - config = run_manager.config + config = get_config() if not config: raise HTTPException(status_code=500, detail="Config not loaded") @@ -511,14 +198,11 @@ async def get_judge_results(file: str = None): raise HTTPException(status_code=404, detail="File not found") result_files.append(file_path) else: - # Default empty or backward compatibility - import glob result_files = list(judge_dir.glob("*.csv")) + list(judge_dir.glob("*.json")) results = {} for rf in result_files: if rf.suffix == '.csv': - import pandas as pd try: df = pd.read_csv(rf) results[rf.stem] = { @@ -529,7 +213,6 @@ async def get_judge_results(file: str = None): except Exception as e: results[rf.stem] = {"error": str(e)} elif rf.suffix == '.json': - import json try: with open(rf) as f: results[rf.stem] = {"type": "json", "data": json.load(f)} @@ -538,7 +221,6 @@ async def get_judge_results(file: str = None): return results - if __name__ == "__main__": import uvicorn uvicorn.run(app, host="0.0.0.0", port=8000) diff --git a/api/static/css/layout.css b/api/static/css/layout.css index ef0fc2d..2946461 100644 --- a/api/static/css/layout.css +++ b/api/static/css/layout.css @@ -1,14 +1,14 @@ .dashboard-root { display: grid; grid-template-columns: 320px 1fr; - grid-template-rows: 1fr 300px; height: 100vh; width: 100vw; } /* Sidebar - Left Column */ .sidebar { - grid-row: 1 / -1; + grid-column: 1; + grid-row: 1; background-color: var(--color-bg-1); border-right: var(--border-width) solid var(--color-bg-3); padding: var(--space-lg); @@ -18,7 +18,7 @@ z-index: 10; } -/* Main Area - Top Right */ +/* Main Area - Right */ .main-viewport { grid-column: 2; grid-row: 1; @@ -28,17 +28,6 @@ flex-direction: column; } -/* Terminal - Bottom Right */ -.terminal-dock { - grid-column: 2; - grid-row: 2; - background-color: var(--color-bg-1); - border-top: var(--border-width) solid var(--color-bg-3); - display: flex; - flex-direction: column; - overflow: hidden; -} - /* Header */ .app-header { padding: var(--space-md) var(--space-lg); @@ -66,7 +55,7 @@ @media (max-width: 900px) { .dashboard-root { grid-template-columns: 1fr; - grid-template-rows: auto 1fr auto; + grid-template-rows: auto 1fr; overflow-y: auto; } .sidebar { @@ -77,11 +66,6 @@ .main-viewport { grid-column: 1; grid-row: 2; - height: 500px; + height: auto; } - .terminal-dock { - grid-column: 1; - grid-row: 3; - height: 300px; - } -} +} \ No newline at end of file diff --git a/api/static/css/terminal.css b/api/static/css/terminal.css deleted file mode 100644 index 7cf79ba..0000000 --- a/api/static/css/terminal.css +++ /dev/null @@ -1,77 +0,0 @@ -.terminal-header { - background-color: var(--color-bg-2); - padding: var(--space-xs) var(--space-md); - border-bottom: var(--border-width) solid var(--color-bg-3); - display: flex; - justify-content: space-between; - align-items: center; -} - -.terminal-title { - font-size: 0.7rem; - font-weight: 800; - text-transform: uppercase; - color: var(--color-text-muted); - letter-spacing: 0.1em; -} - -.terminal-body { - flex: 1; - overflow-y: auto; - padding: var(--space-md); - font-family: 'JetBrains Mono', 'Fira Code', monospace; - font-size: 0.8125rem; - line-height: 1.6; - background-color: #0d0d0f; /* Slightly darker for terminal */ -} - -.terminal-line { - white-space: pre-wrap; - word-break: break-all; - margin-bottom: 2px; -} - -.terminal-line.info { color: var(--color-text-primary); } -.terminal-line.error { color: var(--color-danger); font-weight: 600; } -.terminal-line.warning { color: var(--color-warning); } -.terminal-line.muted { color: var(--color-text-muted); } -.terminal-line.success { color: var(--color-success); } -.terminal-line.reasoning { color: var(--color-accent); border-left: 2px solid var(--color-accent); padding-left: var(--space-sm); margin: var(--space-sm) 0; } - -.btn-terminal { - background: none; - border: none; - color: var(--color-text-muted); - font-size: 0.65rem; - font-weight: 700; - text-transform: uppercase; - cursor: pointer; - padding: 4px 8px; - border-radius: 2px; - transition: var(--transition-fast); -} - -.btn-terminal:hover { - color: var(--color-text-primary); - background-color: var(--color-bg-3); -} - -.btn-terminal.active { - color: var(--color-accent); -} - -.empty-state { - display: flex; - flex-direction: column; - align-items: center; - justify-content: center; - height: 300px; - color: var(--color-text-muted); - font-size: 0.875rem; - text-transform: uppercase; - letter-spacing: 0.05em; - font-weight: 600; - border: 2px dashed var(--color-bg-3); - margin: var(--space-lg); - border-radius: var(--border-radius); -} diff --git a/api/static/index.html b/api/static/index.html index 564618a..4903c06 100644 --- a/api/static/index.html +++ b/api/static/index.html @@ -3,12 +3,11 @@ - AI Reasoning Framework + AI Agent Evaluation Dashboard - @@ -19,95 +18,78 @@
- +
- - + +
-
-
- - -
-
-
No active data.
-
-
- -
+
- - +

Interactive Dashboard

+
+
Select a CSV to generate visuals.
-

Saved Visualizations

+

Static Visualizations (viz/)

No visualizations generated.
+ +
+
+
No active data.
+
+
@@ -122,23 +104,9 @@
- - -
-
- Event Log -
- - -
-
-
-
System ready. Waiting for run...
-
-
- + \ No newline at end of file diff --git a/api/static/js/components/config.js b/api/static/js/components/config.js deleted file mode 100644 index d87c874..0000000 --- a/api/static/js/components/config.js +++ /dev/null @@ -1,107 +0,0 @@ -/** - * config.js - Experiment configuration and execution - */ -import { api } from '../api.js'; -import { ui } from '../ui.js'; -import { terminal } from './terminal.js'; -import { results } from './results.js'; - -let isRunning = false; - -export const config = { - async init() { - await Promise.all([ - this.loadScenarios(), - this.loadModels(), - this.loadDefaults() - ]); - - ui.elements.runBtn.addEventListener('click', () => this.startRun()); - ui.elements.cancelBtn.addEventListener('click', () => this.stopRun()); - - // Scenario detail updates - ui.elements.scenarioSelect.addEventListener('change', (e) => this.updateOversight(e.target.value)); - }, - - async loadScenarios() { - const scenarios = await api.get('/api/scenarios'); - ui.elements.scenarioSelect.innerHTML = scenarios.map(s => - `` - ).join(''); - if (scenarios[0]) this.updateOversight(scenarios[0].name); - }, - - async loadModels() { - const models = await api.get('/api/models'); - ui.elements.modelSelect.innerHTML = models.map(m => - `` - ).join(''); - }, - - async loadDefaults() { - const cfg = await api.get('/api/config'); - if (cfg.defaults?.oversight) { - ui.elements.oversightSelect.value = cfg.defaults.oversight; - } - }, - - async updateOversight(scenarioName) { - if (!scenarioName) return; - const details = await api.get(`/api/scenarios/${scenarioName}`); - const levels = details.oversight_levels || ['low', 'mid', 'high']; - ui.elements.oversightSelect.innerHTML = levels.map(l => - `` - ).join(''); - }, - - async startRun() { - const data = { - scenario: ui.elements.scenarioSelect.value, - model: ui.elements.modelSelect.value, - oversight: ui.elements.oversightSelect.value, - runs: ui.elements.runsInput.value - }; - - this.setRunningState(true); - terminal.clear(); - terminal.append(`[SYSTEM] Starting experiment: ${data.scenario} | ${data.model}`); - - try { - await api.post('/api/run', data); - api.streamLogs( - (log) => terminal.append(log), - () => { - this.setRunningState(false); - ui.updateStatus('complete'); - terminal.append('[SYSTEM] Run completed successfully.'); - results.loadAll(); - }, - (err) => { - this.setRunningState(false); - ui.updateStatus('error'); - terminal.append(`[ERROR] Stream disconnected: ${err}`); - } - ); - ui.updateStatus('running', new Date()); - } catch (err) { - this.setRunningState(false); - ui.updateStatus('error'); - terminal.append(`[ERROR] Failed to start run: ${err.message}`); - } - }, - - async stopRun() { - try { - await api.delete('/api/run'); - terminal.append('[SYSTEM] Cancel request sent.'); - } catch (err) { - terminal.append(`[ERROR] Cancel failed: ${err.message}`); - } - }, - - setRunningState(running) { - isRunning = running; - ui.elements.runBtn.disabled = running; - ui.elements.cancelBtn.disabled = !running; - } -}; diff --git a/api/static/js/components/results.js b/api/static/js/components/results.js index d06d7fc..92514c4 100644 --- a/api/static/js/components/results.js +++ b/api/static/js/components/results.js @@ -4,24 +4,21 @@ import { api } from '../api.js'; import { ui } from '../ui.js'; -let activeCharts = []; // Store chart instances to destroy them before redraw +let activeCharts = []; +let currentCsvData = null; // Store fetched data locally for quick filtering +let currentCsvColumns = []; export const results = { async init() { // Bind events - ui.elements.csvFileSelect.addEventListener('change', (e) => this.loadCSVData(e.target.value)); - ui.elements.chartFileSelect.addEventListener('change', (e) => this.loadChartData(e.target.value)); + ui.elements.csvFileSelect.addEventListener('change', (e) => this.handleCsvChange(e.target.value)); ui.elements.judgeFileSelect.addEventListener('change', (e) => this.loadJudgeData(e.target.value)); + ui.elements.refreshBtn.addEventListener('click', () => this.applyFilters()); - // Initial lists - await this.loadAll(); - }, - - async loadAll() { await Promise.all([ this.updateCSVFileList(), this.updateJudgeFileList(), - this.loadImages() // Static images just load once + this.loadImages() ]); }, @@ -33,19 +30,14 @@ export const results = { : ''; ui.elements.csvFileSelect.innerHTML = options; - ui.elements.chartFileSelect.innerHTML = options; - // Prefer loading results.csv by default if it exists const defaultFile = files.find(f => f.endsWith('results.csv')) || files[0]; - if (defaultFile) { ui.elements.csvFileSelect.value = defaultFile; - ui.elements.chartFileSelect.value = defaultFile; - await this.loadCSVData(defaultFile); - await this.loadChartData(defaultFile); + await this.handleCsvChange(defaultFile); } } catch (error) { - console.error("Failed loading CSV file list", error); + console.error("Failed loading CSV list", error); } }, @@ -57,7 +49,6 @@ export const results = { : ''; ui.elements.judgeFileSelect.innerHTML = options; - if (files.length > 0) { ui.elements.judgeFileSelect.value = files[0]; await this.loadJudgeData(files[0]); @@ -67,92 +58,130 @@ export const results = { } }, - async loadCSVData(filePath) { + async handleCsvChange(filePath) { if (!filePath) return; ui.elements.csvResults.innerHTML = '
Loading data...
'; + ui.elements.interactiveCharts.innerHTML = '
Loading charts...
'; try { const data = await api.get(`/api/results?file=${encodeURIComponent(filePath)}`); - if (Object.keys(data).length === 0 || data.error) { - ui.elements.csvResults.innerHTML = '
Failed to load data.
'; - return; - } - - // Since we asked for a specific file, it's the only key const content = Object.values(data)[0]; - if (content.error) { - ui.elements.csvResults.innerHTML = `
Error: ${content.error}
`; + + if (!content || content.error) { + const err = content?.error || "Unknown error"; + ui.elements.csvResults.innerHTML = `
Error: ${err}
`; + ui.elements.interactiveCharts.innerHTML = `
Error: ${err}
`; return; } - let html = `
-
${filePath}
-
- - ${content.columns.map(c => ``).join('')} - - ${content.data.map(row => ` - ${content.columns.map(col => ``).join('')} - `).join('')} - -
${c}
${row[col] ?? ''}
-
-
`; + currentCsvData = content.data; + currentCsvColumns = content.columns; - ui.elements.csvResults.innerHTML = html; + this.populateFilters(); + this.applyFilters(); // This draws charts and tables + } catch (err) { ui.elements.csvResults.innerHTML = `
Error: ${err.message}
`; + ui.elements.interactiveCharts.innerHTML = `
Error: ${err.message}
`; } }, - async loadChartData(filePath) { - if (!filePath) return; + populateFilters() { + const models = new Set(); + const scenarios = new Set(); + const oversights = new Set(); + + currentCsvData.forEach(row => { + if (row.model) models.add(row.model); + if (row.scenario) scenarios.add(row.scenario); + if (row.oversight) oversights.add(row.oversight); + }); + + ui.elements.filterModel.innerHTML = '' + + Array.from(models).sort().map(m => ``).join(''); + + ui.elements.filterScenario.innerHTML = '' + + Array.from(scenarios).sort().map(s => ``).join(''); + + ui.elements.filterOversight.innerHTML = '' + + Array.from(oversights).sort().map(o => ``).join(''); + }, + + applyFilters() { + if (!currentCsvData) return; + + const modelFilter = ui.elements.filterModel.value; + const scenarioFilter = ui.elements.filterScenario.value; + const oversightFilter = ui.elements.filterOversight.value; + + const filteredData = currentCsvData.filter(row => { + return (modelFilter === 'all' || row.model === modelFilter) && + (scenarioFilter === 'all' || row.scenario === scenarioFilter) && + (oversightFilter === 'all' || row.oversight === oversightFilter); + }); + + ui.elements.dataStats.textContent = `Showing ${filteredData.length} of ${currentCsvData.length} records`; + + this.renderTable(filteredData); + this.renderCharts(filteredData); + }, + + renderTable(data) { + if (data.length === 0) { + ui.elements.csvResults.innerHTML = '
No data matches filters.
'; + return; + } + + let html = `
+
+ + ${currentCsvColumns.map(c => ``).join('')} + + ${data.map(row => ` + ${currentCsvColumns.map(col => ``).join('')} + `).join('')} + +
${c}
${row[col] ?? ''}
+
+
`; - // Clean up old charts + ui.elements.csvResults.innerHTML = html; + }, + + renderCharts(data) { activeCharts.forEach(chart => chart.destroy()); activeCharts = []; - ui.elements.interactiveCharts.innerHTML = '
Generating charts...
'; + ui.elements.interactiveCharts.innerHTML = ''; - try { - const data = await api.get(`/api/results?file=${encodeURIComponent(filePath)}`); - const content = Object.values(data)[0]; - - if (!content || content.error || !content.data || content.data.length === 0) { - ui.elements.interactiveCharts.innerHTML = '
Not enough data to graph.
'; - return; - } + if (data.length === 0) { + ui.elements.interactiveCharts.innerHTML = '
No data to chart.
'; + return; + } - ui.elements.interactiveCharts.innerHTML = ''; - - // Generate charts based on available columns - const cols = content.columns; - - // 1. Blackbox Categories - if (cols.includes('blackbox_category')) { - this.renderPieChart('Blackbox Categories', content.data, 'blackbox_category'); - } - - // 2. Glassbox Categories - if (cols.includes('glassbox_category')) { - this.renderPieChart('Glassbox Categories', content.data, 'glassbox_category'); - } + const cols = currentCsvColumns; + + if (cols.includes('blackbox_category')) { + this.renderDoughnutChart('Blackbox Categories', data, 'blackbox_category'); + } + + if (cols.includes('glassbox_category')) { + this.renderDoughnutChart('Glassbox Categories', data, 'glassbox_category'); + } - // 3. Models vs Blackbox Category - if (cols.includes('model') && cols.includes('blackbox_category')) { - this.renderBarChart('Deception by Model', content.data, 'model', 'blackbox_category'); - } - - // 4. Fallback if no specific columns exist - if (ui.elements.interactiveCharts.innerHTML === '') { - ui.elements.interactiveCharts.innerHTML = '
No plottable categorical columns found in this CSV.
'; - } + if (cols.includes('model') && cols.includes('blackbox_category')) { + this.renderStackedBarChart('Blackbox Category by Model', data, 'model', 'blackbox_category'); + } - } catch (err) { - ui.elements.interactiveCharts.innerHTML = `
Chart Error: ${err.message}
`; + if (cols.includes('oversight') && cols.includes('blackbox_category')) { + this.renderStackedBarChart('Blackbox Category by Oversight', data, 'oversight', 'blackbox_category'); + } + + if (ui.elements.interactiveCharts.innerHTML === '') { + ui.elements.interactiveCharts.innerHTML = '
No plottable categorical columns found.
'; } }, - renderPieChart(title, data, column) { + renderDoughnutChart(title, data, column) { const counts = {}; data.forEach(row => { const val = row[column] || 'Unknown'; @@ -160,10 +189,7 @@ export const results = { }); const canvasId = `chart-${Math.random().toString(36).substr(2, 9)}`; - const container = document.createElement('div'); - container.className = 'chart-container'; - container.innerHTML = ``; - ui.elements.interactiveCharts.appendChild(container); + this.createChartContainer(canvasId); const ctx = document.getElementById(canvasId).getContext('2d'); const chart = new Chart(ctx, { @@ -176,20 +202,12 @@ export const results = { borderWidth: 0 }] }, - options: { - responsive: true, - maintainAspectRatio: false, - plugins: { - title: { display: true, text: title, color: '#f0f0f2', font: { family: 'Inter', size: 14 } }, - legend: { labels: { color: '#9ea0a6' }, position: 'right' } - } - } + options: this.getChartOptions(title) }); activeCharts.push(chart); }, - renderBarChart(title, data, xCol, groupCol) { - // Group by X then GroupCol + renderStackedBarChart(title, data, xCol, groupCol) { const matrix = {}; const groups = new Set(); @@ -203,7 +221,7 @@ export const results = { const labels = Object.keys(matrix); const datasets = Array.from(groups).map((group, i) => { - const colors = ['#ef4444', '#10b981', '#f59e0b', '#3b82f6', '#9d4edd']; + const colors = ['#ef4444', '#10b981', '#f59e0b', '#3b82f6', '#9d4edd', '#6366f1']; return { label: group, data: labels.map(label => matrix[label][group] || 0), @@ -212,31 +230,41 @@ export const results = { }); const canvasId = `chart-${Math.random().toString(36).substr(2, 9)}`; - const container = document.createElement('div'); - container.className = 'chart-container'; - container.innerHTML = ``; - ui.elements.interactiveCharts.appendChild(container); + this.createChartContainer(canvasId); const ctx = document.getElementById(canvasId).getContext('2d'); const chart = new Chart(ctx, { type: 'bar', data: { labels, datasets }, options: { - responsive: true, - maintainAspectRatio: false, + ...this.getChartOptions(title), scales: { x: { stacked: true, ticks: { color: '#9ea0a6' }, grid: { color: '#2a2a2e' } }, y: { stacked: true, ticks: { color: '#9ea0a6', stepSize: 1 }, grid: { color: '#2a2a2e' } } - }, - plugins: { - title: { display: true, text: title, color: '#f0f0f2', font: { family: 'Inter', size: 14 } }, - legend: { labels: { color: '#9ea0a6' } } } } }); activeCharts.push(chart); }, + createChartContainer(canvasId) { + const container = document.createElement('div'); + container.className = 'chart-container'; + container.innerHTML = ``; + ui.elements.interactiveCharts.appendChild(container); + }, + + getChartOptions(title) { + return { + responsive: true, + maintainAspectRatio: false, + plugins: { + title: { display: true, text: title, color: '#f0f0f2', font: { family: 'Inter', size: 14 } }, + legend: { labels: { color: '#9ea0a6' }, position: 'bottom' } + } + }; + }, + async loadImages() { const images = await api.get('/api/results/images'); if (images.length === 0) return; @@ -269,7 +297,6 @@ export const results = { let html = ''; if (content.type === 'csv') { html += `
-
${filePath}
${content.columns.map(c => ``).join('')} @@ -282,9 +309,7 @@ export const results = { `; } else { - // Render JSON as pretty block html += `
-
${filePath}
${JSON.stringify(content.data, null, 2)}
`; } @@ -293,4 +318,4 @@ export const results = { ui.elements.judgeResults.innerHTML = `
Error: ${err.message}
`; } } -}; +}; \ No newline at end of file diff --git a/api/static/js/components/terminal.js b/api/static/js/components/terminal.js deleted file mode 100644 index 77c9794..0000000 --- a/api/static/js/components/terminal.js +++ /dev/null @@ -1,49 +0,0 @@ -/** - * terminal.js - Log rendering logic - */ -import { ui } from '../ui.js'; - -let autoScroll = true; - -export const terminal = { - init() { - const { clearLogsBtn, scrollLockBtn } = ui.elements; - - clearLogsBtn.addEventListener('click', () => this.clear()); - - scrollLockBtn.addEventListener('click', () => { - autoScroll = !autoScroll; - scrollLockBtn.classList.toggle('active', autoScroll); - }); - }, - - append(message) { - const { logsContainer } = ui.elements; - const line = document.createElement('div'); - line.className = 'terminal-line'; - - // High-speed parsing for log levels - if (message.includes('[ERROR]')) line.classList.add('error'); - else if (message.includes('[WARN]')) line.classList.add('warning'); - else if (message.includes('[DEBUG]')) line.classList.add('muted'); - else if (message.includes('SUCCESS') || message.includes('complete')) line.classList.add('success'); - else if (message.includes('Reasoning:')) line.classList.add('reasoning'); - else line.classList.add('info'); - - line.textContent = message; - logsContainer.appendChild(line); - - if (autoScroll) { - logsContainer.scrollTop = logsContainer.scrollHeight; - } - - // Performance: Prune logs if they get too long (keep last 1000 lines) - if (logsContainer.children.length > 1000) { - logsContainer.removeChild(logsContainer.firstChild); - } - }, - - clear() { - ui.elements.logsContainer.innerHTML = '
Logs cleared.
'; - } -}; diff --git a/api/static/js/main.js b/api/static/js/main.js index 21a957a..8c540c3 100644 --- a/api/static/js/main.js +++ b/api/static/js/main.js @@ -1,61 +1,21 @@ -/** - * main.js - Application Entry Point - */ import { ui } from './ui.js'; -import { api } from './api.js'; -import { terminal } from './components/terminal.js'; -import { config } from './components/config.js'; import { results } from './components/results.js'; async function init() { - console.log('Initializing AI Reasoning Framework GUI...'); - - // Init Core Components - terminal.init(); - await config.init(); + console.log('Initializing AI Reasoning Dashboard...'); + await results.init(); // Init UI behaviors ui.initTabs((tabId) => { - // Refresh data when switching to results tabs - if (['csv', 'images', 'judge'].includes(tabId)) { - results.loadAll(); - } + // We could lazy load data here, but for now everything triggers on file/filter change }); - - // Initial Status Check - try { - const status = await api.get('/api/run/status'); - ui.updateStatus(status.status, status.start_time); - - if (status.status === 'running') { - config.setRunningState(true); - // Re-attach to stream if page reloaded - api.streamLogs( - (log) => terminal.append(log), - () => { - config.setRunningState(false); - ui.updateStatus('complete'); - results.loadAll(); - } - ); - } - } catch (e) { - console.warn('Initial status check failed', e); - } } -// Start the app function start() { init().catch(err => { console.error("Initialization error:", err); - const logsContainer = document.getElementById('logs-container'); - if (logsContainer) { - const errorLine = document.createElement('div'); - errorLine.className = 'terminal-line error'; - errorLine.textContent = `[GUI STARTUP ERROR] ${err.message}`; - logsContainer.appendChild(errorLine); - } + alert(`Dashboard Failed to Load: ${err.message}`); }); } @@ -63,4 +23,4 @@ if (document.readyState === 'loading') { document.addEventListener('DOMContentLoaded', start); } else { start(); -} +} \ No newline at end of file diff --git a/api/static/js/ui.js b/api/static/js/ui.js index a2c8b8b..359909b 100644 --- a/api/static/js/ui.js +++ b/api/static/js/ui.js @@ -1,49 +1,26 @@ -/** - * ui.js - Global UI Utilities - */ - export const ui = { get elements() { return { - scenarioSelect: document.getElementById('scenario-select'), - modelSelect: document.getElementById('model-select'), - oversightSelect: document.getElementById('oversight-select'), - runsInput: document.getElementById('runs-input'), - runBtn: document.getElementById('run-btn'), - cancelBtn: document.getElementById('cancel-btn'), - statusBadge: document.getElementById('status-badge'), - statusTime: document.getElementById('status-time'), - logsContainer: document.getElementById('logs-container'), + csvFileSelect: document.getElementById('csv-file-select'), + judgeFileSelect: document.getElementById('judge-file-select'), + + filterModel: document.getElementById('filter-model'), + filterScenario: document.getElementById('filter-scenario'), + filterOversight: document.getElementById('filter-oversight'), + + refreshBtn: document.getElementById('refresh-btn'), csvResults: document.getElementById('csv-results'), imageResults: document.getElementById('image-results'), judgeResults: document.getElementById('judge-results'), interactiveCharts: document.getElementById('interactive-charts'), - - csvFileSelect: document.getElementById('csv-file-select'), - chartFileSelect: document.getElementById('chart-file-select'), - judgeFileSelect: document.getElementById('judge-file-select'), + dataStats: document.getElementById('data-stats'), tabTriggers: document.querySelectorAll('.tab-trigger'), - tabPanels: document.querySelectorAll('.tab-panel'), - clearLogsBtn: document.getElementById('clear-logs-btn'), - scrollLockBtn: document.getElementById('scroll-lock-btn') + tabPanels: document.querySelectorAll('.tab-panel') }; }, - updateStatus(status, time = null) { - const { statusBadge, statusTime } = this.elements; - statusBadge.className = `badge badge-${status}`; - statusBadge.textContent = status.toUpperCase(); - - if (time) { - const date = new Date(time); - statusTime.textContent = date.toLocaleTimeString(); - } else if (status === 'idle') { - statusTime.textContent = ''; - } - }, - initTabs(onTabChange) { this.elements.tabTriggers.forEach(trigger => { trigger.addEventListener('click', () => { @@ -59,4 +36,4 @@ export const ui = { }); }); } -}; +}; \ No newline at end of file -- cgit v1.2.3
${c}