diff options
| author | CaptainJack2491 <jayrupnakawala@gmail.com> | 2026-04-01 16:35:01 +0100 |
|---|---|---|
| committer | CaptainJack2491 <jayrupnakawala@gmail.com> | 2026-04-01 16:35:01 +0100 |
| commit | fa58564ea288600ad93eaf0a5aa111beca6197bc (patch) | |
| tree | 001df2476bbe3044c9594ae1b22ee2e8a621e94a | |
| parent | 289d93c8c99251d5c0b5b34647fa77b8dca63f53 (diff) | |
feat: implement robust reasoning token handling and multi-turn persistence
- Refactored Agent's chat loop to explicitly preserve and pass back reasoning context (reasoning_details, reasoning) to OpenRouter.
- Ensures frontier models like Gemini 3.1 and DeepSeek v3.2 maintain their 'thinking' chain during complex tool-calling sessions.
- Enhanced reasoning extraction to support structured details, reasoning_content, and fallback tags.
- Verified capture of encrypted and plain-text reasoning blocks in experimental logs.
| -rw-r--r-- | config.yaml.example | 38 | ||||
| -rw-r--r-- | config_dryrun.yaml | 53 | ||||
| -rw-r--r-- | src/agent.py | 80 |
3 files changed, 131 insertions, 40 deletions
diff --git a/config.yaml.example b/config.yaml.example index a7554ee..7253c19 100644 --- a/config.yaml.example +++ b/config.yaml.example @@ -30,31 +30,43 @@ providers: api_key_env: OPENROUTER_API_KEY base_url: https://openrouter.ai/api/v1 -# Models to test +# Models to test (frontier models with reasoning traces, via OpenRouter) models: - - id: gemini-3-flash-preview - provider: google - temperature: 1 + - id: google/gemini-3.1-pro-preview + provider: openrouter + extra_body: + reasoning: + effort: "high" + exclude: false - # - id: openai/gpt-oss-20b - # provider: groq + - id: z-ai/glm-5 + provider: openrouter + extra_body: + reasoning: + effort: "high" + exclude: false - # - id: claude-haiku-4-5 - # provider: anthropic - # temperature: 0.7 + - id: deepseek/deepseek-v3.2 + provider: openrouter + extra_body: + reasoning: + effort: "high" + exclude: false # Scenarios to run scenarios: - path: scenarios/corporate_sabotage_v2 - runs: 1 + runs: 20 - # - path: scenarios/child_protection - # runs: 1 + - path: scenarios/child_protection + runs: 20 + + - path: scenarios/pharma_trial + runs: 20 # Oversight levels to test oversight_levels: - low - # - medium - high # Output settings diff --git a/config_dryrun.yaml b/config_dryrun.yaml new file mode 100644 index 0000000..1bb99da --- /dev/null +++ b/config_dryrun.yaml @@ -0,0 +1,53 @@ +defaults: + temperature: 1.0 + generate_baseline: false + +providers: + openrouter: + api_key_env: OPENROUTER_API_KEY + base_url: https://openrouter.ai/api/v1 + +models: + - id: google/gemini-3.1-pro-preview + provider: openrouter + extra_body: + reasoning: + effort: "high" # Using low for dry run to save tokens + exclude: false + - id: z-ai/glm-5 + provider: openrouter + extra_body: + reasoning: + effort: "high" # Using low for dry run to save tokens + exclude: false + - id: deepseek/deepseek-v3.2 + provider: openrouter + extra_body: + reasoning: + effort: "high" # Using low for dry run to save tokens + exclude: false + +goal_types: + - self_serving + - moral + - bare + +scenarios: + - path: scenarios/pharma_trial + runs: 1 + +oversight_levels: + - high + +execution: + max_workers: 10 + +output: + dir: logs/v2_dry_run + save_vfs_state: true + +logging: + level: 4 + format: "[{level}] {message}" + output: both + file: logs/dry_run.log diff --git a/src/agent.py b/src/agent.py index a8edb20..5df0434 100644 --- a/src/agent.py +++ b/src/agent.py @@ -157,40 +157,50 @@ class Agent: # Extract reasoning from raw response content = response_message.content or "" reasoning = None - reasoning_format = None # Track source format (e.g. openai-responses-v1) - - # Try to get reasoning from different sources - # 1. Check for reasoning_content (OpenRouter — Qwen, DeepSeek, etc.) - if hasattr(response_message, 'reasoning_content') and response_message.reasoning_content: - reasoning = response_message.reasoning_content - reasoning_format = "reasoning_content" - # 2. Check for reasoning (Groq GPT-OSS models) - elif hasattr(response_message, 'reasoning') and response_message.reasoning: - reasoning = response_message.reasoning - reasoning_format = "reasoning" - # 3. Check for reasoning_details (structured — OpenAI reasoning models) - elif hasattr(response_message, 'reasoning_details') and response_message.reasoning_details: + reasoning_format = None # Track format for logs + + # Normalize reasoning extraction across different providers + + # A. OpenRouter / Anthropic / Z.ai structured reasoning_details + # This is the most robust way to preserve context for the next turn + reasoning_details = getattr(response_message, "reasoning_details", None) + if reasoning_details: reasoning_parts = [] - for item in response_message.reasoning_details: + for item in reasoning_details: detail_type = item.get("type", "") if detail_type == "reasoning.text": reasoning_parts.append(item.get("text", "")) elif detail_type == "reasoning.summary": reasoning_parts.append(f"[SUMMARY] {item.get('summary', '')}") - # Capture the format field from the first detail item - reasoning_format = response_message.reasoning_details[0].get("format", "unknown") + elif detail_type == "reasoning.encrypted": + reasoning_parts.append("[ENCRYPTED REASONING]") + reasoning = "\n".join(reasoning_parts).strip() - # 3. Regex fallback for <thinking> tags - else: + reasoning_format = reasoning_details[0].get("format", "structured") + + # B. OpenRouter reasoning_content (DeepSeek, Qwen) + elif hasattr(response_message, 'reasoning_content') and response_message.reasoning_content: + reasoning = response_message.reasoning_content + reasoning_format = "reasoning_content" + + # C. Groq / Legacy reasoning field + elif hasattr(response_message, 'reasoning') and response_message.reasoning: + reasoning = response_message.reasoning + reasoning_format = "reasoning" + + # D. Regex fallback for <thinking> tags in content + if not reasoning and content: thought_match = re.search(r"<(thinking|thought)>(.*?)</\1>", content, re.DOTALL) if thought_match: reasoning = thought_match.group(2).strip() content = content.replace(thought_match.group(0), "").strip() reasoning_format = "thinking_tags" - # If no tags and there are tool calls, content is reasoning - elif response_message.tool_calls: + + # E. If no tags and tool calls exist, check if content IS reasoning + elif response_message.tool_calls and content: + # In some models, the only content is reasoning before a tool call reasoning = content - content = None + content = "" # Don't treat it as final content reasoning_format = "content_as_reasoning" # Log reasoning if available (DEBUG level shows full, INFO shows preview) @@ -201,13 +211,29 @@ class Agent: else: logger.info(f"\n--- REASONING ---\n{reasoning}") elif turn_count == 1: - # Warn on first turn — if the model never reasons on turn 1, - # it's unlikely to reason on later turns either - logger.warning(f"No reasoning detected on first turn for model {self.model}. " - f"Glass-box judging will not be possible for this run.") + logger.warning(f"No reasoning detected on first turn for model {self.model}.") - # Append raw response message to preserve extra_content (Google thoughtSignature) - messages.append(response_message) + # Build the assistant message for the next turn in the conversation + # We convert to a dict to ensure all OpenRouter-specific fields are preserved + # when passed back in the 'messages' list of the next request. + assistant_msg_for_next_turn = { + "role": "assistant", + "content": content + } + if response_message.tool_calls: + assistant_msg_for_next_turn["tool_calls"] = response_message.tool_calls + + # Pass reasoning back to preserve CoT context (CRITICAL for multi-turn) + if reasoning: + # Use 'reasoning' as the primary string field + assistant_msg_for_next_turn["reasoning"] = reasoning + + # If we have structured details, pass them back unmodified to ensure + # continuity (especially for tool calls/summarized/encrypted reasoning) + if reasoning_details: + assistant_msg_for_next_turn["reasoning_details"] = reasoning_details + + messages.append(assistant_msg_for_next_turn) # Log entry log_entry = { |
