Skip to content

Commit 0e7af5c

Browse files
committed
Hybrid process reasoning limit fixes
1 parent 31a6a61 commit 0e7af5c

20 files changed

Lines changed: 3304 additions & 226 deletions

‎.gitignore‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -53,3 +53,4 @@ data/accounting.sqlite
5353
data/audit_log.sqlite
5454
.aider.conf.yml
5555
.aider*
56+
.dev/

‎debug_reasoning.py‎

Lines changed: 166 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,166 @@
1+
#!/usr/bin/env python3
2+
"""
3+
Debug script to see what DeepSeek-R1 is actually outputting
4+
"""
5+
6+
import sys
7+
import os
8+
import logging
9+
10+
# Add src to path
11+
sys.path.insert(0, os.path.join(os.path.dirname(__file__), 'src'))
12+
13+
from llm_client import LLMClient
14+
from llm_config import LLMConfig
15+
16+
# Set up logging
17+
logging.basicConfig(level=logging.DEBUG, format='%(levelname)s [%(name)s] %(message)s')
18+
19+
def test_deepseek_output():
20+
"""Test what DeepSeek-R1 actually outputs"""
21+
22+
api_key = os.getenv('OPENROUTER_API_KEY')
23+
if not api_key:
24+
print("❌ Error: OPENROUTER_API_KEY environment variable not set")
25+
return
26+
27+
llm_client = LLMClient(api_key=api_key)
28+
29+
# Test prompt
30+
prompt = """Problem: What is 2+2?
31+
32+
Think step-by-step to solve this problem. When you finish your reasoning, output exactly: <REASONING_COMPLETE>
33+
34+
Reasoning:"""
35+
36+
print("🔍 Testing DeepSeek-R1 output...")
37+
print("📝 Prompt:")
38+
print(prompt)
39+
print("\n" + "="*50 + "\n")
40+
41+
# Test 1: Without stop token
42+
print("🧪 TEST 1: Without stop token")
43+
try:
44+
config = LLMConfig(
45+
temperature=0.1,
46+
max_tokens=800,
47+
stop=None # No stop token
48+
)
49+
50+
output, stats = llm_client.call(
51+
prompt=prompt,
52+
models=["deepseek/deepseek-r1-0528:free"],
53+
config=config
54+
)
55+
56+
print("✅ SUCCESS!")
57+
print("📊 Stats:")
58+
print(f" Model: {stats.model_name}")
59+
print(f" Tokens: {stats.completion_tokens}")
60+
print(f" Duration: {stats.call_duration_seconds:.2f}s")
61+
print("\n🔍 Raw Output:")
62+
print("="*50)
63+
print(repr(output))
64+
print("="*50)
65+
print(output)
66+
print("="*50)
67+
68+
# Test reasoning extraction
69+
print("\n🧠 Testing reasoning extraction...")
70+
71+
# Check if it contains DeepSeek-R1 thinking tags
72+
if "<THINKING>" in output.upper():
73+
print("✅ Found <THINKING> tags")
74+
elif "<think>" in output.lower():
75+
print("✅ Found <think> tags")
76+
else:
77+
print("❌ No DeepSeek thinking tags found")
78+
79+
# Check if it contains the completion token
80+
if "<REASONING_COMPLETE>" in output:
81+
print("✅ Found <REASONING_COMPLETE> token")
82+
else:
83+
print("❌ No <REASONING_COMPLETE> token found")
84+
85+
except Exception as e:
86+
print(f"❌ Error: {e}")
87+
88+
print("\n" + "="*60 + "\n")
89+
90+
# Test 2: With stop token
91+
print("🧪 TEST 2: With stop token")
92+
try:
93+
config = LLMConfig(
94+
temperature=0.1,
95+
max_tokens=800,
96+
stop=["<REASONING_COMPLETE>"]
97+
)
98+
99+
output, stats = llm_client.call(
100+
prompt=prompt,
101+
models=["deepseek/deepseek-r1-0528:free"],
102+
config=config
103+
)
104+
105+
print("✅ SUCCESS!")
106+
print("📊 Stats:")
107+
print(f" Model: {stats.model_name}")
108+
print(f" Tokens: {stats.completion_tokens}")
109+
print(f" Duration: {stats.call_duration_seconds:.2f}s")
110+
print("\n🔍 Raw Output:")
111+
print("="*50)
112+
print(repr(output))
113+
print("="*50)
114+
print(output)
115+
print("="*50)
116+
117+
except Exception as e:
118+
print(f"❌ Error: {e}")
119+
120+
print("\n" + "="*60 + "\n")
121+
122+
# Test 3: Different prompt style for DeepSeek
123+
print("🧪 TEST 3: DeepSeek-specific prompt")
124+
try:
125+
deepseek_prompt = """<THINKING>
126+
Let me solve this step by step:
127+
128+
Problem: What is 2+2?
129+
130+
I need to add 2 and 2 together.
131+
2 + 2 = 4
132+
133+
The answer is 4.
134+
</THINKING>
135+
136+
Looking at this problem, I need to add 2 and 2 together."""
137+
138+
config = LLMConfig(
139+
temperature=0.1,
140+
max_tokens=400,
141+
stop=None
142+
)
143+
144+
output, stats = llm_client.call(
145+
prompt=deepseek_prompt,
146+
models=["deepseek/deepseek-r1-0528:free"],
147+
config=config
148+
)
149+
150+
print("✅ SUCCESS!")
151+
print("📊 Stats:")
152+
print(f" Model: {stats.model_name}")
153+
print(f" Tokens: {stats.completion_tokens}")
154+
print(f" Duration: {stats.call_duration_seconds:.2f}s")
155+
print("\n🔍 Raw Output:")
156+
print("="*50)
157+
print(repr(output))
158+
print("="*50)
159+
print(output)
160+
print("="*50)
161+
162+
except Exception as e:
163+
print(f"❌ Error: {e}")
164+
165+
if __name__ == "__main__":
166+
test_deepseek_output()

‎pytest.ini‎

Lines changed: 19 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,19 @@
1+
[tool:pytest]
2+
markers =
3+
integration: marks tests as integration tests (deselect with '-m "not integration"')
4+
slow: marks tests as slow (deselect with '-m "not slow"')
5+
6+
# By default, skip integration tests
7+
addopts = -m "not integration"
8+
9+
# Test discovery
10+
testpaths = tests
11+
python_files = test_*.py
12+
python_classes = Test*
13+
python_functions = test_*
14+
15+
# Output configuration
16+
console_output_style = progress
17+
filterwarnings =
18+
ignore::DeprecationWarning
19+
ignore::PendingDeprecationWarning

‎run_tests.py‎

Lines changed: 119 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,119 @@
1+
#!/usr/bin/env python3
2+
"""
3+
Test runner script for the LLM AOT Process project.
4+
5+
This script provides convenient ways to run different types of tests:
6+
- Unit tests (default)
7+
- Integration tests (requires API keys)
8+
- All tests
9+
- Specific test modules or functions
10+
11+
Usage:
12+
python run_tests.py # Run unit tests only
13+
python run_tests.py --integration # Run integration tests only
14+
python run_tests.py --all # Run all tests
15+
python run_tests.py --hybrid # Run hybrid-related tests only
16+
python run_tests.py --verbose # Run with verbose output
17+
python run_tests.py --coverage # Run with coverage report
18+
"""
19+
20+
import sys
21+
import os
22+
import subprocess
23+
import argparse
24+
from pathlib import Path
25+
26+
def run_command(cmd, description=""):
27+
"""Run a command and return success status"""
28+
print(f"🔄 {description}")
29+
print(f"Running: {' '.join(cmd)}")
30+
result = subprocess.run(cmd, capture_output=False)
31+
if result.returncode == 0:
32+
print(f"✅ {description} - SUCCESS")
33+
else:
34+
print(f"❌ {description} - FAILED")
35+
return result.returncode == 0
36+
37+
def main():
38+
parser = argparse.ArgumentParser(description="Run tests for LLM AOT Process project")
39+
parser.add_argument("--integration", action="store_true",
40+
help="Run integration tests (requires OPENROUTER_API_KEY)")
41+
parser.add_argument("--all", action="store_true",
42+
help="Run all tests including integration tests")
43+
parser.add_argument("--hybrid", action="store_true",
44+
help="Run hybrid-related tests only")
45+
parser.add_argument("--functionality", action="store_true",
46+
help="Run functionality tests that catch method signature issues")
47+
parser.add_argument("--verbose", "-v", action="store_true",
48+
help="Run with verbose output")
49+
parser.add_argument("--coverage", action="store_true",
50+
help="Run with coverage report")
51+
parser.add_argument("--module", type=str,
52+
help="Run specific test module (e.g., 'tests.hybrid.test_hybrid_processor')")
53+
parser.add_argument("--function", type=str,
54+
help="Run specific test function (e.g., 'test_hybrid_processor_method_signatures')")
55+
56+
args = parser.parse_args()
57+
58+
# Base pytest command - use sys.executable to ensure we use the right Python
59+
cmd = [sys.executable, "-m", "pytest"]
60+
61+
# Add verbosity
62+
if args.verbose:
63+
cmd.append("-v")
64+
65+
# Add coverage
66+
if args.coverage:
67+
cmd.extend(["--cov=src", "--cov-report=html", "--cov-report=term-missing"])
68+
69+
# Determine what tests to run
70+
if args.all:
71+
cmd.append("-m")
72+
cmd.append("integration or not integration")
73+
description = "Running ALL tests (unit + integration)"
74+
elif args.integration:
75+
cmd.append("-m")
76+
cmd.append("integration")
77+
description = "Running INTEGRATION tests only"
78+
# Check for API key
79+
if not os.getenv('OPENROUTER_API_KEY'):
80+
print("⚠️ WARNING: OPENROUTER_API_KEY environment variable not set")
81+
print(" Integration tests may be skipped or fail")
82+
elif args.hybrid:
83+
cmd.append("tests/hybrid/")
84+
if args.functionality:
85+
cmd.append("tests/hybrid/test_hybrid_functionality.py")
86+
description = "Running HYBRID tests only"
87+
elif args.functionality:
88+
cmd.append("tests/hybrid/test_hybrid_method_signatures.py")
89+
description = "Running FUNCTIONALITY tests (method signature validation)"
90+
elif args.module:
91+
cmd.append(args.module.replace(".", "/") + ".py")
92+
description = f"Running module: {args.module}"
93+
elif args.function:
94+
cmd.append("-k")
95+
cmd.append(args.function)
96+
description = f"Running function: {args.function}"
97+
else:
98+
# Default: run unit tests only (integration tests excluded by pytest.ini)
99+
description = "Running UNIT tests only (integration tests excluded)"
100+
101+
# Run the tests
102+
success = run_command(cmd, description)
103+
104+
if not success:
105+
print("\n❌ Tests failed!")
106+
sys.exit(1)
107+
else:
108+
print("\n✅ All tests passed!")
109+
110+
# If running integration tests, provide helpful info
111+
if args.integration or args.all:
112+
print("\n📋 Integration Test Info:")
113+
print(" - These tests make real API calls to OpenRouter")
114+
print(" - They test end-to-end functionality with actual models")
115+
print(" - Requires OPENROUTER_API_KEY environment variable")
116+
print(" - Tests use: deepseek/deepseek-r1-0528:free and openrouter/cypher-alpha:free")
117+
118+
if __name__ == "__main__":
119+
main()

0 commit comments

Comments
 (0)