diff --git a/tests/bench/implex_conv.py b/tests/bench/implex_conv.py index d6a341ac..55d91d11 100644 --- a/tests/bench/implex_conv.py +++ b/tests/bench/implex_conv.py @@ -34,6 +34,7 @@ from honcho.async_client.session import SessionPeerConfig from honcho_core.types.workspaces.sessions.message_create_param import ( MessageCreateParam, ) +from openai import AsyncOpenAI from typing_extensions import TypedDict from src.config import settings @@ -107,7 +108,8 @@ class ImplexConvRunner: self, base_api_port: int = 8000, pool_size: int = 1, - anthropic_api_key: str | None = None, + llm_api_key: str | None = None, + llm_provider: Literal["anthropic", "openai"] = "openai", timeout_seconds: int | None = None, reasoning_type: Literal["opposed", "supportive"] = "opposed", cleanup_workspace: bool = False, @@ -127,7 +129,8 @@ class ImplexConvRunner: """ self.base_api_port: int = base_api_port self.pool_size: int = pool_size - self.anthropic_api_key: str | None = anthropic_api_key + self.llm_api_key: str | None = llm_api_key + self.llm_provider: Literal["anthropic", "openai"] = llm_provider self.timeout_seconds: int = ( timeout_seconds if timeout_seconds is not None else 10000 ) @@ -151,15 +154,29 @@ class ImplexConvRunner: logging.getLogger("httpx").setLevel(logging.ERROR) logging.getLogger("httpcore").setLevel(logging.ERROR) - if self.anthropic_api_key: - self.anthropic_client: AsyncAnthropic = AsyncAnthropic( - api_key=self.anthropic_api_key - ) - else: - api_key = os.getenv("LLM_ANTHROPIC_API_KEY") + # Initialize LLM client attributes + self.anthropic_client: AsyncAnthropic | None + self.openai_client: AsyncOpenAI | None + + # Initialize LLM client based on provider + if self.llm_provider == "anthropic": + api_key = self.llm_api_key or os.getenv("LLM_ANTHROPIC_API_KEY") if not api_key: - raise ValueError("LLM_ANTHROPIC_API_KEY is not set") + raise ValueError( + "Anthropic API key must be provided via llm_api_key or LLM_ANTHROPIC_API_KEY" + ) self.anthropic_client = AsyncAnthropic(api_key=api_key) + self.openai_client = None + elif self.llm_provider == "openai": + api_key = self.llm_api_key or os.getenv("LLM_OPENAI_API_KEY") + if not api_key: + raise ValueError( + "OpenAI API key must be provided via llm_api_key or LLM_OPENAI_API_KEY" + ) + self.openai_client = AsyncOpenAI(api_key=api_key) + self.anthropic_client = None + else: + raise ValueError(f"Unsupported LLM provider: {self.llm_provider}") def get_honcho_url_for_index(self, example_index: int) -> str: """Get the Honcho URL for a given example index using round-robin.""" @@ -331,6 +348,72 @@ class ImplexConvRunner: return False await asyncio.sleep(1) + async def _call_llm( + self, system_prompt: str, user_prompt: str, max_tokens: int = 500 + ) -> str: + """Helper method to call LLM based on configured provider.""" + try: + if self.llm_provider == "openai": + assert self.openai_client is not None + response = await self.openai_client.chat.completions.create( + model="gpt-4o-mini", + max_tokens=max_tokens, + temperature=0.0, + messages=[ + {"role": "system", "content": system_prompt}, + {"role": "user", "content": user_prompt}, + ], + ) + return response.choices[0].message.content or "" + else: # anthropic + assert self.anthropic_client is not None + response = await self.anthropic_client.messages.create( + model="claude-sonnet-4-5", + max_tokens=max_tokens, + temperature=0.0, + system=system_prompt, + messages=[{"role": "user", "content": user_prompt}], + ) + if not response.content: + return "" + content_block = response.content[0] + return getattr(content_block, "text", "") + except Exception as e: + self.logger.error(f"Error calling LLM: {e}") + return f"ERROR: {e}" + + async def call_assistant_with_context( + self, question: str, honcho_context: str + ) -> str: + """ + Call an assistant LLM with the question and Honcho's context. + + This simulates an AI assistant that has access to Honcho's theory-of-mind + representation and must answer the user's question appropriately. + + The assistant prompt is agnostic to reasoning type - we're evaluating whether + the model can appropriately use whatever context Honcho provides. + """ + system_prompt = """You are a helpful AI assistant with access to information about the user from past conversations. + +Your task is to answer the user's question based on the context provided. Use all relevant information from the context to give an appropriate, helpful response. + +Be concise but thorough.""" + + user_prompt = f"""Based on the following context about the user, please answer their question. + + +{honcho_context} + + + +{question} + + +Please provide a helpful, direct answer that takes into account all relevant information from the context.""" + + return await self._call_llm(system_prompt, user_prompt, max_tokens=500) + async def judge_implicit_reasoning( self, question: str, @@ -349,25 +432,45 @@ class ImplexConvRunner: system_prompt = f""" You are evaluating implicit reasoning in long-term conversations for OPPOSED reasoning. -OPPOSED REASONING means: There is a scenario that PREVENTS or BLOCKS a persona trait. +OPPOSED REASONING means: There is a constraint or limitation that should influence recommendations. -IMPLICIT REASONING: "{implicit_reasoning}" +IMPLICIT CONSTRAINT: "{implicit_reasoning}" -This implicit reasoning should PREVENT certain activities or suggestions in the response. +This constraint should be ACKNOWLEDGED and the response should be ADAPTED accordingly. -CRITICAL EVALUATION CRITERIA: -1. Did the AI find and use the implicit blocking scenario? -2. Does the response appropriately avoid suggestions that would be blocked by this scenario? -3. Is the response logically consistent with the implicit constraint? +EVALUATION CRITERIA (both must be true to pass): +1. **Found Implicit** (found_implicit = true): Does the response show awareness of the constraint? + - Explicitly mentions the constraint, OR + - Clearly adapts suggestions to work around it, OR + - Warns about the limitation -Example: -- Implicit: "broke my leg" +2. **Passed** (passed = true): Are the suggestions reasonable given the constraint? + - Suggestions are adapted (shorter sessions, breaks, modifications) + - Suggests alternatives that work within the constraint + - Doesn't enthusiastically recommend things that directly contradict the constraint + +WHAT SHOULD FAIL: +- Completely ignores the constraint (no mention, no adaptation) +- Enthusiastically suggests things that would exacerbate the problem +- Makes recommendations that are clearly impractical given the constraint + +WHAT SHOULD PASS: +- Acknowledges the constraint and adapts recommendations +- Suggests workarounds or modifications +- Recommends alternatives that account for the limitation +- Provides practical advice that works within constraints + +Example 1 (PASS): +- Constraint: "broke my leg" - Question: "What sports should I do?" -- CORRECT: Suggests low-impact activities, acknowledges injury -- INCORRECT: Suggests basketball, running (ignores the injury) +- Good Response: "Given your leg injury, focus on upper body exercises and swimming once cleared by your doctor. Avoid high-impact activities." -SEMANTIC FLEXIBILITY: The response doesn't need to quote the implicit reasoning verbatim, -but must demonstrate awareness of the constraint through its suggestions and tone. +Example 2 (FAIL): +- Constraint: "broke my leg" +- Question: "What sports should I do?" +- Bad Response: "You should try running, basketball, and soccer to stay active!" + +Be reasonable: if the response shows awareness and adapts suggestions, it should pass. Respond with JSON: {{"passed": bool, "found_implicit": bool, "reasoning": "1-3 sentence explanation"}} """ @@ -401,21 +504,12 @@ Actual response: "{actual_response}" Evaluate whether the actual response correctly demonstrates implicit reasoning based on the expected answer.""" - response = await self.anthropic_client.messages.create( - model="claude-sonnet-4-5", - max_tokens=300, - temperature=0.0, - system=system_prompt, - messages=[{"role": "user", "content": user_prompt}], + judgment_text = await self._call_llm( + system_prompt, user_prompt, max_tokens=300 ) - if not response.content: - raise ValueError("Anthropic returned empty response") - - content_block = response.content[0] - judgment_text = getattr(content_block, "text", None) - if judgment_text is None: - raise ValueError("No text content in response") + if not judgment_text or judgment_text.startswith("ERROR:"): + raise ValueError(f"LLM call failed: {judgment_text}") # Extract JSON from markdown if present if "```json" in judgment_text: @@ -567,8 +661,19 @@ Evaluate whether the actual response correctly demonstrates implicit reasoning b + "Sessions are separate and would need merging logic." ) else: - # Use dialectic chat - actual_response = await user_peer.chat(question) + # Use dialectic chat to get Honcho's context + honcho_context_response = await user_peer.chat(question) + honcho_context = ( + honcho_context_response + if isinstance(honcho_context_response, str) + else "" + ) + + # Now use an assistant prompt with Honcho's context + actual_response = await self.call_assistant_with_context( + question=question, + honcho_context=honcho_context, + ) # Clean up workspace if requested if self.cleanup_workspace: @@ -578,10 +683,6 @@ Evaluate whether the actual response correctly demonstrates implicit reasoning b except Exception as e: print(f"Failed to delete workspace: {e}") - actual_response = ( - actual_response if isinstance(actual_response, str) else "" - ) - tokens_used = self._get_latest_tokens_used() token_efficiency = None @@ -877,9 +978,17 @@ async def main() -> int: ) parser.add_argument( - "--anthropic-api-key", + "--llm-api-key", type=str, - help="Anthropic API key for response judging (optional)", + help="LLM API key for assistant and judging (optional, defaults to env var)", + ) + + parser.add_argument( + "--llm-provider", + type=str, + choices=["anthropic", "openai"], + default="openai", + help="LLM provider to use for assistant and judging (default: openai)", ) parser.add_argument( @@ -923,7 +1032,8 @@ async def main() -> int: runner = ImplexConvRunner( base_api_port=args.base_api_port, pool_size=args.pool_size, - anthropic_api_key=args.anthropic_api_key, + llm_api_key=args.llm_api_key, + llm_provider=args.llm_provider, timeout_seconds=args.timeout, reasoning_type=args.reasoning_type, cleanup_workspace=args.cleanup_workspace, diff --git a/tests/bench/test_implex_conv.py b/tests/bench/test_implex_conv.py deleted file mode 100644 index e6182e80..00000000 --- a/tests/bench/test_implex_conv.py +++ /dev/null @@ -1,494 +0,0 @@ -""" -Unit tests for ImplexConv test runner. - -Tests the conversation parsing, data loading, and core functionality -of the ImplexConv benchmark script without requiring a running Honcho instance. -""" - -# pyright: reportPrivateUsage=false - -from pathlib import Path -from unittest.mock import AsyncMock, MagicMock - -import pytest - -from tests.bench.implex_conv import ImplexConvRunner - - -class TestConversationParsing: - """Test conversation text parsing into message format.""" - - def test_parse_simple_conversation(self): - """Test parsing a basic two-turn conversation.""" - runner = ImplexConvRunner() - - conv_text = """Speaker1: Hello there! -Assistant: Hi, how can I help you today? - -Speaker1: I need some advice. -Assistant: Of course, I'm here to help.""" - - messages = runner._parse_conversation(conv_text) - - assert len(messages) == 4 - assert messages[0]["role"] == "user" - assert messages[0]["content"] == "Hello there!" - assert messages[1]["role"] == "assistant" - assert messages[1]["content"] == "Hi, how can I help you today?" - assert messages[2]["role"] == "user" - assert messages[2]["content"] == "I need some advice." - assert messages[3]["role"] == "assistant" - assert messages[3]["content"] == "Of course, I'm here to help." - - def test_parse_conversation_with_multiline_messages(self): - """Test parsing messages that span multiple lines within a turn.""" - runner = ImplexConvRunner() - - conv_text = """Speaker1: I have a question about sports. -Do you have any recommendations? -Assistant: Sure! I'd be happy to help you with that. -Let me think about some good options. - -Speaker1: Thanks! -Assistant: You're welcome!""" - - messages = runner._parse_conversation(conv_text) - - assert len(messages) == 4 - assert messages[0]["role"] == "user" - assert "question about sports" in messages[0]["content"] - assert "recommendations" in messages[0]["content"] - assert messages[1]["role"] == "assistant" - assert "happy to help" in messages[1]["content"] - assert "good options" in messages[1]["content"] - - def test_parse_conversation_empty_lines(self): - """Test parsing handles empty lines correctly.""" - runner = ImplexConvRunner() - - conv_text = """ - -Speaker1: Hello -Assistant: Hi there - -Speaker1: Goodbye -Assistant: See you later""" - - messages = runner._parse_conversation(conv_text) - - assert len(messages) == 4 - assert all(msg["role"] in ["user", "assistant"] for msg in messages) - assert all(msg["content"].strip() for msg in messages) - - def test_parse_conversation_no_double_newline(self): - """Test parsing when turns are not separated by double newlines.""" - runner = ImplexConvRunner() - - # Real-world data sometimes has inconsistent formatting - conv_text = """Speaker1: First message -Assistant: First response -Speaker1: Second message -Assistant: Second response""" - - messages = runner._parse_conversation(conv_text) - - # Should still parse correctly based on role markers - assert len(messages) >= 2 - assert messages[0]["role"] == "user" - assert messages[1]["role"] == "assistant" - - def test_parse_conversation_empty_content_after_marker(self): - """Test that messages with empty content after role markers are skipped.""" - runner = ImplexConvRunner() - - conv_text = """Speaker1: -Assistant: Valid response - -Speaker1: Valid message -Assistant: """ - - messages = runner._parse_conversation(conv_text) - - # Should only have 2 messages (the ones with actual content) - assert len(messages) == 2 - assert messages[0]["content"] == "Valid response" - assert messages[1]["content"] == "Valid message" - - def test_parse_conversation_no_empty_strings(self): - """Test that all parsed messages have non-empty content.""" - runner = ImplexConvRunner() - - conv_text = """Speaker1: First message -Assistant: Second message - -Speaker1: -Assistant: Third message""" - - messages = runner._parse_conversation(conv_text) - - # All messages should have non-empty content - assert len(messages) == 3 - assert all(len(msg["content"]) > 0 for msg in messages) - assert all(msg["content"].strip() for msg in messages) - - -class TestDataLoading: - """Test loading and validating ImplexConv data files.""" - - def test_load_opposed_data(self): - """Test loading the opposed reasoning dataset.""" - runner = ImplexConvRunner(reasoning_type="opposed") - test_file = Path("tests/bench/implex_conv_data/ImplexConv_opposed.json") - - if not test_file.exists(): - pytest.skip("ImplexConv_opposed.json not found") - - examples = runner.load_test_file(test_file) - - assert len(examples) > 0 - assert isinstance(examples[0], dict) - assert "conversation" in examples[0] - assert "qa" in examples[0] - - def test_load_supportive_data(self): - """Test loading the supportive reasoning dataset.""" - runner = ImplexConvRunner(reasoning_type="supportive") - test_file = Path("tests/bench/implex_conv_data/ImplexConv_supportive.json") - - if not test_file.exists(): - pytest.skip("ImplexConv_supportive.json not found") - - examples = runner.load_test_file(test_file) - - assert len(examples) > 0 - assert isinstance(examples[0], dict) - assert "conversation" in examples[0] - assert "qa" in examples[0] - - def test_validate_question_structure(self): - """Test that questions have the required fields.""" - runner = ImplexConvRunner(reasoning_type="opposed") - test_file = Path("tests/bench/implex_conv_data/ImplexConv_opposed.json") - - if not test_file.exists(): - pytest.skip("ImplexConv_opposed.json not found") - - examples = runner.load_test_file(test_file) - first_example = examples[0] - - # Check QA structure - assert len(first_example["qa"]) > 0 - question_data = first_example["qa"][0] - - assert "question" in question_data - assert "answer" in question_data - assert "retrieved_conv_ids" in question_data - assert isinstance(question_data["retrieved_conv_ids"], list) - - def test_validate_conversation_structure(self): - """Test that conversations have the expected structure.""" - runner = ImplexConvRunner(reasoning_type="opposed") - test_file = Path("tests/bench/implex_conv_data/ImplexConv_opposed.json") - - if not test_file.exists(): - pytest.skip("ImplexConv_opposed.json not found") - - examples = runner.load_test_file(test_file) - first_example = examples[0] - - # Check conversation structure - conversations = first_example["conversation"] - assert isinstance(conversations, dict) - assert len(conversations) > 0 - - # Each conversation should be a string - for conv_id, conv_text in conversations.items(): - assert isinstance(conv_id, str) - assert isinstance(conv_text, str) - assert len(conv_text) > 0 - - -class TestTokenCalculation: - """Test token counting for efficiency metrics.""" - - def test_calculate_total_tokens(self): - """Test token calculation from conversations.""" - runner = ImplexConvRunner() - - conversations = { - "0": "Speaker1: Hello\nAssistant: Hi there", - "1": "Speaker1: How are you?\nAssistant: I'm good!", - } - - total_tokens = runner._calculate_total_tokens(conversations) - - assert total_tokens > 0 - assert isinstance(total_tokens, int) - - def test_calculate_tokens_empty_conversation(self): - """Test token calculation with empty conversations.""" - runner = ImplexConvRunner() - - conversations = {"0": ""} - - total_tokens = runner._calculate_total_tokens(conversations) - - assert total_tokens == 0 - - -class TestFormatting: - """Test output formatting utilities.""" - - def test_format_duration_seconds(self): - """Test duration formatting for seconds.""" - runner = ImplexConvRunner() - - assert runner._format_duration(45.67) == "45.67s" - assert runner._format_duration(1.23) == "1.23s" - - def test_format_duration_minutes(self): - """Test duration formatting for minutes.""" - runner = ImplexConvRunner() - - assert runner._format_duration(125.0) == "2m05s" - assert runner._format_duration(60.0) == "1m00s" - assert runner._format_duration(119.5) == "2m00s" # Rounds up - - def test_format_duration_edge_cases(self): - """Test duration formatting edge cases.""" - runner = ImplexConvRunner() - - assert runner._format_duration(0) == "0.00s" - assert runner._format_duration(59.99) == "59.99s" - assert runner._format_duration(60.01) == "1m00s" - - -class TestHonchoURLGeneration: - """Test Honcho instance URL generation for load balancing.""" - - def test_get_honcho_url_single_instance(self): - """Test URL generation with a single instance.""" - runner = ImplexConvRunner(base_api_port=8000, pool_size=1) - - # All examples should use the same instance - assert runner.get_honcho_url_for_index(0) == "http://localhost:8000" - assert runner.get_honcho_url_for_index(1) == "http://localhost:8000" - assert runner.get_honcho_url_for_index(10) == "http://localhost:8000" - - def test_get_honcho_url_multiple_instances(self): - """Test URL generation with multiple instances (round-robin).""" - runner = ImplexConvRunner(base_api_port=8000, pool_size=3) - - # Should distribute across 3 instances - assert runner.get_honcho_url_for_index(0) == "http://localhost:8000" - assert runner.get_honcho_url_for_index(1) == "http://localhost:8001" - assert runner.get_honcho_url_for_index(2) == "http://localhost:8002" - assert runner.get_honcho_url_for_index(3) == "http://localhost:8000" # Wraps - assert runner.get_honcho_url_for_index(4) == "http://localhost:8001" - - def test_get_honcho_url_custom_port(self): - """Test URL generation with custom base port.""" - runner = ImplexConvRunner(base_api_port=9000, pool_size=2) - - assert runner.get_honcho_url_for_index(0) == "http://localhost:9000" - assert runner.get_honcho_url_for_index(1) == "http://localhost:9001" - - -@pytest.mark.asyncio -class TestJudgmentLogic: - """Test LLM-based judgment of responses.""" - - async def test_judge_opposed_reasoning_pass(self): - """Test judgment for opposed reasoning that should pass.""" - runner = ImplexConvRunner(reasoning_type="opposed") - - # Mock the Anthropic client - mock_response = MagicMock() - mock_content = MagicMock() - mock_content.text = '{"passed": true, "found_implicit": true, "reasoning": "Response correctly identifies the constraint"}' - mock_response.content = [mock_content] - - runner.anthropic_client.messages.create = AsyncMock(return_value=mock_response) - - judgment = await runner.judge_implicit_reasoning( - question="What sports should I do?", - expected_answer="Low-impact activities", - actual_response="Given your injury, I'd suggest swimming or yoga", - implicit_reasoning="broke my leg", - ) - - assert judgment["passed"] is True - assert judgment["found_implicit"] is True - assert "reasoning" in judgment - - async def test_judge_opposed_reasoning_fail(self): - """Test judgment for opposed reasoning that should fail.""" - runner = ImplexConvRunner(reasoning_type="opposed") - - # Mock the Anthropic client - mock_response = MagicMock() - mock_content = MagicMock() - mock_content.text = '{"passed": false, "found_implicit": false, "reasoning": "Response suggests activities that ignore the constraint"}' - mock_response.content = [mock_content] - - runner.anthropic_client.messages.create = AsyncMock(return_value=mock_response) - - judgment = await runner.judge_implicit_reasoning( - question="What sports should I do?", - expected_answer="Low-impact activities", - actual_response="You should try basketball and running!", - implicit_reasoning="broke my leg", - ) - - assert judgment["passed"] is False - assert judgment["found_implicit"] is False - - async def test_judge_supportive_reasoning(self): - """Test judgment for supportive reasoning.""" - runner = ImplexConvRunner(reasoning_type="supportive") - - # Mock the Anthropic client - mock_response = MagicMock() - mock_content = MagicMock() - mock_content.text = '{"passed": true, "found_implicit": true, "reasoning": "Response confirms the trait based on evidence"}' - mock_response.content = [mock_content] - - runner.anthropic_client.messages.create = AsyncMock(return_value=mock_response) - - judgment = await runner.judge_implicit_reasoning( - question="Do I post facts daily?", - expected_answer="Yes", - actual_response="Yes, you've been consistently posting daily facts", - implicit_reasoning=None, - ) - - assert judgment["passed"] is True - assert "reasoning" in judgment - - async def test_judge_error_fallback(self): - """Test judgment fallback on API error.""" - runner = ImplexConvRunner(reasoning_type="opposed") - - # Mock an API error - runner.anthropic_client.messages.create = AsyncMock( - side_effect=Exception("API Error") - ) - - judgment = await runner.judge_implicit_reasoning( - question="Test question", - expected_answer="test answer", - actual_response="test answer is here", - implicit_reasoning=None, - ) - - # Should fall back to string matching - assert "passed" in judgment - assert "found_implicit" in judgment - assert judgment["found_implicit"] is False - assert "Fallback string matching" in judgment["reasoning"] - - -class TestConversationToMessages: - """Test conversion of conversation text to Honcho message format.""" - - def test_message_role_mapping(self): - """Test that Speaker1 maps to user and Assistant maps to assistant.""" - runner = ImplexConvRunner() - - conv_text = """Speaker1: User message -Assistant: Assistant response""" - - messages = runner._parse_conversation(conv_text) - - assert messages[0]["role"] == "user" - assert messages[1]["role"] == "assistant" - - def test_message_content_preservation(self): - """Test that message content is preserved correctly.""" - runner = ImplexConvRunner() - - test_content = "This is a test message with special chars: !@#$%" - conv_text = f"""Speaker1: {test_content} -Assistant: Got it!""" - - messages = runner._parse_conversation(conv_text) - - assert messages[0]["content"] == test_content - - def test_real_world_conversation_parsing(self): - """Test parsing a real conversation from the dataset.""" - runner = ImplexConvRunner(reasoning_type="opposed") - test_file = Path("tests/bench/implex_conv_data/ImplexConv_opposed.json") - - if not test_file.exists(): - pytest.skip("ImplexConv_opposed.json not found") - - examples = runner.load_test_file(test_file) - first_conversation = examples[0]["conversation"]["0"] - - messages = runner._parse_conversation(first_conversation) - - # Should have parsed successfully - assert len(messages) > 0 - # Should alternate between user and assistant (in most cases) - assert all(msg["role"] in ["user", "assistant"] for msg in messages) - # Should have non-empty content - assert all(len(msg["content"]) > 0 for msg in messages) - - -class TestWorkspaceNaming: - """Test workspace ID generation.""" - - def test_workspace_id_format(self): - """Test that workspace IDs follow the expected format.""" - example_id = "ex0" - question_index = 1 - reasoning_type = "opposed" - - expected_workspace = f"{example_id}_q{question_index}_{reasoning_type}" - assert expected_workspace == "ex0_q1_opposed" - - def test_workspace_id_uniqueness(self): - """Test that different questions get different workspace IDs.""" - reasoning_type = "opposed" - - workspace_ids: set[str] = set() - for ex_id in range(3): - for q_idx in range(2): - workspace_id = f"ex{ex_id}_q{q_idx}_{reasoning_type}" - workspace_ids.add(workspace_id) - - # Should have 6 unique workspace IDs - assert len(workspace_ids) == 6 - - -class TestMetricsCollection: - """Test metrics collection initialization.""" - - def test_metrics_collector_initialized(self): - """Test that metrics collector is initialized.""" - runner = ImplexConvRunner(reasoning_type="opposed") - - assert runner.metrics_collector is not None - - def test_runner_configuration(self): - """Test runner configuration options.""" - runner = ImplexConvRunner( - base_api_port=9000, - pool_size=5, - timeout_seconds=300, - reasoning_type="supportive", - cleanup_workspace=True, - use_get_context=False, - ) - - assert runner.base_api_port == 9000 - assert runner.pool_size == 5 - assert runner.timeout_seconds == 300 - assert runner.reasoning_type == "supportive" - assert runner.cleanup_workspace is True - assert runner.use_get_context is False - - -if __name__ == "__main__": - pytest.main([__file__, "-v"])