diff --git a/agentic_fde.md b/agentic_fde.md index 1cb93ef4..5b70773e 100644 --- a/agentic_fde.md +++ b/agentic_fde.md @@ -103,7 +103,9 @@ The same primitives (workspaces, peers, sessions, messages, documents) can achie | 4 | ~350 | 514 lines (23 tests) | | **Total** | **~2,950** | **1,559 lines (61 tests)** | -## Testing Checklist +## Testing + +### Unit Tests - [ ] Run full test suite: `uv run pytest tests/` - [ ] Test Phase 1: Create a dialectic query, verify trace is logged @@ -112,3 +114,35 @@ The same primitives (workspaces, peers, sessions, messages, documents) can achie - [ ] Test Phase 4: Submit feedback, verify config updated - [ ] Test interview flow: New workspace + greeting triggers questions - [ ] Test incremental updates: Existing rules preserved when adding new ones + +### Unified End-to-End Tests + +New actions added to the unified test system (`tests/unified/`): + +| Action | Description | +|--------|-------------| +| `set_agent_config` | Set custom deriver_rules and/or dialectic_rules | +| `submit_feedback` | Submit natural language feedback to configure Honcho | +| `trigger_introspection` | Trigger meta-cognitive introspection dream | +| `query_introspection` | Query the latest introspection report | + +Test cases for agentic FDE: + +- `agentic_fde_custom_deriver_rules.json` - Verifies custom deriver rules filter observation extraction +- `agentic_fde_custom_dialectic_rules.json` - Verifies custom dialectic rules change response format +- `agentic_fde_feedback_updates_config.json` - Verifies feedback endpoint updates configuration + +Run unified tests: + +```bash +python -m tests.unified.run --test-dir tests/unified/test_cases +``` + +### API Endpoints + +New endpoints added: + +| Endpoint | Method | Description | +|----------|--------|-------------| +| `/workspaces/{id}/feedback` | POST | Developer feedback channel | +| `/workspaces/{id}/introspection` | GET | Get latest introspection report | diff --git a/src/feedback.py b/src/feedback.py index 69d66ebd..260a3d68 100644 --- a/src/feedback.py +++ b/src/feedback.py @@ -7,11 +7,11 @@ workspace agent settings through conversation. from __future__ import annotations -import json import logging import re -from typing import Any, cast +from typing import Literal +from pydantic import BaseModel, Field from sqlalchemy.ext.asyncio import AsyncSession from src import crud @@ -25,6 +25,24 @@ from src.schemas import ( ) from src.utils.clients import honcho_llm_call + +class FeedbackChange(BaseModel): + """A single configuration change from the feedback LLM.""" + + field: Literal["deriver_rules", "dialectic_rules"] + new_value: str + + +class FeedbackLLMResponse(BaseModel): + """Structured response from the feedback LLM.""" + + message: str = Field(description="Response message to the developer") + understood_intent: str = Field(description="Brief description of understood intent") + changes: list[FeedbackChange] = Field( + default_factory=list, description="Configuration changes to apply" + ) + + logger = logging.getLogger(__name__) @@ -212,54 +230,33 @@ async def process_feedback( prompt=prompt, max_tokens=4096, track_name="feedback_channel", - json_mode=True, + response_model=FeedbackLLMResponse, temperature=0.3, ) - # Parse response - response_text = llm_response.content - try: - response_data: dict[str, object] = json.loads(response_text) - except json.JSONDecodeError as e: - logger.error(f"Failed to parse feedback LLM response: {e}") - return FeedbackResponse( - message="I had trouble processing your request. Could you try rephrasing?", - understood_intent="Error parsing response", - changes_made=[], - current_config=current_config, + # Process changes from structured response + changes_made: list[ConfigChange] = [] + parsed_response = llm_response.content + + for change in parsed_response.changes: + # Get previous value + previous_value = ( + current_config.deriver_rules + if change.field == "deriver_rules" + else current_config.dialectic_rules ) - # Process changes - changes_made: list[ConfigChange] = [] - raw_changes = response_data.get("changes", []) + # Skip if no actual change + if previous_value == change.new_value: + continue - if isinstance(raw_changes, list): - for change_item in cast(list[dict[str, Any]], raw_changes): - change_dict: dict[str, object] = change_item - field = str(change_dict.get("field", "")) - new_value = str(change_dict.get("new_value", "")) - - if field not in ("deriver_rules", "dialectic_rules"): - continue - - # Get previous value - previous_value = ( - current_config.deriver_rules - if field == "deriver_rules" - else current_config.dialectic_rules - ) - - # Skip if no actual change - if previous_value == new_value: - continue - - changes_made.append( - ConfigChange( - field=field, # type: ignore[arg-type] - previous_value=previous_value, - new_value=new_value, - ) + changes_made.append( + ConfigChange( + field=change.field, + previous_value=previous_value, + new_value=change.new_value, ) + ) # Apply changes if any if changes_made: @@ -281,12 +278,9 @@ async def process_feedback( f"Feedback channel: Applied {len(changes_made)} changes to workspace {workspace_name}" ) - message = response_data.get("message", "Configuration updated.") - understood_intent = response_data.get("understood_intent", "Processed feedback") - return FeedbackResponse( - message=str(message), - understood_intent=str(understood_intent), + message=parsed_response.message, + understood_intent=parsed_response.understood_intent, changes_made=changes_made, current_config=current_config, ) diff --git a/src/routers/workspaces.py b/src/routers/workspaces.py index ae730287..5f335c5a 100644 --- a/src/routers/workspaces.py +++ b/src/routers/workspaces.py @@ -290,3 +290,29 @@ async def process_developer_feedback( introspection_report = await get_latest_introspection_report(db, workspace_id) return await process_feedback(db, workspace_id, request, introspection_report) + + +@router.get( + "/{workspace_id}/introspection", + response_model=schemas.IntrospectionReport | None, + dependencies=[Depends(require_auth(workspace_name="workspace_id"))], +) +async def get_introspection_report( + workspace_id: str = Path(...), + db: AsyncSession = db, +): + """ + Get the latest introspection report for a workspace. + + Introspection reports are generated by the meta-cognitive dreamer when + an introspection dream is triggered. They contain: + - Performance signals (query counts, abstention rates, etc.) + - Sample queries and observations + - Suggestions for configuration improvements + + Returns None if no introspection report exists yet. + """ + report = await get_latest_introspection_report(db, workspace_id) + if report is None: + raise HTTPException(status_code=404, detail="No introspection report found") + return report diff --git a/tests/test_feedback.py b/tests/test_feedback.py index 0ae3db2d..3f89534c 100644 --- a/tests/test_feedback.py +++ b/tests/test_feedback.py @@ -1,6 +1,5 @@ """Tests for the developer feedback channel (Phase 4 of Agentic FDE).""" -import json from unittest.mock import AsyncMock, MagicMock, patch import pytest @@ -10,6 +9,8 @@ from sqlalchemy.ext.asyncio import AsyncSession from src import crud, models from src.feedback import ( INTERVIEW_QUESTIONS, + FeedbackChange, + FeedbackLLMResponse, build_feedback_prompt, config_is_empty, is_simple_greeting, @@ -240,14 +241,12 @@ class TestProcessFeedback: config = WorkspaceAgentConfig(deriver_rules="Existing rule") await crud.set_workspace_agent_config(db_session, workspace.name, config) - # Mock the LLM call + # Mock the LLM call (structured output) mock_response = MagicMock() - mock_response.content = json.dumps( - { - "message": "I see you already have rules set up.", - "understood_intent": "Greeting with existing config", - "changes": [], - } + mock_response.content = FeedbackLLMResponse( + message="I see you already have rules set up.", + understood_intent="Greeting with existing config", + changes=[], ) with patch( @@ -270,19 +269,17 @@ class TestProcessFeedback: """Test that config changes are applied correctly.""" workspace, _ = sample_data - # Mock the LLM call to return a config change + # Mock the LLM call to return a config change (structured output) mock_response = MagicMock() - mock_response.content = json.dumps( - { - "message": "I've configured the workspace to focus on emotions.", - "understood_intent": "Configure deriver for emotion tracking", - "changes": [ - { - "field": "deriver_rules", - "new_value": "Focus on emotional content and feelings", - } - ], - } + mock_response.content = FeedbackLLMResponse( + message="I've configured the workspace to focus on emotions.", + understood_intent="Configure deriver for emotion tracking", + changes=[ + FeedbackChange( + field="deriver_rules", + new_value="Focus on emotional content and feelings", + ) + ], ) with patch( @@ -318,14 +315,12 @@ class TestProcessFeedback: config = WorkspaceAgentConfig(deriver_rules="Existing rule") await crud.set_workspace_agent_config(db_session, workspace.name, config) - # Mock LLM to return answer without changes + # Mock LLM to return answer without changes (structured output) mock_response = MagicMock() - mock_response.content = json.dumps( - { - "message": "Your current deriver rule is: 'Existing rule'", - "understood_intent": "Question about current config", - "changes": [], - } + mock_response.content = FeedbackLLMResponse( + message="Your current deriver rule is: 'Existing rule'", + understood_intent="Question about current config", + changes=[], ) with patch( @@ -368,20 +363,21 @@ class TestProcessFeedback: assert response.current_config.deriver_rules == "Existing rule" @pytest.mark.asyncio - async def test_invalid_json_response( + async def test_invalid_response_type( self, db_session: AsyncSession, sample_data: tuple[models.Workspace, models.Peer], ): - """Test handling of invalid JSON from LLM.""" + """Test handling of unexpected response type from LLM.""" workspace, _ = sample_data # Set initial config config = WorkspaceAgentConfig(deriver_rules="Existing rule") await crud.set_workspace_agent_config(db_session, workspace.name, config) + # Mock returns something unexpected (string instead of FeedbackLLMResponse) mock_response = MagicMock() - mock_response.content = "This is not valid JSON" + mock_response.content = "This is not a FeedbackLLMResponse" with patch( "src.feedback.honcho_llm_call", @@ -391,8 +387,8 @@ class TestProcessFeedback: request = FeedbackRequest(message="Update my config") response = await process_feedback(db_session, workspace.name, request) - # Should return error message - assert "trouble" in response.message.lower() + # Should return error message (caught by generic exception handler) + assert "error" in response.message.lower() assert len(response.changes_made) == 0 @pytest.mark.asyncio @@ -408,19 +404,17 @@ class TestProcessFeedback: config = WorkspaceAgentConfig(deriver_rules="Track emotions") await crud.set_workspace_agent_config(db_session, workspace.name, config) - # Mock LLM to append new rule + # Mock LLM to append new rule (structured output) mock_response = MagicMock() - mock_response.content = json.dumps( - { - "message": "Added goal tracking to existing rules.", - "understood_intent": "Add goal tracking while preserving emotion tracking", - "changes": [ - { - "field": "deriver_rules", - "new_value": "Track emotions\nAlso track goals and aspirations", - } - ], - } + mock_response.content = FeedbackLLMResponse( + message="Added goal tracking to existing rules.", + understood_intent="Add goal tracking while preserving emotion tracking", + changes=[ + FeedbackChange( + field="deriver_rules", + new_value="Track emotions\nAlso track goals and aspirations", + ) + ], ) with patch( diff --git a/tests/unified/README.md b/tests/unified/README.md index dd40e038..21124f28 100644 --- a/tests/unified/README.md +++ b/tests/unified/README.md @@ -47,6 +47,20 @@ Tests are defined in JSON files. A test definition consists of a name, optional * `query`: Perform an action and assert on the result. * `target`: "chat", "get_context", "get_peer_card", "get_representation" +5. **Agentic FDE** (Self-adapting Honcho): + * `set_agent_config`: Set custom rules for deriver and/or dialectic prompts. + * `deriver_rules`: Custom rules injected into observation extraction. + * `dialectic_rules`: Custom rules injected into query responses. + * `submit_feedback`: Submit natural language feedback to configure Honcho. + * `message`: The feedback message. + * `include_introspection`: Include latest introspection report (default: true). + * `assertions`: Optional assertions on the feedback response. + * `trigger_introspection`: Trigger a meta-cognitive introspection dream. + * `wait_for_completion`: Wait for introspection to finish (default: true). + * `timeout`: Timeout in seconds (default: 120). + * `query_introspection`: Query the latest introspection report. + * `assertions`: Assertions to run on the report. + ### Assertions * `llm_judge`: Use Claude to evaluate the result against a natural language prompt. diff --git a/tests/unified/runner.py b/tests/unified/runner.py index 104f804c..446cd6cc 100644 --- a/tests/unified/runner.py +++ b/tests/unified/runner.py @@ -44,10 +44,14 @@ from tests.unified.schema import ( LLMJudgeAssertion, NotContainsAssertion, QueryAction, + QueryIntrospectionAction, ScheduleDreamAction, + SetAgentConfigAction, SetSessionConfigAction, SetWorkspaceConfigAction, + SubmitFeedbackAction, TestDefinition, + TriggerIntrospectionAction, WaitAction, ) @@ -183,10 +187,12 @@ class UnifiedTestExecutor: honcho_client: Honcho, anthropic_client: AsyncAnthropic | None, redis_url: str, + base_url: str, ): self.client: Honcho = honcho_client self.anthropic: AsyncAnthropic | None = anthropic_client self.redis_url: str = redis_url + self.base_url: str = base_url async def execute(self, test_def: TestDefinition, test_name: str) -> bool: logger.info(f"Starting test: {test_name}") @@ -304,6 +310,24 @@ class UnifiedTestExecutor: for assertion in step.assertions: await self.check_assertion(result, assertion) + # --- Agentic FDE Actions --- + + elif isinstance(step, SetAgentConfigAction): + await self.set_agent_config(step) + + elif isinstance(step, SubmitFeedbackAction): + result = await self.submit_feedback(step) + for assertion in step.assertions: + await self.check_assertion(result, assertion) + + elif isinstance(step, TriggerIntrospectionAction): + await self.trigger_introspection(step) + + elif isinstance(step, QueryIntrospectionAction): + result = await self.query_introspection() + for assertion in step.assertions: + await self.check_assertion(result, assertion) + async def flush_deriver_queue(self): """Enable deriver flush mode to bypass batch token threshold.""" # Use direct Redis connection to set the flush key @@ -489,6 +513,90 @@ class UnifiedTestExecutor: f"Value mismatch for '{k}': expected {v}, got {result_dict[k]}" ) + # --- Agentic FDE Methods --- + + async def set_agent_config(self, step: SetAgentConfigAction) -> None: + """Set workspace agent configuration via HTTP API.""" + workspace_name = self.client.workspace_id + + # Build agent config + agent_config: dict[str, str] = {} + if step.deriver_rules is not None: + agent_config["deriver_rules"] = step.deriver_rules + if step.dialectic_rules is not None: + agent_config["dialectic_rules"] = step.dialectic_rules + + async with httpx.AsyncClient() as http_client: + # Use PUT to update workspace with _agent_config in metadata + response = await http_client.put( + f"{self.base_url}/v3/workspaces/{workspace_name}", + json={"metadata": {"_agent_config": agent_config}}, + ) + response.raise_for_status() + + logger.info(f"Set agent config: {agent_config}") + + async def submit_feedback(self, step: SubmitFeedbackAction) -> dict[str, Any]: + """Submit feedback to the workspace feedback endpoint.""" + workspace_name = self.client.workspace_id + + async with httpx.AsyncClient(timeout=120.0) as http_client: + response = await http_client.post( + f"{self.base_url}/v3/workspaces/{workspace_name}/feedback", + json={ + "message": step.message, + "include_introspection": step.include_introspection, + }, + ) + response.raise_for_status() + result = response.json() + + logger.info(f"Feedback response: {result.get('message', '')[:100]}...") + return result + + async def trigger_introspection(self, step: TriggerIntrospectionAction) -> None: + """Trigger an introspection dream.""" + workspace_name = self.client.workspace_id + + async with httpx.AsyncClient() as http_client: + # Introspection ignores observer/observed but endpoint requires them + response = await http_client.post( + f"{self.base_url}/v3/workspaces/{workspace_name}/schedule_dream", + json={ + "observer": "_system", + "observed": "_introspection", + "dream_type": "introspection", + }, + ) + response.raise_for_status() + + logger.info("Triggered introspection dream") + + if step.wait_for_completion: + # Wait for dream to complete - introspection goes through dream queue + await asyncio.sleep(2) # Give it time to enqueue + # For now, just wait a fixed time since dream queue is separate + await asyncio.sleep(step.timeout) + logger.info("Introspection wait period complete") + + async def query_introspection(self) -> dict[str, Any]: + """Query the latest introspection report.""" + workspace_name = self.client.workspace_id + + async with httpx.AsyncClient() as http_client: + response = await http_client.get( + f"{self.base_url}/v3/workspaces/{workspace_name}/introspection" + ) + if response.status_code == 404: + return {"error": "No introspection report found"} + response.raise_for_status() + result = response.json() + + logger.info( + f"Retrieved introspection report from {result.get('generated_at', 'unknown')}" + ) + return result + class UnifiedTestRunner: def __init__( @@ -577,13 +685,14 @@ class UnifiedTestRunner: logger.info(f"Found {len(test_files)} test(s)") # 3. Execute Tests + base_url = f"http://localhost:{self.harness.api_port}" client = Honcho( - base_url=f"http://localhost:{self.harness.api_port}", + base_url=base_url, workspace_id="default", # Will be overridden per test ) redis_url = f"redis://localhost:{self.harness.redis_port}/0" - executor = UnifiedTestExecutor(client, self.anthropic, redis_url) + executor = UnifiedTestExecutor(client, self.anthropic, redis_url, base_url) suite_start_time = time.time() diff --git a/tests/unified/schema.py b/tests/unified/schema.py index aa4c78ad..6e1f609d 100644 --- a/tests/unified/schema.py +++ b/tests/unified/schema.py @@ -92,6 +92,53 @@ class ScheduleDreamAction(TestStep): dream_type: DreamType = Field(..., description="Type of dream to schedule") +# --- Agentic FDE Actions --- + + +class SetAgentConfigAction(TestStep): + """Set workspace agent configuration for prompt customization.""" + + step_type: Literal["set_agent_config"] = "set_agent_config" + deriver_rules: str | None = Field( + None, description="Custom rules to inject into the deriver prompt" + ) + dialectic_rules: str | None = Field( + None, description="Custom rules to inject into the dialectic prompt" + ) + + +class SubmitFeedbackAction(TestStep): + """Submit natural language feedback to configure Honcho.""" + + step_type: Literal["submit_feedback"] = "submit_feedback" + message: str = Field(..., description="Natural language feedback message") + include_introspection: bool = Field( + True, description="Include latest introspection report in context" + ) + assertions: list["AssertionType"] = Field( + default_factory=list, description="Assertions to run on the feedback response" + ) + + +class TriggerIntrospectionAction(TestStep): + """Trigger an introspection dream to analyze workspace usage.""" + + step_type: Literal["trigger_introspection"] = "trigger_introspection" + wait_for_completion: bool = Field( + True, description="Wait for introspection to complete" + ) + timeout: int = Field(120, description="Timeout in seconds when waiting") + + +class QueryIntrospectionAction(TestStep): + """Query the latest introspection report and run assertions.""" + + step_type: Literal["query_introspection"] = "query_introspection" + assertions: list["AssertionType"] = Field( + default_factory=list, description="Assertions to run on the report" + ) + + # --- Assertions --- @@ -128,6 +175,17 @@ class JsonMatchAssertion(Assertion): key_value_pairs: dict[str, Any] | None = None +# --- Assertion Type Alias --- + +AssertionType = ( + LLMJudgeAssertion + | ContainsAssertion + | NotContainsAssertion + | ExactMatchAssertion + | JsonMatchAssertion +) + + # --- Query/Assertion Actions --- @@ -149,13 +207,7 @@ class QueryAction(TestStep): # for chat - reasoning level reasoning_level: ReasoningLevel | None = None - assertions: list[ - LLMJudgeAssertion - | ContainsAssertion - | NotContainsAssertion - | ExactMatchAssertion - | JsonMatchAssertion - ] + assertions: list[AssertionType] # --- Unified Step Type --- @@ -173,7 +225,12 @@ class TestDefinition(BaseModel): | AddMessagesAction | WaitAction | ScheduleDreamAction - | QueryAction, + | QueryAction + # Agentic FDE actions + | SetAgentConfigAction + | SubmitFeedbackAction + | TriggerIntrospectionAction + | QueryIntrospectionAction, Field(discriminator="step_type"), ] ] diff --git a/tests/unified/test_cases/agentic_fde_custom_deriver_rules.json b/tests/unified/test_cases/agentic_fde_custom_deriver_rules.json new file mode 100644 index 00000000..6756776e --- /dev/null +++ b/tests/unified/test_cases/agentic_fde_custom_deriver_rules.json @@ -0,0 +1,92 @@ +{ + "description": "Test that custom deriver rules modify observation extraction behavior", + "steps": [ + { + "step_type": "create_session", + "session_id": "deriver_rules_test", + "peer_configs": { + "user": { + "observe_me": true, + "observe_others": false + }, + "assistant": { + "observe_me": false, + "observe_others": true + } + } + }, + { + "step_type": "set_agent_config", + "description": "Configure deriver to ONLY extract food-related observations", + "deriver_rules": "ONLY extract observations about food preferences, dietary restrictions, and cooking. Completely ignore all other topics like work, location, hobbies, and personal information." + }, + { + "step_type": "add_messages", + "session_id": "deriver_rules_test", + "messages": [ + { + "peer_id": "user", + "content": "I love pizza and hate sushi. Raw fish just isn't for me." + }, + { + "peer_id": "assistant", + "content": "That's interesting! Pizza is definitely a popular choice. Any particular toppings you prefer?" + }, + { + "peer_id": "user", + "content": "I work as a software engineer in Seattle. Been there for 5 years now." + }, + { + "peer_id": "assistant", + "content": "Seattle is a great tech hub! The food scene there is amazing too." + }, + { + "peer_id": "user", + "content": "Yeah, I'm vegetarian so I appreciate all the plant-based options." + } + ] + }, + { + "step_type": "wait", + "target": "queue_empty", + "timeout": 120, + "flush": true + }, + { + "step_type": "query", + "description": "Verify food observations were extracted", + "target": "get_representation", + "observer_peer_id": "assistant", + "observed_peer_id": "user", + "session_id": "deriver_rules_test", + "assertions": [ + { + "assertion_type": "llm_judge", + "prompt": "Does this representation mention food-related facts like pizza, sushi, or vegetarian? It should contain at least one food-related observation." + } + ] + }, + { + "step_type": "query", + "description": "Verify non-food observations were NOT extracted", + "target": "get_representation", + "observer_peer_id": "assistant", + "observed_peer_id": "user", + "session_id": "deriver_rules_test", + "assertions": [ + { + "assertion_type": "not_contains", + "text": "software engineer" + }, + { + "assertion_type": "not_contains", + "text": "Seattle" + }, + { + "assertion_type": "not_contains", + "text": "5 years" + } + ] + } + ] +} diff --git a/tests/unified/test_cases/agentic_fde_custom_dialectic_rules.json b/tests/unified/test_cases/agentic_fde_custom_dialectic_rules.json new file mode 100644 index 00000000..e84ed0e2 --- /dev/null +++ b/tests/unified/test_cases/agentic_fde_custom_dialectic_rules.json @@ -0,0 +1,67 @@ +{ + "description": "Test that custom dialectic rules modify chat response behavior", + "steps": [ + { + "step_type": "create_session", + "session_id": "dialectic_rules_test", + "peer_configs": { + "user": { + "observe_me": true, + "observe_others": false + }, + "assistant": { + "observe_me": false, + "observe_others": true + } + } + }, + { + "step_type": "add_messages", + "session_id": "dialectic_rules_test", + "messages": [ + { + "peer_id": "user", + "content": "I started a new job at Google last week as a senior engineer." + }, + { + "peer_id": "assistant", + "content": "Congratulations on the new role! That's exciting." + }, + { + "peer_id": "user", + "content": "Thanks! I'm working on their cloud infrastructure team." + } + ] + }, + { + "step_type": "wait", + "target": "queue_empty", + "timeout": 120, + "flush": true + }, + { + "step_type": "set_agent_config", + "description": "Configure dialectic to respond in bullet points only", + "dialectic_rules": "CRITICAL: Always format your response as exactly 3 bullet points using '•' characters. Never use prose paragraphs. Be extremely concise - each bullet should be under 10 words." + }, + { + "step_type": "query", + "description": "Test that response follows bullet point format", + "target": "chat", + "session_id": "dialectic_rules_test", + "observer_peer_id": "assistant", + "observed_peer_id": "user", + "input": "What do you know about this user's job?", + "assertions": [ + { + "assertion_type": "llm_judge", + "prompt": "Does this response mention the user's job at Google or on a cloud infrastructure team? It should reference their employment." + }, + { + "assertion_type": "llm_judge", + "prompt": "Is this response formatted primarily as bullet points (using • or - or * characters) rather than flowing prose paragraphs? The response should have a list-like structure." + } + ] + } + ] +} diff --git a/tests/unified/test_cases/agentic_fde_feedback_updates_config.json b/tests/unified/test_cases/agentic_fde_feedback_updates_config.json new file mode 100644 index 00000000..ce888da3 --- /dev/null +++ b/tests/unified/test_cases/agentic_fde_feedback_updates_config.json @@ -0,0 +1,95 @@ +{ + "description": "Test that the feedback endpoint updates workspace agent configuration", + "steps": [ + { + "step_type": "create_session", + "session_id": "feedback_test", + "peer_configs": { + "user": { + "observe_me": true, + "observe_others": false + }, + "assistant": { + "observe_me": false, + "observe_others": true + } + } + }, + { + "step_type": "submit_feedback", + "description": "Submit feedback to configure for a cooking app", + "message": "I'm building a cooking and recipe app. Please configure Honcho to focus on extracting food preferences, dietary restrictions, allergies, and favorite cuisines. Ignore any non-food related information.", + "include_introspection": false, + "assertions": [ + { + "assertion_type": "llm_judge", + "prompt": "Does this response acknowledge the cooking/food app use case and indicate that configuration was updated? It should mention something about food preferences, dietary info, or cooking." + } + ] + }, + { + "step_type": "add_messages", + "session_id": "feedback_test", + "messages": [ + { + "peer_id": "user", + "content": "I'm allergic to peanuts and shellfish. Very serious allergies." + }, + { + "peer_id": "assistant", + "content": "That's important to know! I'll make sure to avoid those ingredients." + }, + { + "peer_id": "user", + "content": "I work from home as a freelance writer. Mostly do tech blogs." + }, + { + "peer_id": "assistant", + "content": "That sounds like a flexible lifestyle!" + }, + { + "peer_id": "user", + "content": "My favorite cuisine is Thai food, especially pad thai and green curry." + } + ] + }, + { + "step_type": "wait", + "target": "queue_empty", + "timeout": 120, + "flush": true + }, + { + "step_type": "query", + "description": "Verify food-related observations were extracted", + "target": "get_representation", + "observer_peer_id": "assistant", + "observed_peer_id": "user", + "session_id": "feedback_test", + "assertions": [ + { + "assertion_type": "llm_judge", + "prompt": "Does this representation contain information about food allergies (peanuts, shellfish) or cuisine preferences (Thai food, pad thai, curry)? It should focus on food-related facts." + } + ] + }, + { + "step_type": "query", + "description": "Verify non-food observations were filtered out", + "target": "get_representation", + "observer_peer_id": "assistant", + "observed_peer_id": "user", + "session_id": "feedback_test", + "assertions": [ + { + "assertion_type": "not_contains", + "text": "freelance writer" + }, + { + "assertion_type": "not_contains", + "text": "tech blogs" + } + ] + } + ] +}