From 166d75e8843bffc7215b324f3f4b9eb1f5836079 Mon Sep 17 00:00:00 2001 From: Rajat Ahuja Date: Tue, 18 Nov 2025 11:22:36 -0500 Subject: [PATCH] feat: add prompt to 'reverse engineer' honcho --- scripts/reverse_engineer_example.py | 153 ++++++++++++++ src/utils/reverse_engineer.py | 305 ++++++++++++++++++++++++++++ 2 files changed, 458 insertions(+) create mode 100644 scripts/reverse_engineer_example.py create mode 100644 src/utils/reverse_engineer.py diff --git a/scripts/reverse_engineer_example.py b/scripts/reverse_engineer_example.py new file mode 100644 index 00000000..39bd836b --- /dev/null +++ b/scripts/reverse_engineer_example.py @@ -0,0 +1,153 @@ +""" +Example usage of the reverse_engineer module. + +This script demonstrates how to use the reverse engineering functionality +to generate ground truth atomic propositions and deductions from a conversation +history, question, and correct answer. +""" + +import asyncio +import json +import sys +from pathlib import Path + +# Add the project root to the Python path +project_root = Path(__file__).parent.parent +sys.path.insert(0, str(project_root)) + +from src.utils.reverse_engineer import reverse_engineer_trace + + +async def main(): + # Example from LongMemEval dataset - Seco de Cordero recipe + messages = [ + { + "role": "user", + "content": "Could you suggest a recipe for a classic dish from Ancash?", + }, + { + "role": "assistant", + "content": """Sure! How about "Seco de Cordero" (Lamb Stew) from Ancash: + +Ingredients: +- 2 lbs. lamb, cut into small pieces +- 1 onion, chopped +- 3 garlic cloves, minced +- 1 tsp. cumin +- 1 tsp. dried oregano +- 2 tbsp. vegetable oil +- 2 tbsp. ají amarillo paste +- 1 cup beer +- 2 cups water +- 4 potatoes, peeled and cut into wedges +- Salt and pepper, to taste +- 1/4 cup chopped cilantro + +Instructions: +1. In a large pot, heat the oil over medium-high heat. Add the lamb pieces and cook until browned on all sides, about 8-10 minutes. +2. Remove the lamb from the pot and set it aside on a plate. +3. In the same pot, add the chopped onion and garlic. Cook until translucent, about 5-7 minutes. +4. Add the cumin, oregano, and ají amarillo paste to the pot. Stir to combine and cook for another minute or two. +5. Add the beer, water, and lamb back to the pot. Bring the mixture to a boil and then reduce the heat to low. Cover and let simmer for about 1 hour, or until the lamb is tender. +6. Add the potatoes to the pot and season with salt and pepper to taste. Cook until the potatoes are soft and the stew has thickened, about 30-45 minutes. +7. To finish, stir in the chopped cilantro and serve hot with a side of rice. + +Enjoy your Seco de Cordero, a delicious and hearty dish from Ancash!""", + }, + { + "role": "user", + "content": "That sounds amazing! I've never tried a lamb stew before. Do you have any recommendations for a good type of beer to use in the recipe?", + }, + { + "role": "assistant", + "content": "I don't have personal preferences, but in general, it's recommended to use a light or medium-bodied beer for this recipe to avoid overpowering the flavors of the lamb and spices. A Pilsner or Lager would work well, but you can also use a local beer from Ancash to give the dish a more authentic touch. Ultimately, it comes down to your personal preference, so you can choose a beer that you enjoy drinking and experiment with different types to find your favorite one. Enjoy your delicious Seco de Cordero!", + }, + ] + + question = "I'm looking back at our previous conversation about the Seco de Cordero recipe from Ancash. You mentioned using a light or medium-bodied beer, but I was wondering if you could remind me what type of beer you specifically recommended?" + + answer = "I recommended using a Pilsner or Lager for the recipe." + + print("=" * 80) + print("REVERSE ENGINEERING EXAMPLE") + print("=" * 80) + print("\nQuestion:") + print(question) + print("\nCorrect Answer:") + print(answer) + print("\nGenerating minimal set of atomic propositions and deductions...") + print("=" * 80) + + # Call the reverse engineering function (single unified LLM call) + ( + explicit_response, + deductive_response, + observer_card, + observed_card, + ) = await reverse_engineer_trace( + messages=messages, + question=question, + answer=answer, + observer="user", + observed="assistant", + ) + + print("\n--- EXPLICIT OBSERVATIONS ---") + if explicit_response.explicit: + for i, obs in enumerate(explicit_response.explicit, 1): + print(f"{i}. {obs.content}") + else: + print("(none)") + + print("\n--- IMPLICIT OBSERVATIONS ---") + if explicit_response.implicit: + for i, obs in enumerate(explicit_response.implicit, 1): + print(f"{i}. {obs.content}") + else: + print("(none)") + + print("\n--- DEDUCTIVE OBSERVATIONS ---") + if deductive_response.deductions: + for i, obs in enumerate(deductive_response.deductions, 1): + print(f"{i}. Conclusion: {obs.conclusion}") + if obs.premises: + print(" Premises:") + for premise in obs.premises: + print(f" - {premise}") + else: + print("(none)") + + print("\n--- OBSERVER PEER CARD ---") + if observer_card: + for i, card_entry in enumerate(observer_card, 1): + print(f"{i}. {card_entry}") + else: + print("(none)") + + print("\n--- OBSERVED PEER CARD ---") + if observed_card: + for i, card_entry in enumerate(observed_card, 1): + print(f"{i}. {card_entry}") + else: + print("(none)") + + print("\n" + "=" * 80) + print("JSON OUTPUT") + print("=" * 80) + + output = { + "explicit": [obs.content for obs in explicit_response.explicit], + "implicit": [obs.content for obs in explicit_response.implicit], + "deductions": [ + {"premises": obs.premises, "conclusion": obs.conclusion} + for obs in deductive_response.deductions + ], + "observer_card": observer_card, + "observed_card": observed_card, + } + + print(json.dumps(output, indent=2)) + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/src/utils/reverse_engineer.py b/src/utils/reverse_engineer.py new file mode 100644 index 00000000..8b6f8409 --- /dev/null +++ b/src/utils/reverse_engineer.py @@ -0,0 +1,305 @@ +""" +Reverse engineer atomic propositions and deductions from message history. + +Given a conversation, a question, and the correct answer, this module generates +the minimal set of atomic propositions (explicit + implicit) and deductive +conclusions that would be sufficient to answer the question correctly. + +This is primarily used for generating ground truth training data when the +dialectic produces incorrect answers. +""" + +import logging + +from pydantic import BaseModel, Field + +from src.config import settings +from src.utils.clients import HonchoLLMCallResponse, honcho_llm_call +from src.utils.representation import DeductiveResponse, ExplicitResponse +from src.utils.types import SupportedProviders + +logger = logging.getLogger(__name__) + + +class DeductionItem(BaseModel): + """A single deduction with premises and conclusion.""" + + premises: list[str] = Field( + description="Premises supporting this deduction", + default_factory=list, + ) + conclusion: str = Field( + description="The deductive conclusion", + ) + + +class ReverseEngineerResponse(BaseModel): + """Combined response containing propositions, deductions, and peer cards.""" + + explicit: list[str] = Field( + description="Explicit facts directly stated in the conversation", + default_factory=list, + ) + implicit: list[str] = Field( + description="Facts clearly implied by the conversation", + default_factory=list, + ) + deductions: list[DeductionItem] = Field( + description="Deductive conclusions with premises and conclusion", + default_factory=list, + ) + observer_card: list[str] | None = Field( + description="Biographical card for the observer (peer asking the question)", + default=None, + ) + observed_card: list[str] | None = Field( + description="Biographical card for the observed peer (peer being asked about)", + default=None, + ) + + +def reverse_engineer_prompt( + messages: list[dict[str, str]], + question: str, + answer: str, + *, + observer: str | None = None, + observed: str | None = None, +) -> str: + """ + Generate prompt for reverse-engineering minimal trace from conversation. + + This prompt is strictly aligned with the deriver's explicit reasoning, + deductive reasoning, and peer card extraction templates to ensure the + reverse-engineered trace follows the same extraction rules. + + Args: + messages: List of message dicts with 'role' and 'content' keys + question: The question that was asked + answer: The correct/ground truth answer to the question + observer: Optional name of the peer asking the question + observed: Optional name of the peer being asked about + + Returns: + Formatted prompt string for LLM + """ + # Format the conversation history + conversation_lines: list[str] = [] + for msg in messages: + role = msg.get("role", "unknown") + content = msg.get("content", "") + conversation_lines.append(f"{role.upper()}: {content}") + + conversation_text = "\n\n".join(conversation_lines) + + # Build perspective context if provided + perspective_context = "" + if observer and observed: + if observer == observed: + perspective_context = ( + f"\nThe conversation involves {observer} (observing themselves)." + ) + else: + perspective_context = ( + f"\nThe question is asked by {observer} about {observed}." + ) + + prompt = f"""You are a knowledge extraction system that reverse-engineers the MINIMAL set of observations needed to answer a question. + +# Conversation History +{conversation_text} +{perspective_context} + +# Question +{question} + +# Correct Answer +{answer} + +# Task +Extract the MINIMAL set of atomic propositions, deductions, and peer biographical information from the conversation that would be SUFFICIENT to answer the question correctly. + +You must follow the EXACT extraction rules used by the deriver system: + +## PART 1: ATOMIC PROPOSITIONS (Explicit + Implicit) + +An atomic proposition is: +1. A statement with a SINGLE TRUTH VALUE (evaluable as true or false independently) +2. Contains NO LOGICAL CONNECTIVES (no AND, OR, IF-THEN, UNLESS, etc.) +3. SUFFICIENTLY CONTEXTUALIZED to be meaningful standing alone + +**EXPLICIT EXTRACTION** - Directly stated facts: +- Extract propositions directly asserted in the conversation +- Each claim becomes a separate atomic proposition +- Word-for-word or clear paraphrase only + +**IMPLICIT EXTRACTION** - Clearly implied facts: +- Extract propositions that are obviously implied by the conversation +- Only include implications that are CERTAIN, not speculative +- Examples: + * "I graduated from college" → IMPLIES: "X attended college" + * "I'm taking my dog to the vet" → IMPLIES: "X has a dog" + +ONLY extract propositions that are NECESSARY to answer the question. + +## PART 2: DEDUCTIVE REASONING + +A deductive inference is valid when: +1. The conclusion NECESSARILY follows from the premises +2. If all premises are true, the conclusion MUST be true +3. The reasoning follows the laws of formal logic + +**SUBSTANTIVE THRESHOLD:** +Only generate deductions that add meaningful, non-obvious information that is semantically differentiated from the atomic propositions. + +- ❌ TRIVIAL: "Maria spoke" → "Maria is alive" (biological necessity, assumed) +- ❌ DEFINITIONAL RESTATEMENT: "Liam went to the store" → "Liam visited a retail establishment" (just rewording) +- ✓ SUBSTANTIVE: "Maria attended college" → "Maria completed high school or equivalent education" (non-obvious precondition) + +**PERMITTED PREMISE TYPES:** +1. Atomic propositions (explicit/implicit extracted above) +2. General knowledge - widely accepted facts +3. Temporal information +4. Logical principles + +ONLY generate deductions that are NECESSARY to answer the question. + +## PART 3: PEER BIOGRAPHICAL CARDS + +Extract minimal biographical information that would be SUFFICIENT or HELPFUL to answer the question. A biographical card contains essential PERMANENT information: +- Name, nicknames +- Age, location +- Occupation +- Core interests/hobbies +- Key likes/dislikes +- Other permanent traits + +**Guidelines:** +- ONLY extract from conversation messages (NOT from the answer) +- ONLY include info that helps answer the question +- Value permanent properties over transient ones ("is a software engineer" not "wrote Python today") +- Value concision over detail +- Never infer traits from one-off behaviors +- Format: "Name: Alice", "Occupation: Artist", etc. +- Set to null if no relevant biographical info exists + +# Response Format +Return JSON with five keys: +- "explicit": array of explicit atomic proposition strings (ONLY those needed to answer question) +- "implicit": array of implicit atomic proposition strings (ONLY those needed to answer question) +- "deductions": array of objects with "premises" (array) and "conclusion" (string) (ONLY those needed to answer question) +- "observer_card": array of biographical strings for {observer or "the observer"}, or null +- "observed_card": array of biographical strings for {observed or "the observed"}, or null + +For self-queries (observer == observed), only populate "observer_card". + +Example: +{{ + "explicit": ["The assistant recommended using a Pilsner or Lager for the recipe"], + "implicit": ["A Pilsner is a type of light-bodied beer", "A Lager is a type of light-bodied beer"], + "deductions": [], + "observer_card": null, + "observed_card": null +}} +""" + + return prompt + + +async def reverse_engineer_trace( + messages: list[dict[str, str]], + question: str, + answer: str, + *, + observer: str | None = None, + observed: str | None = None, + provider: SupportedProviders | None = None, + model: str | None = None, + max_tokens: int | None = None, +) -> tuple[ExplicitResponse, DeductiveResponse, list[str] | None, list[str] | None]: + """ + Call LLM to reverse-engineer atomic propositions, deductions, and peer cards. + + This makes a single LLM call that extracts: + 1. Explicit and implicit atomic propositions + 2. Deductive conclusions with premises + 3. Observer and observed peer biographical cards + + All outputs represent the MINIMAL information sufficient to answer the question. + + Args: + messages: List of message dicts with 'role' and 'content' keys + question: The question that was asked + answer: The correct/ground truth answer + observer: Optional name of the peer asking the question + observed: Optional name of the peer being asked about + provider: LLM provider to use (defaults to DERIVER settings) + model: Model name to use (defaults to DERIVER settings) + max_tokens: Max output tokens (defaults to DERIVER settings) + + Returns: + Tuple of (ExplicitResponse, DeductiveResponse, observer_card, observed_card) + where peer cards are list[str] or None + """ + prompt = reverse_engineer_prompt( + messages, + question, + answer, + observer=observer, + observed=observed, + ) + + response: HonchoLLMCallResponse[ReverseEngineerResponse] = await honcho_llm_call( + provider=provider or settings.DERIVER.PROVIDER, + model=model or settings.DERIVER.MODEL, + prompt=prompt, + max_tokens=max_tokens or settings.DERIVER.MAX_OUTPUT_TOKENS, + track_name="Reverse Engineer Trace", + response_model=ReverseEngineerResponse, + json_mode=True, + enable_retry=True, + retry_attempts=3, + ) + + # Convert to existing schema formats + from src.utils.representation import ( + DeductiveObservationBase, + ExplicitObservationBase, + ImplicitObservationBase, + ) + + # Parse the response + result: ReverseEngineerResponse = response.content + + # Build ExplicitResponse + explicit_response = ExplicitResponse( + explicit=[ExplicitObservationBase(content=e) for e in result.explicit], + implicit=[ImplicitObservationBase(content=i) for i in result.implicit], + ) + + # Build DeductiveResponse + deductive_response = DeductiveResponse( + deductions=[ + DeductiveObservationBase( + premises=d.premises, + conclusion=d.conclusion, + ) + for d in result.deductions + ] + ) + + logger.debug( + "Reverse engineered trace: %d explicit, %d implicit, %d deductive, observer_card=%s, observed_card=%s", + len(explicit_response.explicit), + len(explicit_response.implicit), + len(deductive_response.deductions), + result.observer_card is not None, + result.observed_card is not None, + ) + + return ( + explicit_response, + deductive_response, + result.observer_card, + result.observed_card, + )