honcho/src/deriver/prompts.py

114 lines
3.8 KiB
Python

"""
Minimal prompts for the deriver module optimized for speed.
This module contains simplified prompt templates focused only on observation extraction.
NO peer card instructions, NO working representation - just extract observations.
"""
from functools import cache
from inspect import cleandoc as c
from src.utils.tokens import estimate_tokens
def _normalized_custom_instructions(custom_instructions: str | None) -> str | None:
"""Return stripped custom instructions, if any."""
if custom_instructions is None:
return None
normalized = custom_instructions.strip()
return normalized or None
def _custom_instructions_section(custom_instructions: str | None) -> str:
"""Render optional custom instructions for the deriver prompt."""
normalized_custom_instructions = _normalized_custom_instructions(
custom_instructions
)
if normalized_custom_instructions is None:
return ""
return c(
f"""
CUSTOM INSTRUCTIONS:
These instructions apply to the target peer.
{normalized_custom_instructions}
"""
)
def minimal_deriver_prompt(
peer_id: str,
messages: str,
custom_instructions: str | None = None,
) -> str:
"""
Generate minimal prompt for fast observation extraction.
Args:
peer_id: The ID of the user being analyzed.
messages: All messages in the range (interleaving messages and new turns combined).
Returns:
Formatted prompt string for observation extraction.
"""
custom_instructions_section = _custom_instructions_section(custom_instructions)
return c(
f"""
Analyze messages to extract **explicit atomic facts** about the target peer.
[EXPLICIT] DEFINITION: Facts about the target peer that can be derived directly from their messages.
- Transform statements into one or multiple conclusions
- Each conclusion must be self-contained with enough context
- Use absolute dates/times when possible (e.g. "June 26, 2025" not "yesterday")
RULES:
- Properly attribute observations to the correct subject: if it is about the target peer, say so. If the target peer is referencing someone or something else, make that clear.
- Observations should make sense on their own. Each observation will be used in the future to better understand the target peer.
- Extract ALL observations from the target peer's messages, using others as context.
- Contextualize each observation sufficiently (e.g. "Ann is nervous about the job interview at the pharmacy" not just "Ann is nervous")
EXAMPLES:
- EXPLICIT: "I just had my 25th birthday last Saturday""The target peer is 25 years old", "The target peer's birthday is June 21st"
- EXPLICIT: "I took my dog for a walk in NYC""The target peer has a dog", "The target peer lives in NYC"
- EXPLICIT: "The target peer attended college" + general knowledge → "The target peer completed high school or equivalent"
{custom_instructions_section}
Target peer:
{peer_id}
Messages to analyze:
<messages>
{messages}
</messages>
"""
)
@cache
def estimate_minimal_deriver_prompt_tokens() -> int:
"""Estimate the static minimal deriver prompt without custom instructions."""
prompt = minimal_deriver_prompt(
peer_id="",
messages="",
custom_instructions=None,
)
return estimate_tokens(prompt)
def estimate_deriver_prompt_tokens(custom_instructions: str | None) -> int:
"""Estimate minimal deriver prompt tokens, including custom instructions if present."""
normalized_custom_instructions = _normalized_custom_instructions(
custom_instructions
)
if normalized_custom_instructions is None:
return estimate_minimal_deriver_prompt_tokens()
prompt = minimal_deriver_prompt(
peer_id="",
messages="",
custom_instructions=normalized_custom_instructions,
)
return estimate_tokens(prompt)