Skip to content
Open
Changes from 2 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
58 changes: 56 additions & 2 deletions src/utils/summarizer.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
from src.config import ConfiguredModelSettings, settings
from src.crud.session import session_cache_key
from src.dependencies import tracked_db
from src.deriver.prompts import _custom_instructions_section
from src.exceptions import ResourceNotFoundException
from src.llm import HonchoLLMCallResponse, honcho_llm_call
from src.llm.types import LLMTelemetryContext
Expand Down Expand Up @@ -57,6 +58,7 @@ class Summary(TypedDict):


def to_schema_summary(s: Summary) -> schemas.Summary:
"""Convert a Summary TypedDict to a Pydantic Summary schema object."""
return schemas.Summary(
content=s["content"],
message_id=s["message_id"],
Expand All @@ -81,6 +83,7 @@ def to_schema_summary(s: Summary) -> schemas.Summary:


def _get_summary_model_config() -> ConfiguredModelSettings:
"""Return the configured model settings for summary generation."""
return settings.SUMMARY.MODEL_CONFIG


Expand All @@ -101,8 +104,10 @@ def short_summary_prompt(
formatted_messages: str,
output_words: int,
previous_summary_text: str,
custom_instructions: str | None = None,
) -> str:
"""Generate the short summary prompt."""
custom_instructions_section = _custom_instructions_section(custom_instructions)
return c(f"""
You are a system that summarizes parts of a conversation to create a concise and accurate summary. Focus on capturing:

Expand All @@ -117,6 +122,7 @@ def short_summary_prompt(

Return only the summary without any explanation or meta-commentary.

{custom_instructions_section}
<previous_summary>
{previous_summary_text}
</previous_summary>
Expand All @@ -133,8 +139,10 @@ def long_summary_prompt(
formatted_messages: str,
output_words: int,
previous_summary_text: str,
custom_instructions: str | None = None,
) -> str:
"""Generate the long summary prompt."""
custom_instructions_section = _custom_instructions_section(custom_instructions)
return c(f"""
You are a system that creates thorough, comprehensive summaries of conversations. Focus on capturing:

Expand All @@ -151,6 +159,7 @@ def long_summary_prompt(

Return only the summary without any explanation or meta-commentary.

{custom_instructions_section}
<previous_summary>
{previous_summary_text}
</previous_summary>
Expand All @@ -172,6 +181,7 @@ def estimate_short_summary_prompt_tokens() -> int:
formatted_messages="",
output_words=0,
previous_summary_text="",
custom_instructions=None,
)
)
except Exception:
Expand All @@ -188,6 +198,7 @@ def estimate_long_summary_prompt_tokens() -> int:
formatted_messages="",
output_words=0,
previous_summary_text="",
custom_instructions=None,

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

should be custom_instructions=custom_instructions

)
)
except Exception:
Expand All @@ -200,9 +211,23 @@ async def create_short_summary(
formatted_messages: str,
input_tokens: int,
previous_summary: str | None = None,
custom_instructions: str | None = None,
*,
workspace_name: str | None = None,
) -> HonchoLLMCallResponse[str]:
"""
Generate a short summary via an LLM call.

Args:
formatted_messages: Pre-formatted conversation messages.
input_tokens: Token count of the input (messages + previous summary).
previous_summary: Previous summary text for continuity, if any.
custom_instructions: Optional custom instructions from configuration.
workspace_name: Workspace name for telemetry attribution.

Returns:
The LLM response containing the short summary text and token counts.
"""
# input_tokens indicates how many tokens the message list + previous summary take up
# we want to optimize short summaries to be smaller than the actual content being summarized
# so we ask the agent to produce a word count roughly equal to either the input, or the max
Expand All @@ -216,7 +241,10 @@ async def create_short_summary(
previous_summary_text = "There is no previous summary -- the messages are the beginning of the conversation."

prompt = short_summary_prompt(
formatted_messages, output_words, previous_summary_text
formatted_messages,
output_words,
previous_summary_text,
custom_instructions=custom_instructions,
)

return await honcho_llm_call(
Expand All @@ -235,9 +263,22 @@ async def create_short_summary(
async def create_long_summary(
formatted_messages: str,
previous_summary: str | None = None,
custom_instructions: str | None = None,
*,
workspace_name: str | None = None,
) -> HonchoLLMCallResponse[str]:
"""
Generate a comprehensive long summary via an LLM call.

Args:
formatted_messages: Pre-formatted conversation messages.
previous_summary: Previous summary text for continuity, if any.
custom_instructions: Optional custom instructions from configuration.
workspace_name: Workspace name for telemetry attribution.

Returns:
The LLM response containing the long summary text and token counts.
"""
# the word/token ratio is roughly 4:3 so we multiply by 0.75.
# LLMs *seem* to respond better to getting asked for a word count but should workshop this.
output_words = int(settings.SUMMARY.MAX_TOKENS_LONG * 0.75)
Expand All @@ -248,7 +289,10 @@ async def create_long_summary(
previous_summary_text = "There is no previous summary -- the messages are the beginning of the conversation."

prompt = long_summary_prompt(
formatted_messages, output_words, previous_summary_text
formatted_messages,
output_words,
previous_summary_text,
custom_instructions=custom_instructions,
)

return await honcho_llm_call(
Expand Down Expand Up @@ -439,6 +483,12 @@ async def _create_and_save_summary(
previous_summary_tokens = latest_summary["token_count"] if latest_summary else 0
input_tokens = messages_tokens + previous_summary_tokens

# Extract custom_instructions from configuration for the summarizer prompt.
# This mirrors the deriver pattern in src/deriver/prompts.py.
custom_instructions: str | None = None
if configuration.reasoning and configuration.reasoning.custom_instructions:
custom_instructions = configuration.reasoning.custom_instructions

(
new_summary,
is_fallback,
Expand All @@ -453,6 +503,7 @@ async def _create_and_save_summary(
last_message_id=last_message_id,
last_message_content_preview=last_message_content_preview,
message_count=message_count,
custom_instructions=custom_instructions,
workspace_name=workspace_name,
)

Expand Down Expand Up @@ -552,6 +603,7 @@ async def _create_summary(
last_message_id: int,
last_message_content_preview: str,
message_count: int,
custom_instructions: str | None = None,
*,
workspace_name: str | None = None,
) -> tuple[Summary, bool, int, int]:
Expand Down Expand Up @@ -585,12 +637,14 @@ async def _create_summary(
formatted_messages,
input_tokens,
previous_summary_text,
custom_instructions=custom_instructions,
workspace_name=workspace_name,
)
else:
response = await create_long_summary(
formatted_messages,
previous_summary_text,
custom_instructions=custom_instructions,
workspace_name=workspace_name,
)

Expand Down