From 035fc9d7f3dfba87dfb4ad14faa9db6aa0e08ba9 Mon Sep 17 00:00:00 2001 From: Matt Johnson Date: Thu, 2 Jul 2026 18:48:14 +0000 Subject: [PATCH] feat(reply): cap interactive LLM replies to 3 mesh packets Add terse-answer guidance to the interactive system prompt and a hard ceiling of 3 packets (3 x connector.max_chars) on LLM replies, with an "ask for more" indicator when truncated. Protects LoRa airtime from runaway replies. Broadcast chunking unchanged. Co-Authored-By: Claude Opus 4.8 (1M context) --- work/meshai/chunker.py | 29 ++++++++ work/meshai/router.py | 15 +++- work/tests/test_llm_reply_cap.py | 117 +++++++++++++++++++++++++++++++ 3 files changed, 160 insertions(+), 1 deletion(-) create mode 100644 work/tests/test_llm_reply_cap.py diff --git a/work/meshai/chunker.py b/work/meshai/chunker.py index 3b6e6d5..88ac6a5 100644 --- a/work/meshai/chunker.py +++ b/work/meshai/chunker.py @@ -13,6 +13,35 @@ import re logger = logging.getLogger(__name__) +# Hard ceiling on interactive LLM replies — protects LoRa airtime. +# Broadcast/notification chunking is NOT subject to this cap. +MAX_REPLY_PACKETS = 3 + +_TRUNCATION_INDICATOR = " …(ask for more)" + + +def cap_reply_chunks(chunks: list[str], max_packets: int, max_chars: int) -> list[str]: + """Cap LLM reply chunks to max_packets; append truncation indicator if cut. + + Args: + chunks: List of message chunks produced by chunk_response. + max_packets: Maximum number of packets to deliver (e.g. MAX_REPLY_PACKETS). + max_chars: Per-packet character budget (connector.max_chars). + + Returns: + Original list when len(chunks) <= max_packets; otherwise a capped + copy of length max_packets with the truncation indicator appended + (and the last chunk trimmed if necessary to stay within max_chars). + """ + if len(chunks) <= max_packets: + return chunks + capped = list(chunks[:max_packets]) + last = capped[-1] + if len(last) + len(_TRUNCATION_INDICATOR) > max_chars: + last = last[: max_chars - len(_TRUNCATION_INDICATOR)] + capped[-1] = last + _TRUNCATION_INDICATOR + return capped + def strip_markdown(text: str) -> str: """Remove markdown formatting from LLM output. diff --git a/work/meshai/router.py b/work/meshai/router.py index f6db43f..e6e674d 100644 --- a/work/meshai/router.py +++ b/work/meshai/router.py @@ -13,7 +13,7 @@ from .config import Config from .connector import MeshConnector, MeshMessage from .context import MeshContext from .history import ConversationHistory -from .chunker import chunk_response, ContinuationState +from .chunker import chunk_response, cap_reply_chunks, MAX_REPLY_PACKETS, ContinuationState logger = logging.getLogger(__name__) @@ -889,6 +889,13 @@ class MessageRouter: if env_summary: system_prompt += "\n\n" + env_summary + # Terse-reply guidance for slow LoRa links (always appended last for + # interactive replies so it isn't buried by mesh-data blocks). + system_prompt += ( + "\n\nYou're replying over a slow LoRa mesh. " + "Answer in 1-2 short messages. Be terse and direct; omit preamble and filler." + ) + # DEBUG: Log system prompt status logger.debug(f"System prompt length: {len(system_prompt)} chars") @@ -929,6 +936,12 @@ class MessageRouter: max_messages=self.config.response.max_messages, ) + # Hard cap: LLM interactive replies must not exceed MAX_REPLY_PACKETS. + # This is a safety ceiling on top of config.response.max_messages so + # LoRa airtime is protected even if config grows the message budget. + # Broadcast/notification chunking is NOT affected by this cap. + messages = cap_reply_chunks(messages, MAX_REPLY_PACKETS, self.connector.max_chars) + # Store remaining content for continuation if remaining: logger.debug(f"Storing continuation for {message.sender_id}: {len(remaining)} chars remaining") diff --git a/work/tests/test_llm_reply_cap.py b/work/tests/test_llm_reply_cap.py new file mode 100644 index 0000000..97471fa --- /dev/null +++ b/work/tests/test_llm_reply_cap.py @@ -0,0 +1,117 @@ +"""Tests for the interactive LLM reply cap (MAX_REPLY_PACKETS). + +All tests are hermetic — no real LLM, no real socket. +""" + +import pytest + +from meshai.chunker import cap_reply_chunks, chunk_response, MAX_REPLY_PACKETS, _TRUNCATION_INDICATOR + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +def _make_chunks(n: int, chars_each: int = 50) -> list[str]: + """Build n chunks of exactly chars_each ASCII characters.""" + return [f"chunk{i}" + "x" * (chars_each - len(f"chunk{i}")) for i in range(n)] + + +# --------------------------------------------------------------------------- +# Test 1 — long reply produces exactly 3 chunks with truncation indicator +# --------------------------------------------------------------------------- + +def test_long_reply_capped_to_three_chunks(): + """A reply long enough to produce >3 chunks results in exactly 3 delivered + chunks, and the truncation indicator is present in the last chunk.""" + max_chars = 140 + + # Build a response that will produce >3 chunks at 140 chars each. + # Each sentence is ~130 chars so 5 sentences → 5 chunks. + sentence = "A" * 130 + "." + text = " ".join([sentence] * 5) + + # chunk_response with a high max_messages so it happily returns 5 chunks + raw_chunks, _ = chunk_response(text, max_chars=max_chars, max_messages=10) + assert len(raw_chunks) > MAX_REPLY_PACKETS, ( + f"Precondition: expected >3 raw chunks, got {len(raw_chunks)}" + ) + + capped = cap_reply_chunks(raw_chunks, MAX_REPLY_PACKETS, max_chars) + + assert len(capped) == MAX_REPLY_PACKETS + assert _TRUNCATION_INDICATOR in capped[-1], ( + f"Expected truncation indicator in last chunk, got: {capped[-1]!r}" + ) + + +# --------------------------------------------------------------------------- +# Test 2 — short reply is delivered unchanged, no indicator +# --------------------------------------------------------------------------- + +def test_short_reply_unchanged(): + """A short reply (<=3 chunks) is delivered unchanged with no indicator.""" + max_chars = 140 + + # One chunk well under the limit + chunks_1 = ["Hello, this is a short reply."] + result_1 = cap_reply_chunks(chunks_1, MAX_REPLY_PACKETS, max_chars) + assert result_1 == chunks_1 + assert _TRUNCATION_INDICATOR not in result_1[-1] + + # Exactly 3 chunks — boundary condition + chunks_3 = _make_chunks(3, chars_each=50) + result_3 = cap_reply_chunks(chunks_3, MAX_REPLY_PACKETS, max_chars) + assert result_3 == chunks_3 + assert _TRUNCATION_INDICATOR not in result_3[-1] + + +# --------------------------------------------------------------------------- +# Test 3 — cap respects connector.max_chars=140 +# --------------------------------------------------------------------------- + +def test_cap_respects_max_chars_140(): + """cap_reply_chunks uses the max_chars budget (140) from connector.max_chars.""" + max_chars = 140 + + # 5 chunks of 80 chars each (well under max_chars individually) + chunks = _make_chunks(5, chars_each=80) + + capped = cap_reply_chunks(chunks, MAX_REPLY_PACKETS, max_chars) + + assert len(capped) == 3 + # Last chunk must not exceed max_chars in total length + assert len(capped[-1]) <= max_chars + assert _TRUNCATION_INDICATOR in capped[-1] + + +# --------------------------------------------------------------------------- +# Test 4 — 3rd chunk trimmed correctly when at max_chars limit +# --------------------------------------------------------------------------- + +def test_third_chunk_trimmed_to_fit_indicator(): + """When the 3rd chunk fills max_chars exactly, it is trimmed so that + trimmed_chunk + indicator == max_chars exactly.""" + max_chars = 140 + indicator = _TRUNCATION_INDICATOR # len 16 Python chars (" …(ask for more)") + + # 3rd chunk that is exactly max_chars long + full_chunk = "B" * max_chars + chunks = [ + "First chunk.", + "Second chunk.", + full_chunk, + "Fourth chunk (should be dropped).", + ] + + capped = cap_reply_chunks(chunks, MAX_REPLY_PACKETS, max_chars) + + assert len(capped) == MAX_REPLY_PACKETS + last = capped[-1] + assert last.endswith(indicator), f"Expected indicator at end, got: {last!r}" + assert len(last) == max_chars, ( + f"Expected trimmed chunk + indicator == {max_chars}, got len={len(last)}" + ) + # The trimmed body is max_chars - len(indicator) Bs + expected_body = "B" * (max_chars - len(indicator)) + assert last == expected_body + indicator