From e15f427cdeb9fc9c3c50f9ebf2e47e1d2ecfaf23 Mon Sep 17 00:00:00 2001 From: "Julianne H." Date: Fri, 25 Sep 2026 17:06:10 +0200 Subject: [PATCH] feat: add per-request compression toggles --- edgee/__init__.py | 35 ++++++++++-- example/compression.py | 114 ++++++++++++++++----------------------- tests/test_edgee.py | 120 +++++++++++++++++++++++++++++++++++++++++ 3 files changed, 199 insertions(+), 70 deletions(-) diff --git a/edgee/__init__.py b/edgee/__init__.py index ad0bd25..c9323ec 100644 --- a/edgee/__init__.py +++ b/edgee/__init__.py @@ -3,6 +3,7 @@ import json import os import ssl +import warnings from dataclasses import dataclass from urllib.error import HTTPError from urllib.request import Request, urlopen @@ -11,6 +12,22 @@ DEFAULT_BASE_URL = "https://edgee.io" API_ENDPOINT = "/v1/chat/completions" +# Per-request compression toggles, keyed by their InputObject field name. +COMPRESSION_HEADERS = { + "tool_result_trimming": "X-Edgee-Compression-Tool-Result-Trimming", + "tool_surface_reduction": "X-Edgee-Compression-Tool-Surface-Reduction", + "output_brevity": "X-Edgee-Compression-Brevity", +} + + +def _compression_headers(toggles: dict) -> dict[str, str]: + """Headers for the toggles the caller set. Only real bools count; None means keep the key setting.""" + return { + COMPRESSION_HEADERS[field]: "true" if value else "false" + for field, value in toggles.items() + if isinstance(value, bool) + } + def _ssl_context() -> ssl.SSLContext: """Create SSL context. Uses certifi's CA bundle when available (fixes cert issues on macOS).""" @@ -59,9 +76,12 @@ class InputObject: tools: list[dict] | None = None tool_choice: str | dict | None = None tags: list[str] | None = None - compression_model: str | None = ( - None # Compression model: claude, opencode, cursor, customer (gateway-internal) - ) + # Deprecated: any value turns tool-result trimming on. Use tool_result_trimming instead. + compression_model: str | None = None + # Per-request overrides of the API key settings. None keeps the key setting. + tool_result_trimming: bool | None = None + tool_surface_reduction: bool | None = None + output_brevity: bool | None = None @dataclass @@ -216,18 +236,21 @@ def send( tool_choice = None tags = None compression_model = None + toggles = {} elif isinstance(input, InputObject): messages = input.messages tools = input.tools tool_choice = input.tool_choice tags = input.tags compression_model = input.compression_model + toggles = {field: getattr(input, field) for field in COMPRESSION_HEADERS} else: messages = input.get("messages", []) tools = input.get("tools") tool_choice = input.get("tool_choice") tags = input.get("tags") compression_model = input.get("compression_model") + toggles = {field: input.get(field) for field in COMPRESSION_HEADERS} body: dict = {"model": model, "messages": messages} if stream: @@ -239,6 +262,11 @@ def send( if tags: body["tags"] = tags if compression_model is not None: + warnings.warn( + "compression_model is deprecated; use tool_result_trimming=True instead", + DeprecationWarning, + stacklevel=2, + ) body["compression_model"] = compression_model request = Request( @@ -247,6 +275,7 @@ def send( headers={ "Content-Type": "application/json", "Authorization": f"Bearer {self.api_key}", + **_compression_headers(toggles), }, method="POST", ) diff --git a/example/compression.py b/example/compression.py index b9a5788..236c59c 100644 --- a/example/compression.py +++ b/example/compression.py @@ -1,13 +1,14 @@ """Example: Token compression with Edgee Gateway SDK This example demonstrates how to: -1. Enable compression for a request with a large input context -2. Set a custom compression rate -3. Access compression metrics from the response - -IMPORTANT: Only USER messages are compressed. System messages are not compressed. -This example includes a large context in the user message to demonstrate meaningful -compression savings. +1. Turn tool-result trimming on for a single request +2. Access compression metrics from the response + +Tool-result trimming shortens the output of tool calls (here a long `ls -la` +listing) before it reaches the model. The per-request toggles +(`tool_result_trimming`, `tool_surface_reduction`, `output_brevity`) override +the API key settings for this request only; leave one out to keep the key's +setting. """ import os @@ -21,78 +22,57 @@ # Initialize the client edgee = Edgee(os.environ.get("EDGEE_API_KEY")) -# Large context document to demonstrate input compression -LARGE_CONTEXT = """ -The History and Impact of Artificial Intelligence - -Artificial intelligence (AI) has evolved from a theoretical concept to a -transformative technology that influences nearly every aspect of modern life. -The field began in earnest in the 1950s when pioneers like Alan Turing and -John McCarthy laid the groundwork for machine intelligence. - -Early developments focused on symbolic reasoning and expert systems. These -rule-based approaches dominated the field through the 1970s and 1980s, with -systems like MYCIN demonstrating practical applications in medical diagnosis. -However, these early systems were limited by their inability to learn from data -and adapt to new situations. - -The resurgence of neural networks in the 1980s and 1990s, particularly with -backpropagation algorithms, opened new possibilities. Yet it wasn't until the -2010s, with the advent of deep learning and the availability of massive datasets -and computational power, that AI truly began to revolutionize industries. - -Modern AI applications span numerous domains: -- Natural language processing enables machines to understand and generate human language -- Computer vision allows machines to interpret visual information from the world -- Robotics combines AI with mechanical systems for autonomous operation -- Healthcare uses AI for diagnosis, drug discovery, and personalized treatment -- Finance leverages AI for fraud detection, algorithmic trading, and risk assessment -- Transportation is being transformed by autonomous vehicles and traffic optimization - -The development of large language models like GPT, BERT, and others has -particularly accelerated progress in natural language understanding and generation. -These models, trained on vast amounts of text data, can perform a wide range of -language tasks with remarkable proficiency. - -Despite remarkable progress, significant challenges remain. Issues of bias, -interpretability, safety, and ethical considerations continue to be areas of -active research and debate. The AI community is working to ensure that these -powerful technologies are developed and deployed responsibly, with consideration -for their societal impact. - -Looking forward, AI is expected to continue advancing rapidly, with potential -breakthroughs in areas like artificial general intelligence, quantum machine -learning, and brain-computer interfaces. The integration of AI into daily life -will likely deepen, raising important questions about human-AI collaboration, -workforce transformation, and the future of human cognition itself. -""" +# A long directory listing, the kind of tool output coding agents send back. +LS_OUTPUT = "total 800\n" + "\n".join( + f"-rw-r--r-- 1 user staff {1000 + i} Jan 1 12:00 src/components/module_{i:03}.tsx" + for i in range(200) +) print("=" * 70) print("Edgee Token Compression Example") print("=" * 70) print() -# Example: Request with compression enabled and large input -print("Example: Large user message with compression enabled") +print("Example: Large tool result with tool-result trimming turned on") print("-" * 70) -print(f"Input context length: {len(LARGE_CONTEXT)} characters") +print(f"Tool output length: {len(LS_OUTPUT)} characters") print() -# NOTE: Only USER messages are compressed -# Put the large context in the user message to demonstrate compression -user_message = f"""Here is some context about AI: - -{LARGE_CONTEXT} - -Based on this context, summarize the key milestones in AI development in 3 bullet points.""" - response = edgee.send( model="anthropic/claude-haiku-4-5", input={ "messages": [ - {"role": "user", "content": user_message}, + {"role": "user", "content": "How many files are in src/components?"}, + { + "role": "assistant", + "tool_calls": [ + { + "id": "call_1", + "type": "function", + "function": { + "name": "Bash", + "arguments": '{"command":"ls -la src/components"}', + }, + } + ], + }, + {"role": "tool", "tool_call_id": "call_1", "content": LS_OUTPUT}, + ], + "tools": [ + { + "type": "function", + "function": { + "name": "Bash", + "description": "Run a shell command and return its output.", + "parameters": { + "type": "object", + "properties": {"command": {"type": "string"}}, + "required": ["command"], + }, + }, + } ], - "compression_model": "claude", + "tool_result_trimming": True, }, ) @@ -123,8 +103,8 @@ print(f" With compression, only {tokens_after} tokens were processed!") else: print("No compression data available in response.") - print("Note: Compression data is only returned when compression is enabled") - print(" and supported by your API key configuration.") + print("Note: Compression data is only returned when trimming actually shortened") + print(" a tool result.") print() print("=" * 70) diff --git a/tests/test_edgee.py b/tests/test_edgee.py index 6c6e2b8..0a9fef5 100644 --- a/tests/test_edgee.py +++ b/tests/test_edgee.py @@ -355,3 +355,123 @@ def test_send_without_compression_response(self, mock_urlopen): result = client.send(model="gpt-4", input="Test") assert result.compression is None + + +class TestCompressionOverrides: + """Per-request compression toggles are sent as headers, never in the body""" + + TRIM = "X-Edgee-Compression-Tool-Result-Trimming" + SURFACE = "X-Edgee-Compression-Tool-Surface-Reduction" + BREVITY = "X-Edgee-Compression-Brevity" + + def _mock_response(self, data: dict): + mock = MagicMock() + mock.read.return_value = json.dumps(data).encode("utf-8") + mock.__enter__ = MagicMock(return_value=mock) + mock.__exit__ = MagicMock(return_value=False) + return mock + + def _ok(self): + return self._mock_response( + { + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok"}, + "finish_reason": "stop", + } + ] + } + ) + + def _sent(self, mock_urlopen): + request = mock_urlopen.call_args[0][0] + return request, json.loads(request.data.decode("utf-8")) + + @patch("edgee.urlopen") + def test_input_object_toggles_become_headers(self, mock_urlopen): + from edgee import InputObject + + mock_urlopen.return_value = self._ok() + Edgee("test-api-key").send( + model="gpt-4", + input=InputObject( + messages=[{"role": "user", "content": "Hello"}], + tool_result_trimming=True, + tool_surface_reduction=False, + output_brevity=True, + ), + ) + + request, body = self._sent(mock_urlopen) + # urllib normalizes header names with str.capitalize(). + assert request.get_header(self.TRIM.capitalize()) == "true" + assert request.get_header(self.SURFACE.capitalize()) == "false" + assert request.get_header(self.BREVITY.capitalize()) == "true" + for field in ("tool_result_trimming", "tool_surface_reduction", "output_brevity"): + assert field not in body + + @patch("edgee.urlopen") + def test_dict_toggles_and_unset_fields(self, mock_urlopen): + mock_urlopen.return_value = self._ok() + Edgee("test-api-key").send( + model="gpt-4", + input={ + "messages": [{"role": "user", "content": "Hello"}], + "tool_result_trimming": False, + # Not a bool: ignored, so the key setting applies. + "output_brevity": "yes", + }, + ) + + request, _ = self._sent(mock_urlopen) + assert request.get_header(self.TRIM.capitalize()) == "false" + assert not request.has_header(self.SURFACE.capitalize()) + assert not request.has_header(self.BREVITY.capitalize()) + + @patch("edgee.urlopen") + def test_string_input_sends_no_compression_headers(self, mock_urlopen): + mock_urlopen.return_value = self._ok() + Edgee("test-api-key").send(model="gpt-4", input="Hello") + + request, _ = self._sent(mock_urlopen) + assert not any(name.startswith("X-edgee-compression") for name in request.headers) + + @patch("edgee.urlopen") + def test_streaming_request_sends_toggles(self, mock_urlopen): + stream = MagicMock() + stream.__iter__ = MagicMock(return_value=iter([b"data: [DONE]\n"])) + stream.__enter__ = MagicMock(return_value=stream) + stream.__exit__ = MagicMock(return_value=False) + mock_urlopen.return_value = stream + + chunks = list( + Edgee("test-api-key").send( + model="gpt-4", + input={ + "messages": [{"role": "user", "content": "Hello"}], + "tool_surface_reduction": True, + }, + stream=True, + ) + ) + + assert chunks == [] + request, _ = self._sent(mock_urlopen) + assert request.get_header(self.SURFACE.capitalize()) == "true" + assert not request.has_header(self.TRIM.capitalize()) + + @patch("edgee.urlopen") + def test_compression_model_is_deprecated_but_still_sent(self, mock_urlopen): + mock_urlopen.return_value = self._ok() + with pytest.warns(DeprecationWarning, match="tool_result_trimming"): + Edgee("test-api-key").send( + model="gpt-4", + input={ + "messages": [{"role": "user", "content": "Hello"}], + "compression_model": "claude", + }, + ) + + _, body = self._sent(mock_urlopen) + assert body["compression_model"] == "claude"