Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
35 changes: 32 additions & 3 deletions edgee/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
import json
import os
import ssl
import warnings
from dataclasses import dataclass
from urllib.error import HTTPError
from urllib.request import Request, urlopen
Expand All @@ -11,6 +12,22 @@
DEFAULT_BASE_URL = "https://edgee.io"
API_ENDPOINT = "/v1/chat/completions"

# Per-request compression toggles, keyed by their InputObject field name.
COMPRESSION_HEADERS = {
"tool_result_trimming": "X-Edgee-Compression-Tool-Result-Trimming",
"tool_surface_reduction": "X-Edgee-Compression-Tool-Surface-Reduction",
"output_brevity": "X-Edgee-Compression-Brevity",
}


def _compression_headers(toggles: dict) -> dict[str, str]:
"""Headers for the toggles the caller set. Only real bools count; None means keep the key setting."""
return {
COMPRESSION_HEADERS[field]: "true" if value else "false"
for field, value in toggles.items()
if isinstance(value, bool)
}


def _ssl_context() -> ssl.SSLContext:
"""Create SSL context. Uses certifi's CA bundle when available (fixes cert issues on macOS)."""
Expand Down Expand Up @@ -59,9 +76,12 @@ class InputObject:
tools: list[dict] | None = None
tool_choice: str | dict | None = None
tags: list[str] | None = None
compression_model: str | None = (
None # Compression model: claude, opencode, cursor, customer (gateway-internal)
)
# Deprecated: any value turns tool-result trimming on. Use tool_result_trimming instead.
compression_model: str | None = None
# Per-request overrides of the API key settings. None keeps the key setting.
tool_result_trimming: bool | None = None
tool_surface_reduction: bool | None = None
output_brevity: bool | None = None


@dataclass
Expand Down Expand Up @@ -216,18 +236,21 @@ def send(
tool_choice = None
tags = None
compression_model = None
toggles = {}
elif isinstance(input, InputObject):
messages = input.messages
tools = input.tools
tool_choice = input.tool_choice
tags = input.tags
compression_model = input.compression_model
toggles = {field: getattr(input, field) for field in COMPRESSION_HEADERS}
else:
messages = input.get("messages", [])
tools = input.get("tools")
tool_choice = input.get("tool_choice")
tags = input.get("tags")
compression_model = input.get("compression_model")
toggles = {field: input.get(field) for field in COMPRESSION_HEADERS}

body: dict = {"model": model, "messages": messages}
if stream:
Expand All @@ -239,6 +262,11 @@ def send(
if tags:
body["tags"] = tags
if compression_model is not None:
warnings.warn(
"compression_model is deprecated; use tool_result_trimming=True instead",
DeprecationWarning,
stacklevel=2,
)
body["compression_model"] = compression_model

request = Request(
Expand All @@ -247,6 +275,7 @@ def send(
headers={
"Content-Type": "application/json",
"Authorization": f"Bearer {self.api_key}",
**_compression_headers(toggles),
},
method="POST",
)
Expand Down
114 changes: 47 additions & 67 deletions example/compression.py
Original file line number Diff line number Diff line change
@@ -1,13 +1,14 @@
"""Example: Token compression with Edgee Gateway SDK

This example demonstrates how to:
1. Enable compression for a request with a large input context
2. Set a custom compression rate
3. Access compression metrics from the response

IMPORTANT: Only USER messages are compressed. System messages are not compressed.
This example includes a large context in the user message to demonstrate meaningful
compression savings.
1. Turn tool-result trimming on for a single request
2. Access compression metrics from the response

Tool-result trimming shortens the output of tool calls (here a long `ls -la`
listing) before it reaches the model. The per-request toggles
(`tool_result_trimming`, `tool_surface_reduction`, `output_brevity`) override
the API key settings for this request only; leave one out to keep the key's
setting.
"""

import os
Expand All @@ -21,78 +22,57 @@
# Initialize the client
edgee = Edgee(os.environ.get("EDGEE_API_KEY"))

# Large context document to demonstrate input compression
LARGE_CONTEXT = """
The History and Impact of Artificial Intelligence

Artificial intelligence (AI) has evolved from a theoretical concept to a
transformative technology that influences nearly every aspect of modern life.
The field began in earnest in the 1950s when pioneers like Alan Turing and
John McCarthy laid the groundwork for machine intelligence.

Early developments focused on symbolic reasoning and expert systems. These
rule-based approaches dominated the field through the 1970s and 1980s, with
systems like MYCIN demonstrating practical applications in medical diagnosis.
However, these early systems were limited by their inability to learn from data
and adapt to new situations.

The resurgence of neural networks in the 1980s and 1990s, particularly with
backpropagation algorithms, opened new possibilities. Yet it wasn't until the
2010s, with the advent of deep learning and the availability of massive datasets
and computational power, that AI truly began to revolutionize industries.

Modern AI applications span numerous domains:
- Natural language processing enables machines to understand and generate human language
- Computer vision allows machines to interpret visual information from the world
- Robotics combines AI with mechanical systems for autonomous operation
- Healthcare uses AI for diagnosis, drug discovery, and personalized treatment
- Finance leverages AI for fraud detection, algorithmic trading, and risk assessment
- Transportation is being transformed by autonomous vehicles and traffic optimization

The development of large language models like GPT, BERT, and others has
particularly accelerated progress in natural language understanding and generation.
These models, trained on vast amounts of text data, can perform a wide range of
language tasks with remarkable proficiency.

Despite remarkable progress, significant challenges remain. Issues of bias,
interpretability, safety, and ethical considerations continue to be areas of
active research and debate. The AI community is working to ensure that these
powerful technologies are developed and deployed responsibly, with consideration
for their societal impact.

Looking forward, AI is expected to continue advancing rapidly, with potential
breakthroughs in areas like artificial general intelligence, quantum machine
learning, and brain-computer interfaces. The integration of AI into daily life
will likely deepen, raising important questions about human-AI collaboration,
workforce transformation, and the future of human cognition itself.
"""
# A long directory listing, the kind of tool output coding agents send back.
LS_OUTPUT = "total 800\n" + "\n".join(
f"-rw-r--r-- 1 user staff {1000 + i} Jan 1 12:00 src/components/module_{i:03}.tsx"
for i in range(200)
)

print("=" * 70)
print("Edgee Token Compression Example")
print("=" * 70)
print()

# Example: Request with compression enabled and large input
print("Example: Large user message with compression enabled")
print("Example: Large tool result with tool-result trimming turned on")
print("-" * 70)
print(f"Input context length: {len(LARGE_CONTEXT)} characters")
print(f"Tool output length: {len(LS_OUTPUT)} characters")
print()

# NOTE: Only USER messages are compressed
# Put the large context in the user message to demonstrate compression
user_message = f"""Here is some context about AI:

{LARGE_CONTEXT}

Based on this context, summarize the key milestones in AI development in 3 bullet points."""

response = edgee.send(
model="anthropic/claude-haiku-4-5",
input={
"messages": [
{"role": "user", "content": user_message},
{"role": "user", "content": "How many files are in src/components?"},
{
"role": "assistant",
"tool_calls": [
{
"id": "call_1",
"type": "function",
"function": {
"name": "Bash",
"arguments": '{"command":"ls -la src/components"}',
},
}
],
},
{"role": "tool", "tool_call_id": "call_1", "content": LS_OUTPUT},
],
"tools": [
{
"type": "function",
"function": {
"name": "Bash",
"description": "Run a shell command and return its output.",
"parameters": {
"type": "object",
"properties": {"command": {"type": "string"}},
"required": ["command"],
},
},
}
],
"compression_model": "claude",
"tool_result_trimming": True,
},
)

Expand Down Expand Up @@ -123,8 +103,8 @@
print(f" With compression, only {tokens_after} tokens were processed!")
else:
print("No compression data available in response.")
print("Note: Compression data is only returned when compression is enabled")
print(" and supported by your API key configuration.")
print("Note: Compression data is only returned when trimming actually shortened")
print(" a tool result.")

print()
print("=" * 70)
120 changes: 120 additions & 0 deletions tests/test_edgee.py
Original file line number Diff line number Diff line change
Expand Up @@ -355,3 +355,123 @@ def test_send_without_compression_response(self, mock_urlopen):
result = client.send(model="gpt-4", input="Test")

assert result.compression is None


class TestCompressionOverrides:
"""Per-request compression toggles are sent as headers, never in the body"""

TRIM = "X-Edgee-Compression-Tool-Result-Trimming"
SURFACE = "X-Edgee-Compression-Tool-Surface-Reduction"
BREVITY = "X-Edgee-Compression-Brevity"

def _mock_response(self, data: dict):
mock = MagicMock()
mock.read.return_value = json.dumps(data).encode("utf-8")
mock.__enter__ = MagicMock(return_value=mock)
mock.__exit__ = MagicMock(return_value=False)
return mock

def _ok(self):
return self._mock_response(
{
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "ok"},
"finish_reason": "stop",
}
]
}
)

def _sent(self, mock_urlopen):
request = mock_urlopen.call_args[0][0]
return request, json.loads(request.data.decode("utf-8"))

@patch("edgee.urlopen")
def test_input_object_toggles_become_headers(self, mock_urlopen):
from edgee import InputObject

mock_urlopen.return_value = self._ok()
Edgee("test-api-key").send(
model="gpt-4",
input=InputObject(
messages=[{"role": "user", "content": "Hello"}],
tool_result_trimming=True,
tool_surface_reduction=False,
output_brevity=True,
),
)

request, body = self._sent(mock_urlopen)
# urllib normalizes header names with str.capitalize().
assert request.get_header(self.TRIM.capitalize()) == "true"
assert request.get_header(self.SURFACE.capitalize()) == "false"
assert request.get_header(self.BREVITY.capitalize()) == "true"
for field in ("tool_result_trimming", "tool_surface_reduction", "output_brevity"):
assert field not in body

@patch("edgee.urlopen")
def test_dict_toggles_and_unset_fields(self, mock_urlopen):
mock_urlopen.return_value = self._ok()
Edgee("test-api-key").send(
model="gpt-4",
input={
"messages": [{"role": "user", "content": "Hello"}],
"tool_result_trimming": False,
# Not a bool: ignored, so the key setting applies.
"output_brevity": "yes",
},
)

request, _ = self._sent(mock_urlopen)
assert request.get_header(self.TRIM.capitalize()) == "false"
assert not request.has_header(self.SURFACE.capitalize())
assert not request.has_header(self.BREVITY.capitalize())

@patch("edgee.urlopen")
def test_string_input_sends_no_compression_headers(self, mock_urlopen):
mock_urlopen.return_value = self._ok()
Edgee("test-api-key").send(model="gpt-4", input="Hello")

request, _ = self._sent(mock_urlopen)
assert not any(name.startswith("X-edgee-compression") for name in request.headers)

@patch("edgee.urlopen")
def test_streaming_request_sends_toggles(self, mock_urlopen):
stream = MagicMock()
stream.__iter__ = MagicMock(return_value=iter([b"data: [DONE]\n"]))
stream.__enter__ = MagicMock(return_value=stream)
stream.__exit__ = MagicMock(return_value=False)
mock_urlopen.return_value = stream

chunks = list(
Edgee("test-api-key").send(
model="gpt-4",
input={
"messages": [{"role": "user", "content": "Hello"}],
"tool_surface_reduction": True,
},
stream=True,
)
)

assert chunks == []
request, _ = self._sent(mock_urlopen)
assert request.get_header(self.SURFACE.capitalize()) == "true"
assert not request.has_header(self.TRIM.capitalize())

@patch("edgee.urlopen")
def test_compression_model_is_deprecated_but_still_sent(self, mock_urlopen):
mock_urlopen.return_value = self._ok()
with pytest.warns(DeprecationWarning, match="tool_result_trimming"):
Edgee("test-api-key").send(
model="gpt-4",
input={
"messages": [{"role": "user", "content": "Hello"}],
"compression_model": "claude",
},
)

_, body = self._sent(mock_urlopen)
assert body["compression_model"] == "claude"
Loading