Skip to content

Commit 5c0dac2

Browse files
committed
fix(streaming): preserve unknown usage instead of fabricating input_tokens=0
When both message_start and message_delta.input_tokens omit the input count, the accumulator reported input_tokens=0, which under-reports accounting to callers. Per review feedback, leave the count unset when it is genuinely unknown: construct_type already leaves fields the delta did not supply as None, matching how the rest of the SDK represents wire-omitted values. A later delta that does supply the count still fills it in via the accumulate branch.
1 parent d2dc3fe commit 5c0dac2

6 files changed

Lines changed: 120 additions & 10 deletions

File tree

‎src/anthropic/lib/streaming/_beta_messages.py‎

Lines changed: 5 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -563,11 +563,11 @@ def accumulate_event(
563563
# `message_start` may omit usage (see the streaming docs), in which
564564
# case the snapshot has no usage yet. Initialize it from the delta
565565
# so the final message still carries token counts, and tolerate
566-
# streams that never supply usage.
567-
usage = event.usage.to_dict()
568-
if event.usage.input_tokens is None:
569-
usage["input_tokens"] = 0
570-
current_snapshot.usage = construct_type(type_=BetaUsage, value=usage)
566+
# streams that never supply usage. Anything the delta leaves out
567+
# (e.g. `input_tokens`) stays unset instead of being fabricated
568+
# as 0 so an unknown count is never reported to callers; a later
569+
# delta that does supply it fills it in below.
570+
current_snapshot.usage = construct_type(type_=BetaUsage, value=event.usage.to_dict())
571571
else:
572572
current_snapshot.usage.output_tokens = event.usage.output_tokens
573573

‎src/anthropic/lib/streaming/_messages.py‎

Lines changed: 5 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -523,11 +523,11 @@ def accumulate_event(
523523
# `message_start` may omit usage (see the streaming docs), in which
524524
# case the snapshot has no usage yet. Initialize it from the delta
525525
# so the final message still carries token counts, and tolerate
526-
# streams that never supply usage.
527-
usage = event.usage.to_dict()
528-
if event.usage.input_tokens is None:
529-
usage["input_tokens"] = 0
530-
current_snapshot.usage = construct_type(type_=Usage, value=usage)
526+
# streams that never supply usage. Anything the delta leaves out
527+
# (e.g. `input_tokens`) stays unset instead of being fabricated
528+
# as 0 so an unknown count is never reported to callers; a later
529+
# delta that does supply it fills it in below.
530+
current_snapshot.usage = construct_type(type_=Usage, value=event.usage.to_dict())
531531
else:
532532
current_snapshot.usage.output_tokens = event.usage.output_tokens
533533

Lines changed: 17 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,17 @@
1+
event: message_start
2+
data: {"type":"message_start","message":{"id":"msg_usage_omitted_input","type":"message","role":"assistant","content":[],"model":"claude-test","stop_reason":null,"stop_sequence":null}}
3+
4+
event: content_block_start
5+
data: {"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}
6+
7+
event: content_block_delta
8+
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"Hello"}}
9+
10+
event: content_block_stop
11+
data: {"type":"content_block_stop","index":0}
12+
13+
event: message_delta
14+
data: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"output_tokens":6}}
15+
16+
event: message_stop
17+
data: {"type":"message_stop"}
Lines changed: 17 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,17 @@
1+
event: message_start
2+
data: {"type":"message_start","message":{"id":"msg_usage_omitted_input","type":"message","role":"assistant","content":[],"model":"claude-test","stop_reason":null,"stop_sequence":null}}
3+
4+
event: content_block_start
5+
data: {"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}
6+
7+
event: content_block_delta
8+
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"Hello"}}
9+
10+
event: content_block_stop
11+
data: {"type":"content_block_stop","index":0}
12+
13+
event: message_delta
14+
data: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"output_tokens":6}}
15+
16+
event: message_stop
17+
data: {"type":"message_stop"}

‎tests/lib/streaming/test_beta_messages.py‎

Lines changed: 23 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -590,6 +590,29 @@ def test_usage_omitted_at_message_start_uses_beta_usage_and_preserves_delta_opti
590590
assert message.usage.fallback_credit is not None
591591
assert message.usage.fallback_credit.status.type == "redeemed"
592592

593+
@pytest.mark.respx(base_url=base_url)
594+
def test_usage_omitted_at_message_start_preserves_unknown_input_tokens(self, respx_mock: MockRouter) -> None:
595+
# If both `message_start` and `message_delta` omit the input count then
596+
# it is genuinely unknown; the accumulator must preserve that rather
597+
# than fabricate `input_tokens=0`, which would under-report accounting.
598+
respx_mock.post("/v1/messages").mock(
599+
return_value=httpx2.Response(200, content=get_response("beta_usage_omitted_input_tokens_response.txt"))
600+
)
601+
602+
client = Anthropic(base_url=base_url, api_key=api_key)
603+
604+
with client.beta.messages.stream(
605+
max_tokens=1024,
606+
messages=[{"role": "user", "content": "Say hello there!"}],
607+
model="claude-test",
608+
) as stream:
609+
message = stream.get_final_message()
610+
611+
assert isinstance(message.usage, BetaUsage)
612+
# unknown counts stay unset instead of being reported as 0
613+
assert message.usage.input_tokens is None # pyright: ignore[reportUnnecessaryComparison]
614+
assert message.usage.output_tokens == 6
615+
593616
@pytest.mark.respx(base_url=base_url)
594617
@pytest.mark.parametrize("fixture, expected", INPUT_TRANSFORMATIONS_CASES)
595618
def test_input_transformations_propagated(

‎tests/lib/streaming/test_messages.py‎

Lines changed: 53 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -411,6 +411,31 @@ def test_usage_omitted_at_message_start(self, respx_mock: MockRouter) -> None:
411411
assert message.usage.input_tokens == 12
412412
assert message.usage.output_tokens == 6
413413

414+
@pytest.mark.respx(base_url=base_url)
415+
def test_usage_omitted_at_message_start_preserves_unknown_input_tokens(self, respx_mock: MockRouter) -> None:
416+
# If both `message_start` and `message_delta` omit the input count then
417+
# it is genuinely unknown; the accumulator must preserve that rather
418+
# than fabricate `input_tokens=0`, which would under-report accounting.
419+
respx_mock.post("/v1/messages").mock(
420+
return_value=httpx2.Response(200, content=get_response("usage_omitted_input_tokens_response.txt"))
421+
)
422+
423+
# A default (non-strict) client mirrors how the docs' event sequence
424+
# reaches the accumulator without response-validation rejecting it.
425+
client = Anthropic(base_url=base_url, api_key=api_key)
426+
427+
with client.messages.stream(
428+
max_tokens=1024,
429+
messages=[{"role": "user", "content": "Say hello there!"}],
430+
model="claude-test",
431+
) as stream:
432+
message = stream.get_final_message()
433+
434+
assert message.usage is not None
435+
# unknown counts stay unset instead of being reported as 0
436+
assert message.usage.input_tokens is None # pyright: ignore[reportUnnecessaryComparison]
437+
assert message.usage.output_tokens == 6
438+
414439
@pytest.mark.respx(base_url=base_url)
415440
def test_usage_omitted_at_message_start_preserves_delta_optional_usage_fields(
416441
self, respx_mock: MockRouter
@@ -662,6 +687,34 @@ async def test_usage_omitted_at_message_start(self, respx_mock: MockRouter) -> N
662687
assert message.usage.input_tokens == 12
663688
assert message.usage.output_tokens == 6
664689

690+
@pytest.mark.asyncio
691+
@pytest.mark.respx(base_url=base_url)
692+
async def test_usage_omitted_at_message_start_preserves_unknown_input_tokens(self, respx_mock: MockRouter) -> None:
693+
# If both `message_start` and `message_delta` omit the input count then
694+
# it is genuinely unknown; the accumulator must preserve that rather
695+
# than fabricate `input_tokens=0`, which would under-report accounting.
696+
respx_mock.post("/v1/messages").mock(
697+
return_value=httpx2.Response(
698+
200, content=to_async_iter(get_response("usage_omitted_input_tokens_response.txt"))
699+
)
700+
)
701+
702+
# A default (non-strict) client mirrors how the docs' event sequence
703+
# reaches the accumulator without response-validation rejecting it.
704+
client = AsyncAnthropic(base_url=base_url, api_key=api_key)
705+
706+
async with client.messages.stream(
707+
max_tokens=1024,
708+
messages=[{"role": "user", "content": "Say hello there!"}],
709+
model="claude-test",
710+
) as stream:
711+
message = await stream.get_final_message()
712+
713+
assert message.usage is not None
714+
# unknown counts stay unset instead of being reported as 0
715+
assert message.usage.input_tokens is None # pyright: ignore[reportUnnecessaryComparison]
716+
assert message.usage.output_tokens == 6
717+
665718

666719
def test_message_delta_fields_are_all_accumulated() -> None:
667720
# tripwire: handle a new field in accumulate_event (src/anthropic/lib/streaming/_messages.py), then list it here

0 commit comments

Comments
 (0)