From 11056d775b6c492e0283e6535be998316041fb7f Mon Sep 17 00:00:00 2001 From: VihaanAgarwal Date: Mon, 17 Aug 2026 12:15:23 +0530 Subject: [PATCH] fix(litellm): Set operation name from call type instead of chat fallback The litellm integration always fell back to the chat operation name, so text completions, embeddings and responses calls were all recorded as chat. Map the operation from the call type instead, and skip instrumentation for call types we do not know. --- sentry_sdk/integrations/litellm.py | 43 ++++---- tests/integrations/litellm/test_litellm.py | 116 ++++++++++++++++++++- 2 files changed, 140 insertions(+), 19 deletions(-) diff --git a/sentry_sdk/integrations/litellm.py b/sentry_sdk/integrations/litellm.py index 74adddb7ec..794803f78f 100644 --- a/sentry_sdk/integrations/litellm.py +++ b/sentry_sdk/integrations/litellm.py @@ -22,7 +22,7 @@ if TYPE_CHECKING: from datetime import datetime - from typing import Any, Dict, List + from typing import Any, Dict, List, Tuple try: import litellm # type: ignore[import-not-found] @@ -35,6 +35,19 @@ # to every callback, so it lives and dies with the request. _SPAN_KEY = "_sentry_span" +# Call types whose gen_ai operation name we can determine accurately. Everything +# else is not instrumented, since guessing records wrong data. +_CALL_TYPE_OPERATIONS: "Dict[Any, Tuple[str, str]]" = { + "completion": ("chat", consts.OP.GEN_AI_CHAT), + "acompletion": ("chat", consts.OP.GEN_AI_CHAT), + "text_completion": ("text_completion", consts.OP.GEN_AI_TEXT_COMPLETION), + "atext_completion": ("text_completion", consts.OP.GEN_AI_TEXT_COMPLETION), + "embedding": ("embeddings", consts.OP.GEN_AI_EMBEDDINGS), + "aembedding": ("embeddings", consts.OP.GEN_AI_EMBEDDINGS), + "responses": ("responses", consts.OP.GEN_AI_RESPONSES), + "aresponses": ("responses", consts.OP.GEN_AI_RESPONSES), +} + def _store_span(kwargs: "Dict[str, Any]", span: "Any") -> None: kwargs[_SPAN_KEY] = span @@ -83,6 +96,12 @@ def _input_callback(kwargs: "Dict[str, Any]") -> None: if integration is None: return + call_type = kwargs.get("call_type", None) + if call_type not in _CALL_TYPE_OPERATIONS: + return + + operation, span_op = _CALL_TYPE_OPERATIONS[call_type] + # Get key parameters full_model = kwargs.get("model", "") try: @@ -91,33 +110,21 @@ def _input_callback(kwargs: "Dict[str, Any]") -> None: model = full_model provider = "unknown" - call_type = kwargs.get("call_type", None) - if call_type == "embedding" or call_type == "aembedding": - operation = "embeddings" - else: - operation = "chat" + span_name = f"{operation} {model}" # Start a new span/transaction if has_span_streaming_enabled(client.options): span = sentry_sdk.traces.start_span( - name=f"{operation} {model}", + name=span_name, attributes={ - "sentry.op": ( - consts.OP.GEN_AI_CHAT - if operation == "chat" - else consts.OP.GEN_AI_EMBEDDINGS - ), + "sentry.op": span_op, "sentry.origin": LiteLLMIntegration.origin, }, ) else: span = get_start_span_function()( - op=( - consts.OP.GEN_AI_CHAT - if operation == "chat" - else consts.OP.GEN_AI_EMBEDDINGS - ), - name=f"{operation} {model}", + op=span_op, + name=span_name, origin=LiteLLMIntegration.origin, ) span.__enter__() diff --git a/tests/integrations/litellm/test_litellm.py b/tests/integrations/litellm/test_litellm.py index 8358151580..8ba0745010 100644 --- a/tests/integrations/litellm/test_litellm.py +++ b/tests/integrations/litellm/test_litellm.py @@ -34,7 +34,8 @@ async def __call__(self, *args, **kwargs): from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from openai import AsyncOpenAI, OpenAI -from openai.types import CompletionUsage +from openai.types import Completion, CompletionUsage +from openai.types.completion_choice import CompletionChoice from sentry_sdk import start_transaction from sentry_sdk._types import BLOB_DATA_SUBSTITUTE @@ -2651,6 +2652,7 @@ def test_response_without_usage( kwargs = { "model": "gpt-3.5-turbo", "messages": messages, + "call_type": "completion", } _input_callback(kwargs) @@ -2674,6 +2676,7 @@ def test_response_without_usage( kwargs = { "model": "gpt-3.5-turbo", "messages": messages, + "call_type": "completion", } _input_callback(kwargs) @@ -2733,6 +2736,7 @@ def test_litellm_message_truncation(sentry_init, capture_events): kwargs = { "model": "gpt-3.5-turbo", "messages": messages, + "call_type": "completion", } _input_callback(kwargs) @@ -3847,3 +3851,113 @@ def test_embeddings_data_collection( assert span_data[SPANDATA.GEN_AI_OPERATION_NAME] == "embeddings" assert span_data[SPANDATA.GEN_AI_REQUEST_MODEL] == "text-embedding-ada-002" assert span_data[SPANDATA.GEN_AI_USAGE_INPUT_TOKENS] == 5 + + +def test_text_completion_operation_name( + sentry_init, + capture_events, + get_model_response, + reset_litellm_executor, +): + """text_completion calls get the text_completion op and record their prompt.""" + sentry_init( + integrations=[LiteLLMIntegration(include_prompts=True)], + disabled_integrations=[StdlibIntegration], + traces_sample_rate=1.0, + send_default_pii=True, + stream_gen_ai_spans=False, + ) + events = capture_events() + + client = OpenAI(api_key="test-key") + + model_response = get_model_response( + Completion( + id="cmpl-test", + choices=[ + CompletionChoice(finish_reason="stop", index=0, text="Test response") + ], + created=1234567890, + model="gpt-3.5-turbo-instruct", + object="text_completion", + usage=CompletionUsage( + prompt_tokens=10, + completion_tokens=20, + total_tokens=30, + ), + ), + serialize_pydantic=True, + request_headers={"X-Stainless-Raw-Response": "true"}, + ) + + with mock.patch.object( + client.completions._client._client, + "send", + return_value=model_response, + ), start_transaction(name="litellm test"): + litellm.text_completion( + model="gpt-3.5-turbo-instruct", + prompt="Hello!", + client=client, + ) + + litellm_utils.executor.shutdown(wait=True) + + (event,) = events + (span,) = [s for s in event["spans"] if s["origin"] == "auto.ai.litellm"] + + assert span["op"] == OP.GEN_AI_TEXT_COMPLETION + assert span["description"] == "text_completion gpt-3.5-turbo-instruct" + assert span["data"][SPANDATA.GEN_AI_OPERATION_NAME] == "text_completion" + assert json.loads(span["data"][SPANDATA.GEN_AI_REQUEST_MESSAGES]) == [ + {"role": "user", "content": "Hello!"} + ] + + +def test_responses_operation_name( + sentry_init, + capture_events, + get_model_response, + nonstreaming_responses_model_response, + reset_litellm_executor, +): + """Responses API calls get the responses op and record their input.""" + sentry_init( + integrations=[LiteLLMIntegration(include_prompts=True)], + disabled_integrations=[StdlibIntegration], + traces_sample_rate=1.0, + send_default_pii=True, + stream_gen_ai_spans=False, + ) + events = capture_events() + + client = HTTPHandler() + + model_response = get_model_response( + nonstreaming_responses_model_response, + serialize_pydantic=True, + ) + + with mock.patch.object( + client, + "post", + return_value=model_response, + ), start_transaction(name="litellm test"): + litellm.responses( + model="gpt-4", + input="Hello!", + client=client, + api_key="test-key", + ) + + litellm_utils.executor.shutdown(wait=True) + + (event,) = events + (span,) = [s for s in event["spans"] if s["origin"] == "auto.ai.litellm"] + + assert span["op"] == OP.GEN_AI_RESPONSES + assert span["description"] == "responses gpt-4" + assert span["data"][SPANDATA.GEN_AI_OPERATION_NAME] == "responses" + assert json.loads(span["data"][SPANDATA.GEN_AI_REQUEST_MESSAGES]) == [ + {"role": "user", "content": "Hello!"} + ]