Skip to content

Commit 0c39434

Browse files
feat(typesafe): Record gen_ai.input.messages
1 parent 0fdca7e commit 0c39434

2 files changed

Lines changed: 328 additions & 2 deletions

File tree

‎sentry_sdk/integrations/typesafe.py‎

Lines changed: 151 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,6 @@
1+
import json
12
from functools import wraps
2-
from typing import TYPE_CHECKING
3+
from typing import TYPE_CHECKING, cast
34

45
import sentry_sdk
56
from sentry_sdk.ai.utils import get_start_span_function
@@ -8,9 +9,40 @@
89
from sentry_sdk.tracing_utils import has_span_streaming_enabled
910

1011
if TYPE_CHECKING:
11-
from typing import Any, Callable
12+
from typing import (
13+
Any,
14+
Callable,
15+
Literal,
16+
Mapping,
17+
NotRequired,
18+
Sequence,
19+
TypedDict,
20+
Union,
21+
)
22+
23+
from typesafe_sdk import JSONContent
24+
from typesafe_sdk._core.question_types import Questions
25+
26+
class NoulModel(TypedDict):
27+
type: Literal["noul"]
28+
name: str
29+
instructions: NotRequired[JSONContent | None]
30+
31+
class ChoiceModel(TypedDict):
32+
type: Literal["choice"]
33+
name: str
34+
criteria: NotRequired[Mapping[str, JSONContent | None]]
35+
instructions: NotRequired[JSONContent | None]
36+
37+
class ScoreModel(TypedDict):
38+
type: Literal["score"]
39+
name: str
40+
criteria: NotRequired[Sequence[JSONContent]]
41+
instructions: NotRequired[JSONContent | None]
42+
1243

1344
try:
45+
from typesafe_sdk import Choice, Noul, Score
1446
from typesafe_sdk._core.client.aio.client import AsyncTypeSafeClient
1547
from typesafe_sdk._core.client.sync.client import TypeSafeClient
1648
from typesafe_sdk._core.response_types import SystemOneResponse
@@ -30,6 +62,95 @@ def setup_once() -> None:
3062
)
3163

3264

65+
def _transform_questions(
66+
questions: "Questions",
67+
) -> "list[Union[NoulModel, ChoiceModel, ScoreModel]]":
68+
transformed_questions: "list[Union[NoulModel, ChoiceModel, ScoreModel]]" = []
69+
for name, question in questions.items():
70+
if isinstance(question, Noul):
71+
noul: "NoulModel" = {
72+
"type": "noul",
73+
"name": name,
74+
}
75+
76+
if question.instructions is not None:
77+
noul["instructions"] = question.instructions
78+
79+
transformed_questions.append(noul)
80+
continue
81+
82+
if isinstance(question, Choice):
83+
choice: "ChoiceModel" = {
84+
"type": "choice",
85+
"name": name,
86+
"criteria": question.criteria,
87+
}
88+
89+
if question.instructions is not None:
90+
choice["instructions"] = question.instructions
91+
92+
transformed_questions.append(choice)
93+
continue
94+
95+
if isinstance(question, Score):
96+
score: "ScoreModel" = {
97+
"type": "score",
98+
"name": name,
99+
"criteria": question.criteria,
100+
}
101+
102+
if question.instructions is not None:
103+
score["instructions"] = question.instructions
104+
105+
transformed_questions.append(score)
106+
continue
107+
108+
if not isinstance(question, dict):
109+
continue
110+
111+
question_type = question.get("type")
112+
if question_type == "noul":
113+
noul = {
114+
"type": question_type,
115+
"name": name,
116+
}
117+
if "instructions" in question:
118+
noul["instructions"] = question["instructions"]
119+
120+
transformed_questions.append(noul)
121+
continue
122+
123+
if question_type == "choice":
124+
choice = {
125+
"type": question_type,
126+
"name": name,
127+
}
128+
if "criteria" in question:
129+
choice["criteria"] = cast(
130+
"Mapping[str, JSONContent | None]", question["criteria"]
131+
)
132+
if "instructions" in question:
133+
choice["instructions"] = question["instructions"]
134+
135+
transformed_questions.append(choice)
136+
continue
137+
138+
if question_type == "score":
139+
score = {
140+
"type": question_type,
141+
"name": name,
142+
}
143+
if "criteria" in question:
144+
score["criteria"] = cast("Sequence[JSONContent]", question["criteria"])
145+
if "instructions" in question:
146+
score["instructions"] = question["instructions"]
147+
148+
transformed_questions.append(score)
149+
continue
150+
151+
return transformed_questions
152+
153+
33154
def _wrap_system_one(f: "Callable[..., Any]") -> "Callable[..., Any]":
34155
@wraps(f)
35156
def wrap_system_one(self: "TypeSafeClient", *args: "Any", **kwargs: "Any") -> "Any":
@@ -72,6 +193,20 @@ def wrap_system_one(self: "TypeSafeClient", *args: "Any", **kwargs: "Any") -> "A
72193
if model is not None:
73194
set_on_span(SPANDATA.GEN_AI_REQUEST_MODEL, model)
74195

196+
state = kwargs.get("state")
197+
questions = kwargs.get("questions")
198+
if state is not None and questions:
199+
set_on_span(
200+
SPANDATA.GEN_AI_INPUT_MESSAGES,
201+
json.dumps(
202+
{
203+
"type": "evaluation",
204+
"state": state,
205+
"questions": _transform_questions(questions),
206+
}
207+
),
208+
)
209+
75210
response = f(self, *args, **kwargs)
76211

77212
if not isinstance(response, SystemOneResponse):
@@ -126,6 +261,20 @@ async def wrap_system_one_async(
126261
if model is not None:
127262
set_on_span(SPANDATA.GEN_AI_REQUEST_MODEL, model)
128263

264+
state = kwargs.get("state")
265+
questions = kwargs.get("questions")
266+
if state is not None and questions:
267+
set_on_span(
268+
SPANDATA.GEN_AI_INPUT_MESSAGES,
269+
json.dumps(
270+
{
271+
"type": "evaluation",
272+
"state": state,
273+
"questions": _transform_questions(questions),
274+
}
275+
),
276+
)
277+
129278
response = await f(self, *args, **kwargs)
130279

131280
if not isinstance(response, SystemOneResponse):

‎tests/integrations/typesafe/test_typesafe.py‎

Lines changed: 177 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
1+
import json
12
from unittest import mock
23

34
import pytest
@@ -119,6 +120,50 @@ def test_system_one(
119120
assert span["attributes"][SPANDATA.GEN_AI_REQUEST_MODEL] == "jev-latest"
120121

121122
assert span["attributes"][SPANDATA.GEN_AI_RESPONSE_MODEL] == "jev-latest"
123+
124+
assert json.loads(span["attributes"][SPANDATA.GEN_AI_INPUT_MESSAGES]) == {
125+
"type": "evaluation",
126+
"state": {
127+
"subject": "Charged twice this month",
128+
"body": "I see two charges of $49. I only have one account. Please fix this ASAP.",
129+
},
130+
"questions": [
131+
{
132+
"type": "noul",
133+
"name": "spam",
134+
"instructions": "Spam?",
135+
},
136+
{
137+
"type": "choice",
138+
"name": "tone",
139+
"instructions": "Tone?",
140+
"criteria": {"friendly": None, "hostile": None},
141+
},
142+
{
143+
"type": "score",
144+
"name": "quality",
145+
"instructions": "Quality?",
146+
"criteria": ["bad", "ok", "great"],
147+
},
148+
{
149+
"type": "noul",
150+
"name": "spam_obj",
151+
"instructions": "Spam?",
152+
},
153+
{
154+
"type": "choice",
155+
"name": "tone_obj",
156+
"instructions": "Tone?",
157+
"criteria": {"friendly": None, "hostile": None},
158+
},
159+
{
160+
"type": "score",
161+
"name": "quality_obj",
162+
"instructions": "Quality?",
163+
"criteria": ["bad", "ok", "great"],
164+
},
165+
],
166+
}
122167
else:
123168
items = capture_items("transaction")
124169

@@ -163,6 +208,50 @@ def test_system_one(
163208

164209
assert span["data"][SPANDATA.GEN_AI_RESPONSE_MODEL] == "jev-latest"
165210

211+
assert json.loads(span["data"][SPANDATA.GEN_AI_INPUT_MESSAGES]) == {
212+
"type": "evaluation",
213+
"state": {
214+
"subject": "Charged twice this month",
215+
"body": "I see two charges of $49. I only have one account. Please fix this ASAP.",
216+
},
217+
"questions": [
218+
{
219+
"type": "noul",
220+
"name": "spam",
221+
"instructions": "Spam?",
222+
},
223+
{
224+
"type": "choice",
225+
"name": "tone",
226+
"instructions": "Tone?",
227+
"criteria": {"friendly": None, "hostile": None},
228+
},
229+
{
230+
"type": "score",
231+
"name": "quality",
232+
"instructions": "Quality?",
233+
"criteria": ["bad", "ok", "great"],
234+
},
235+
{
236+
"type": "noul",
237+
"name": "spam_obj",
238+
"instructions": "Spam?",
239+
},
240+
{
241+
"type": "choice",
242+
"name": "tone_obj",
243+
"instructions": "Tone?",
244+
"criteria": {"friendly": None, "hostile": None},
245+
},
246+
{
247+
"type": "score",
248+
"name": "quality_obj",
249+
"instructions": "Quality?",
250+
"criteria": ["bad", "ok", "great"],
251+
},
252+
],
253+
}
254+
166255

167256
@pytest.mark.asyncio
168257
@pytest.mark.parametrize("span_streaming", [True, False])
@@ -234,6 +323,50 @@ async def test_system_one_async(
234323
assert span["attributes"][SPANDATA.GEN_AI_REQUEST_MODEL] == "jev-latest"
235324

236325
assert span["attributes"][SPANDATA.GEN_AI_RESPONSE_MODEL] == "jev-latest"
326+
327+
assert json.loads(span["attributes"][SPANDATA.GEN_AI_INPUT_MESSAGES]) == {
328+
"type": "evaluation",
329+
"state": {
330+
"subject": "Charged twice this month",
331+
"body": "I see two charges of $49. I only have one account. Please fix this ASAP.",
332+
},
333+
"questions": [
334+
{
335+
"type": "noul",
336+
"name": "spam",
337+
"instructions": "Spam?",
338+
},
339+
{
340+
"type": "choice",
341+
"name": "tone",
342+
"instructions": "Tone?",
343+
"criteria": {"friendly": None, "hostile": None},
344+
},
345+
{
346+
"type": "score",
347+
"name": "quality",
348+
"instructions": "Quality?",
349+
"criteria": ["bad", "ok", "great"],
350+
},
351+
{
352+
"type": "noul",
353+
"name": "spam_obj",
354+
"instructions": "Spam?",
355+
},
356+
{
357+
"type": "choice",
358+
"name": "tone_obj",
359+
"instructions": "Tone?",
360+
"criteria": {"friendly": None, "hostile": None},
361+
},
362+
{
363+
"type": "score",
364+
"name": "quality_obj",
365+
"instructions": "Quality?",
366+
"criteria": ["bad", "ok", "great"],
367+
},
368+
],
369+
}
237370
else:
238371
items = capture_items("transaction")
239372

@@ -280,3 +413,47 @@ async def test_system_one_async(
280413
assert span["data"][SPANDATA.GEN_AI_REQUEST_MODEL] == "jev-latest"
281414

282415
assert span["data"][SPANDATA.GEN_AI_RESPONSE_MODEL] == "jev-latest"
416+
417+
assert json.loads(span["data"][SPANDATA.GEN_AI_INPUT_MESSAGES]) == {
418+
"type": "evaluation",
419+
"state": {
420+
"subject": "Charged twice this month",
421+
"body": "I see two charges of $49. I only have one account. Please fix this ASAP.",
422+
},
423+
"questions": [
424+
{
425+
"type": "noul",
426+
"name": "spam",
427+
"instructions": "Spam?",
428+
},
429+
{
430+
"type": "choice",
431+
"name": "tone",
432+
"instructions": "Tone?",
433+
"criteria": {"friendly": None, "hostile": None},
434+
},
435+
{
436+
"type": "score",
437+
"name": "quality",
438+
"instructions": "Quality?",
439+
"criteria": ["bad", "ok", "great"],
440+
},
441+
{
442+
"type": "noul",
443+
"name": "spam_obj",
444+
"instructions": "Spam?",
445+
},
446+
{
447+
"type": "choice",
448+
"name": "tone_obj",
449+
"instructions": "Tone?",
450+
"criteria": {"friendly": None, "hostile": None},
451+
},
452+
{
453+
"type": "score",
454+
"name": "quality_obj",
455+
"instructions": "Quality?",
456+
"criteria": ["bad", "ok", "great"],
457+
},
458+
],
459+
}

0 commit comments

Comments
 (0)