diff --git a/manifests/java.yml b/manifests/java.yml index 4d1cd0cc716..4a4322e5d03 100644 --- a/manifests/java.yml +++ b/manifests/java.yml @@ -3286,6 +3286,7 @@ manifest: "*": irrelevant spring-boot: v1.63.0 tests/integration_frameworks/llm/anthropic/test_anthropic_llmobs.py::TestAnthropicLlmObsMessages::test_create_error: bug (MLOB-1234) + tests/integration_frameworks/llm/openai/test_openai_ai_guard.py: missing_feature (APPSEC-68977, AI Guard/OpenAI integration not yet wired into INTEGRATION_FRAMEWORKS) tests/integration_frameworks/llm/openai/test_openai_apm.py: v1.61.0 tests/integration_frameworks/llm/openai/test_openai_llmobs.py: v1.61.0 tests/integrations/crossed_integrations/test_kafka.py::Test_Kafka: diff --git a/manifests/nodejs.yml b/manifests/nodejs.yml index 8051d6a7457..bf1caf4d3bd 100644 --- a/manifests/nodejs.yml +++ b/manifests/nodejs.yml @@ -1752,6 +1752,7 @@ manifest: : bug (MLOB-5070) ? tests/integration_frameworks/llm/google_genai/test_google_genai_llmobs.py::TestGoogleGenAiGenerateContentWithTools::test_generate_content_with_tools : missing_feature (Node.js LLM Observability Google GenAI integration does not submit tool definitions) + tests/integration_frameworks/llm/openai/test_openai_ai_guard.py::TestOpenAiAiGuard: missing_feature (APPSEC-68977, AI Guard/OpenAI integration not yet wired into INTEGRATION_FRAMEWORKS) tests/integration_frameworks/llm/openai/test_openai_apm.py::TestOpenAiApmChatCompletions: *ref_5_76_0 tests/integration_frameworks/llm/openai/test_openai_apm.py::TestOpenAiApmCompletions: *ref_5_76_0 tests/integration_frameworks/llm/openai/test_openai_apm.py::TestOpenAiApmEmbeddings: *ref_5_76_0 diff --git a/manifests/python.yml b/manifests/python.yml index 3445067e84d..04b3e210e61 100644 --- a/manifests/python.yml +++ b/manifests/python.yml @@ -1366,6 +1366,7 @@ manifest: tests/integration_frameworks/llm/google_genai/test_google_genai_llmobs.py::TestGoogleGenAiGenerateContentWithTools: v3.13.0 ? tests/integration_frameworks/llm/google_genai/test_google_genai_llmobs.py::TestGoogleGenAiGenerateContentWithTools::test_generate_content_with_tools : bug (MLOB-5071) + tests/integration_frameworks/llm/openai/test_openai_ai_guard.py::TestOpenAiAiGuard: v4.10.0 tests/integration_frameworks/llm/openai/test_openai_apm.py::TestOpenAiApmChatCompletions: v3.13.0 tests/integration_frameworks/llm/openai/test_openai_apm.py::TestOpenAiApmCompletions: v3.13.0 tests/integration_frameworks/llm/openai/test_openai_apm.py::TestOpenAiApmEmbeddings: v3.13.0 diff --git a/tests/integration_frameworks/llm/openai/test_openai_ai_guard.py b/tests/integration_frameworks/llm/openai/test_openai_ai_guard.py new file mode 100644 index 00000000000..4c8aba64122 --- /dev/null +++ b/tests/integration_frameworks/llm/openai/test_openai_ai_guard.py @@ -0,0 +1,183 @@ +"""AI Guard <-> OpenAI integration tests, run under the INTEGRATION_FRAMEWORKS scenario. + +Investigation task: https://datadoghq.atlassian.net/browse/APPSEC-68977 + +Unlike ``tests/ai_guard/test_ai_guard_sdk.py`` (which drives the AI Guard SDK directly via +``/ai_guard/evaluate``), this suite exercises the *integration* between AI Guard and the +OpenAI client, the same way the LLM Observability suite does: it calls the OpenAI SDK +directly through the existing weblog endpoints (``/chat/completions``). When +``DD_AI_GUARD_ENABLED=true``, ``ai_guard_listen()`` auto-wires into the OpenAI SDK, so AI +Guard evaluates the call at three points with no manual ``evaluate()`` call: + +- **before-model**: the request/prompt is evaluated before the model is called; +- **tool-call**: tool calls produced by the model are evaluated. + +The after-model evaluation is intentionally not covered here: exercising it end-to-end needs +the streamed-response path (``DD_AI_GUARD_ANALYZE_STREAM_RESPONSES_ENABLED``), which is not +implemented across the other tracer libraries yet, so we keep this suite at cross-language +parity. + +We assert that the integration wires each evaluation point: that it emits an ``ai_guard`` +span for the specific evaluation being exercised (identified by ``ai_guard.target`` and by +the messages captured in ``meta_struct.ai_guard``) and tags the local root span with +``ai_guard.event:true``. We do not assert trace *linkage* to the ``openai.request`` span: +the tracer does not deterministically nest the ``ai_guard`` span in the OpenAI trace (it +may be emitted as its own trace), so a shared ``trace_id`` is not guaranteed. The +evaluation *outcome* (ALLOW / DENY / ABORT) is already covered by the ``AI_GUARD`` scenario +and is intentionally not re-asserted here. +""" + +import time + +import pytest +import requests + +from utils import features, scenarios +from utils.docker_fixtures import FrameworkTestClientApi, TestAgentAPI + +from .utils import TOOLS, BaseOpenaiTest + + +@pytest.fixture +def library_env() -> dict[str, str]: + # The AI Guard client also needs DD_API_KEY / DD_APP_KEY, but those are injected via the + # scenario environment (see IntegrationFrameworksScenario._required_cassette_generation_api_keys) + # rather than here: library_env is copied into the JSON report metadata, so keeping secrets + # out of it prevents real keys from leaking into logs/artifacts during cassette generation. + return { + "DD_AI_GUARD_ENABLED": "true", + } + + +def _ai_guard_spans(traces: list[list[dict]]) -> list[dict]: + return [span for trace in traces for span in trace if span.get("resource") == "ai_guard"] + + +def _guard_messages(span: dict) -> list[dict]: + """The messages AI Guard evaluated, as captured in ``meta_struct.ai_guard.messages``.""" + return span.get("meta_struct", {}).get("ai_guard", {}).get("messages", []) + + +def _ai_guard_event_root_spans(traces: list[list[dict]]) -> list[dict]: + """Local root (service-entry) spans tagged ``ai_guard.event:true``. + + When AI Guard evaluates a call it tags the trace's local root span with + ``ai_guard.event:true`` (dd-trace-py ``appsec/ai_guard/_api_client.py``). This is a + tracer-emitted marker that AI Guard ran on the trace, and is what we assert on here. + + Note: the ``_dd.ai_guard.enabled:1`` facet that is searchable in the Datadog UI is NOT + present in the raw payloads captured by the test agent (it is not emitted by the tracer; + it is produced somewhere in intake), so it cannot be asserted on directly. + """ + return [ + span + for trace in traces + for span in trace + if span.get("parent_id") in (0, None) and span.get("meta", {}).get("ai_guard.event", False) in (True, "true") + ] + + +def _wait_for_ai_guard_spans( + test_agent: TestAgentAPI, *, target: str | None = None, wait_loops: int = 30 +) -> list[dict]: + """Poll the test agent until at least one matching ``ai_guard`` span is received. + + We assert on the presence of the ``ai_guard`` span rather than on a fixed number of + traces: the tracer does not deterministically group the ``ai_guard`` span with the + OpenAI span. The ``ai_guard`` span may be emitted either nested in the OpenAI trace + (1 trace) or as its own trace (2 traces), so ``wait_for_num_traces`` with a hard-coded + count is inherently racy. When ``target`` is given, only spans whose ``ai_guard.target`` + matches are considered (so we keep polling until the specific evaluation point we care + about has arrived). + """ + spans: list[dict] = [] + for _ in range(wait_loops): + try: + traces = test_agent.traces(clear=False) + except requests.exceptions.RequestException: + pass + else: + spans = _ai_guard_spans(traces) + if target is not None: + spans = [span for span in spans if span["meta"].get("ai_guard.target") == target] + if spans: + return spans + time.sleep(0.1) + return spans + + +def _wait_for_ai_guard_event_root_spans(test_agent: TestAgentAPI, *, wait_loops: int = 30) -> list[dict]: + """Poll the test agent until at least one root span tagged ``ai_guard.event:true`` arrives. + + Like the ``ai_guard`` span itself, the tagged local root span may land in a later trace + chunk than the evaluation span, so we poll rather than reading a single snapshot. + """ + spans: list[dict] = [] + for _ in range(wait_loops): + try: + traces = test_agent.traces(clear=False) + except requests.exceptions.RequestException: + pass + else: + spans = _ai_guard_event_root_spans(traces) + if spans: + return spans + time.sleep(0.1) + return spans + + +@features.ai_guard +@scenarios.integration_frameworks +class TestOpenAiAiGuard(BaseOpenaiTest): + """AI Guard evaluation triggered through the auto-instrumented OpenAI integration.""" + + def test_before_model_validation(self, test_agent: TestAgentAPI, test_client: FrameworkTestClientApi): + """The prompt is evaluated by AI Guard before the OpenAI model is called.""" + with test_agent.vcr_context(): + test_client.request( + "POST", + "/chat/completions", + dict( + model="gpt-4o-mini", + messages=[{"role": "user", "content": "What is the weather like today?"}], + parameters=dict(max_tokens=35), + ), + ) + + guard_spans = _wait_for_ai_guard_spans(test_agent, target="prompt") + assert guard_spans, "expected a before-model ai_guard span with target 'prompt'" + + event_root_spans = _wait_for_ai_guard_event_root_spans(test_agent) + assert event_root_spans, "expected a local root span tagged ai_guard.event:true" + + def test_tool_call_validation(self, test_agent: TestAgentAPI, test_client: FrameworkTestClientApi): + """Tool calls produced by the model are evaluated by AI Guard.""" + with test_agent.vcr_context(): + test_client.request( + "POST", + "/chat/completions", + dict( + model="gpt-4o-mini", + messages=[ + { + "role": "user", + "content": "Bob is a student at Stanford University. He is studying computer science.", + } + ], + parameters=dict(tool_choice="auto", tools=TOOLS), + ), + ) + + guard_spans = _wait_for_ai_guard_spans(test_agent, target="tool") + assert guard_spans, "expected a tool-call ai_guard span with target 'tool'" + # ``target == "tool"`` alone can also come from an ordinary after-model eval of an + # assistant response, so require the assistant tool_calls entry to actually be in the + # payload sent to AI Guard - that is what proves the tool-call path was forwarded. + assert any( + msg.get("role") == "assistant" and msg.get("tool_calls") + for span in guard_spans + for msg in _guard_messages(span) + ), "expected the assistant tool_calls entry in the ai_guard evaluation payload" + + event_root_spans = _wait_for_ai_guard_event_root_spans(test_agent) + assert event_root_spans, "expected a local root span tagged ai_guard.event:true" diff --git a/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_before_model_validation_aiguard_evaluate_post_ca44130a.json b/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_before_model_validation_aiguard_evaluate_post_ca44130a.json new file mode 100644 index 00000000000..c555d5cebbf --- /dev/null +++ b/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_before_model_validation_aiguard_evaluate_post_ca44130a.json @@ -0,0 +1,38 @@ +{ + "request": { + "method": "POST", + "url": "https://app.datadoghq.com/api/v2/ai-guard/evaluate", + "headers": { + "Accept-Encoding": "identity", + "Content-Length": "148", + "Content-Type": "application/json", + "DD-AI-GUARD-VERSION": "4.10.6", + "DD-AI-GUARD-SOURCE": "SDK", + "DD-AI-GUARD-LANGUAGE": "python", + "Datadog-Entity-ID": "in-115200" + }, + "body": "{\"data\": {\"attributes\": {\"messages\": [{\"role\": \"user\", \"content\": \"What is the weather like today?\"}], \"meta\": {\"service\": \"openai\", \"env\": null}}}}" + }, + "response": { + "status": { + "code": 200, + "message": "OK" + }, + "headers": { + "content-security-policy": "frame-ancestors 'self'; report-uri https://logs.browser-intake-datadoghq.com/api/v2/logs?dd-api-key=pube4f163c23bbf91c16b8f57f56af9fc58&dd-evp-origin=content-security-policy&ddsource=csp-report&ddtags=site%3Adatadoghq.com", + "content-type": "application/vnd.api+json", + "vary": "Accept-Encoding", + "x-frame-options": "SAMEORIGIN", + "content-length": "216", + "date": "Fri, 03 Jul 2026 14:49:53 GMT", + "x-content-type-options": "nosniff", + "strict-transport-security": "max-age=31536000; includeSubDomains; preload", + "x-ratelimit-limit": "5000", + "x-ratelimit-period": "60", + "x-ratelimit-remaining": "4999", + "x-ratelimit-reset": "7", + "x-ratelimit-name": "ai_guard_evaluate_per_org" + }, + "body": "{\"data\":{\"id\":\"f0863f0c-9c8c-4041-b178-d932f449d0f2\",\"type\":\"evaluations\",\"attributes\":{\"action\":\"ALLOW\",\"global_prob\":0.00244140625,\"is_blocking_enabled\":false,\"reason\":\"No rule match.\",\"tag_probs\":null,\"tags\":[]}}}" + } +} \ No newline at end of file diff --git a/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_before_model_validation_aiguard_evaluate_post_d13cc790.json b/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_before_model_validation_aiguard_evaluate_post_d13cc790.json new file mode 100644 index 00000000000..81871716f6c --- /dev/null +++ b/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_before_model_validation_aiguard_evaluate_post_d13cc790.json @@ -0,0 +1,38 @@ +{ + "request": { + "method": "POST", + "url": "https://app.datadoghq.com/api/v2/ai-guard/evaluate", + "headers": { + "Accept-Encoding": "identity", + "Content-Length": "394", + "Content-Type": "application/json", + "DD-AI-GUARD-VERSION": "4.10.6", + "DD-AI-GUARD-SOURCE": "SDK", + "DD-AI-GUARD-LANGUAGE": "python", + "Datadog-Entity-ID": "in-115200" + }, + "body": "{\"data\": {\"attributes\": {\"messages\": [{\"role\": \"user\", \"content\": \"What is the weather like today?\"}, {\"role\": \"assistant\", \"content\": \"I'm unable to provide real-time information, including current weather updates. I recommend checking a reliable weather website or app for the most accurate and up-to-date information about today's weather in\"}], \"meta\": {\"service\": \"openai\", \"env\": null}}}}" + }, + "response": { + "status": { + "code": 200, + "message": "OK" + }, + "headers": { + "content-security-policy": "frame-ancestors 'self'; report-uri https://logs.browser-intake-datadoghq.com/api/v2/logs?dd-api-key=pube4f163c23bbf91c16b8f57f56af9fc58&dd-evp-origin=content-security-policy&ddsource=csp-report&ddtags=site%3Adatadoghq.com", + "content-type": "application/vnd.api+json", + "vary": "Accept-Encoding", + "x-frame-options": "SAMEORIGIN", + "content-length": "221", + "date": "Fri, 03 Jul 2026 14:49:58 GMT", + "x-content-type-options": "nosniff", + "strict-transport-security": "max-age=31536000; includeSubDomains; preload", + "x-ratelimit-limit": "5000", + "x-ratelimit-period": "60", + "x-ratelimit-remaining": "4998", + "x-ratelimit-reset": "2", + "x-ratelimit-name": "ai_guard_evaluate_per_org" + }, + "body": "{\"data\":{\"id\":\"6f8dceb7-054d-4d1f-8e35-7c4487977669\",\"type\":\"evaluations\",\"attributes\":{\"action\":\"ALLOW\",\"global_prob\":0.0013275146484375,\"is_blocking_enabled\":false,\"reason\":\"No rule match.\",\"tag_probs\":null,\"tags\":[]}}}" + } +} \ No newline at end of file diff --git a/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_tool_call_validation_aiguard_evaluate_post_9e556a11.json b/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_tool_call_validation_aiguard_evaluate_post_9e556a11.json new file mode 100644 index 00000000000..c0d782f31d9 --- /dev/null +++ b/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_tool_call_validation_aiguard_evaluate_post_9e556a11.json @@ -0,0 +1,38 @@ +{ + "request": { + "method": "POST", + "url": "https://app.datadoghq.com/api/v2/ai-guard/evaluate", + "headers": { + "Accept-Encoding": "identity", + "Content-Length": "417", + "Content-Type": "application/json", + "DD-AI-GUARD-VERSION": "4.10.6", + "DD-AI-GUARD-SOURCE": "SDK", + "DD-AI-GUARD-LANGUAGE": "python", + "Datadog-Entity-ID": "in-115716" + }, + "body": "{\"data\": {\"attributes\": {\"messages\": [{\"role\": \"user\", \"content\": \"Bob is a student at Stanford University. He is studying computer science.\"}, {\"role\": \"assistant\", \"tool_calls\": [{\"id\": \"call_C5dXIRBkTQjYoQrBr3WrkQYY\", \"function\": {\"name\": \"extract_student_info\", \"arguments\": \"{\\\"name\\\":\\\"Bob\\\",\\\"major\\\":\\\"computer science\\\",\\\"school\\\":\\\"Stanford University\\\"}\"}}]}], \"meta\": {\"service\": \"openai\", \"env\": null}}}}" + }, + "response": { + "status": { + "code": 200, + "message": "OK" + }, + "headers": { + "content-security-policy": "frame-ancestors 'self'; report-uri https://logs.browser-intake-datadoghq.com/api/v2/logs?dd-api-key=pube4f163c23bbf91c16b8f57f56af9fc58&dd-evp-origin=content-security-policy&ddsource=csp-report&ddtags=site%3Adatadoghq.com", + "content-type": "application/vnd.api+json", + "vary": "Accept-Encoding", + "x-frame-options": "SAMEORIGIN", + "content-length": "219", + "date": "Fri, 03 Jul 2026 14:50:13 GMT", + "x-content-type-options": "nosniff", + "strict-transport-security": "max-age=31536000; includeSubDomains; preload", + "x-ratelimit-limit": "5000", + "x-ratelimit-period": "60", + "x-ratelimit-remaining": "4997", + "x-ratelimit-reset": "47", + "x-ratelimit-name": "ai_guard_evaluate_per_org" + }, + "body": "{\"data\":{\"id\":\"4aefdbd6-761e-4bb5-ba69-f543586e8b4e\",\"type\":\"evaluations\",\"attributes\":{\"action\":\"ALLOW\",\"global_prob\":0.00628662109375,\"is_blocking_enabled\":false,\"reason\":\"No rule match.\",\"tag_probs\":null,\"tags\":[]}}}" + } +} \ No newline at end of file diff --git a/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_tool_call_validation_aiguard_evaluate_post_a304603c.json b/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_tool_call_validation_aiguard_evaluate_post_a304603c.json new file mode 100644 index 00000000000..87774a918b2 --- /dev/null +++ b/tests/integration_frameworks/utils/vcr-cassettes/aiguard/test_tool_call_validation_aiguard_evaluate_post_a304603c.json @@ -0,0 +1,38 @@ +{ + "request": { + "method": "POST", + "url": "https://app.datadoghq.com/api/v2/ai-guard/evaluate", + "headers": { + "Accept-Encoding": "identity", + "Content-Length": "190", + "Content-Type": "application/json", + "DD-AI-GUARD-VERSION": "4.10.6", + "DD-AI-GUARD-SOURCE": "SDK", + "DD-AI-GUARD-LANGUAGE": "python", + "Datadog-Entity-ID": "in-115716" + }, + "body": "{\"data\": {\"attributes\": {\"messages\": [{\"role\": \"user\", \"content\": \"Bob is a student at Stanford University. He is studying computer science.\"}], \"meta\": {\"service\": \"openai\", \"env\": null}}}}" + }, + "response": { + "status": { + "code": 200, + "message": "OK" + }, + "headers": { + "content-security-policy": "frame-ancestors 'self'; report-uri https://logs.browser-intake-datadoghq.com/api/v2/logs?dd-api-key=pube4f163c23bbf91c16b8f57f56af9fc58&dd-evp-origin=content-security-policy&ddsource=csp-report&ddtags=site%3Adatadoghq.com", + "content-type": "application/vnd.api+json", + "vary": "Accept-Encoding", + "x-frame-options": "SAMEORIGIN", + "content-length": "219", + "date": "Fri, 03 Jul 2026 14:50:11 GMT", + "x-content-type-options": "nosniff", + "strict-transport-security": "max-age=31536000; includeSubDomains; preload", + "x-ratelimit-limit": "5000", + "x-ratelimit-period": "60", + "x-ratelimit-remaining": "4998", + "x-ratelimit-reset": "49", + "x-ratelimit-name": "ai_guard_evaluate_per_org" + }, + "body": "{\"data\":{\"id\":\"c3372620-7e64-47c1-bbd6-63a2c664334c\",\"type\":\"evaluations\",\"attributes\":{\"action\":\"ALLOW\",\"global_prob\":0.00885009765625,\"is_blocking_enabled\":false,\"reason\":\"No rule match.\",\"tag_probs\":null,\"tags\":[]}}}" + } +} \ No newline at end of file diff --git a/tests/integration_frameworks/utils/vcr-cassettes/openai/test_before_model_validation_openai_chat_completions_post_eb8df755.json b/tests/integration_frameworks/utils/vcr-cassettes/openai/test_before_model_validation_openai_chat_completions_post_eb8df755.json new file mode 100644 index 00000000000..3db865a00a4 --- /dev/null +++ b/tests/integration_frameworks/utils/vcr-cassettes/openai/test_before_model_validation_openai_chat_completions_post_eb8df755.json @@ -0,0 +1,63 @@ +{ + "request": { + "method": "POST", + "url": "https://api.openai.com/v1/chat/completions", + "headers": { + "Accept-Encoding": "gzip, deflate", + "Connection": "keep-alive", + "Accept": "application/json", + "Content-Type": "application/json", + "User-Agent": "OpenAI/Python 2.0.0", + "X-Stainless-Lang": "python", + "X-Stainless-Package-Version": "2.0.0", + "X-Stainless-OS": "Linux", + "X-Stainless-Arch": "x64", + "X-Stainless-Runtime": "CPython", + "X-Stainless-Runtime-Version": "3.11.15", + "X-Stainless-Async": "false", + "x-stainless-retry-count": "0", + "x-stainless-read-timeout": "600", + "Content-Length": "112", + "x-datadog-trace-id": "11307714882692163746", + "x-datadog-parent-id": "2798732504086163091", + "x-datadog-sampling-priority": "2", + "x-datadog-tags": "_dd.p.dm=-4,_dd.p.tid=6a47cc1100000000", + "traceparent": "00-6a47cc11000000009ced12c66443a8a2-26d7186257652e93-01", + "tracestate": "dd=p:26d7186257652e93;s:2;t.dm:-4;t.tid:6a47cc1100000000" + }, + "body": "{\"messages\":[{\"role\":\"user\",\"content\":\"What is the weather like today?\"}],\"model\":\"gpt-4o-mini\",\"max_tokens\":35}" + }, + "response": { + "status": { + "code": 200, + "message": "OK" + }, + "headers": { + "Date": "Fri, 03 Jul 2026 14:49:58 GMT", + "Content-Type": "application/json", + "Transfer-Encoding": "chunked", + "Connection": "keep-alive", + "CF-Ray": "a156b3111ac3ae85-MAD", + "CF-Cache-Status": "DYNAMIC", + "Server": "cloudflare", + "Strict-Transport-Security": "max-age=31536000; includeSubDomains; preload", + "X-Content-Type-Options": "nosniff", + "access-control-expose-headers": "X-Request-ID, CF-Ray, CF-Ray", + "openai-processing-ms": "1215", + "openai-project": "proj_gt6TQZPRbZfoY2J9AQlEJMpd", + "openai-version": "2020-10-01", + "x-openai-proxy-wasm": "v0.1", + "x-ratelimit-limit-requests": "30000", + "x-ratelimit-limit-tokens": "150000000", + "x-ratelimit-remaining-requests": "29999", + "x-ratelimit-remaining-tokens": "149999990", + "x-ratelimit-reset-requests": "2ms", + "x-ratelimit-reset-tokens": "0s", + "x-request-id": "req_9263056278434d2696c121eb01e0a54f", + "set-cookie": "__cf_bm=Lg_OtVyDALy0TiXdy_TD_RwWwe69PHvFgttVvQIfG3M-1783090194.0993586-1.0.1.1-KoSSM0CzYR9uaV2VpejfgbplM9KWTnym3Z2bra1WakC0HAaN3YJpAZQT8DX__hhXuwI3ce1Va7ppxNf4OEeBgpJLoHRzP_4oHta5ndbd7L.IAwnCIz2Ba0jRF2cCLS.5; HttpOnly; SameSite=None; Secure; Path=/; Domain=api.openai.com; Expires=Fri, 03 Jul 2026 15:19:58 GMT", + "Content-Encoding": "gzip", + "alt-svc": "h3=\":443\"; ma=86400" + }, + "body": "{\n \"id\": \"chatcmpl-DxZVFdYaeYYcW0e9vLEFWDoRazh3p\",\n \"object\": \"chat.completion\",\n \"created\": 1783090197,\n \"model\": \"gpt-4o-mini-2024-07-18\",\n \"choices\": [\n {\n \"index\": 0,\n \"message\": {\n \"role\": \"assistant\",\n \"content\": \"I'm unable to provide real-time information, including current weather updates. I recommend checking a reliable weather website or app for the most accurate and up-to-date information about today's weather in\",\n \"refusal\": null,\n \"annotations\": []\n },\n \"logprobs\": null,\n \"finish_reason\": \"length\"\n }\n ],\n \"usage\": {\n \"prompt_tokens\": 14,\n \"completion_tokens\": 35,\n \"total_tokens\": 49,\n \"prompt_tokens_details\": {\n \"cached_tokens\": 0,\n \"audio_tokens\": 0\n },\n \"completion_tokens_details\": {\n \"reasoning_tokens\": 0,\n \"audio_tokens\": 0,\n \"accepted_prediction_tokens\": 0,\n \"rejected_prediction_tokens\": 0\n }\n },\n \"service_tier\": \"default\",\n \"system_fingerprint\": \"fp_b4b9deb02a\"\n}\n" + } +} \ No newline at end of file diff --git a/tests/integration_frameworks/utils/vcr-cassettes/openai/test_tool_call_validation_openai_chat_completions_post_f2dffea2.json b/tests/integration_frameworks/utils/vcr-cassettes/openai/test_tool_call_validation_openai_chat_completions_post_f2dffea2.json new file mode 100644 index 00000000000..8427c9e7cf7 --- /dev/null +++ b/tests/integration_frameworks/utils/vcr-cassettes/openai/test_tool_call_validation_openai_chat_completions_post_f2dffea2.json @@ -0,0 +1,63 @@ +{ + "request": { + "method": "POST", + "url": "https://api.openai.com/v1/chat/completions", + "headers": { + "Accept-Encoding": "gzip, deflate", + "Connection": "keep-alive", + "Accept": "application/json", + "Content-Type": "application/json", + "User-Agent": "OpenAI/Python 2.0.0", + "X-Stainless-Lang": "python", + "X-Stainless-Package-Version": "2.0.0", + "X-Stainless-OS": "Linux", + "X-Stainless-Arch": "x64", + "X-Stainless-Runtime": "CPython", + "X-Stainless-Runtime-Version": "3.11.15", + "X-Stainless-Async": "false", + "x-stainless-retry-count": "0", + "x-stainless-read-timeout": "600", + "Content-Length": "535", + "x-datadog-trace-id": "11679812058723169118", + "x-datadog-parent-id": "17144347262350322105", + "x-datadog-sampling-priority": "2", + "x-datadog-tags": "_dd.p.dm=-4,_dd.p.tid=6a47cc2300000000", + "traceparent": "00-6a47cc2300000000a217072f6384b35e-edecf5001e54e9b9-01", + "tracestate": "dd=p:edecf5001e54e9b9;s:2;t.dm:-4;t.tid:6a47cc2300000000" + }, + "body": "{\"messages\":[{\"role\":\"user\",\"content\":\"Bob is a student at Stanford University. He is studying computer science.\"}],\"model\":\"gpt-4o-mini\",\"tool_choice\":\"auto\",\"tools\":[{\"type\":\"function\",\"function\":{\"name\":\"extract_student_info\",\"description\":\"Get the student information from the body of the input text\",\"parameters\":{\"type\":\"object\",\"properties\":{\"name\":{\"type\":\"string\",\"description\":\"Name of the person\"},\"major\":{\"type\":\"string\",\"description\":\"Major subject.\"},\"school\":{\"type\":\"string\",\"description\":\"The university name.\"}}}}}]}" + }, + "response": { + "status": { + "code": 200, + "message": "OK" + }, + "headers": { + "Date": "Fri, 03 Jul 2026 14:50:12 GMT", + "Content-Type": "application/json", + "Transfer-Encoding": "chunked", + "Connection": "keep-alive", + "CF-Ray": "a156b3815dbe0142-MAD", + "CF-Cache-Status": "DYNAMIC", + "Server": "cloudflare", + "Strict-Transport-Security": "max-age=31536000; includeSubDomains; preload", + "X-Content-Type-Options": "nosniff", + "access-control-expose-headers": "X-Request-ID, CF-Ray, CF-Ray", + "openai-processing-ms": "690", + "openai-project": "proj_gt6TQZPRbZfoY2J9AQlEJMpd", + "openai-version": "2020-10-01", + "x-openai-proxy-wasm": "v0.1", + "x-ratelimit-limit-requests": "30000", + "x-ratelimit-limit-tokens": "150000000", + "x-ratelimit-remaining-requests": "29999", + "x-ratelimit-remaining-tokens": "149999980", + "x-ratelimit-reset-requests": "2ms", + "x-ratelimit-reset-tokens": "0s", + "x-request-id": "req_9b5983560f2c46209300bbc7dc7d1398", + "set-cookie": "__cf_bm=lFOi_XfqoatTdp8UUQviTl2ntLZoPNYZNNUPOmYoIS4-1783090212.0518818-1.0.1.1-8LZKca4JkZ_Z6AmofdkXp3ls.q2mUD2uefWtfGYruqEynmongOkfGw5EZnfsFEASpchoGpnXbCKtqo0tXp49uTvHNVExyvAYfGy1SUoqiGwyC7GwLjkxauXRvOdNrwjZ; HttpOnly; SameSite=None; Secure; Path=/; Domain=api.openai.com; Expires=Fri, 03 Jul 2026 15:20:12 GMT", + "Content-Encoding": "gzip", + "alt-svc": "h3=\":443\"; ma=86400" + }, + "body": "{\n \"id\": \"chatcmpl-DxZVUSEzvfuZVXYVyOwL7CSxKngPg\",\n \"object\": \"chat.completion\",\n \"created\": 1783090212,\n \"model\": \"gpt-4o-mini-2024-07-18\",\n \"choices\": [\n {\n \"index\": 0,\n \"message\": {\n \"role\": \"assistant\",\n \"content\": null,\n \"tool_calls\": [\n {\n \"id\": \"call_C5dXIRBkTQjYoQrBr3WrkQYY\",\n \"type\": \"function\",\n \"function\": {\n \"name\": \"extract_student_info\",\n \"arguments\": \"{\\\"name\\\":\\\"Bob\\\",\\\"major\\\":\\\"computer science\\\",\\\"school\\\":\\\"Stanford University\\\"}\"\n }\n }\n ],\n \"refusal\": null,\n \"annotations\": []\n },\n \"logprobs\": null,\n \"finish_reason\": \"tool_calls\"\n }\n ],\n \"usage\": {\n \"prompt_tokens\": 85,\n \"completion_tokens\": 26,\n \"total_tokens\": 111,\n \"prompt_tokens_details\": {\n \"cached_tokens\": 0,\n \"audio_tokens\": 0\n },\n \"completion_tokens_details\": {\n \"reasoning_tokens\": 0,\n \"audio_tokens\": 0,\n \"accepted_prediction_tokens\": 0,\n \"rejected_prediction_tokens\": 0\n }\n },\n \"service_tier\": \"default\",\n \"system_fingerprint\": \"fp_8d97543466\"\n}\n" + } +} \ No newline at end of file diff --git a/utils/_context/_scenarios/integration_frameworks.py b/utils/_context/_scenarios/integration_frameworks.py index 6783a6b672e..1247456a05b 100644 --- a/utils/_context/_scenarios/integration_frameworks.py +++ b/utils/_context/_scenarios/integration_frameworks.py @@ -21,7 +21,12 @@ class IntegrationFrameworksScenario(DockerFixturesScenario): _test_client_factory: FrameworkTestClientFactory _required_cassette_generation_api_keys: dict[str, list[str]] = { - "openai": ["OPENAI_API_KEY"], + # DD_API_KEY / DD_APP_KEY are needed by the AI Guard client, which calls the real + # AI Guard API while recording. Declaring them here means they are injected into the + # container via the scenario environment (never through ``library_env``, which is + # serialized into the JSON report) and that a missing key fails fast at configure() + # time instead of surfacing as a silent xfail during cassette generation. + "openai": ["OPENAI_API_KEY", "DD_API_KEY", "DD_APP_KEY"], "anthropic": ["ANTHROPIC_API_KEY"], "google_genai": ["GEMINI_API_KEY"], } diff --git a/utils/docker_fixtures/_test_agent.py b/utils/docker_fixtures/_test_agent.py index a8e3a2593f8..c6c89355e8b 100644 --- a/utils/docker_fixtures/_test_agent.py +++ b/utils/docker_fixtures/_test_agent.py @@ -110,6 +110,10 @@ def start_agent( "ENABLED_CHECKS": "trace_count_header", "OTLP_HTTP_PORT": str(container_otlp_http_port), "OTLP_GRPC_PORT": str(container_otlp_grpc_port), + # Register the AI Guard REST API as a VCR upstream so the AI Guard/OpenAI integration + # can be replayed. Additive: extends the image's built-in providers (openai, anthropic, + # ...) without overriding them (see dd-apm-test-agent vcr_proxy.PROVIDER_BASE_URLS). + "VCR_PROVIDER_MAP": "aiguard=https://app.datadoghq.com/api/v2/ai-guard", "VCR_CASSETTES_DIRECTORY": "/vcr-cassettes", } if os.getenv("DEV_MODE") is not None: diff --git a/utils/docker_fixtures/_test_clients/_test_client_framework_integrations.py b/utils/docker_fixtures/_test_clients/_test_client_framework_integrations.py index 39bfa5692a6..16c7f0b4fb6 100644 --- a/utils/docker_fixtures/_test_clients/_test_client_framework_integrations.py +++ b/utils/docker_fixtures/_test_clients/_test_client_framework_integrations.py @@ -57,6 +57,9 @@ def get_client( environment["DD_AGENT_HOST"] = test_agent.container_name environment["DD_TRACE_AGENT_PORT"] = str(test_agent.container_port) environment["FRAMEWORK_TEST_CLIENT_SERVER_PORT"] = str(container_port) + # AI Guard evaluations (when DD_AI_GUARD_ENABLED=true) are replayed by the test-agent VCR proxy. + # Harmless when AI Guard is disabled, which is the default for framework tests. + environment["DD_AI_GUARD_ENDPOINT"] = f"{environment['DD_TRACE_AGENT_URL']}/vcr/aiguard" # overwrite env with the one provided by the test environment |= library_env