diff --git a/sentry_sdk/consts.py b/sentry_sdk/consts.py index e8874cd184..96f7b873ac 100644 --- a/sentry_sdk/consts.py +++ b/sentry_sdk/consts.py @@ -369,6 +369,42 @@ class SPANDATA: Example: "79b9da39-b7ae-508a-a6bc-864b2829c622" """ + AWS_S3_BUCKET = "aws.s3.bucket" + """ + The S3 bucket name the request refers to. + Example: "ot-demo-test" + """ + + AWS_S3_COPY_SOURCE = "aws.s3.copy_source" + """ + The source object (in the form bucket/key) for the copy operation. + Example: "someFile.yml" + """ + + AWS_S3_DELETE = "aws.s3.delete" + """ + The delete request container that specifies the objects to be deleted. + Example: "Objects=[{Key=string,VersionId=string},{Key=string,VersionId=string}],Quiet=boolean" + """ + + AWS_S3_KEY = "aws.s3.key" + """ + The S3 object key the request refers to. Corresponds to the --key parameter of the S3 API operations. + Example: "someFile.yml" + """ + + AWS_S3_PART_NUMBER = "aws.s3.part_number" + """ + The part number of the part being uploaded in a multipart-upload operation. This is a positive integer between 1 and 10,000. + Example: 3456 + """ + + AWS_S3_UPLOAD_ID = "aws.s3.upload_id" + """ + Upload ID that identifies the multipart upload. + Example: "dfRtDYWFbkRONycy.Yxwh66Yjlx.cph0gtNBtJ" + """ + CACHE_HIT = "cache.hit" """ A boolean indicating whether the requested data was found in the cache. @@ -539,6 +575,12 @@ class SPANDATA: Example: "timeout" """ + FILE_SIZE = "file.size" + """ + File size in bytes. + Example: 1024 + """ + GEN_AI_AGENT_NAME = "gen_ai.agent.name" """ The name of the agent being used. @@ -908,6 +950,12 @@ class SPANDATA: Example: ?foo=bar&bar=baz """ + HTTP_BODY_SIZE = "http.response.body.size" + """ + The encoded body size of the response (in bytes). + Example: 123 + """ + HTTP_STATUS_CODE = "http.response.status_code" """ The HTTP status code as an integer. diff --git a/sentry_sdk/integrations/boto3/_client.py b/sentry_sdk/integrations/boto3/_client.py index a1ef105613..10a5bb5c93 100644 --- a/sentry_sdk/integrations/boto3/_client.py +++ b/sentry_sdk/integrations/boto3/_client.py @@ -100,11 +100,9 @@ def sentry_patched_make_api_call( return orig_make_api_call(self, operation_name, api_params) # activate without finishing; a streaming response may outlive the call. - span_ctx = _activate_client_span(span) - attributes: "Attributes" = {} try: - with span_ctx: + with _activate_client_span(span): try: parsed = orig_make_api_call(self, operation_name, api_params) except BaseException as error: diff --git a/sentry_sdk/integrations/boto3/_instrumentation.py b/sentry_sdk/integrations/boto3/_instrumentation.py index 8a05e3c544..e3ee4fd7d6 100644 --- a/sentry_sdk/integrations/boto3/_instrumentation.py +++ b/sentry_sdk/integrations/boto3/_instrumentation.py @@ -207,7 +207,13 @@ def _instrument_streaming_body(span: "Span", parsed: "Dict[str, Any]") -> bool: if isinstance(span, NoOpSpan): return False - body = parsed.get("Body") + # botocore uses service-specific response key for streaming payload; e.g. `Payload` + # for Lambda Invoke, `Body` for S3 GetObject. Find it by type so every streaming + # response is handled. + body = next( + (value for value in parsed.values() if isinstance(value, StreamingBody)), + None, + ) if not isinstance(body, StreamingBody): return False diff --git a/sentry_sdk/integrations/boto3/_services/registry.py b/sentry_sdk/integrations/boto3/_services/registry.py index b387db0f26..f9291d04df 100644 --- a/sentry_sdk/integrations/boto3/_services/registry.py +++ b/sentry_sdk/integrations/boto3/_services/registry.py @@ -1,5 +1,7 @@ from typing import TYPE_CHECKING +from sentry_sdk.integrations.boto3._services.s3 import _S3Extension + if TYPE_CHECKING: from typing import Dict, Optional @@ -10,7 +12,9 @@ # _SERVICE_EXTENSIONS = {"s3": _S3Extension()} # when py 3.15 drops, we might want to take a look at using # a lazy-loading approach using the new `lazy` keyword. -_SERVICE_EXTENSIONS: "Dict[str, _ServiceExtension]" = {} +_SERVICE_EXTENSIONS: "Dict[str, _ServiceExtension]" = { + "s3": _S3Extension(), +} def _resolve_service( diff --git a/sentry_sdk/integrations/boto3/_services/s3.py b/sentry_sdk/integrations/boto3/_services/s3.py new file mode 100644 index 0000000000..1c2a6626e1 --- /dev/null +++ b/sentry_sdk/integrations/boto3/_services/s3.py @@ -0,0 +1,107 @@ +import json +from datetime import datetime +from typing import TYPE_CHECKING + +from sentry_sdk.consts import SPANDATA +from sentry_sdk.integrations.boto3._services.base import _ServiceExtension +from sentry_sdk.integrations.boto3._utils import ( + _extract_attributes, +) +from sentry_sdk.utils import capture_internal_exceptions + +if TYPE_CHECKING: + from typing import Any, Sequence + + from sentry_sdk._types import Attributes + from sentry_sdk.integrations.boto3._context import AwsCallContext + from sentry_sdk.integrations.boto3._utils import ( + _AttributeSpec, + ) + +# maps s3 operations to the response field that contains the complete file size. +_RESPONSE_FILE_SIZE_FIELDS = { + "GetObjectAttributes": "ObjectSize", + "PutObject": "Size", +} + +# s3 request attributes that are extracted when present, regardless of operation. +_REQUEST_ATTRIBUTES: "Sequence[_AttributeSpec]" = ( + ("Bucket", SPANDATA.AWS_S3_BUCKET), + ("Key", SPANDATA.AWS_S3_KEY), + ("UploadId", SPANDATA.AWS_S3_UPLOAD_ID), +) + + +class _S3Extension(_ServiceExtension): + __slots__ = () + + def get_request_attributes(self, ctx: "AwsCallContext") -> "Attributes": + attributes: "Attributes" = _extract_attributes(ctx.params, _REQUEST_ATTRIBUTES) + + if "CopySource" in ctx.params: + with capture_internal_exceptions(): + # boto3 accepts either "bucket/key" or a dict {"bucket": ..., "Key": ..., "VersionId": ...} for `CopySource`. + # https://docs.aws.amazon.com/boto3/latest/reference/services/s3/client/upload_part_copy.html + copy_source = ctx.params["CopySource"] + if isinstance(copy_source, str): + attributes[SPANDATA.AWS_S3_COPY_SOURCE] = copy_source + else: + value = f"{copy_source['Bucket']}/{copy_source['Key']}" + if "VersionId" in copy_source: + value += f"?versionId={copy_source['VersionId']}" + attributes[SPANDATA.AWS_S3_COPY_SOURCE] = value + + # OTel defines `PartNumber` for `UploadPart` and `UploadPartCopy` only. + # https://opentelemetry.io/docs/specs/semconv/object-stores/s3/#attributes + if ( + ctx.operation_name in ("UploadPart", "UploadPartCopy") + and "PartNumber" in ctx.params + ): + attributes[SPANDATA.AWS_S3_PART_NUMBER] = ctx.params["PartNumber"] + + if "Delete" in ctx.params: + with capture_internal_exceptions(): + attributes[SPANDATA.AWS_S3_DELETE] = json.dumps( + ctx.params["Delete"], + default=lambda value: ( + value.isoformat() if isinstance(value, datetime) else str(value) + ), + separators=(",", ":"), + sort_keys=True, + ) + + if ( + ctx.operation_name == "CompleteMultipartUpload" + and "MpuObjectSize" in ctx.params + ): + attributes[SPANDATA.FILE_SIZE] = ctx.params["MpuObjectSize"] + + return attributes + + def get_response_attributes( + self, ctx: "AwsCallContext", response: "Any" + ) -> "Attributes": + attributes: "Attributes" = {} + + if ( + ctx.operation_name in ("GetObject", "GetObjectAnnotation") + and "ContentLength" in response + ): + # `ContentLength` is the size of the HTTP body returned, which may be a range. + attributes[SPANDATA.HTTP_BODY_SIZE] = response["ContentLength"] + + # report the complete file size, not just the HTTP body size. + file_size_field = _RESPONSE_FILE_SIZE_FIELDS.get(ctx.operation_name) + if file_size_field is not None and file_size_field in response: + attributes[SPANDATA.FILE_SIZE] = response[file_size_field] + + if ( + ctx.operation_name == "HeadObject" + and "Range" not in ctx.params + and "PartNumber" not in ctx.params + and "ContentLength" in response + ): + # an un-ranged `HEAD` has no body, so `ContentLength` is the file size. + attributes[SPANDATA.FILE_SIZE] = response["ContentLength"] + + return attributes diff --git a/sentry_sdk/integrations/boto3/_utils.py b/sentry_sdk/integrations/boto3/_utils.py new file mode 100644 index 0000000000..73d1031c0f --- /dev/null +++ b/sentry_sdk/integrations/boto3/_utils.py @@ -0,0 +1,19 @@ +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from typing import Any, Dict, Sequence, Tuple + + from sentry_sdk._types import Attributes + + # tuple of (request param name, span attribute name). + _AttributeSpec = Tuple[str, str] + + +def _extract_attributes( + source: "Dict[str, Any]", specs: "Sequence[_AttributeSpec]" +) -> "Attributes": + attributes = {} + for param, attribute in specs: + if param in source: + attributes[attribute] = source[param] + return attributes diff --git a/tests/integrations/boto3/helpers.py b/tests/integrations/boto3/helpers.py new file mode 100644 index 0000000000..ec9d146c4d --- /dev/null +++ b/tests/integrations/boto3/helpers.py @@ -0,0 +1,115 @@ +from types import SimpleNamespace +from typing import TYPE_CHECKING, Dict, List, Optional + +import boto3 +import pytest +from botocore.config import Config + +import sentry_sdk +from sentry_sdk.consts import SPANDATA +from sentry_sdk.integrations.boto3 import Boto3Integration +from sentry_sdk.integrations.boto3.consts import ORIGIN + +if TYPE_CHECKING: + from sentry_sdk._types import SpanJSON + + +@pytest.fixture +def client_factory(sentry_init, monkeypatch): + sentry_init( + traces_sample_rate=1.0, + integrations=[Boto3Integration()], + ) + session = boto3.Session( # type: ignore + aws_access_key_id="-", + aws_secret_access_key="-", + region_name="eu-north-1", + ) + clients = [] + + def make_client(service_name="s3", attempt_count=1, **client_kwargs): + client = session.client( + service_name, + config=Config( + retries={"total_max_attempts": attempt_count, "mode": "standard"} + ), + **client_kwargs, + ) # type: ignore + clients.append(client) + return client + + yield make_client + + for client in clients: + # older supported botocore versions do not expose `BaseClient.close()`. + close = getattr(client, "close", None) + if close is not None: + close() + + +@pytest.fixture +def no_botocore_retry_delay(monkeypatch): + # remove request retry delays without replacing botocore's retry handling. + monkeypatch.setattr( + "botocore.endpoint.time", + SimpleNamespace(sleep=lambda delay: None), + ) + + +@pytest.fixture +def s3_client(client_factory): + return client_factory("s3") + + +def require_botocore_model_fields( + client, + method, + input_fields=(), + output_fields=(), +): + """Skip tests when botocore lacks required fields, including nested paths.""" + + def has_field(shape, field): + for part in field.split("."): + while shape is not None and shape.type_name == "list": + shape = shape.member + if shape is None or part not in getattr(shape, "members", {}): + return False + shape = shape.members[part] + return True + + operation_name = client.meta.method_to_api_mapping.get(method) + if operation_name is None: + pytest.skip("%s is absent from this botocore model; skipping test" % method) + model = client.meta.service_model.operation_model(operation_name) # type: ignore + for shape, fields in ( + (model.input_shape, input_fields), + (model.output_shape, output_fields), + ): + for field in fields: + if not has_field(shape, field): + pytest.skip( + "%s.%s is absent from this botocore model; skipping test" + % (method, field) + ) + + +def capture_spans_by_op( + invoke_client_method, + capture_items, + expected_origin=ORIGIN, +): + items = capture_items("span") + + with sentry_sdk.start_span(name="parent"): + invoke_client_method() + + sentry_sdk.flush() + spans_by_op: Dict[Optional[str], List["SpanJSON"]] = {} + for item in items: + span = item.payload + if span["attributes"].get(SPANDATA.SENTRY_ORIGIN) == expected_origin: + spans_by_op.setdefault( + span["attributes"].get(SPANDATA.SENTRY_OP), [] + ).append(span) + return spans_by_op diff --git a/tests/integrations/boto3/s3_list.xml b/tests/integrations/boto3/s3_list.xml index 10d5b16340..e3ca2073b3 100644 --- a/tests/integrations/boto3/s3_list.xml +++ b/tests/integrations/boto3/s3_list.xml @@ -1,2 +1,29 @@ -marshalls-furious-bucket1000urlfalsefoo.txt2020-10-24T00:13:39.000Z"a895ba674b4abd01b5d67cfd7074b827"2064537bef397f7e536914d1ff1bbdb105ed90bcfd06269456bf4a06c6e2e54564daf7STANDARDbar.txt2020-10-02T15:15:20.000Z"a895ba674b4abd01b5d67cfd7074b827"2064537bef397f7e536914d1ff1bbdb105ed90bcfd06269456bf4a06c6e2e54564daf7STANDARD + + marshalls-furious-bucket + + + 1000 + url + false + + foo.txt + 2020-10-24T00:13:39.000Z + "a895ba674b4abd01b5d67cfd7074b827" + 206453 + + 7bef397f7e536914d1ff1bbdb105ed90bcfd06269456bf4a06c6e2e54564daf7 + + STANDARD + + + bar.txt + 2020-10-02T15:15:20.000Z + "a895ba674b4abd01b5d67cfd7074b827" + 206453 + + 7bef397f7e536914d1ff1bbdb105ed90bcfd06269456bf4a06c6e2e54564daf7 + + STANDARD + + diff --git a/tests/integrations/boto3/test_client.py b/tests/integrations/boto3/test_client.py index 1113833a80..3d1d1b8dd4 100644 --- a/tests/integrations/boto3/test_client.py +++ b/tests/integrations/boto3/test_client.py @@ -1,5 +1,6 @@ from http.server import BaseHTTPRequestHandler, HTTPServer from threading import Thread +from unittest import mock import boto3 import pytest @@ -10,6 +11,7 @@ from botocore.stub import Stubber import sentry_sdk +from sentry_sdk import capture_message from sentry_sdk.consts import OP, SPANDATA from sentry_sdk.integrations.boto3 import Boto3Integration from sentry_sdk.integrations.boto3._services.base import _ServiceExtension @@ -21,7 +23,21 @@ ) from sentry_sdk.integrations.stdlib import StdlibIntegration from sentry_sdk.traces import Span +from tests.conftest import ApproxDict +from tests.integrations.boto3 import read_fixture from tests.integrations.boto3.aws_mock import Body, MockResponse +from tests.integrations.boto3.helpers import ( + capture_spans_by_op as _capture_boto3_spans_by_op, +) +from tests.integrations.boto3.helpers import ( + client_factory as client_factory, +) +from tests.integrations.boto3.helpers import ( + no_botocore_retry_delay as no_botocore_retry_delay, +) +from tests.integrations.boto3.helpers import ( + require_botocore_model_fields, +) session = boto3.Session( # type: ignore[attr-defined] aws_access_key_id="-", @@ -41,7 +57,7 @@ def do_GET(self): self.wfile.write(b"x") self.wfile.flush() - def log_message(self, *args): + def log_message(self, *args): # type: ignore pass server = HTTPServer(("127.0.0.1", 0), StreamingS3Handler) @@ -129,22 +145,23 @@ def record_client_span(request, **kwargs): for span in spans if span["name"] == "S3.GetObject" and ( - span["attributes"].get(SPANDATA.SENTRY_ORIGIN) == ORIGIN - and span["attributes"].get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT + span.get("attributes", {}).get(SPANDATA.SENTRY_ORIGIN) == ORIGIN + and span.get("attributes", {}).get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT ) ] http_spans = [ span for span in spans - if ( - span["attributes"].get(SPANDATA.SENTRY_ORIGIN) == "auto.http.stdlib.httplib" - ) + if span.get("attributes", {}).get(SPANDATA.SENTRY_ORIGIN) + == "auto.http.stdlib.httplib" ] stream_spans = [ span for span in spans if span["name"] == "S3.GetObject" - and (span["attributes"].get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT_STREAM) + and ( + span.get("attributes", {}).get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT_STREAM + ) ] assert len(client_spans) == 1 assert len(http_spans) == 1 @@ -153,72 +170,218 @@ def record_client_span(request, **kwargs): http_span = http_spans[0] stream_span = stream_spans[0] - assert http_span["parent_span_id"] == client_span["span_id"] - assert stream_span["parent_span_id"] == client_span["span_id"] + assert http_span.get("parent_span_id") == client_span["span_id"] + assert stream_span.get("parent_span_id") == client_span["span_id"] assert client_span["span_id"] == request_client_span.span_id for span in (client_span, http_span, stream_span): - assert span["end_timestamp"] is not None + assert span.get("end_timestamp") is not None + + +@pytest.mark.parametrize( + "service_name,method_name,api_params,payload_field", + [ + pytest.param( + "lambda", + "invoke", + {"FunctionName": "function"}, + "Payload", + id="payload", + ), + pytest.param( + "s3", + "get_object_annotation", + { + "Bucket": "bucket", + "Key": "file.txt", + "AnnotationName": "annotation", + }, + "AnnotationPayload", + id="annotation-payload", + ), + ], +) +def test_non_body_stream_delays_client_span( + capture_items, + client_factory, + service_name, + method_name, + api_params, + payload_field, +): + client = client_factory(service_name) + require_botocore_model_fields(client, method_name, output_fields=(payload_field,)) + request_client_spans = [] + + def record_client_span(request, **kwargs): + request_client_spans.append(request.context["_sentrysdk_span"]) + + client.meta.events.register("request-created", record_client_span) + + def invoke(): + parent = sentry_sdk.get_current_span() + with MockResponse(client, 200, {"content-length": "1"}, b"x"): + response = getattr(client, method_name)(**api_params) + body = response[payload_field] + assert isinstance(body, StreamingBody) + (request_client_span,) = request_client_spans + assert request_client_span.end_timestamp is None + assert sentry_sdk.get_current_span() is parent + body.close() + assert request_client_span.end_timestamp is not None + assert sentry_sdk.get_current_span() is parent + + spans_by_op = _capture_boto3_spans_by_op(invoke, capture_items) + (client_span,) = spans_by_op[OP.HTTP_CLIENT] + (stream_span,) = spans_by_op[OP.HTTP_CLIENT_STREAM] + assert stream_span.get("parent_span_id") == client_span["span_id"] + + +@pytest.mark.tests_internal_exceptions +def test_omit_url_data_if_parsing_fails(capture_items, client_factory): + client = client_factory() + + with mock.patch( + "sentry_sdk.integrations.boto3._instrumentation.parse_url", + side_effect=ValueError, + ) as parse_url: + with MockResponse(client, 200, {}, b""): + spans_by_op = _capture_boto3_spans_by_op( + lambda: client.head_object(Bucket="bucket", Key="file.txt"), + capture_items, + ) + + parse_url.assert_called() + (span,) = spans_by_op[OP.HTTP_CLIENT] + attributes = span.get("attributes", {}) + assert SPANDATA.URL_FULL not in attributes + assert SPANDATA.URL_FRAGMENT not in attributes + assert SPANDATA.URL_QUERY not in attributes + + +BUCKET_URL = "https://bucket.s3.eu-north-1.amazonaws.com/" + +URL_QUERY_PARAMS = [ + pytest.param( + {}, + "list-type=2&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=url", + id="defaults", + ), + pytest.param( + { + "data_collection": { + "url_query_params": {"mode": "denylist", "terms": ["prefix"]} + } + }, + "list-type=2&prefix=%5BFiltered%5D&continuation-token=%5BFiltered%5D&encoding-type=url", + id="data_collection_denylist_custom_terms", + ), + pytest.param( + { + "data_collection": { + "url_query_params": {"mode": "allowlist", "terms": ["prefix"]} + } + }, + "list-type=%5BFiltered%5D&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=%5BFiltered%5D", + id="data_collection_allowlist", + ), + pytest.param( + { + "data_collection": { + "url_query_params": { + "mode": "allowlist", + "terms": ["continuation-token"], + } + } + }, + "list-type=%5BFiltered%5D&prefix=%5BFiltered%5D&continuation-token=%5BFiltered%5D&encoding-type=%5BFiltered%5D", + id="data_collection_allowlist_sensitive_term", + ), + pytest.param( + {"data_collection": {"url_query_params": {"mode": "off"}}}, + "", + id="data_collection_off", + ), +] -def test_non_body_stream_does_not_delay_client_span(sentry_init, capture_items): +@pytest.mark.parametrize("init_kwargs, expected_query", URL_QUERY_PARAMS) +def test_url_query_data_collection( + sentry_init, capture_items, init_kwargs, expected_query +): sentry_init( traces_sample_rate=1.0, integrations=[Boto3Integration()], - server_name="", + default_integrations=False, + **init_kwargs, ) - client = session.client("lambda") - - def respond(request, **kwargs): - return AWSResponse( - request.url, - 200, - {"content-length": "1"}, - Body(b"x"), - ) - - client.meta.events.register("before-send", respond) + client = session.client("s3") items = capture_items("span") - with sentry_sdk.start_span(name="parent") as parent: # type: ignore[attr-defined] - response = client.invoke(FunctionName="function") - assert isinstance(response["Payload"], StreamingBody) - assert sentry_sdk.get_current_span() is parent # type: ignore[attr-defined] + with sentry_sdk.start_span(name="parent"), MockResponse( + client, 200, {}, read_fixture("s3_list.xml") + ): + client.list_objects_v2(Bucket="bucket", Prefix="foo", ContinuationToken="abc") sentry_sdk.flush() - - spans = [item.payload for item in items] - boto_spans = [ - span - for span in spans - if span["attributes"].get(SPANDATA.SENTRY_ORIGIN) == ORIGIN + (span,) = [ + item.payload + for item in items + if item.payload.get("attributes", {}).get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT ] - assert len(boto_spans) == 1 - assert boto_spans[0]["attributes"].get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT - response["Payload"].close() + attributes = span.get("attributes", {}) + if expected_query == "": + assert SPANDATA.URL_QUERY not in attributes + assert attributes[SPANDATA.URL_FULL] == BUCKET_URL + else: + assert attributes[SPANDATA.URL_QUERY] == expected_query + assert attributes[SPANDATA.URL_FULL] == BUCKET_URL + "?" + expected_query -@pytest.fixture -def client_factory(sentry_init, monkeypatch): + +@pytest.mark.parametrize( + "data_collection,expected_query", + [ + pytest.param( + {}, + "list-type=2&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=url", + id="default", + ), + pytest.param( + {"url_query_params": {"mode": "off"}}, + None, + id="query-collection-off", + ), + ], +) +def test_breadcrumb(sentry_init, capture_events, data_collection, expected_query): sentry_init( - traces_sample_rate=1.0, integrations=[Boto3Integration()], - # avoid SDK's machine hostname being used as server name. - server_name="", + default_integrations=False, + data_collection=data_collection, ) - # remove retry delay to speed up tests - monkeypatch.setattr("botocore.endpoint.time.sleep", lambda delay: None) - - def make_client(service_name="s3", attempt_count=1, **client_kwargs): - return session.client( - service_name, - config=Config( - # `total_max_attempts` includes the initial request. - retries={"total_max_attempts": attempt_count, "mode": "standard"} - ), - **client_kwargs, - ) + client = session.client("s3") + events = capture_events() - return make_client + with MockResponse(client, 200, {}, read_fixture("s3_list.xml")): + client.list_objects_v2(Bucket="bucket", Prefix="foo", ContinuationToken="abc") + + capture_message("Testing!") + (event,) = events + (crumb,) = event["breadcrumbs"]["values"] + assert crumb["type"] == "http" + assert crumb["category"] == "httplib" + assert SPANDATA.URL_FRAGMENT not in crumb["data"] + + if expected_query is None: + assert SPANDATA.URL_QUERY not in crumb["data"] + assert crumb["data"][SPANDATA.URL_FULL] == BUCKET_URL + else: + assert crumb["data"] == ApproxDict( + { + SPANDATA.URL_FULL: BUCKET_URL + "?" + expected_query, + SPANDATA.URL_QUERY: expected_query, + } + ) def _mock_responses(client, status_codes): @@ -233,44 +396,19 @@ def respond(request, **kwargs): # `request_created` runs before `before_send`, so use zero-based index for current # attempt; `min(..., len(status_codes) - 1)` clamps to last status to avoid `IndexError`. response_index = min(len(request_span_ids) - 1, len(status_codes) - 1) - return AWSResponse(request.url, status_codes[response_index], {}, Body(b"")) + return AWSResponse(request.url, status_codes[response_index], {}, Body(b"")) # type: ignore client.meta.events.register("request-created", record_request) client.meta.events.register("before-send", respond) return request_span_ids -def _capture_boto3_spans_by_op( - invoke_client_method, - capture_items, - expected_origin=ORIGIN, -): - items = capture_items() - - with sentry_sdk.start_span(name="parent"): - invoke_client_method() - - sentry_sdk.flush() - spans = [ - item.payload - for item in items - if item.type == "span" - and item.payload["attributes"].get(SPANDATA.SENTRY_ORIGIN) == expected_origin - ] - - spans_by_op = {} - for span in spans: - spans_by_op.setdefault(span["attributes"].get(SPANDATA.SENTRY_OP), []).append( - span - ) - return spans_by_op - - def _assert_one_failed_span(spans): assert len(spans) == 1 assert spans[0]["status"] == "error" - assert spans[0]["attributes"][SPANDATA.ERROR_TYPE] - assert spans[0]["end_timestamp"] is not None + attributes = spans[0].get("attributes", {}) + assert attributes[SPANDATA.ERROR_TYPE] + assert spans[0].get("end_timestamp") is not None def _capture_stubbed_client_span( @@ -288,6 +426,7 @@ def _capture_stubbed_client_span( lambda: getattr(client, method_name)(**api_params), capture_items, ) + stubber.assert_no_pending_responses() client_spans = spans_by_op.get(OP.HTTP_CLIENT, []) assert len(client_spans) == 1 @@ -342,15 +481,24 @@ def get_response_attributes(self, ctx, response): spans = spans_by_op.get("aws.test", []) assert len(spans) == 1 - attributes = spans[0]["attributes"] - assert attributes["aws.test.request"] == "foo" - assert attributes["aws.test.response"] == "request-id" + span = spans[0] + attributes = span.get("attributes", {}) + assert span["name"] == "S3.HeadObject" + assert span.get("end_timestamp") is not None + assert attributes[SPANDATA.SENTRY_OP] == "aws.test" + assert attributes[SPANDATA.SENTRY_ORIGIN] == "auto.aws.test" assert attributes[SPANDATA.SENTRY_KIND] == "producer" + assert attributes[SPANDATA.CLOUD_PROVIDER] == CLOUD_PROVIDER + assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME + assert attributes[SPANDATA.RPC_SERVICE] == "S3" assert attributes[SPANDATA.RPC_METHOD] == "HeadObject" + assert attributes[SPANDATA.CLOUD_REGION] == "eu-north-1" + assert attributes[SPANDATA.SERVER_ADDRESS] == "s3.eu-north-1.amazonaws.com" + assert attributes[SPANDATA.SERVER_PORT] == 443 + assert attributes["aws.test.request"] == "foo" + assert attributes["aws.test.response"] == "request-id" assert attributes[SPANDATA.HTTP_STATUS_CODE] == 200 assert attributes[SPANDATA.AWS_EXTENDED_REQUEST_ID] == "extended-request-id" - assert spans[0]["end_timestamp"] is not None - assert attributes[SPANDATA.SENTRY_ORIGIN] == "auto.aws.test" @pytest.mark.parametrize( @@ -428,22 +576,24 @@ def test_client_call_has_common_attributes( } }, ) - attributes = span["attributes"] + attributes = span.get("attributes", {}) assert span["name"] == span_name + assert span.get("end_timestamp") is not None + assert attributes[SPANDATA.SENTRY_OP] == OP.HTTP_CLIENT + assert attributes[SPANDATA.SENTRY_ORIGIN] == ORIGIN + assert attributes[SPANDATA.SENTRY_KIND] == "client" + assert attributes[SPANDATA.CLOUD_PROVIDER] == CLOUD_PROVIDER + assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME assert attributes[SPANDATA.RPC_SERVICE] == rpc_service assert attributes[SPANDATA.RPC_METHOD] == rpc_method - assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME - assert attributes[SPANDATA.SENTRY_KIND] == "client" assert attributes[SPANDATA.CLOUD_REGION] == "eu-north-1" - assert attributes[SPANDATA.CLOUD_PROVIDER] == CLOUD_PROVIDER assert attributes[SPANDATA.SERVER_ADDRESS] == server_address assert attributes[SPANDATA.SERVER_PORT] == server_port assert attributes[SPANDATA.HTTP_STATUS_CODE] == 200 assert attributes[SPANDATA.AWS_REQUEST_ID] == "request-id" assert SPANDATA.HTTP_REQUEST_RESEND_COUNT not in attributes assert SPANDATA.ERROR_TYPE not in attributes - assert span["end_timestamp"] is not None def test_client_call_attributes_are_available_at_span_creation( @@ -482,7 +632,7 @@ def test_client_call_attributes_are_available_at_span_creation( client_spans = [ item.payload for item in items - if item.payload["attributes"].get(SPANDATA.SENTRY_ORIGIN) == ORIGIN + if item.payload.get("attributes", {}).get(SPANDATA.SENTRY_ORIGIN) == ORIGIN ] assert client_spans == [] @@ -503,14 +653,16 @@ def test_client_call_has_response_header_attributes( spans = spans_by_op[OP.HTTP_CLIENT] assert len(spans) == 1 - attributes = spans[0]["attributes"] + attributes = spans[0].get("attributes", {}) assert attributes[SPANDATA.HTTP_STATUS_CODE] == 200 assert attributes[SPANDATA.AWS_REQUEST_ID] == "request-id" assert attributes[SPANDATA.AWS_EXTENDED_REQUEST_ID] == "extended-request-id" assert SPANDATA.HTTP_REQUEST_RESEND_COUNT not in attributes -def test_retry_attempts_share_one_client_span(capture_items, client_factory): +def test_retry_attempts_share_one_client_span( + capture_items, client_factory, no_botocore_retry_delay +): attempt_count = 3 client = client_factory(attempt_count=attempt_count) request_span_ids = _mock_responses(client, [500] * (attempt_count - 1) + [200]) @@ -524,11 +676,13 @@ def test_retry_attempts_share_one_client_span(capture_items, client_factory): # all `AWSRequest` instances created during retries reference the same client span. assert len(set(request_span_ids)) == 1 assert len(client_spans) == 1 - attributes = client_spans[0]["attributes"] + attributes = client_spans[0].get("attributes", {}) assert attributes[SPANDATA.HTTP_REQUEST_RESEND_COUNT] == attempt_count - 1 -def test_retries_exhausted_has_one_failed_client_span(capture_items, client_factory): +def test_retries_exhausted_has_one_failed_client_span( + capture_items, client_factory, no_botocore_retry_delay +): client = client_factory(attempt_count=2) request_span_ids = _mock_responses(client, [500]) @@ -544,7 +698,7 @@ def attempt_failed_head_object_call(): assert len(request_span_ids) == 2 assert len(set(request_span_ids)) == 1 _assert_one_failed_span(client_spans) - attributes = client_spans[0]["attributes"] + attributes = client_spans[0].get("attributes", {}) assert attributes[SPANDATA.HTTP_STATUS_CODE] == 500 assert attributes[SPANDATA.HTTP_REQUEST_RESEND_COUNT] == 1 @@ -578,7 +732,7 @@ def get_response_attributes(self, ctx, response): "HTTPStatusCode": 403, "RetryAttempts": 1, }, - }, + }, # type: ignore "HeadObject", ) @@ -597,8 +751,21 @@ def invoke_failing_client_method(): ) client_spans = spans_by_op.get(OP.HTTP_CLIENT, []) _assert_one_failed_span(client_spans) - attributes = client_spans[0]["attributes"] + attributes = client_spans[0].get("attributes", {}) + span = client_spans[0] + assert span["name"] == "S3.HeadObject" + assert span.get("end_timestamp") is not None + assert attributes[SPANDATA.SENTRY_OP] == OP.HTTP_CLIENT + assert attributes[SPANDATA.SENTRY_ORIGIN] == ORIGIN + assert attributes[SPANDATA.SENTRY_KIND] == "client" + assert attributes[SPANDATA.CLOUD_PROVIDER] == CLOUD_PROVIDER + assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME + assert attributes[SPANDATA.RPC_SERVICE] == "S3" + assert attributes[SPANDATA.RPC_METHOD] == "HeadObject" + assert attributes[SPANDATA.CLOUD_REGION] == "eu-north-1" + assert attributes[SPANDATA.SERVER_ADDRESS] == "s3.eu-north-1.amazonaws.com" + assert attributes[SPANDATA.SERVER_PORT] == 443 assert attributes[SPANDATA.AWS_REQUEST_ID] == "request-id" assert attributes[SPANDATA.HTTP_STATUS_CODE] == 403 assert attributes[SPANDATA.HTTP_REQUEST_RESEND_COUNT] == 1 @@ -651,7 +818,8 @@ def invoke_failing_client_method(): if event_name == "before-send" else "ValueError" ) - assert client_spans[0]["attributes"][SPANDATA.ERROR_TYPE] == expected_error_type + attributes = client_spans[0].get("attributes", {}) + assert attributes[SPANDATA.ERROR_TYPE] == expected_error_type @pytest.mark.tests_internal_exceptions @@ -693,7 +861,7 @@ def invoke_client_method(): assert returned_responses[0] is original_response if failing_instrumentation == "_get_response_attributes": assert len(client_spans) == 1 - assert client_spans[0]["end_timestamp"] is not None + assert client_spans[0].get("end_timestamp") is not None else: assert client_spans == [] @@ -729,7 +897,7 @@ def invoke_failing_client_method(): assert len(client_spans) == 1 assert client_spans[0]["status"] == "error" - assert client_spans[0]["end_timestamp"] is not None + assert client_spans[0].get("end_timestamp") is not None def test_streaming_response_attributes_belong_to_client_span( @@ -744,7 +912,7 @@ def respond(request, **kwargs): { "content-length": "5", "x-amz-request-id": "request-id", - }, + }, # type: ignore Body(b"hello"), ) @@ -763,8 +931,8 @@ def invoke_client_method_and_read_body(): assert len(client_spans) == 1 assert len(stream_spans) == 1 - client_attributes = client_spans[0]["attributes"] - stream_attributes = stream_spans[0]["attributes"] + client_attributes = client_spans[0].get("attributes", {}) + stream_attributes = stream_spans[0].get("attributes", {}) assert client_attributes[SPANDATA.AWS_REQUEST_ID] == "request-id" assert client_attributes[SPANDATA.HTTP_STATUS_CODE] == 200 assert SPANDATA.HTTP_REQUEST_RESEND_COUNT not in client_attributes @@ -792,7 +960,7 @@ def respond(request, **kwargs): return AWSResponse( request.url, 200, - {"content-length": "1"}, + {"content-length": "1"}, # type: ignore _FailingBody(original_exception), ) @@ -813,4 +981,5 @@ def invoke_client_method_and_read_body(): assert len(client_spans) == 1 _assert_one_failed_span(client_spans) _assert_one_failed_span(stream_spans) - assert stream_spans[0]["attributes"][SPANDATA.ERROR_TYPE] == "OSError" + attributes = stream_spans[0].get("attributes", {}) + assert attributes[SPANDATA.ERROR_TYPE] == "OSError" diff --git a/tests/integrations/boto3/test_s3.py b/tests/integrations/boto3/test_s3.py index 4cf66d26c2..918ea10063 100644 --- a/tests/integrations/boto3/test_s3.py +++ b/tests/integrations/boto3/test_s3.py @@ -1,381 +1,278 @@ -from unittest import mock +import json +from copy import deepcopy +from datetime import datetime, timezone -import boto3 import pytest +from botocore.stub import Stubber -import sentry_sdk -from sentry_sdk import capture_message -from sentry_sdk.consts import SPANDATA -from sentry_sdk.integrations.boto3 import Boto3Integration -from sentry_sdk.integrations.boto3.consts import ORIGIN -from tests.conftest import ApproxDict -from tests.integrations.boto3 import read_fixture -from tests.integrations.boto3.aws_mock import MockResponse - -session = boto3.Session( - aws_access_key_id="-", - aws_secret_access_key="-", +from sentry_sdk.consts import OP, SPANDATA +from sentry_sdk.integrations.boto3.consts import AWS_RPC_SYSTEM_NAME, ORIGIN +from tests.integrations.boto3.helpers import ( + capture_spans_by_op, + require_botocore_model_fields, +) +from tests.integrations.boto3.helpers import ( + client_factory as client_factory, +) +from tests.integrations.boto3.helpers import ( + s3_client as s3_client, ) -def test_basic( - sentry_init, - capture_items, -): - sentry_init( - traces_sample_rate=1.0, - integrations=[Boto3Integration()], - # disabled because session.resource() or s3.Bucket() result in a subprocess span for a - # shell that runs "uname -p 2> /dev/null" on Python 3.7 with boto3 version 1.12.49. - default_integrations=False, - ) - - s3 = session.resource("s3") - bucket = s3.Bucket("bucket") - items = capture_items("span") - - with sentry_sdk.start_span(name="custom parent") as span, MockResponse( - s3.meta.client, 200, {}, read_fixture("s3_list.xml") - ): - objects = [obj for obj in bucket.objects.all()] - assert len(objects) == 2 - assert objects[0].key == "foo.txt" - assert objects[1].key == "bar.txt" - span.end() - - sentry_sdk.flush() - spans = [item.payload for item in items] - assert len(spans) == 2 - span = spans[0] - assert span["attributes"]["sentry.op"] == "http.client" - assert span["name"] == "S3.ListObjects" - - -def test_streaming(sentry_init, capture_items): - sentry_init( - traces_sample_rate=1.0, - integrations=[Boto3Integration()], - ) - - s3 = session.resource("s3") - obj = s3.Bucket("bucket").Object("foo.pdf") - - items = capture_items("span") - - with sentry_sdk.start_span(name="custom parent") as span, MockResponse( - s3.meta.client, 200, {}, b"hello" - ): - body = obj.get()["Body"] - assert body.read(1) == b"h" - assert body.read(2) == b"el" - assert body.read(3) == b"lo" - assert body.read(1) == b"" - span.end() - - sentry_sdk.flush() - spans = [item.payload for item in items] - assert len(spans) == 3 - - stream_span, client_span, parent_span = spans - assert stream_span["attributes"]["sentry.op"] == "http.client.stream" - assert stream_span["name"] == "S3.GetObject" - assert stream_span["parent_span_id"] == client_span["span_id"] - - assert client_span["attributes"]["sentry.op"] == "http.client" - assert client_span["name"] == "S3.GetObject" - assert client_span["parent_span_id"] == parent_span["span_id"] - - assert parent_span["name"] == "custom parent" - assert parent_span["start_timestamp"] <= client_span["start_timestamp"] - assert client_span["start_timestamp"] <= stream_span["start_timestamp"] - assert stream_span["end_timestamp"] <= client_span["end_timestamp"] - - expected_attrs = { - "http.request.method": "GET", - "rpc.method": "GetObject", - "rpc.service": "S3", - "sentry.environment": "production", - "sentry.op": "http.client", - "sentry.origin": ORIGIN, - "sentry.release": mock.ANY, - "sentry.sdk.name": "sentry.python", - "sentry.sdk.version": mock.ANY, - "sentry.segment.id": mock.ANY, - "sentry.segment.name": "custom parent", - "server.address": mock.ANY, - "thread.id": mock.ANY, - "thread.name": mock.ANY, - "url.full": "https://bucket.s3.amazonaws.com/foo.pdf", +def _stubbed_span(client, capture_items, method, params, response): + with Stubber(client) as stubber: + stubber.add_response(method, response, expected_params=params) + spans = capture_spans_by_op( + lambda: getattr(client, method)(**params), capture_items + ) + stubber.assert_no_pending_responses() + (span,) = spans[OP.HTTP_CLIENT] + operation = client.meta.method_to_api_mapping[method] + assert span["name"] == "S3.%s" % operation + assert span.get("end_timestamp") is not None + attributes = span.get("attributes", {}) + assert attributes[SPANDATA.SENTRY_OP] == OP.HTTP_CLIENT + assert attributes[SPANDATA.SENTRY_ORIGIN] == ORIGIN + assert attributes[SPANDATA.SENTRY_KIND] == "client" + assert attributes[SPANDATA.CLOUD_PROVIDER] == "aws" + assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME + assert attributes[SPANDATA.RPC_SERVICE] == "S3" + assert attributes[SPANDATA.RPC_METHOD] == operation + assert attributes[SPANDATA.CLOUD_REGION] == "eu-north-1" + assert attributes[SPANDATA.SERVER_ADDRESS] == "s3.eu-north-1.amazonaws.com" + assert attributes[SPANDATA.SERVER_PORT] == 443 + return span + + +def test_request_attributes(s3_client, capture_items): + params = { + "Bucket": "bucket", + "Key": "file.txt", + "UploadId": "upload-id", + "PartNumber": 1, + "Body": b"private-content", } + original = deepcopy(params) + + span = _stubbed_span(s3_client, capture_items, "upload_part", params, {}) + + attributes = span.get("attributes", {}) + assert attributes[SPANDATA.AWS_S3_BUCKET] == "bucket" + assert attributes[SPANDATA.AWS_S3_KEY] == "file.txt" + assert attributes[SPANDATA.AWS_S3_UPLOAD_ID] == "upload-id" + assert attributes[SPANDATA.AWS_S3_PART_NUMBER] == 1 + assert SPANDATA.AWS_S3_COPY_SOURCE not in attributes + assert SPANDATA.AWS_S3_DELETE not in attributes + assert SPANDATA.FILE_SIZE not in attributes + assert SPANDATA.HTTP_BODY_SIZE not in attributes + assert params == original + assert "private-content" not in json.dumps(span) + + +@pytest.mark.parametrize( + "method,copy_source,expected", + [ + pytest.param( + "copy_object", + "source/path/file.txt", + "source/path/file.txt", + id="string", + ), + pytest.param( + "copy_object", + {"Bucket": "source", "Key": "path/file.txt"}, + "source/path/file.txt", + id="dictionary", + ), + pytest.param( + "upload_part_copy", + {"Bucket": "source", "Key": "path/file.txt", "VersionId": "version-1"}, + "source/path/file.txt?versionId=version-1", + id="versioned-dictionary", + ), + ], +) +def test_copy_source(s3_client, capture_items, method, copy_source, expected): + source = deepcopy(copy_source) + params = {"Bucket": "bucket", "Key": "file.txt", "CopySource": source} + expected_attributes = { + SPANDATA.AWS_S3_BUCKET: "bucket", + SPANDATA.AWS_S3_KEY: "file.txt", + SPANDATA.AWS_S3_COPY_SOURCE: expected, + } + if method == "upload_part_copy": + params.update(UploadId="upload-id", PartNumber=1) + expected_attributes.update( + {SPANDATA.AWS_S3_UPLOAD_ID: "upload-id", SPANDATA.AWS_S3_PART_NUMBER: 1} + ) - assert client_span["attributes"] == ApproxDict(expected_attrs) - - -def test_streaming_close(sentry_init, capture_items): - sentry_init( - traces_sample_rate=1.0, - integrations=[Boto3Integration()], - ) - - s3 = session.resource("s3") - obj = s3.Bucket("bucket").Object("foo.pdf") - - items = capture_items("span") - - with sentry_sdk.start_span(name="custom parent") as span, MockResponse( - s3.meta.client, 200, {}, b"hello" - ): - body = obj.get()["Body"] - assert body.read(1) == b"h" - body.close() # close partially-read stream - span.end() - - sentry_sdk.flush() - spans = [item.payload for item in items] - assert len(spans) == 3 - - stream_span, client_span, parent_span = spans - assert stream_span["attributes"]["sentry.op"] == "http.client.stream" - assert stream_span["name"] == "S3.GetObject" - assert stream_span["parent_span_id"] == client_span["span_id"] - - assert client_span["attributes"]["sentry.op"] == "http.client" - assert client_span["name"] == "S3.GetObject" - assert client_span["parent_span_id"] == parent_span["span_id"] - - assert parent_span["name"] == "custom parent" - assert parent_span["start_timestamp"] <= client_span["start_timestamp"] - assert client_span["start_timestamp"] <= stream_span["start_timestamp"] - assert stream_span["end_timestamp"] <= client_span["end_timestamp"] - - -@pytest.mark.tests_internal_exceptions -def test_omit_url_data_if_parsing_fails(sentry_init, capture_items): - sentry_init( - traces_sample_rate=1.0, - integrations=[Boto3Integration()], - ) - - s3 = session.resource("s3") - bucket = s3.Bucket("bucket") - - items = capture_items("span") - - with mock.patch( - "sentry_sdk.integrations.boto3._instrumentation.parse_url", - side_effect=ValueError, - ): - with sentry_sdk.start_span(name="custom parent") as span, MockResponse( - s3.meta.client, 200, {}, read_fixture("s3_list.xml") - ): - objects = [obj for obj in bucket.objects.all()] - assert len(objects) == 2 - assert objects[0].key == "foo.txt" - assert objects[1].key == "bar.txt" - span.end() - - sentry_sdk.flush() - spans = [item.payload for item in items] - assert spans[0]["attributes"] == ApproxDict( - { - "http.request.method": "GET", - "rpc.method": "ListObjects", - "rpc.service": "S3", - "sentry.environment": "production", - "sentry.op": "http.client", - "sentry.origin": ORIGIN, - "sentry.release": mock.ANY, - "sentry.sdk.name": "sentry.python", - "sentry.sdk.version": mock.ANY, - "sentry.segment.id": mock.ANY, - "sentry.segment.name": "custom parent", - "server.address": mock.ANY, - "thread.id": mock.ANY, - "thread.name": mock.ANY, - } - ) - - assert "url.full" not in spans[0]["attributes"] - assert "url.fragment" not in spans[0]["attributes"] - assert "url.query" not in spans[0]["attributes"] - - -def test_span_origin(sentry_init, capture_items): - sentry_init( - traces_sample_rate=1.0, - integrations=[Boto3Integration()], - ) - - s3 = session.resource("s3") - bucket = s3.Bucket("bucket") - items = capture_items("span") - - with sentry_sdk.start_span(name="custom parent"), MockResponse( - s3.meta.client, 200, {}, read_fixture("s3_list.xml") - ): - _ = [obj for obj in bucket.objects.all()] - - sentry_sdk.flush() - - spans = [item.payload for item in items] - - assert spans[1]["attributes"]["sentry.origin"] == "manual" - assert spans[0]["attributes"]["sentry.origin"] == ORIGIN - - -def test_breadcrumb(sentry_init, capture_events): - sentry_init( - integrations=[Boto3Integration()], - default_integrations=False, - ) - - s3 = session.resource("s3") - bucket = s3.Bucket("bucket") - - events = capture_events() - - with sentry_sdk.start_span(name="custom parent"), MockResponse( - s3.meta.client, 200, {}, read_fixture("s3_list.xml") + span = _stubbed_span(s3_client, capture_items, method, params, {}) + + attributes = span.get("attributes", {}) + for key, value in expected_attributes.items(): + assert attributes[key] == value + for key in ( + SPANDATA.AWS_S3_UPLOAD_ID, + SPANDATA.AWS_S3_PART_NUMBER, + SPANDATA.AWS_S3_DELETE, + SPANDATA.FILE_SIZE, + SPANDATA.HTTP_BODY_SIZE, ): - _ = [obj for obj in bucket.objects.all()] - - capture_message("Testing!") - - (event,) = events - (crumb,) = event["breadcrumbs"]["values"] - assert crumb["type"] == "http" - assert crumb["category"] == "httplib" - - assert crumb["data"] == ApproxDict( - { - SPANDATA.URL_FULL: mock.ANY, - SPANDATA.HTTP_REQUEST_METHOD: "GET", - SPANDATA.URL_QUERY: mock.ANY, - } - ) - assert SPANDATA.URL_FRAGMENT not in crumb["data"] - - -BUCKET_URL = "https://bucket.s3.amazonaws.com/" - -# ``expected_query`` of ``None`` means no URL data is recorded at all; ``""`` -# means the URL is recorded without a query string. -# Structure of the parameters is "init_kwargs, expected_query" -URL_QUERY_PARAMS = [ - pytest.param( - {}, - "list-type=2&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=url", - id="defaults", - ), - pytest.param( - { - "data_collection": { - "url_query_params": {"mode": "denylist", "terms": ["prefix"]} - } - }, - "list-type=2&prefix=%5BFiltered%5D&continuation-token=%5BFiltered%5D&encoding-type=url", - id="data_collection_denylist_custom_terms", - ), - pytest.param( - { - "data_collection": { - "url_query_params": {"mode": "allowlist", "terms": ["prefix"]} - } - }, - "list-type=%5BFiltered%5D&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=%5BFiltered%5D", - id="data_collection_allowlist", - ), - pytest.param( - { - "data_collection": { - "url_query_params": { - "mode": "allowlist", - "terms": ["continuation-token"], - } - } - }, - "list-type=%5BFiltered%5D&prefix=%5BFiltered%5D&continuation-token=%5BFiltered%5D&encoding-type=%5BFiltered%5D", - id="data_collection_allowlist_sensitive_term", - ), - pytest.param( - {"data_collection": {"url_query_params": {"mode": "off"}}}, - "", - id="data_collection_off", - ), -] - - -@pytest.mark.parametrize("init_kwargs, expected_query", URL_QUERY_PARAMS) -def test_url_query_data_collection( - sentry_init, capture_items, init_kwargs, expected_query + if key not in expected_attributes: + assert key not in attributes + assert params["CopySource"] is source + assert source == copy_source + + +@pytest.mark.parametrize( + "delete, input_fields, expected_serialized_delete", + [ + pytest.param( + {"Quiet": True, "Objects": [{"VersionId": "version-1", "Key": "file.txt"}]}, + (), + '{"Objects":[{"Key":"file.txt","VersionId":"version-1"}],"Quiet":true}', + id="basic", + ), + pytest.param( + { + "Objects": [ + { + "Key": "file.txt", + "VersionId": "version-1", + "ETag": "etag", + "LastModifiedTime": datetime( + 2026, 10, 8, 12, 34, 56, tzinfo=timezone.utc + ), + "Size": 123, + } + ] + }, + ("Delete.Objects.LastModifiedTime",), + '{"Objects":[{"ETag":"etag","Key":"file.txt","LastModifiedTime":"2026-10-08T12:34:56+00:00","Size":123,"VersionId":"version-1"}]}', + id="last-modified-time", + ), + ], +) +def test_delete_serialization( + s3_client, capture_items, delete, input_fields, expected_serialized_delete ): - sentry_init( - traces_sample_rate=1.0, - integrations=[Boto3Integration()], - default_integrations=False, - **init_kwargs, + require_botocore_model_fields( + s3_client, + "delete_objects", + input_fields=input_fields, ) - client = session.client("s3") - - items = capture_items("span") - - with sentry_sdk.start_span(name="custom parent"), MockResponse( - client, 200, {}, read_fixture("s3_list.xml") - ): - client.list_objects_v2(Bucket="bucket", Prefix="foo", ContinuationToken="abc") - - sentry_sdk.flush() - - (span,) = [ - item.payload - for item in items - if item.payload["attributes"].get("sentry.op") == "http.client" - ] - - if expected_query is None: - assert SPANDATA.URL_QUERY not in span["attributes"] - assert SPANDATA.URL_FULL not in span["attributes"] - elif expected_query == "": - assert SPANDATA.URL_QUERY not in span["attributes"] - assert span["attributes"][SPANDATA.URL_FULL] == BUCKET_URL - else: - assert span["attributes"][SPANDATA.URL_QUERY] == expected_query - assert ( - span["attributes"][SPANDATA.URL_FULL] == BUCKET_URL + "?" + expected_query - ) - - -@pytest.mark.parametrize("init_kwargs, expected_query", URL_QUERY_PARAMS) -def test_url_query_data_collection_breadcrumb( - sentry_init, capture_events, init_kwargs, expected_query + params = {"Bucket": "bucket", "Delete": deepcopy(delete)} + original = deepcopy(params) + caller_delete = params["Delete"] + caller_objects = caller_delete["Objects"] + + span = _stubbed_span(s3_client, capture_items, "delete_objects", params, {}) + + attributes = span.get("attributes", {}) + assert attributes[SPANDATA.AWS_S3_BUCKET] == "bucket" + assert attributes[SPANDATA.AWS_S3_DELETE] == expected_serialized_delete + assert SPANDATA.AWS_S3_KEY not in attributes + assert SPANDATA.AWS_S3_UPLOAD_ID not in attributes + assert SPANDATA.AWS_S3_COPY_SOURCE not in attributes + assert SPANDATA.AWS_S3_PART_NUMBER not in attributes + assert SPANDATA.FILE_SIZE not in attributes + assert SPANDATA.HTTP_BODY_SIZE not in attributes + assert params == original + assert params["Delete"] is caller_delete + assert caller_delete["Objects"] is caller_objects + + +@pytest.mark.parametrize( + "method,extra_params,response,expected", + [ + pytest.param( + "head_object", + {}, + {"ContentLength": 1024}, + {SPANDATA.FILE_SIZE: 1024}, + id="head-whole-object", + ), + pytest.param( + "head_object", + {}, + {"ContentLength": 0}, + {SPANDATA.FILE_SIZE: 0}, + id="head-empty-object", + ), + pytest.param( + "head_object", + {"Range": "bytes=0-3"}, + {"ContentLength": 4}, + {}, + id="head-range", + ), + pytest.param( + "head_object", {"PartNumber": 1}, {"ContentLength": 4}, {}, id="head-part" + ), + pytest.param( + "get_object", + {"Range": "bytes=0-3"}, + {"ContentLength": 4}, + {SPANDATA.HTTP_BODY_SIZE: 4}, + id="get-range", + ), + pytest.param( + "get_object_attributes", + {"ObjectAttributes": ["ObjectSize"]}, + {"ObjectSize": 1024}, + {SPANDATA.FILE_SIZE: 1024}, + id="object-attributes-size", + ), + pytest.param( + "put_object", + {"Body": b"data", "WriteOffsetBytes": 1020}, + {"Size": 1024}, + {SPANDATA.FILE_SIZE: 1024}, + id="put-append-size", + ), + pytest.param("put_object", {"Body": b"data"}, {}, {}, id="put-missing-size"), + pytest.param( + "complete_multipart_upload", + {"UploadId": "upload-id", "MpuObjectSize": 1024}, + {}, + {SPANDATA.FILE_SIZE: 1024, SPANDATA.AWS_S3_UPLOAD_ID: "upload-id"}, + id="multipart-request-size", + ), + ], +) +def test_size_attributes( + s3_client, capture_items, method, extra_params, response, expected ): - sentry_init( - integrations=[Boto3Integration()], - default_integrations=False, - **init_kwargs, + require_botocore_model_fields( + s3_client, + method, + input_fields=tuple( + field + for field in ("MpuObjectSize", "WriteOffsetBytes") + if field in extra_params + ), + output_fields=tuple(response), ) + params = {"Bucket": "bucket", "Key": "file.txt", **deepcopy(extra_params)} - client = session.client("s3") + span = _stubbed_span(s3_client, capture_items, method, params, response) - events = capture_events() - - with sentry_sdk.start_span(name="custom parent"), MockResponse( - client, 200, {}, read_fixture("s3_list.xml") + attributes = span.get("attributes", {}) + expected_attributes = { + SPANDATA.AWS_S3_BUCKET: "bucket", + SPANDATA.AWS_S3_KEY: "file.txt", + **expected, + } + for key, value in expected_attributes.items(): + assert attributes[key] == value + for key in ( + SPANDATA.AWS_S3_UPLOAD_ID, + SPANDATA.AWS_S3_COPY_SOURCE, + SPANDATA.AWS_S3_PART_NUMBER, + SPANDATA.AWS_S3_DELETE, + SPANDATA.FILE_SIZE, + SPANDATA.HTTP_BODY_SIZE, ): - client.list_objects_v2(Bucket="bucket", Prefix="foo", ContinuationToken="abc") - - capture_message("Testing!") - - (event,) = events - (crumb,) = event["breadcrumbs"]["values"] - - if expected_query is None: - assert SPANDATA.URL_QUERY not in crumb["data"] - assert SPANDATA.URL_FULL not in crumb["data"] - elif expected_query == "": - assert SPANDATA.URL_QUERY not in crumb["data"] - assert crumb["data"][SPANDATA.URL_FULL] == BUCKET_URL - else: - assert crumb["data"][SPANDATA.URL_QUERY] == expected_query - assert crumb["data"][SPANDATA.URL_FULL] == BUCKET_URL + "?" + expected_query + if key not in expected_attributes: + assert key not in attributes