diff --git a/sentry_sdk/consts.py b/sentry_sdk/consts.py
index e8874cd184..96f7b873ac 100644
--- a/sentry_sdk/consts.py
+++ b/sentry_sdk/consts.py
@@ -369,6 +369,42 @@ class SPANDATA:
Example: "79b9da39-b7ae-508a-a6bc-864b2829c622"
"""
+ AWS_S3_BUCKET = "aws.s3.bucket"
+ """
+ The S3 bucket name the request refers to.
+ Example: "ot-demo-test"
+ """
+
+ AWS_S3_COPY_SOURCE = "aws.s3.copy_source"
+ """
+ The source object (in the form bucket/key) for the copy operation.
+ Example: "someFile.yml"
+ """
+
+ AWS_S3_DELETE = "aws.s3.delete"
+ """
+ The delete request container that specifies the objects to be deleted.
+ Example: "Objects=[{Key=string,VersionId=string},{Key=string,VersionId=string}],Quiet=boolean"
+ """
+
+ AWS_S3_KEY = "aws.s3.key"
+ """
+ The S3 object key the request refers to. Corresponds to the --key parameter of the S3 API operations.
+ Example: "someFile.yml"
+ """
+
+ AWS_S3_PART_NUMBER = "aws.s3.part_number"
+ """
+ The part number of the part being uploaded in a multipart-upload operation. This is a positive integer between 1 and 10,000.
+ Example: 3456
+ """
+
+ AWS_S3_UPLOAD_ID = "aws.s3.upload_id"
+ """
+ Upload ID that identifies the multipart upload.
+ Example: "dfRtDYWFbkRONycy.Yxwh66Yjlx.cph0gtNBtJ"
+ """
+
CACHE_HIT = "cache.hit"
"""
A boolean indicating whether the requested data was found in the cache.
@@ -539,6 +575,12 @@ class SPANDATA:
Example: "timeout"
"""
+ FILE_SIZE = "file.size"
+ """
+ File size in bytes.
+ Example: 1024
+ """
+
GEN_AI_AGENT_NAME = "gen_ai.agent.name"
"""
The name of the agent being used.
@@ -908,6 +950,12 @@ class SPANDATA:
Example: ?foo=bar&bar=baz
"""
+ HTTP_BODY_SIZE = "http.response.body.size"
+ """
+ The encoded body size of the response (in bytes).
+ Example: 123
+ """
+
HTTP_STATUS_CODE = "http.response.status_code"
"""
The HTTP status code as an integer.
diff --git a/sentry_sdk/integrations/boto3/_client.py b/sentry_sdk/integrations/boto3/_client.py
index a1ef105613..10a5bb5c93 100644
--- a/sentry_sdk/integrations/boto3/_client.py
+++ b/sentry_sdk/integrations/boto3/_client.py
@@ -100,11 +100,9 @@ def sentry_patched_make_api_call(
return orig_make_api_call(self, operation_name, api_params)
# activate without finishing; a streaming response may outlive the call.
- span_ctx = _activate_client_span(span)
-
attributes: "Attributes" = {}
try:
- with span_ctx:
+ with _activate_client_span(span):
try:
parsed = orig_make_api_call(self, operation_name, api_params)
except BaseException as error:
diff --git a/sentry_sdk/integrations/boto3/_instrumentation.py b/sentry_sdk/integrations/boto3/_instrumentation.py
index 8a05e3c544..e3ee4fd7d6 100644
--- a/sentry_sdk/integrations/boto3/_instrumentation.py
+++ b/sentry_sdk/integrations/boto3/_instrumentation.py
@@ -207,7 +207,13 @@ def _instrument_streaming_body(span: "Span", parsed: "Dict[str, Any]") -> bool:
if isinstance(span, NoOpSpan):
return False
- body = parsed.get("Body")
+ # botocore uses service-specific response key for streaming payload; e.g. `Payload`
+ # for Lambda Invoke, `Body` for S3 GetObject. Find it by type so every streaming
+ # response is handled.
+ body = next(
+ (value for value in parsed.values() if isinstance(value, StreamingBody)),
+ None,
+ )
if not isinstance(body, StreamingBody):
return False
diff --git a/sentry_sdk/integrations/boto3/_services/registry.py b/sentry_sdk/integrations/boto3/_services/registry.py
index b387db0f26..f9291d04df 100644
--- a/sentry_sdk/integrations/boto3/_services/registry.py
+++ b/sentry_sdk/integrations/boto3/_services/registry.py
@@ -1,5 +1,7 @@
from typing import TYPE_CHECKING
+from sentry_sdk.integrations.boto3._services.s3 import _S3Extension
+
if TYPE_CHECKING:
from typing import Dict, Optional
@@ -10,7 +12,9 @@
# _SERVICE_EXTENSIONS = {"s3": _S3Extension()}
# when py 3.15 drops, we might want to take a look at using
# a lazy-loading approach using the new `lazy` keyword.
-_SERVICE_EXTENSIONS: "Dict[str, _ServiceExtension]" = {}
+_SERVICE_EXTENSIONS: "Dict[str, _ServiceExtension]" = {
+ "s3": _S3Extension(),
+}
def _resolve_service(
diff --git a/sentry_sdk/integrations/boto3/_services/s3.py b/sentry_sdk/integrations/boto3/_services/s3.py
new file mode 100644
index 0000000000..1c2a6626e1
--- /dev/null
+++ b/sentry_sdk/integrations/boto3/_services/s3.py
@@ -0,0 +1,107 @@
+import json
+from datetime import datetime
+from typing import TYPE_CHECKING
+
+from sentry_sdk.consts import SPANDATA
+from sentry_sdk.integrations.boto3._services.base import _ServiceExtension
+from sentry_sdk.integrations.boto3._utils import (
+ _extract_attributes,
+)
+from sentry_sdk.utils import capture_internal_exceptions
+
+if TYPE_CHECKING:
+ from typing import Any, Sequence
+
+ from sentry_sdk._types import Attributes
+ from sentry_sdk.integrations.boto3._context import AwsCallContext
+ from sentry_sdk.integrations.boto3._utils import (
+ _AttributeSpec,
+ )
+
+# maps s3 operations to the response field that contains the complete file size.
+_RESPONSE_FILE_SIZE_FIELDS = {
+ "GetObjectAttributes": "ObjectSize",
+ "PutObject": "Size",
+}
+
+# s3 request attributes that are extracted when present, regardless of operation.
+_REQUEST_ATTRIBUTES: "Sequence[_AttributeSpec]" = (
+ ("Bucket", SPANDATA.AWS_S3_BUCKET),
+ ("Key", SPANDATA.AWS_S3_KEY),
+ ("UploadId", SPANDATA.AWS_S3_UPLOAD_ID),
+)
+
+
+class _S3Extension(_ServiceExtension):
+ __slots__ = ()
+
+ def get_request_attributes(self, ctx: "AwsCallContext") -> "Attributes":
+ attributes: "Attributes" = _extract_attributes(ctx.params, _REQUEST_ATTRIBUTES)
+
+ if "CopySource" in ctx.params:
+ with capture_internal_exceptions():
+ # boto3 accepts either "bucket/key" or a dict {"bucket": ..., "Key": ..., "VersionId": ...} for `CopySource`.
+ # https://docs.aws.amazon.com/boto3/latest/reference/services/s3/client/upload_part_copy.html
+ copy_source = ctx.params["CopySource"]
+ if isinstance(copy_source, str):
+ attributes[SPANDATA.AWS_S3_COPY_SOURCE] = copy_source
+ else:
+ value = f"{copy_source['Bucket']}/{copy_source['Key']}"
+ if "VersionId" in copy_source:
+ value += f"?versionId={copy_source['VersionId']}"
+ attributes[SPANDATA.AWS_S3_COPY_SOURCE] = value
+
+ # OTel defines `PartNumber` for `UploadPart` and `UploadPartCopy` only.
+ # https://opentelemetry.io/docs/specs/semconv/object-stores/s3/#attributes
+ if (
+ ctx.operation_name in ("UploadPart", "UploadPartCopy")
+ and "PartNumber" in ctx.params
+ ):
+ attributes[SPANDATA.AWS_S3_PART_NUMBER] = ctx.params["PartNumber"]
+
+ if "Delete" in ctx.params:
+ with capture_internal_exceptions():
+ attributes[SPANDATA.AWS_S3_DELETE] = json.dumps(
+ ctx.params["Delete"],
+ default=lambda value: (
+ value.isoformat() if isinstance(value, datetime) else str(value)
+ ),
+ separators=(",", ":"),
+ sort_keys=True,
+ )
+
+ if (
+ ctx.operation_name == "CompleteMultipartUpload"
+ and "MpuObjectSize" in ctx.params
+ ):
+ attributes[SPANDATA.FILE_SIZE] = ctx.params["MpuObjectSize"]
+
+ return attributes
+
+ def get_response_attributes(
+ self, ctx: "AwsCallContext", response: "Any"
+ ) -> "Attributes":
+ attributes: "Attributes" = {}
+
+ if (
+ ctx.operation_name in ("GetObject", "GetObjectAnnotation")
+ and "ContentLength" in response
+ ):
+ # `ContentLength` is the size of the HTTP body returned, which may be a range.
+ attributes[SPANDATA.HTTP_BODY_SIZE] = response["ContentLength"]
+
+ # report the complete file size, not just the HTTP body size.
+ file_size_field = _RESPONSE_FILE_SIZE_FIELDS.get(ctx.operation_name)
+ if file_size_field is not None and file_size_field in response:
+ attributes[SPANDATA.FILE_SIZE] = response[file_size_field]
+
+ if (
+ ctx.operation_name == "HeadObject"
+ and "Range" not in ctx.params
+ and "PartNumber" not in ctx.params
+ and "ContentLength" in response
+ ):
+ # an un-ranged `HEAD` has no body, so `ContentLength` is the file size.
+ attributes[SPANDATA.FILE_SIZE] = response["ContentLength"]
+
+ return attributes
diff --git a/sentry_sdk/integrations/boto3/_utils.py b/sentry_sdk/integrations/boto3/_utils.py
new file mode 100644
index 0000000000..73d1031c0f
--- /dev/null
+++ b/sentry_sdk/integrations/boto3/_utils.py
@@ -0,0 +1,19 @@
+from typing import TYPE_CHECKING
+
+if TYPE_CHECKING:
+ from typing import Any, Dict, Sequence, Tuple
+
+ from sentry_sdk._types import Attributes
+
+ # tuple of (request param name, span attribute name).
+ _AttributeSpec = Tuple[str, str]
+
+
+def _extract_attributes(
+ source: "Dict[str, Any]", specs: "Sequence[_AttributeSpec]"
+) -> "Attributes":
+ attributes = {}
+ for param, attribute in specs:
+ if param in source:
+ attributes[attribute] = source[param]
+ return attributes
diff --git a/tests/integrations/boto3/helpers.py b/tests/integrations/boto3/helpers.py
new file mode 100644
index 0000000000..ec9d146c4d
--- /dev/null
+++ b/tests/integrations/boto3/helpers.py
@@ -0,0 +1,115 @@
+from types import SimpleNamespace
+from typing import TYPE_CHECKING, Dict, List, Optional
+
+import boto3
+import pytest
+from botocore.config import Config
+
+import sentry_sdk
+from sentry_sdk.consts import SPANDATA
+from sentry_sdk.integrations.boto3 import Boto3Integration
+from sentry_sdk.integrations.boto3.consts import ORIGIN
+
+if TYPE_CHECKING:
+ from sentry_sdk._types import SpanJSON
+
+
+@pytest.fixture
+def client_factory(sentry_init, monkeypatch):
+ sentry_init(
+ traces_sample_rate=1.0,
+ integrations=[Boto3Integration()],
+ )
+ session = boto3.Session( # type: ignore
+ aws_access_key_id="-",
+ aws_secret_access_key="-",
+ region_name="eu-north-1",
+ )
+ clients = []
+
+ def make_client(service_name="s3", attempt_count=1, **client_kwargs):
+ client = session.client(
+ service_name,
+ config=Config(
+ retries={"total_max_attempts": attempt_count, "mode": "standard"}
+ ),
+ **client_kwargs,
+ ) # type: ignore
+ clients.append(client)
+ return client
+
+ yield make_client
+
+ for client in clients:
+ # older supported botocore versions do not expose `BaseClient.close()`.
+ close = getattr(client, "close", None)
+ if close is not None:
+ close()
+
+
+@pytest.fixture
+def no_botocore_retry_delay(monkeypatch):
+ # remove request retry delays without replacing botocore's retry handling.
+ monkeypatch.setattr(
+ "botocore.endpoint.time",
+ SimpleNamespace(sleep=lambda delay: None),
+ )
+
+
+@pytest.fixture
+def s3_client(client_factory):
+ return client_factory("s3")
+
+
+def require_botocore_model_fields(
+ client,
+ method,
+ input_fields=(),
+ output_fields=(),
+):
+ """Skip tests when botocore lacks required fields, including nested paths."""
+
+ def has_field(shape, field):
+ for part in field.split("."):
+ while shape is not None and shape.type_name == "list":
+ shape = shape.member
+ if shape is None or part not in getattr(shape, "members", {}):
+ return False
+ shape = shape.members[part]
+ return True
+
+ operation_name = client.meta.method_to_api_mapping.get(method)
+ if operation_name is None:
+ pytest.skip("%s is absent from this botocore model; skipping test" % method)
+ model = client.meta.service_model.operation_model(operation_name) # type: ignore
+ for shape, fields in (
+ (model.input_shape, input_fields),
+ (model.output_shape, output_fields),
+ ):
+ for field in fields:
+ if not has_field(shape, field):
+ pytest.skip(
+ "%s.%s is absent from this botocore model; skipping test"
+ % (method, field)
+ )
+
+
+def capture_spans_by_op(
+ invoke_client_method,
+ capture_items,
+ expected_origin=ORIGIN,
+):
+ items = capture_items("span")
+
+ with sentry_sdk.start_span(name="parent"):
+ invoke_client_method()
+
+ sentry_sdk.flush()
+ spans_by_op: Dict[Optional[str], List["SpanJSON"]] = {}
+ for item in items:
+ span = item.payload
+ if span["attributes"].get(SPANDATA.SENTRY_ORIGIN) == expected_origin:
+ spans_by_op.setdefault(
+ span["attributes"].get(SPANDATA.SENTRY_OP), []
+ ).append(span)
+ return spans_by_op
diff --git a/tests/integrations/boto3/s3_list.xml b/tests/integrations/boto3/s3_list.xml
index 10d5b16340..e3ca2073b3 100644
--- a/tests/integrations/boto3/s3_list.xml
+++ b/tests/integrations/boto3/s3_list.xml
@@ -1,2 +1,29 @@
-marshalls-furious-bucket1000urlfalsefoo.txt2020-10-24T00:13:39.000Z"a895ba674b4abd01b5d67cfd7074b827"2064537bef397f7e536914d1ff1bbdb105ed90bcfd06269456bf4a06c6e2e54564daf7STANDARDbar.txt2020-10-02T15:15:20.000Z"a895ba674b4abd01b5d67cfd7074b827"2064537bef397f7e536914d1ff1bbdb105ed90bcfd06269456bf4a06c6e2e54564daf7STANDARD
+
+ marshalls-furious-bucket
+
+
+ 1000
+ url
+ false
+
+ foo.txt
+ 2020-10-24T00:13:39.000Z
+ "a895ba674b4abd01b5d67cfd7074b827"
+ 206453
+
+ 7bef397f7e536914d1ff1bbdb105ed90bcfd06269456bf4a06c6e2e54564daf7
+
+ STANDARD
+
+
+ bar.txt
+ 2020-10-02T15:15:20.000Z
+ "a895ba674b4abd01b5d67cfd7074b827"
+ 206453
+
+ 7bef397f7e536914d1ff1bbdb105ed90bcfd06269456bf4a06c6e2e54564daf7
+
+ STANDARD
+
+
diff --git a/tests/integrations/boto3/test_client.py b/tests/integrations/boto3/test_client.py
index 1113833a80..3d1d1b8dd4 100644
--- a/tests/integrations/boto3/test_client.py
+++ b/tests/integrations/boto3/test_client.py
@@ -1,5 +1,6 @@
from http.server import BaseHTTPRequestHandler, HTTPServer
from threading import Thread
+from unittest import mock
import boto3
import pytest
@@ -10,6 +11,7 @@
from botocore.stub import Stubber
import sentry_sdk
+from sentry_sdk import capture_message
from sentry_sdk.consts import OP, SPANDATA
from sentry_sdk.integrations.boto3 import Boto3Integration
from sentry_sdk.integrations.boto3._services.base import _ServiceExtension
@@ -21,7 +23,21 @@
)
from sentry_sdk.integrations.stdlib import StdlibIntegration
from sentry_sdk.traces import Span
+from tests.conftest import ApproxDict
+from tests.integrations.boto3 import read_fixture
from tests.integrations.boto3.aws_mock import Body, MockResponse
+from tests.integrations.boto3.helpers import (
+ capture_spans_by_op as _capture_boto3_spans_by_op,
+)
+from tests.integrations.boto3.helpers import (
+ client_factory as client_factory,
+)
+from tests.integrations.boto3.helpers import (
+ no_botocore_retry_delay as no_botocore_retry_delay,
+)
+from tests.integrations.boto3.helpers import (
+ require_botocore_model_fields,
+)
session = boto3.Session( # type: ignore[attr-defined]
aws_access_key_id="-",
@@ -41,7 +57,7 @@ def do_GET(self):
self.wfile.write(b"x")
self.wfile.flush()
- def log_message(self, *args):
+ def log_message(self, *args): # type: ignore
pass
server = HTTPServer(("127.0.0.1", 0), StreamingS3Handler)
@@ -129,22 +145,23 @@ def record_client_span(request, **kwargs):
for span in spans
if span["name"] == "S3.GetObject"
and (
- span["attributes"].get(SPANDATA.SENTRY_ORIGIN) == ORIGIN
- and span["attributes"].get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT
+ span.get("attributes", {}).get(SPANDATA.SENTRY_ORIGIN) == ORIGIN
+ and span.get("attributes", {}).get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT
)
]
http_spans = [
span
for span in spans
- if (
- span["attributes"].get(SPANDATA.SENTRY_ORIGIN) == "auto.http.stdlib.httplib"
- )
+ if span.get("attributes", {}).get(SPANDATA.SENTRY_ORIGIN)
+ == "auto.http.stdlib.httplib"
]
stream_spans = [
span
for span in spans
if span["name"] == "S3.GetObject"
- and (span["attributes"].get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT_STREAM)
+ and (
+ span.get("attributes", {}).get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT_STREAM
+ )
]
assert len(client_spans) == 1
assert len(http_spans) == 1
@@ -153,72 +170,218 @@ def record_client_span(request, **kwargs):
http_span = http_spans[0]
stream_span = stream_spans[0]
- assert http_span["parent_span_id"] == client_span["span_id"]
- assert stream_span["parent_span_id"] == client_span["span_id"]
+ assert http_span.get("parent_span_id") == client_span["span_id"]
+ assert stream_span.get("parent_span_id") == client_span["span_id"]
assert client_span["span_id"] == request_client_span.span_id
for span in (client_span, http_span, stream_span):
- assert span["end_timestamp"] is not None
+ assert span.get("end_timestamp") is not None
+
+
+@pytest.mark.parametrize(
+ "service_name,method_name,api_params,payload_field",
+ [
+ pytest.param(
+ "lambda",
+ "invoke",
+ {"FunctionName": "function"},
+ "Payload",
+ id="payload",
+ ),
+ pytest.param(
+ "s3",
+ "get_object_annotation",
+ {
+ "Bucket": "bucket",
+ "Key": "file.txt",
+ "AnnotationName": "annotation",
+ },
+ "AnnotationPayload",
+ id="annotation-payload",
+ ),
+ ],
+)
+def test_non_body_stream_delays_client_span(
+ capture_items,
+ client_factory,
+ service_name,
+ method_name,
+ api_params,
+ payload_field,
+):
+ client = client_factory(service_name)
+ require_botocore_model_fields(client, method_name, output_fields=(payload_field,))
+ request_client_spans = []
+
+ def record_client_span(request, **kwargs):
+ request_client_spans.append(request.context["_sentrysdk_span"])
+
+ client.meta.events.register("request-created", record_client_span)
+
+ def invoke():
+ parent = sentry_sdk.get_current_span()
+ with MockResponse(client, 200, {"content-length": "1"}, b"x"):
+ response = getattr(client, method_name)(**api_params)
+ body = response[payload_field]
+ assert isinstance(body, StreamingBody)
+ (request_client_span,) = request_client_spans
+ assert request_client_span.end_timestamp is None
+ assert sentry_sdk.get_current_span() is parent
+ body.close()
+ assert request_client_span.end_timestamp is not None
+ assert sentry_sdk.get_current_span() is parent
+
+ spans_by_op = _capture_boto3_spans_by_op(invoke, capture_items)
+ (client_span,) = spans_by_op[OP.HTTP_CLIENT]
+ (stream_span,) = spans_by_op[OP.HTTP_CLIENT_STREAM]
+ assert stream_span.get("parent_span_id") == client_span["span_id"]
+
+
+@pytest.mark.tests_internal_exceptions
+def test_omit_url_data_if_parsing_fails(capture_items, client_factory):
+ client = client_factory()
+
+ with mock.patch(
+ "sentry_sdk.integrations.boto3._instrumentation.parse_url",
+ side_effect=ValueError,
+ ) as parse_url:
+ with MockResponse(client, 200, {}, b""):
+ spans_by_op = _capture_boto3_spans_by_op(
+ lambda: client.head_object(Bucket="bucket", Key="file.txt"),
+ capture_items,
+ )
+
+ parse_url.assert_called()
+ (span,) = spans_by_op[OP.HTTP_CLIENT]
+ attributes = span.get("attributes", {})
+ assert SPANDATA.URL_FULL not in attributes
+ assert SPANDATA.URL_FRAGMENT not in attributes
+ assert SPANDATA.URL_QUERY not in attributes
+
+
+BUCKET_URL = "https://bucket.s3.eu-north-1.amazonaws.com/"
+
+URL_QUERY_PARAMS = [
+ pytest.param(
+ {},
+ "list-type=2&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=url",
+ id="defaults",
+ ),
+ pytest.param(
+ {
+ "data_collection": {
+ "url_query_params": {"mode": "denylist", "terms": ["prefix"]}
+ }
+ },
+ "list-type=2&prefix=%5BFiltered%5D&continuation-token=%5BFiltered%5D&encoding-type=url",
+ id="data_collection_denylist_custom_terms",
+ ),
+ pytest.param(
+ {
+ "data_collection": {
+ "url_query_params": {"mode": "allowlist", "terms": ["prefix"]}
+ }
+ },
+ "list-type=%5BFiltered%5D&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=%5BFiltered%5D",
+ id="data_collection_allowlist",
+ ),
+ pytest.param(
+ {
+ "data_collection": {
+ "url_query_params": {
+ "mode": "allowlist",
+ "terms": ["continuation-token"],
+ }
+ }
+ },
+ "list-type=%5BFiltered%5D&prefix=%5BFiltered%5D&continuation-token=%5BFiltered%5D&encoding-type=%5BFiltered%5D",
+ id="data_collection_allowlist_sensitive_term",
+ ),
+ pytest.param(
+ {"data_collection": {"url_query_params": {"mode": "off"}}},
+ "",
+ id="data_collection_off",
+ ),
+]
-def test_non_body_stream_does_not_delay_client_span(sentry_init, capture_items):
+@pytest.mark.parametrize("init_kwargs, expected_query", URL_QUERY_PARAMS)
+def test_url_query_data_collection(
+ sentry_init, capture_items, init_kwargs, expected_query
+):
sentry_init(
traces_sample_rate=1.0,
integrations=[Boto3Integration()],
- server_name="",
+ default_integrations=False,
+ **init_kwargs,
)
- client = session.client("lambda")
-
- def respond(request, **kwargs):
- return AWSResponse(
- request.url,
- 200,
- {"content-length": "1"},
- Body(b"x"),
- )
-
- client.meta.events.register("before-send", respond)
+ client = session.client("s3")
items = capture_items("span")
- with sentry_sdk.start_span(name="parent") as parent: # type: ignore[attr-defined]
- response = client.invoke(FunctionName="function")
- assert isinstance(response["Payload"], StreamingBody)
- assert sentry_sdk.get_current_span() is parent # type: ignore[attr-defined]
+ with sentry_sdk.start_span(name="parent"), MockResponse(
+ client, 200, {}, read_fixture("s3_list.xml")
+ ):
+ client.list_objects_v2(Bucket="bucket", Prefix="foo", ContinuationToken="abc")
sentry_sdk.flush()
-
- spans = [item.payload for item in items]
- boto_spans = [
- span
- for span in spans
- if span["attributes"].get(SPANDATA.SENTRY_ORIGIN) == ORIGIN
+ (span,) = [
+ item.payload
+ for item in items
+ if item.payload.get("attributes", {}).get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT
]
- assert len(boto_spans) == 1
- assert boto_spans[0]["attributes"].get(SPANDATA.SENTRY_OP) == OP.HTTP_CLIENT
- response["Payload"].close()
+ attributes = span.get("attributes", {})
+ if expected_query == "":
+ assert SPANDATA.URL_QUERY not in attributes
+ assert attributes[SPANDATA.URL_FULL] == BUCKET_URL
+ else:
+ assert attributes[SPANDATA.URL_QUERY] == expected_query
+ assert attributes[SPANDATA.URL_FULL] == BUCKET_URL + "?" + expected_query
-@pytest.fixture
-def client_factory(sentry_init, monkeypatch):
+
+@pytest.mark.parametrize(
+ "data_collection,expected_query",
+ [
+ pytest.param(
+ {},
+ "list-type=2&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=url",
+ id="default",
+ ),
+ pytest.param(
+ {"url_query_params": {"mode": "off"}},
+ None,
+ id="query-collection-off",
+ ),
+ ],
+)
+def test_breadcrumb(sentry_init, capture_events, data_collection, expected_query):
sentry_init(
- traces_sample_rate=1.0,
integrations=[Boto3Integration()],
- # avoid SDK's machine hostname being used as server name.
- server_name="",
+ default_integrations=False,
+ data_collection=data_collection,
)
- # remove retry delay to speed up tests
- monkeypatch.setattr("botocore.endpoint.time.sleep", lambda delay: None)
-
- def make_client(service_name="s3", attempt_count=1, **client_kwargs):
- return session.client(
- service_name,
- config=Config(
- # `total_max_attempts` includes the initial request.
- retries={"total_max_attempts": attempt_count, "mode": "standard"}
- ),
- **client_kwargs,
- )
+ client = session.client("s3")
+ events = capture_events()
- return make_client
+ with MockResponse(client, 200, {}, read_fixture("s3_list.xml")):
+ client.list_objects_v2(Bucket="bucket", Prefix="foo", ContinuationToken="abc")
+
+ capture_message("Testing!")
+ (event,) = events
+ (crumb,) = event["breadcrumbs"]["values"]
+ assert crumb["type"] == "http"
+ assert crumb["category"] == "httplib"
+ assert SPANDATA.URL_FRAGMENT not in crumb["data"]
+
+ if expected_query is None:
+ assert SPANDATA.URL_QUERY not in crumb["data"]
+ assert crumb["data"][SPANDATA.URL_FULL] == BUCKET_URL
+ else:
+ assert crumb["data"] == ApproxDict(
+ {
+ SPANDATA.URL_FULL: BUCKET_URL + "?" + expected_query,
+ SPANDATA.URL_QUERY: expected_query,
+ }
+ )
def _mock_responses(client, status_codes):
@@ -233,44 +396,19 @@ def respond(request, **kwargs):
# `request_created` runs before `before_send`, so use zero-based index for current
# attempt; `min(..., len(status_codes) - 1)` clamps to last status to avoid `IndexError`.
response_index = min(len(request_span_ids) - 1, len(status_codes) - 1)
- return AWSResponse(request.url, status_codes[response_index], {}, Body(b""))
+ return AWSResponse(request.url, status_codes[response_index], {}, Body(b"")) # type: ignore
client.meta.events.register("request-created", record_request)
client.meta.events.register("before-send", respond)
return request_span_ids
-def _capture_boto3_spans_by_op(
- invoke_client_method,
- capture_items,
- expected_origin=ORIGIN,
-):
- items = capture_items()
-
- with sentry_sdk.start_span(name="parent"):
- invoke_client_method()
-
- sentry_sdk.flush()
- spans = [
- item.payload
- for item in items
- if item.type == "span"
- and item.payload["attributes"].get(SPANDATA.SENTRY_ORIGIN) == expected_origin
- ]
-
- spans_by_op = {}
- for span in spans:
- spans_by_op.setdefault(span["attributes"].get(SPANDATA.SENTRY_OP), []).append(
- span
- )
- return spans_by_op
-
-
def _assert_one_failed_span(spans):
assert len(spans) == 1
assert spans[0]["status"] == "error"
- assert spans[0]["attributes"][SPANDATA.ERROR_TYPE]
- assert spans[0]["end_timestamp"] is not None
+ attributes = spans[0].get("attributes", {})
+ assert attributes[SPANDATA.ERROR_TYPE]
+ assert spans[0].get("end_timestamp") is not None
def _capture_stubbed_client_span(
@@ -288,6 +426,7 @@ def _capture_stubbed_client_span(
lambda: getattr(client, method_name)(**api_params),
capture_items,
)
+ stubber.assert_no_pending_responses()
client_spans = spans_by_op.get(OP.HTTP_CLIENT, [])
assert len(client_spans) == 1
@@ -342,15 +481,24 @@ def get_response_attributes(self, ctx, response):
spans = spans_by_op.get("aws.test", [])
assert len(spans) == 1
- attributes = spans[0]["attributes"]
- assert attributes["aws.test.request"] == "foo"
- assert attributes["aws.test.response"] == "request-id"
+ span = spans[0]
+ attributes = span.get("attributes", {})
+ assert span["name"] == "S3.HeadObject"
+ assert span.get("end_timestamp") is not None
+ assert attributes[SPANDATA.SENTRY_OP] == "aws.test"
+ assert attributes[SPANDATA.SENTRY_ORIGIN] == "auto.aws.test"
assert attributes[SPANDATA.SENTRY_KIND] == "producer"
+ assert attributes[SPANDATA.CLOUD_PROVIDER] == CLOUD_PROVIDER
+ assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME
+ assert attributes[SPANDATA.RPC_SERVICE] == "S3"
assert attributes[SPANDATA.RPC_METHOD] == "HeadObject"
+ assert attributes[SPANDATA.CLOUD_REGION] == "eu-north-1"
+ assert attributes[SPANDATA.SERVER_ADDRESS] == "s3.eu-north-1.amazonaws.com"
+ assert attributes[SPANDATA.SERVER_PORT] == 443
+ assert attributes["aws.test.request"] == "foo"
+ assert attributes["aws.test.response"] == "request-id"
assert attributes[SPANDATA.HTTP_STATUS_CODE] == 200
assert attributes[SPANDATA.AWS_EXTENDED_REQUEST_ID] == "extended-request-id"
- assert spans[0]["end_timestamp"] is not None
- assert attributes[SPANDATA.SENTRY_ORIGIN] == "auto.aws.test"
@pytest.mark.parametrize(
@@ -428,22 +576,24 @@ def test_client_call_has_common_attributes(
}
},
)
- attributes = span["attributes"]
+ attributes = span.get("attributes", {})
assert span["name"] == span_name
+ assert span.get("end_timestamp") is not None
+ assert attributes[SPANDATA.SENTRY_OP] == OP.HTTP_CLIENT
+ assert attributes[SPANDATA.SENTRY_ORIGIN] == ORIGIN
+ assert attributes[SPANDATA.SENTRY_KIND] == "client"
+ assert attributes[SPANDATA.CLOUD_PROVIDER] == CLOUD_PROVIDER
+ assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME
assert attributes[SPANDATA.RPC_SERVICE] == rpc_service
assert attributes[SPANDATA.RPC_METHOD] == rpc_method
- assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME
- assert attributes[SPANDATA.SENTRY_KIND] == "client"
assert attributes[SPANDATA.CLOUD_REGION] == "eu-north-1"
- assert attributes[SPANDATA.CLOUD_PROVIDER] == CLOUD_PROVIDER
assert attributes[SPANDATA.SERVER_ADDRESS] == server_address
assert attributes[SPANDATA.SERVER_PORT] == server_port
assert attributes[SPANDATA.HTTP_STATUS_CODE] == 200
assert attributes[SPANDATA.AWS_REQUEST_ID] == "request-id"
assert SPANDATA.HTTP_REQUEST_RESEND_COUNT not in attributes
assert SPANDATA.ERROR_TYPE not in attributes
- assert span["end_timestamp"] is not None
def test_client_call_attributes_are_available_at_span_creation(
@@ -482,7 +632,7 @@ def test_client_call_attributes_are_available_at_span_creation(
client_spans = [
item.payload
for item in items
- if item.payload["attributes"].get(SPANDATA.SENTRY_ORIGIN) == ORIGIN
+ if item.payload.get("attributes", {}).get(SPANDATA.SENTRY_ORIGIN) == ORIGIN
]
assert client_spans == []
@@ -503,14 +653,16 @@ def test_client_call_has_response_header_attributes(
spans = spans_by_op[OP.HTTP_CLIENT]
assert len(spans) == 1
- attributes = spans[0]["attributes"]
+ attributes = spans[0].get("attributes", {})
assert attributes[SPANDATA.HTTP_STATUS_CODE] == 200
assert attributes[SPANDATA.AWS_REQUEST_ID] == "request-id"
assert attributes[SPANDATA.AWS_EXTENDED_REQUEST_ID] == "extended-request-id"
assert SPANDATA.HTTP_REQUEST_RESEND_COUNT not in attributes
-def test_retry_attempts_share_one_client_span(capture_items, client_factory):
+def test_retry_attempts_share_one_client_span(
+ capture_items, client_factory, no_botocore_retry_delay
+):
attempt_count = 3
client = client_factory(attempt_count=attempt_count)
request_span_ids = _mock_responses(client, [500] * (attempt_count - 1) + [200])
@@ -524,11 +676,13 @@ def test_retry_attempts_share_one_client_span(capture_items, client_factory):
# all `AWSRequest` instances created during retries reference the same client span.
assert len(set(request_span_ids)) == 1
assert len(client_spans) == 1
- attributes = client_spans[0]["attributes"]
+ attributes = client_spans[0].get("attributes", {})
assert attributes[SPANDATA.HTTP_REQUEST_RESEND_COUNT] == attempt_count - 1
-def test_retries_exhausted_has_one_failed_client_span(capture_items, client_factory):
+def test_retries_exhausted_has_one_failed_client_span(
+ capture_items, client_factory, no_botocore_retry_delay
+):
client = client_factory(attempt_count=2)
request_span_ids = _mock_responses(client, [500])
@@ -544,7 +698,7 @@ def attempt_failed_head_object_call():
assert len(request_span_ids) == 2
assert len(set(request_span_ids)) == 1
_assert_one_failed_span(client_spans)
- attributes = client_spans[0]["attributes"]
+ attributes = client_spans[0].get("attributes", {})
assert attributes[SPANDATA.HTTP_STATUS_CODE] == 500
assert attributes[SPANDATA.HTTP_REQUEST_RESEND_COUNT] == 1
@@ -578,7 +732,7 @@ def get_response_attributes(self, ctx, response):
"HTTPStatusCode": 403,
"RetryAttempts": 1,
},
- },
+ }, # type: ignore
"HeadObject",
)
@@ -597,8 +751,21 @@ def invoke_failing_client_method():
)
client_spans = spans_by_op.get(OP.HTTP_CLIENT, [])
_assert_one_failed_span(client_spans)
- attributes = client_spans[0]["attributes"]
+ attributes = client_spans[0].get("attributes", {})
+ span = client_spans[0]
+ assert span["name"] == "S3.HeadObject"
+ assert span.get("end_timestamp") is not None
+ assert attributes[SPANDATA.SENTRY_OP] == OP.HTTP_CLIENT
+ assert attributes[SPANDATA.SENTRY_ORIGIN] == ORIGIN
+ assert attributes[SPANDATA.SENTRY_KIND] == "client"
+ assert attributes[SPANDATA.CLOUD_PROVIDER] == CLOUD_PROVIDER
+ assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME
+ assert attributes[SPANDATA.RPC_SERVICE] == "S3"
+ assert attributes[SPANDATA.RPC_METHOD] == "HeadObject"
+ assert attributes[SPANDATA.CLOUD_REGION] == "eu-north-1"
+ assert attributes[SPANDATA.SERVER_ADDRESS] == "s3.eu-north-1.amazonaws.com"
+ assert attributes[SPANDATA.SERVER_PORT] == 443
assert attributes[SPANDATA.AWS_REQUEST_ID] == "request-id"
assert attributes[SPANDATA.HTTP_STATUS_CODE] == 403
assert attributes[SPANDATA.HTTP_REQUEST_RESEND_COUNT] == 1
@@ -651,7 +818,8 @@ def invoke_failing_client_method():
if event_name == "before-send"
else "ValueError"
)
- assert client_spans[0]["attributes"][SPANDATA.ERROR_TYPE] == expected_error_type
+ attributes = client_spans[0].get("attributes", {})
+ assert attributes[SPANDATA.ERROR_TYPE] == expected_error_type
@pytest.mark.tests_internal_exceptions
@@ -693,7 +861,7 @@ def invoke_client_method():
assert returned_responses[0] is original_response
if failing_instrumentation == "_get_response_attributes":
assert len(client_spans) == 1
- assert client_spans[0]["end_timestamp"] is not None
+ assert client_spans[0].get("end_timestamp") is not None
else:
assert client_spans == []
@@ -729,7 +897,7 @@ def invoke_failing_client_method():
assert len(client_spans) == 1
assert client_spans[0]["status"] == "error"
- assert client_spans[0]["end_timestamp"] is not None
+ assert client_spans[0].get("end_timestamp") is not None
def test_streaming_response_attributes_belong_to_client_span(
@@ -744,7 +912,7 @@ def respond(request, **kwargs):
{
"content-length": "5",
"x-amz-request-id": "request-id",
- },
+ }, # type: ignore
Body(b"hello"),
)
@@ -763,8 +931,8 @@ def invoke_client_method_and_read_body():
assert len(client_spans) == 1
assert len(stream_spans) == 1
- client_attributes = client_spans[0]["attributes"]
- stream_attributes = stream_spans[0]["attributes"]
+ client_attributes = client_spans[0].get("attributes", {})
+ stream_attributes = stream_spans[0].get("attributes", {})
assert client_attributes[SPANDATA.AWS_REQUEST_ID] == "request-id"
assert client_attributes[SPANDATA.HTTP_STATUS_CODE] == 200
assert SPANDATA.HTTP_REQUEST_RESEND_COUNT not in client_attributes
@@ -792,7 +960,7 @@ def respond(request, **kwargs):
return AWSResponse(
request.url,
200,
- {"content-length": "1"},
+ {"content-length": "1"}, # type: ignore
_FailingBody(original_exception),
)
@@ -813,4 +981,5 @@ def invoke_client_method_and_read_body():
assert len(client_spans) == 1
_assert_one_failed_span(client_spans)
_assert_one_failed_span(stream_spans)
- assert stream_spans[0]["attributes"][SPANDATA.ERROR_TYPE] == "OSError"
+ attributes = stream_spans[0].get("attributes", {})
+ assert attributes[SPANDATA.ERROR_TYPE] == "OSError"
diff --git a/tests/integrations/boto3/test_s3.py b/tests/integrations/boto3/test_s3.py
index 4cf66d26c2..918ea10063 100644
--- a/tests/integrations/boto3/test_s3.py
+++ b/tests/integrations/boto3/test_s3.py
@@ -1,381 +1,278 @@
-from unittest import mock
+import json
+from copy import deepcopy
+from datetime import datetime, timezone
-import boto3
import pytest
+from botocore.stub import Stubber
-import sentry_sdk
-from sentry_sdk import capture_message
-from sentry_sdk.consts import SPANDATA
-from sentry_sdk.integrations.boto3 import Boto3Integration
-from sentry_sdk.integrations.boto3.consts import ORIGIN
-from tests.conftest import ApproxDict
-from tests.integrations.boto3 import read_fixture
-from tests.integrations.boto3.aws_mock import MockResponse
-
-session = boto3.Session(
- aws_access_key_id="-",
- aws_secret_access_key="-",
+from sentry_sdk.consts import OP, SPANDATA
+from sentry_sdk.integrations.boto3.consts import AWS_RPC_SYSTEM_NAME, ORIGIN
+from tests.integrations.boto3.helpers import (
+ capture_spans_by_op,
+ require_botocore_model_fields,
+)
+from tests.integrations.boto3.helpers import (
+ client_factory as client_factory,
+)
+from tests.integrations.boto3.helpers import (
+ s3_client as s3_client,
)
-def test_basic(
- sentry_init,
- capture_items,
-):
- sentry_init(
- traces_sample_rate=1.0,
- integrations=[Boto3Integration()],
- # disabled because session.resource() or s3.Bucket() result in a subprocess span for a
- # shell that runs "uname -p 2> /dev/null" on Python 3.7 with boto3 version 1.12.49.
- default_integrations=False,
- )
-
- s3 = session.resource("s3")
- bucket = s3.Bucket("bucket")
- items = capture_items("span")
-
- with sentry_sdk.start_span(name="custom parent") as span, MockResponse(
- s3.meta.client, 200, {}, read_fixture("s3_list.xml")
- ):
- objects = [obj for obj in bucket.objects.all()]
- assert len(objects) == 2
- assert objects[0].key == "foo.txt"
- assert objects[1].key == "bar.txt"
- span.end()
-
- sentry_sdk.flush()
- spans = [item.payload for item in items]
- assert len(spans) == 2
- span = spans[0]
- assert span["attributes"]["sentry.op"] == "http.client"
- assert span["name"] == "S3.ListObjects"
-
-
-def test_streaming(sentry_init, capture_items):
- sentry_init(
- traces_sample_rate=1.0,
- integrations=[Boto3Integration()],
- )
-
- s3 = session.resource("s3")
- obj = s3.Bucket("bucket").Object("foo.pdf")
-
- items = capture_items("span")
-
- with sentry_sdk.start_span(name="custom parent") as span, MockResponse(
- s3.meta.client, 200, {}, b"hello"
- ):
- body = obj.get()["Body"]
- assert body.read(1) == b"h"
- assert body.read(2) == b"el"
- assert body.read(3) == b"lo"
- assert body.read(1) == b""
- span.end()
-
- sentry_sdk.flush()
- spans = [item.payload for item in items]
- assert len(spans) == 3
-
- stream_span, client_span, parent_span = spans
- assert stream_span["attributes"]["sentry.op"] == "http.client.stream"
- assert stream_span["name"] == "S3.GetObject"
- assert stream_span["parent_span_id"] == client_span["span_id"]
-
- assert client_span["attributes"]["sentry.op"] == "http.client"
- assert client_span["name"] == "S3.GetObject"
- assert client_span["parent_span_id"] == parent_span["span_id"]
-
- assert parent_span["name"] == "custom parent"
- assert parent_span["start_timestamp"] <= client_span["start_timestamp"]
- assert client_span["start_timestamp"] <= stream_span["start_timestamp"]
- assert stream_span["end_timestamp"] <= client_span["end_timestamp"]
-
- expected_attrs = {
- "http.request.method": "GET",
- "rpc.method": "GetObject",
- "rpc.service": "S3",
- "sentry.environment": "production",
- "sentry.op": "http.client",
- "sentry.origin": ORIGIN,
- "sentry.release": mock.ANY,
- "sentry.sdk.name": "sentry.python",
- "sentry.sdk.version": mock.ANY,
- "sentry.segment.id": mock.ANY,
- "sentry.segment.name": "custom parent",
- "server.address": mock.ANY,
- "thread.id": mock.ANY,
- "thread.name": mock.ANY,
- "url.full": "https://bucket.s3.amazonaws.com/foo.pdf",
+def _stubbed_span(client, capture_items, method, params, response):
+ with Stubber(client) as stubber:
+ stubber.add_response(method, response, expected_params=params)
+ spans = capture_spans_by_op(
+ lambda: getattr(client, method)(**params), capture_items
+ )
+ stubber.assert_no_pending_responses()
+ (span,) = spans[OP.HTTP_CLIENT]
+ operation = client.meta.method_to_api_mapping[method]
+ assert span["name"] == "S3.%s" % operation
+ assert span.get("end_timestamp") is not None
+ attributes = span.get("attributes", {})
+ assert attributes[SPANDATA.SENTRY_OP] == OP.HTTP_CLIENT
+ assert attributes[SPANDATA.SENTRY_ORIGIN] == ORIGIN
+ assert attributes[SPANDATA.SENTRY_KIND] == "client"
+ assert attributes[SPANDATA.CLOUD_PROVIDER] == "aws"
+ assert attributes[SPANDATA.RPC_SYSTEM_NAME] == AWS_RPC_SYSTEM_NAME
+ assert attributes[SPANDATA.RPC_SERVICE] == "S3"
+ assert attributes[SPANDATA.RPC_METHOD] == operation
+ assert attributes[SPANDATA.CLOUD_REGION] == "eu-north-1"
+ assert attributes[SPANDATA.SERVER_ADDRESS] == "s3.eu-north-1.amazonaws.com"
+ assert attributes[SPANDATA.SERVER_PORT] == 443
+ return span
+
+
+def test_request_attributes(s3_client, capture_items):
+ params = {
+ "Bucket": "bucket",
+ "Key": "file.txt",
+ "UploadId": "upload-id",
+ "PartNumber": 1,
+ "Body": b"private-content",
}
+ original = deepcopy(params)
+
+ span = _stubbed_span(s3_client, capture_items, "upload_part", params, {})
+
+ attributes = span.get("attributes", {})
+ assert attributes[SPANDATA.AWS_S3_BUCKET] == "bucket"
+ assert attributes[SPANDATA.AWS_S3_KEY] == "file.txt"
+ assert attributes[SPANDATA.AWS_S3_UPLOAD_ID] == "upload-id"
+ assert attributes[SPANDATA.AWS_S3_PART_NUMBER] == 1
+ assert SPANDATA.AWS_S3_COPY_SOURCE not in attributes
+ assert SPANDATA.AWS_S3_DELETE not in attributes
+ assert SPANDATA.FILE_SIZE not in attributes
+ assert SPANDATA.HTTP_BODY_SIZE not in attributes
+ assert params == original
+ assert "private-content" not in json.dumps(span)
+
+
+@pytest.mark.parametrize(
+ "method,copy_source,expected",
+ [
+ pytest.param(
+ "copy_object",
+ "source/path/file.txt",
+ "source/path/file.txt",
+ id="string",
+ ),
+ pytest.param(
+ "copy_object",
+ {"Bucket": "source", "Key": "path/file.txt"},
+ "source/path/file.txt",
+ id="dictionary",
+ ),
+ pytest.param(
+ "upload_part_copy",
+ {"Bucket": "source", "Key": "path/file.txt", "VersionId": "version-1"},
+ "source/path/file.txt?versionId=version-1",
+ id="versioned-dictionary",
+ ),
+ ],
+)
+def test_copy_source(s3_client, capture_items, method, copy_source, expected):
+ source = deepcopy(copy_source)
+ params = {"Bucket": "bucket", "Key": "file.txt", "CopySource": source}
+ expected_attributes = {
+ SPANDATA.AWS_S3_BUCKET: "bucket",
+ SPANDATA.AWS_S3_KEY: "file.txt",
+ SPANDATA.AWS_S3_COPY_SOURCE: expected,
+ }
+ if method == "upload_part_copy":
+ params.update(UploadId="upload-id", PartNumber=1)
+ expected_attributes.update(
+ {SPANDATA.AWS_S3_UPLOAD_ID: "upload-id", SPANDATA.AWS_S3_PART_NUMBER: 1}
+ )
- assert client_span["attributes"] == ApproxDict(expected_attrs)
-
-
-def test_streaming_close(sentry_init, capture_items):
- sentry_init(
- traces_sample_rate=1.0,
- integrations=[Boto3Integration()],
- )
-
- s3 = session.resource("s3")
- obj = s3.Bucket("bucket").Object("foo.pdf")
-
- items = capture_items("span")
-
- with sentry_sdk.start_span(name="custom parent") as span, MockResponse(
- s3.meta.client, 200, {}, b"hello"
- ):
- body = obj.get()["Body"]
- assert body.read(1) == b"h"
- body.close() # close partially-read stream
- span.end()
-
- sentry_sdk.flush()
- spans = [item.payload for item in items]
- assert len(spans) == 3
-
- stream_span, client_span, parent_span = spans
- assert stream_span["attributes"]["sentry.op"] == "http.client.stream"
- assert stream_span["name"] == "S3.GetObject"
- assert stream_span["parent_span_id"] == client_span["span_id"]
-
- assert client_span["attributes"]["sentry.op"] == "http.client"
- assert client_span["name"] == "S3.GetObject"
- assert client_span["parent_span_id"] == parent_span["span_id"]
-
- assert parent_span["name"] == "custom parent"
- assert parent_span["start_timestamp"] <= client_span["start_timestamp"]
- assert client_span["start_timestamp"] <= stream_span["start_timestamp"]
- assert stream_span["end_timestamp"] <= client_span["end_timestamp"]
-
-
-@pytest.mark.tests_internal_exceptions
-def test_omit_url_data_if_parsing_fails(sentry_init, capture_items):
- sentry_init(
- traces_sample_rate=1.0,
- integrations=[Boto3Integration()],
- )
-
- s3 = session.resource("s3")
- bucket = s3.Bucket("bucket")
-
- items = capture_items("span")
-
- with mock.patch(
- "sentry_sdk.integrations.boto3._instrumentation.parse_url",
- side_effect=ValueError,
- ):
- with sentry_sdk.start_span(name="custom parent") as span, MockResponse(
- s3.meta.client, 200, {}, read_fixture("s3_list.xml")
- ):
- objects = [obj for obj in bucket.objects.all()]
- assert len(objects) == 2
- assert objects[0].key == "foo.txt"
- assert objects[1].key == "bar.txt"
- span.end()
-
- sentry_sdk.flush()
- spans = [item.payload for item in items]
- assert spans[0]["attributes"] == ApproxDict(
- {
- "http.request.method": "GET",
- "rpc.method": "ListObjects",
- "rpc.service": "S3",
- "sentry.environment": "production",
- "sentry.op": "http.client",
- "sentry.origin": ORIGIN,
- "sentry.release": mock.ANY,
- "sentry.sdk.name": "sentry.python",
- "sentry.sdk.version": mock.ANY,
- "sentry.segment.id": mock.ANY,
- "sentry.segment.name": "custom parent",
- "server.address": mock.ANY,
- "thread.id": mock.ANY,
- "thread.name": mock.ANY,
- }
- )
-
- assert "url.full" not in spans[0]["attributes"]
- assert "url.fragment" not in spans[0]["attributes"]
- assert "url.query" not in spans[0]["attributes"]
-
-
-def test_span_origin(sentry_init, capture_items):
- sentry_init(
- traces_sample_rate=1.0,
- integrations=[Boto3Integration()],
- )
-
- s3 = session.resource("s3")
- bucket = s3.Bucket("bucket")
- items = capture_items("span")
-
- with sentry_sdk.start_span(name="custom parent"), MockResponse(
- s3.meta.client, 200, {}, read_fixture("s3_list.xml")
- ):
- _ = [obj for obj in bucket.objects.all()]
-
- sentry_sdk.flush()
-
- spans = [item.payload for item in items]
-
- assert spans[1]["attributes"]["sentry.origin"] == "manual"
- assert spans[0]["attributes"]["sentry.origin"] == ORIGIN
-
-
-def test_breadcrumb(sentry_init, capture_events):
- sentry_init(
- integrations=[Boto3Integration()],
- default_integrations=False,
- )
-
- s3 = session.resource("s3")
- bucket = s3.Bucket("bucket")
-
- events = capture_events()
-
- with sentry_sdk.start_span(name="custom parent"), MockResponse(
- s3.meta.client, 200, {}, read_fixture("s3_list.xml")
+ span = _stubbed_span(s3_client, capture_items, method, params, {})
+
+ attributes = span.get("attributes", {})
+ for key, value in expected_attributes.items():
+ assert attributes[key] == value
+ for key in (
+ SPANDATA.AWS_S3_UPLOAD_ID,
+ SPANDATA.AWS_S3_PART_NUMBER,
+ SPANDATA.AWS_S3_DELETE,
+ SPANDATA.FILE_SIZE,
+ SPANDATA.HTTP_BODY_SIZE,
):
- _ = [obj for obj in bucket.objects.all()]
-
- capture_message("Testing!")
-
- (event,) = events
- (crumb,) = event["breadcrumbs"]["values"]
- assert crumb["type"] == "http"
- assert crumb["category"] == "httplib"
-
- assert crumb["data"] == ApproxDict(
- {
- SPANDATA.URL_FULL: mock.ANY,
- SPANDATA.HTTP_REQUEST_METHOD: "GET",
- SPANDATA.URL_QUERY: mock.ANY,
- }
- )
- assert SPANDATA.URL_FRAGMENT not in crumb["data"]
-
-
-BUCKET_URL = "https://bucket.s3.amazonaws.com/"
-
-# ``expected_query`` of ``None`` means no URL data is recorded at all; ``""``
-# means the URL is recorded without a query string.
-# Structure of the parameters is "init_kwargs, expected_query"
-URL_QUERY_PARAMS = [
- pytest.param(
- {},
- "list-type=2&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=url",
- id="defaults",
- ),
- pytest.param(
- {
- "data_collection": {
- "url_query_params": {"mode": "denylist", "terms": ["prefix"]}
- }
- },
- "list-type=2&prefix=%5BFiltered%5D&continuation-token=%5BFiltered%5D&encoding-type=url",
- id="data_collection_denylist_custom_terms",
- ),
- pytest.param(
- {
- "data_collection": {
- "url_query_params": {"mode": "allowlist", "terms": ["prefix"]}
- }
- },
- "list-type=%5BFiltered%5D&prefix=foo&continuation-token=%5BFiltered%5D&encoding-type=%5BFiltered%5D",
- id="data_collection_allowlist",
- ),
- pytest.param(
- {
- "data_collection": {
- "url_query_params": {
- "mode": "allowlist",
- "terms": ["continuation-token"],
- }
- }
- },
- "list-type=%5BFiltered%5D&prefix=%5BFiltered%5D&continuation-token=%5BFiltered%5D&encoding-type=%5BFiltered%5D",
- id="data_collection_allowlist_sensitive_term",
- ),
- pytest.param(
- {"data_collection": {"url_query_params": {"mode": "off"}}},
- "",
- id="data_collection_off",
- ),
-]
-
-
-@pytest.mark.parametrize("init_kwargs, expected_query", URL_QUERY_PARAMS)
-def test_url_query_data_collection(
- sentry_init, capture_items, init_kwargs, expected_query
+ if key not in expected_attributes:
+ assert key not in attributes
+ assert params["CopySource"] is source
+ assert source == copy_source
+
+
+@pytest.mark.parametrize(
+ "delete, input_fields, expected_serialized_delete",
+ [
+ pytest.param(
+ {"Quiet": True, "Objects": [{"VersionId": "version-1", "Key": "file.txt"}]},
+ (),
+ '{"Objects":[{"Key":"file.txt","VersionId":"version-1"}],"Quiet":true}',
+ id="basic",
+ ),
+ pytest.param(
+ {
+ "Objects": [
+ {
+ "Key": "file.txt",
+ "VersionId": "version-1",
+ "ETag": "etag",
+ "LastModifiedTime": datetime(
+ 2026, 10, 8, 12, 34, 56, tzinfo=timezone.utc
+ ),
+ "Size": 123,
+ }
+ ]
+ },
+ ("Delete.Objects.LastModifiedTime",),
+ '{"Objects":[{"ETag":"etag","Key":"file.txt","LastModifiedTime":"2026-10-08T12:34:56+00:00","Size":123,"VersionId":"version-1"}]}',
+ id="last-modified-time",
+ ),
+ ],
+)
+def test_delete_serialization(
+ s3_client, capture_items, delete, input_fields, expected_serialized_delete
):
- sentry_init(
- traces_sample_rate=1.0,
- integrations=[Boto3Integration()],
- default_integrations=False,
- **init_kwargs,
+ require_botocore_model_fields(
+ s3_client,
+ "delete_objects",
+ input_fields=input_fields,
)
- client = session.client("s3")
-
- items = capture_items("span")
-
- with sentry_sdk.start_span(name="custom parent"), MockResponse(
- client, 200, {}, read_fixture("s3_list.xml")
- ):
- client.list_objects_v2(Bucket="bucket", Prefix="foo", ContinuationToken="abc")
-
- sentry_sdk.flush()
-
- (span,) = [
- item.payload
- for item in items
- if item.payload["attributes"].get("sentry.op") == "http.client"
- ]
-
- if expected_query is None:
- assert SPANDATA.URL_QUERY not in span["attributes"]
- assert SPANDATA.URL_FULL not in span["attributes"]
- elif expected_query == "":
- assert SPANDATA.URL_QUERY not in span["attributes"]
- assert span["attributes"][SPANDATA.URL_FULL] == BUCKET_URL
- else:
- assert span["attributes"][SPANDATA.URL_QUERY] == expected_query
- assert (
- span["attributes"][SPANDATA.URL_FULL] == BUCKET_URL + "?" + expected_query
- )
-
-
-@pytest.mark.parametrize("init_kwargs, expected_query", URL_QUERY_PARAMS)
-def test_url_query_data_collection_breadcrumb(
- sentry_init, capture_events, init_kwargs, expected_query
+ params = {"Bucket": "bucket", "Delete": deepcopy(delete)}
+ original = deepcopy(params)
+ caller_delete = params["Delete"]
+ caller_objects = caller_delete["Objects"]
+
+ span = _stubbed_span(s3_client, capture_items, "delete_objects", params, {})
+
+ attributes = span.get("attributes", {})
+ assert attributes[SPANDATA.AWS_S3_BUCKET] == "bucket"
+ assert attributes[SPANDATA.AWS_S3_DELETE] == expected_serialized_delete
+ assert SPANDATA.AWS_S3_KEY not in attributes
+ assert SPANDATA.AWS_S3_UPLOAD_ID not in attributes
+ assert SPANDATA.AWS_S3_COPY_SOURCE not in attributes
+ assert SPANDATA.AWS_S3_PART_NUMBER not in attributes
+ assert SPANDATA.FILE_SIZE not in attributes
+ assert SPANDATA.HTTP_BODY_SIZE not in attributes
+ assert params == original
+ assert params["Delete"] is caller_delete
+ assert caller_delete["Objects"] is caller_objects
+
+
+@pytest.mark.parametrize(
+ "method,extra_params,response,expected",
+ [
+ pytest.param(
+ "head_object",
+ {},
+ {"ContentLength": 1024},
+ {SPANDATA.FILE_SIZE: 1024},
+ id="head-whole-object",
+ ),
+ pytest.param(
+ "head_object",
+ {},
+ {"ContentLength": 0},
+ {SPANDATA.FILE_SIZE: 0},
+ id="head-empty-object",
+ ),
+ pytest.param(
+ "head_object",
+ {"Range": "bytes=0-3"},
+ {"ContentLength": 4},
+ {},
+ id="head-range",
+ ),
+ pytest.param(
+ "head_object", {"PartNumber": 1}, {"ContentLength": 4}, {}, id="head-part"
+ ),
+ pytest.param(
+ "get_object",
+ {"Range": "bytes=0-3"},
+ {"ContentLength": 4},
+ {SPANDATA.HTTP_BODY_SIZE: 4},
+ id="get-range",
+ ),
+ pytest.param(
+ "get_object_attributes",
+ {"ObjectAttributes": ["ObjectSize"]},
+ {"ObjectSize": 1024},
+ {SPANDATA.FILE_SIZE: 1024},
+ id="object-attributes-size",
+ ),
+ pytest.param(
+ "put_object",
+ {"Body": b"data", "WriteOffsetBytes": 1020},
+ {"Size": 1024},
+ {SPANDATA.FILE_SIZE: 1024},
+ id="put-append-size",
+ ),
+ pytest.param("put_object", {"Body": b"data"}, {}, {}, id="put-missing-size"),
+ pytest.param(
+ "complete_multipart_upload",
+ {"UploadId": "upload-id", "MpuObjectSize": 1024},
+ {},
+ {SPANDATA.FILE_SIZE: 1024, SPANDATA.AWS_S3_UPLOAD_ID: "upload-id"},
+ id="multipart-request-size",
+ ),
+ ],
+)
+def test_size_attributes(
+ s3_client, capture_items, method, extra_params, response, expected
):
- sentry_init(
- integrations=[Boto3Integration()],
- default_integrations=False,
- **init_kwargs,
+ require_botocore_model_fields(
+ s3_client,
+ method,
+ input_fields=tuple(
+ field
+ for field in ("MpuObjectSize", "WriteOffsetBytes")
+ if field in extra_params
+ ),
+ output_fields=tuple(response),
)
+ params = {"Bucket": "bucket", "Key": "file.txt", **deepcopy(extra_params)}
- client = session.client("s3")
+ span = _stubbed_span(s3_client, capture_items, method, params, response)
- events = capture_events()
-
- with sentry_sdk.start_span(name="custom parent"), MockResponse(
- client, 200, {}, read_fixture("s3_list.xml")
+ attributes = span.get("attributes", {})
+ expected_attributes = {
+ SPANDATA.AWS_S3_BUCKET: "bucket",
+ SPANDATA.AWS_S3_KEY: "file.txt",
+ **expected,
+ }
+ for key, value in expected_attributes.items():
+ assert attributes[key] == value
+ for key in (
+ SPANDATA.AWS_S3_UPLOAD_ID,
+ SPANDATA.AWS_S3_COPY_SOURCE,
+ SPANDATA.AWS_S3_PART_NUMBER,
+ SPANDATA.AWS_S3_DELETE,
+ SPANDATA.FILE_SIZE,
+ SPANDATA.HTTP_BODY_SIZE,
):
- client.list_objects_v2(Bucket="bucket", Prefix="foo", ContinuationToken="abc")
-
- capture_message("Testing!")
-
- (event,) = events
- (crumb,) = event["breadcrumbs"]["values"]
-
- if expected_query is None:
- assert SPANDATA.URL_QUERY not in crumb["data"]
- assert SPANDATA.URL_FULL not in crumb["data"]
- elif expected_query == "":
- assert SPANDATA.URL_QUERY not in crumb["data"]
- assert crumb["data"][SPANDATA.URL_FULL] == BUCKET_URL
- else:
- assert crumb["data"][SPANDATA.URL_QUERY] == expected_query
- assert crumb["data"][SPANDATA.URL_FULL] == BUCKET_URL + "?" + expected_query
+ if key not in expected_attributes:
+ assert key not in attributes