From 2fd62e6e192cb33199e5a2db02586aa77e07ef6a Mon Sep 17 00:00:00 2001 From: Louis Parkin Date: Mon, 14 Sep 2026 11:27:55 +0200 Subject: [PATCH 1/2] Cache failed entity lookups for the duration of a check run Only successful lookups were cached, so every event referencing an unreachable entity cost a request. Over 97h of customer logs that was 12286 requests where 7211 would have done. The cache is reset each run, so an entity that becomes reachable is picked up on the next check. Persistent negative caching was declined for that reason and is not what this does. Removes the PROCESS_GROUP_INSTANCE skip this makes redundant. That special case also dropped events for entities that had not themselves failed, so one deleted entity blinded the run to the whole type. --- .../dynatrace_health/dynatrace_health.py | 34 +++++------- .../tests/test_dynatrace_health.py | 52 +++++++++++++++++++ 2 files changed, 64 insertions(+), 22 deletions(-) diff --git a/dynatrace_health/stackstate_checks/dynatrace_health/dynatrace_health.py b/dynatrace_health/stackstate_checks/dynatrace_health/dynatrace_health.py index b9d82c53..532fcce5 100644 --- a/dynatrace_health/stackstate_checks/dynatrace_health/dynatrace_health.py +++ b/dynatrace_health/stackstate_checks/dynatrace_health/dynatrace_health.py @@ -185,8 +185,6 @@ def _process_events(self, dynatrace_client, instance_info): # Dictionary to accumulate 404 errors per entity type entity_404_errors = {} - # Set to track entity types that have caused 404 errors in this run - problematic_entity_types = set() self.health.start_snapshot() for event in events: @@ -219,12 +217,6 @@ def _process_events(self, dynatrace_client, instance_info): continue entity_id = event.entityId.entityId.id or 'unknown' - # Skip PROCESS_GROUP_INSTANCE entities if they've caused 404 errors in this run - entity_type = self._extract_entity_type(entity_id) - if entity_type == 'PROCESS_GROUP_INSTANCE' and entity_type in problematic_entity_types: - self.log.debug(f"Skipping PROCESS_GROUP_INSTANCE entity {entity_id} due to previous 404 errors") - continue - try: entity_data = self._get_entity_definition( dynatrace_client, str(instance_info.url), entity_id @@ -240,8 +232,6 @@ def _process_events(self, dynatrace_client, instance_info): if entity_type not in entity_404_errors: entity_404_errors[entity_type] = 0 entity_404_errors[entity_type] += 1 - # Mark this entity type as problematic for this run - problematic_entity_types.add(entity_type) else: # Log non-404 errors as warnings self.log.info( @@ -265,14 +255,6 @@ def _process_events(self, dynatrace_client, instance_info): entity_id = event.entityId.entityId.id or 'unknown' - # Skip PROCESS_GROUP_INSTANCE entities if they've caused 404 errors in this run - entity_type = self._extract_entity_type(entity_id) - if entity_type == 'PROCESS_GROUP_INSTANCE' and entity_type in problematic_entity_types: - self.log.debug( - f"Skipping PROCESS_GROUP_INSTANCE entity {entity_id} for health state creation due to " - f"previous 404 errors") - continue - identifier = Identifiers.create_custom_identifier("dynatrace", entity_id) self.health.check_state( check_state_id=entity_id, @@ -296,8 +278,6 @@ def _process_events(self, dynatrace_client, instance_info): if entity_type not in entity_404_errors: entity_404_errors[entity_type] = 0 entity_404_errors[entity_type] += 1 - # Mark this entity type as problematic for this run - problematic_entity_types.add(entity_type) else: # Log non-404 errors as warnings self.log.warning( @@ -330,13 +310,23 @@ def _get_event_type_definition(self, dynatrace_client, base_url, event_type): def _get_entity_definition(self, dynatrace_client, base_url, entity_id): """ Return the entity definition from cache if present, otherwise fetch and cache it. + Failures are cached too, so several events referencing the same unreachable entity + cost one request. The cache is per run, so the next run retries and picks up an + entity that has since become reachable. """ if not hasattr(self, '_entity_cache'): self._entity_cache = {} if entity_id in self._entity_cache: - return self._entity_cache[entity_id] + cached = self._entity_cache[entity_id] + if isinstance(cached, Exception): + raise cached + return cached endpoint = f"{base_url}/api/v2/entities/{entity_id}" - data = dynatrace_client.get_dynatrace_json_response(endpoint, None) + try: + data = dynatrace_client.get_dynatrace_json_response(endpoint, None) + except Exception as e: + self._entity_cache[entity_id] = e + raise minimal = {"displayName": data.get("displayName")} self._entity_cache[entity_id] = minimal return minimal diff --git a/dynatrace_health/tests/test_dynatrace_health.py b/dynatrace_health/tests/test_dynatrace_health.py index 5965da61..ca41fc2b 100644 --- a/dynatrace_health/tests/test_dynatrace_health.py +++ b/dynatrace_health/tests/test_dynatrace_health.py @@ -246,6 +246,58 @@ def test_rejected_entity_does_not_suppress_remaining_events(dynatrace_check, tes assert telemetry._topology_events[0]['msg_title'] == "Custom Info on billing-worker" +@freeze_time('2025-07-22 08:26:24') +def test_failed_entity_lookup_is_cached_for_the_run(dynatrace_check, test_instance, requests_mock, aggregator): + """ + Several events referencing the same unreachable entity must cost one request. + """ + os.environ["JWT_AUTH"] = "false" + entity_id = 'PROCESS_GROUP_INSTANCE-ABCDEF0123456789' + event_response = { + "totalCount": 3, + "pageSize": 3, + "events": [_pgi_info_event('e-%d' % n, entity_id, 'checkout-worker') for n in (1, 2, 3)], + } + set_http_responses(requests_mock, custom_info_event=read_file('event_type_custom_info.json', 'samples')) + requests_mock.get("{}/api/v2/entities/{}".format(test_instance['url'], entity_id), + text='{"detail": "denied by policy"}', status_code=403) + _mock_events_endpoint(requests_mock, test_instance, event_response) + + dynatrace_check.run() + + entity_calls = [r for r in requests_mock.request_history if entity_id.lower() in r.url.lower()] + assert len(entity_calls) == 1 + + +@freeze_time('2025-07-22 08:26:24') +def test_missing_entity_does_not_suppress_type(dynatrace_check, test_instance, requests_mock, aggregator, telemetry): + """ + A deleted entity must not blind the run to other entities of the same type. + """ + os.environ["JWT_AUTH"] = "false" + missing_id = 'PROCESS_GROUP_INSTANCE-AAAAAAAAAAAAAAAA' + present_id = 'PROCESS_GROUP_INSTANCE-BBBBBBBBBBBBBBBB' + event_response = { + "totalCount": 2, + "pageSize": 2, + "events": [ + _pgi_info_event('missing-1', missing_id, 'gone-worker'), + _pgi_info_event('present-1', present_id, 'billing-worker'), + ], + } + set_http_responses(requests_mock, custom_info_event=read_file('event_type_custom_info.json', 'samples')) + requests_mock.get("{}/api/v2/entities/{}".format(test_instance['url'], missing_id), + text='{"error": {"message": "Entity not found"}}', status_code=404) + requests_mock.get("{}/api/v2/entities/{}".format(test_instance['url'], present_id), + text=json.dumps({"displayName": "billing-worker"})) + _mock_events_endpoint(requests_mock, test_instance, event_response) + + dynatrace_check.run() + + assert len(telemetry._topology_events) == 1 + assert telemetry._topology_events[0]['msg_title'] == "Custom Info on billing-worker" + + @freeze_time('2025-07-22 08:26:24') def test_marked_for_termination_event(dynatrace_check, test_instance, requests_mock, health, aggregator, telemetry): event_type = "MARKED_FOR_TERMINATION" From 66c1aa51e9e3fc97b820c211bb4f1ef873892fc0 Mon Sep 17 00:00:00 2001 From: Louis Parkin Date: Mon, 14 Sep 2026 14:38:27 +0200 Subject: [PATCH 2/2] Bound cached Dynatrace failure tracebacks Clear the previous traceback before raising a cached lookup failure so repeated events do not retain a growing chain of frames. Extend the request-deduplication regression test to check that only one lookup frame remains. --- .../stackstate_checks/dynatrace_health/dynatrace_health.py | 2 +- dynatrace_health/tests/test_dynatrace_health.py | 3 +++ 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/dynatrace_health/stackstate_checks/dynatrace_health/dynatrace_health.py b/dynatrace_health/stackstate_checks/dynatrace_health/dynatrace_health.py index 532fcce5..4dc9839d 100644 --- a/dynatrace_health/stackstate_checks/dynatrace_health/dynatrace_health.py +++ b/dynatrace_health/stackstate_checks/dynatrace_health/dynatrace_health.py @@ -319,7 +319,7 @@ def _get_entity_definition(self, dynatrace_client, base_url, entity_id): if entity_id in self._entity_cache: cached = self._entity_cache[entity_id] if isinstance(cached, Exception): - raise cached + raise cached.with_traceback(None) return cached endpoint = f"{base_url}/api/v2/entities/{entity_id}" try: diff --git a/dynatrace_health/tests/test_dynatrace_health.py b/dynatrace_health/tests/test_dynatrace_health.py index ca41fc2b..bcd83b43 100644 --- a/dynatrace_health/tests/test_dynatrace_health.py +++ b/dynatrace_health/tests/test_dynatrace_health.py @@ -6,6 +6,7 @@ import json import os import re +import traceback import requests from freezegun import freeze_time @@ -267,6 +268,8 @@ def test_failed_entity_lookup_is_cached_for_the_run(dynatrace_check, test_instan entity_calls = [r for r in requests_mock.request_history if entity_id.lower() in r.url.lower()] assert len(entity_calls) == 1 + frames = traceback.extract_tb(dynatrace_check._entity_cache[entity_id].__traceback__) + assert sum(frame.name == '_get_entity_definition' for frame in frames) == 1 @freeze_time('2025-07-22 08:26:24')