Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
171 changes: 171 additions & 0 deletions src/sentry/monitors/logic/incident_occurrence.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,171 @@
from __future__ import annotations

import logging
import uuid
from collections import Counter
from collections.abc import Mapping, Sequence
from datetime import datetime, timezone
from typing import TYPE_CHECKING

from django.utils.text import get_text_list
from django.utils.translation import gettext_lazy as _

from sentry.issues.grouptype import MonitorIncidentType
from sentry.monitors.models import (
CheckInStatus,
MonitorCheckIn,
MonitorEnvironment,
MonitorIncident,
)
from sentry.monitors.types import SimpleCheckIn

if TYPE_CHECKING:
from django.utils.functional import _StrPromise

logger = logging.getLogger(__name__)


def create_incident_occurrence(
failed_checkins: Sequence[SimpleCheckIn],
failed_checkin: MonitorCheckIn,
incident: MonitorIncident,
received: datetime | None,
) -> None:
from sentry.issues.issue_occurrence import IssueEvidence, IssueOccurrence
from sentry.issues.producer import PayloadType, produce_occurrence_to_kafka

monitor_env = failed_checkin.monitor_environment

if monitor_env is None:
return

current_timestamp = datetime.now(timezone.utc)

# Get last successful check-in to show in evidence display
last_successful_checkin_timestamp = "Never"
last_successful_checkin = monitor_env.get_last_successful_checkin()
if last_successful_checkin:
last_successful_checkin_timestamp = last_successful_checkin.date_added.isoformat()

occurrence = IssueOccurrence(
id=uuid.uuid4().hex,
resource_id=None,
project_id=monitor_env.monitor.project_id,
event_id=uuid.uuid4().hex,
fingerprint=[incident.grouphash],
type=MonitorIncidentType,
issue_title=f"Monitor failure: {monitor_env.monitor.name}",
subtitle="Your monitor has reached its failure threshold.",
evidence_display=[
IssueEvidence(
name="Failure reason",
value=str(get_failure_reason(failed_checkins)),
important=True,
),
IssueEvidence(
name="Environment",
value=monitor_env.get_environment().name,
important=False,
),
IssueEvidence(
name="Last successful check-in",
value=last_successful_checkin_timestamp,
important=False,
),
],
evidence_data={},
culprit="",
detection_time=current_timestamp,
level="error",
assignee=monitor_env.monitor.owner_actor,
)

if failed_checkin.trace_id:
trace_id = failed_checkin.trace_id.hex
else:
trace_id = None

event_data = {
"contexts": {"monitor": get_monitor_environment_context(monitor_env)},
"environment": monitor_env.get_environment().name,
"event_id": occurrence.event_id,
"fingerprint": [incident.grouphash],
"platform": "other",
"project_id": monitor_env.monitor.project_id,
# We set this to the time that the checkin that triggered the occurrence was written to relay if available
"received": (received if received else current_timestamp).isoformat(),
"sdk": None,
"tags": {
"monitor.id": str(monitor_env.monitor.guid),
"monitor.slug": str(monitor_env.monitor.slug),
"monitor.incident": str(incident.id),
},
"timestamp": current_timestamp.isoformat(),
}

if trace_id:
event_data["contexts"]["trace"] = {"trace_id": trace_id, "span_id": None}

produce_occurrence_to_kafka(
payload_type=PayloadType.OCCURRENCE,
occurrence=occurrence,
event_data=event_data,
)


HUMAN_FAILURE_STATUS_MAP: Mapping[int, _StrPromise] = {
CheckInStatus.ERROR: _("error"),
CheckInStatus.MISSED: _("missed"),
CheckInStatus.TIMEOUT: _("timeout"),
}

# Exists due to the vowel differences (A vs An) in the statuses
SINGULAR_HUMAN_FAILURE_MAP: Mapping[int, _StrPromise] = {
CheckInStatus.ERROR: _("An error check-in was detected"),
CheckInStatus.MISSED: _("A missed check-in was detected"),
CheckInStatus.TIMEOUT: _("A timeout check-in was detected"),
}


def get_failure_reason(failed_checkins: Sequence[SimpleCheckIn]):
"""
Builds a humam readible string from a list of failed check-ins.

"3 missed check-ins detected"
"2 missed check-ins, 1 timeout check-in and 1 error check-in were detected"
"A failed check-in was detected"
"""

status_counts = Counter(
checkin["status"]
for checkin in failed_checkins
if checkin["status"] in HUMAN_FAILURE_STATUS_MAP.keys()
)

if sum(status_counts.values()) == 1:
return SINGULAR_HUMAN_FAILURE_MAP[list(status_counts.keys())[0]]

human_status = get_text_list(
[
"%(count)d %(status)s" % {"count": count, "status": HUMAN_FAILURE_STATUS_MAP[status]}
for status, count in status_counts.items()
],
last_word=_("and"),
)

return _("%(problem_checkins)s check-ins detected") % {"problem_checkins": human_status}


def get_monitor_environment_context(monitor_environment: MonitorEnvironment):
config = monitor_environment.monitor.config.copy()
if "schedule_type" in config:
config["schedule_type"] = monitor_environment.monitor.get_schedule_type_display()

return {
"id": str(monitor_environment.monitor.guid),
"slug": str(monitor_environment.monitor.slug),
"name": monitor_environment.monitor.name,
"config": monitor_environment.monitor.config,
"status": monitor_environment.get_status_display(),
"type": monitor_environment.monitor.get_type_display(),
}
Comment on lines +159 to +171

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Potential issue | 🟠 Major | ⚡ Quick win

🧩 Analysis chain

🏁 Script executed:

#!/bin/bash
git log -p -S 'get_monitor_environment_context' -- src/sentry/monitors/logic/mark_failed.py | grep -nA20 'def get_monitor_environment_context' | head -60

Repository: test-code-tool/sentry__sentry__coderabbit__PR80528__20260407

Length of output: 86


🏁 Script executed:

#!/bin/bash
set -euo pipefail

# 1) Show the full current implementation of get_monitor_environment_context
rg -n "def get_monitor_environment_context" -S .
FILE="$(rg -n "def get_monitor_environment_context" -S . | head -n1 | cut -d: -f1)"
echo "== file: $FILE =="
sed -n '1,260p' "$FILE" | nl -ba | sed -n '120,220p'

# 2) Find where schedule_type is stored/used
echo "== schedule_type references (limited) =="
rg -n "schedule_type" src/sentry/monitors -S | head -n 80

# 3) Inspect get_schedule_type_display implementation
echo "== get_schedule_type_display implementation =="
rg -n "def get_schedule_type_display" -S src/sentry | head -n 50
# print surrounding code for the first match (if any)
DISP_FILE="$(rg -n "def get_schedule_type_display" -S src/sentry | head -n1 | cut -d: -f1)"
if [ -n "${DISP_FILE:-}" ]; then
  sed -n '1,260p' "$DISP_FILE" | nl -ba | sed -n '1,260p' | head -n 220
fi

# 4) Inspect MonitorEnvironment / monitor.config types if available
echo "== monitor_environment.monitor.config usage =="
rg -n "monitor_environment\.monitor\.config" -S src/sentry | head -n 50

# 5) History: try locating the function in git history anywhere (not just mark_failed.py)
echo "== git history for get_monitor_environment_context across repo =="
git log -p -S 'get_monitor_environment_context' -- src/sentry/monitors/logic 2>/dev/null | head -n 200

Repository: test-code-tool/sentry__sentry__coderabbit__PR80528__20260407

Length of output: 361


🏁 Script executed:

#!/bin/bash
set -euo pipefail

FILE="$(rg -n "def get_monitor_environment_context" -S src/sentry | head -n1 | cut -d: -f1)"
echo "== get_monitor_environment_context file: $FILE =="
sed -n '130,220p' "$FILE" | cat -n

echo "== schedule_type references (limited) =="
rg -n "schedule_type" src/sentry/monitors -S | head -n 80

echo "== get_schedule_type_display implementation =="
DISP_FILE="$(rg -n "def get_schedule_type_display" -S src/sentry | head -n1 | cut -d: -f1 || true)"
if [ -n "${DISP_FILE:-}" ]; then
  echo "== file: $DISP_FILE =="
  sed -n '1,260p' "$DISP_FILE" | cat -n | head -n 220
else
  echo "No def get_schedule_type_display found under src/sentry"
fi

echo "== monitor_environment.monitor.config usage (first 50 matches) =="
rg -n "monitor_environment\.monitor\.config" -S src/sentry | head -n 50 || true

echo "== git history for get_monitor_environment_context (any path) =="
git log -p -S 'get_monitor_environment_context' -- src/sentry/monitors/logic 2>/dev/null | head -n 200 || true

Repository: test-code-tool/sentry__sentry__coderabbit__PR80528__20260407

Length of output: 14210


🏁 Script executed:

#!/bin/bash
set -euo pipefail

# 1) Inspect monitors/serializers.py around where schedule_type display substitution happens
SER_FILE="src/sentry/monitors/serializers.py"
if [ -f "$SER_FILE" ]; then
  sed -n '200,280p' "$SER_FILE" | cat -n
fi

# 2) Find usages of get_monitor_environment_context
rg -n "get_monitor_environment_context\(" -S src/sentry | head -n 50

# 3) If there is an event schema/consumer expecting config.schedule_type as string, find it
rg -n "\"config\".*schedule_type|schedule_type.*config" -S src/sentry/monitors src/sentry/sentry* 2>/dev/null | head -n 80 || true

# 4) Search for Kafka event payload builder/contract area around incident occurrence
rg -n "incident_occurrence|get_monitor_environment_context|IssueOccurrence|event_data|kafka" -S src/sentry/monitors/logic/incident_occurrence.py | cat -n

# 5) Check ruff warning RUF015 definition (best-effort): find ruff config mentioning RUF015
if [ -f ".ruff.toml" ]; then
  rg -n "RUF015" .ruff.toml || true
fi
if [ -f "pyproject.toml" ]; then
  rg -n "RUF015" pyproject.toml || true
fi

Repository: test-code-tool/sentry__sentry__coderabbit__PR80528__20260407

Length of output: 4731


🏁 Script executed:

#!/bin/bash
set -euo pipefail

# 1) Locate mark_failed.py (if present) and any other similar history targets
echo "== mark_failed.py files =="
git ls-files | rg -n "mark_failed\.py" || true

# 2) Show current incident_occurrence.py around create_incident_occurrence and context usage
sed -n '1,220p' src/sentry/monitors/logic/incident_occurrence.py | cat -n

# 3) Find any code that reads event contexts["monitor"] / "contexts"."monitor"
echo "== reads of contexts monitor in codebase (best-effort) =="
rg -n "\"contexts\"\\s*\\[\"monitor\"\\]|contexts\"\\]\\[\"monitor\"\\]|contexts\\]\\[\\\"monitor\\\"\\]|contexts\\.monitor" -S src/sentry | head -n 80 || true

echo "== reads of get_monitor_environment_context result shape (contexts monitor) =="
rg -n "get_status_display\\(\\)|get_type_display\\(\\)|schedule_type\"\\s*:\\s*|schedule_type" -S src/sentry/sentry* src/sentry | head -n 80 || true

# 4) History search: look for schedule_type substitution inside get_monitor_environment_context in git history
echo "== git history for incident_occurrence context function (best-effort, limited) =="
git log -p -S 'def get_monitor_environment_context' -- src/sentry/monitors/logic/incident_occurrence.py 2>/dev/null | head -n 200 || true

git log -p -S '"config": monitor_environment.monitor.config' -- src/sentry/monitors/logic/incident_occurrence.py 2>/dev/null | head -n 200 || true
git log -p -S '"schedule_type" = monitor_environment.monitor.get_schedule_type_display' -- src/sentry/monitors/logic/incident_occurrence.py 2>/dev/null | head -n 200 || true

Repository: test-code-tool/sentry__sentry__coderabbit__PR80528__20260407

Length of output: 13484


🏁 Script executed:

#!/bin/bash
set -euo pipefail

# 1) Inspect mark_failed.py implementation (likely previous/parallel code path)
MARK_FILE="src/sentry/monitors/logic/mark_failed.py"
if [ -f "$MARK_FILE" ]; then
  echo "== mark_failed.py get_monitor_environment_context =="
  sed -n '1,260p' "$MARK_FILE" | cat -n | rg -n "def get_monitor_environment_context|get_monitor_environment_context" || true
  # Show neighborhood around the function if present
  FN_LINE="$(rg -n "def get_monitor_environment_context" "$MARK_FILE" | head -n1 | cut -d: -f1 || true)"
  if [ -n "${FN_LINE:-}" ]; then
    START=$((FN_LINE-30))
    END=$((FN_LINE+60))
    sed -n "${START},${END}p" "$MARK_FILE" | cat -n
  fi

  echo "== mark_failed.py schedule_type display usage =="
  rg -n "schedule_type.*get_schedule_type_display|get_schedule_type_display\(" "$MARK_FILE" || true
fi

# 2) Find consumers that read contexts.monitor.config or specifically schedule_type from the event contexts
echo "== Reads of contexts.monitor.config (best-effort) =="
rg -n "contexts\.(monitor\.config|monitor\['config'\]|monitor\]\['config'\]|monitor\"\s*:\s*\{[^}]*config|contexts.*monitor.*config)" -S src/sentry tests | head -n 80 || true

echo "== Reads of schedule_type from monitor context (best-effort) =="
rg -n "contexts.*monitor.*schedule_type|schedule_type.*contexts.*monitor" -S src/sentry tests | head -n 80 || true

# 3) git history search for the specific return config key/value
echo "== git history: config returned from get_monitor_environment_context =="
git log -p -S '"config": monitor_environment.monitor.config' -- src/sentry/monitors/logic/incident_occurrence.py 2>/dev/null | head -n 200 || true
git log -p -S '"config": config' -- src/sentry/monitors/logic/incident_occurrence.py 2>/dev/null | head -n 200 || true
git log -p -S 'get_schedule_type_display' -- src/sentry/monitors/logic/incident_occurrence.py 2>/dev/null | head -n 200 || true

Repository: test-code-tool/sentry__sentry__coderabbit__PR80528__20260407

Length of output: 395


🏁 Script executed:

#!/bin/bash
set -euo pipefail

echo "== Search for incident/monitor failure event payload usage of schedule_type in contexts =="
rg -n "contexts.*monitor.*schedule_type|schedule_type.*contexts.*monitor|monitor.*config.*schedule_type|config.*schedule_type" -S src/sentry tests | head -n 120

echo "== Search for get_monitor_environment_context in older/other files =="
rg -n "get_monitor_environment_context" -S src/sentry tests

echo "== Inspect analytics monitor_mark_failed.py for monitor context shape =="
FILE="src/sentry/analytics/events/monitor_mark_failed.py"
if [ -f "$FILE" ]; then
  rg -n "contexts|monitor" "$FILE" | head -n 120
  sed -n '1,260p' "$FILE" | cat -n
fi

echo "== Inspect monitors/logic/mark_failed.py for any monitor context building =="
FILE="src/sentry/monitors/logic/mark_failed.py"
if [ -f "$FILE" ]; then
  rg -n "contexts|monitor.*context|get_monitor_environment_context" "$FILE" | head -n 120
  sed -n '1,340p' "$FILE" | cat -n
fi

Repository: test-code-tool/sentry__sentry__coderabbit__PR80528__20260407

Length of output: 6195


🏁 Script executed:

#!/bin/bash
set -euo pipefail

echo "== Search for literal 'contexts': {'monitor' usages =="
rg -n "\"contexts\"\\s*:\\s*\\{\\s*\"monitor\"|\"contexts\"\\s*:\\s*{\\s*\"monitor\"" -S src tests | head -n 80 || true

echo "== Search for contexts.monitor.config reads =="
rg -n "contexts\\.monitor\\.config|contexts\\]\\[\"monitor\"\\].*config|contexts\\]\\['monitor'\\].*config|contexts\\.monitor.*config" -S src tests | head -n 120 || true

echo "== Search for schedule_type inside contexts.monitor.config reads =="
rg -n "monitor.*config.*schedule_type|contexts\\.monitor\\.config.*schedule_type|contexts\\]\\[\"monitor\"\\].*schedule_type|contexts\\]\\['monitor'\\].*schedule_type" -S src tests | head -n 120 || true

echo "== Search for consumers of IssueOccurrence event contexts =="
rg -n "IssueOccurrence|produce_occurrence_to_kafka|event_data\\[\"contexts\"\\]" -S src | head -n 80

echo "== Search for 'monitor' context key usage in issue/occurrence handling =="
rg -n "\"monitor\"\\s*:\\s*get_monitor_environment_context|contexts\"\\]\\[\"monitor\"\\]|contexts\\\"\\]\\[\\\"monitor\\\"\\]" -S src | head -n 80 || true

Repository: test-code-tool/sentry__sentry__coderabbit__PR80528__20260407

Length of output: 9627


Fix dead code in get_monitor_environment_context: return the modified config

The function copies monitor_environment.monitor.config into config and replaces schedule_type with monitor_environment.monitor.get_schedule_type_display(), but the returned context still uses monitor_environment.monitor.config, so the human-readable substitution never reaches the emitted contexts.monitor.config.

🐛 Proposed fix
     return {
         "id": str(monitor_environment.monitor.guid),
         "slug": str(monitor_environment.monitor.slug),
         "name": monitor_environment.monitor.name,
-        "config": monitor_environment.monitor.config,
+        "config": config,
         "status": monitor_environment.get_status_display(),
         "type": monitor_environment.monitor.get_type_display(),
     }
📝 Committable suggestion

‼️ IMPORTANT
Carefully review the code before committing. Ensure that it accurately replaces the highlighted code, contains no missing lines, and has no issues with indentation. Thoroughly test & benchmark the code to ensure it meets the requirements.

Suggested change
def get_monitor_environment_context(monitor_environment: MonitorEnvironment):
config = monitor_environment.monitor.config.copy()
if "schedule_type" in config:
config["schedule_type"] = monitor_environment.monitor.get_schedule_type_display()
return {
"id": str(monitor_environment.monitor.guid),
"slug": str(monitor_environment.monitor.slug),
"name": monitor_environment.monitor.name,
"config": monitor_environment.monitor.config,
"status": monitor_environment.get_status_display(),
"type": monitor_environment.monitor.get_type_display(),
}
def get_monitor_environment_context(monitor_environment: MonitorEnvironment):
config = monitor_environment.monitor.config.copy()
if "schedule_type" in config:
config["schedule_type"] = monitor_environment.monitor.get_schedule_type_display()
return {
"id": str(monitor_environment.monitor.guid),
"slug": str(monitor_environment.monitor.slug),
"name": monitor_environment.monitor.name,
"config": config,
"status": monitor_environment.get_status_display(),
"type": monitor_environment.monitor.get_type_display(),
}
🤖 Prompt for AI Agents
Verify each finding against current code. Fix only still-valid issues, skip the
rest with a brief reason, keep changes minimal, and validate.

In `@src/sentry/monitors/logic/incident_occurrence.py` around lines 159 - 171,
get_monitor_environment_context copies monitor_environment.monitor.config into a
local variable config and mutates it, but then returns
monitor_environment.monitor.config instead of the modified config; update the
return value in get_monitor_environment_context so the "config" field uses the
local config variable (i.e., "config": config) so the human-readable
schedule_type substitution
(monitor_environment.monitor.get_schedule_type_display()) is propagated to the
emitted context.

104 changes: 104 additions & 0 deletions src/sentry/monitors/logic/incidents.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,104 @@
from __future__ import annotations

import logging
from datetime import datetime
from typing import cast

from sentry.monitors.logic.incident_occurrence import create_incident_occurrence
from sentry.monitors.models import CheckInStatus, MonitorCheckIn, MonitorIncident, MonitorStatus
from sentry.monitors.types import SimpleCheckIn

logger = logging.getLogger(__name__)


def try_incident_threshold(
failed_checkin: MonitorCheckIn,
failure_issue_threshold: int,
received: datetime | None,
) -> bool:
from sentry.signals import monitor_environment_failed

monitor_env = failed_checkin.monitor_environment

if monitor_env is None:
return False

# check to see if we need to update the status
if monitor_env.status in [MonitorStatus.OK, MonitorStatus.ACTIVE]:
if failure_issue_threshold == 1:
previous_checkins: list[SimpleCheckIn] = [
{
"id": failed_checkin.id,
"date_added": failed_checkin.date_added,
"status": failed_checkin.status,
}
]
else:
previous_checkins = cast(
list[SimpleCheckIn],
# Using .values for performance reasons
MonitorCheckIn.objects.filter(
monitor_environment=monitor_env, date_added__lte=failed_checkin.date_added
)
.order_by("-date_added")
.values("id", "date_added", "status"),
)

# reverse the list after slicing in order to start with oldest check-in
previous_checkins = list(reversed(previous_checkins[:failure_issue_threshold]))

# If we have any successful check-ins within the threshold of
# commits we have NOT reached an incident state
if any([checkin["status"] == CheckInStatus.OK for checkin in previous_checkins]):
return False

# change monitor status + update fingerprint timestamp
monitor_env.status = MonitorStatus.ERROR
monitor_env.save(update_fields=("status",))

starting_checkin = previous_checkins[0]

incident: MonitorIncident | None
incident, _ = MonitorIncident.objects.get_or_create(
monitor_environment=monitor_env,
resolving_checkin=None,
defaults={
"monitor": monitor_env.monitor,
"starting_checkin_id": starting_checkin["id"],
"starting_timestamp": starting_checkin["date_added"],
},
)

elif monitor_env.status == MonitorStatus.ERROR:
# if monitor environment has a failed status, use the failed
# check-in and send occurrence
previous_checkins = [
{
"id": failed_checkin.id,
"date_added": failed_checkin.date_added,
"status": failed_checkin.status,
}
]

# get the active incident from the monitor environment
incident = monitor_env.active_incident
else:
# don't send occurrence for other statuses
return False

# Only create an occurrence if:
# - We have an active incident and fingerprint
# - The monitor and env are not muted
if not monitor_env.monitor.is_muted and not monitor_env.is_muted and incident:
checkins = MonitorCheckIn.objects.filter(id__in=[c["id"] for c in previous_checkins])
for checkin in checkins:
create_incident_occurrence(
previous_checkins,
checkin,
incident,
received=received,
)

monitor_environment_failed.send(monitor_environment=monitor_env, sender=type(monitor_env))

return True
Loading