"""장애 알림 — 영구 저장 + 재시도 + 중복 억제.""" import os import re from datetime import timedelta from common.database.db_session_manager import DB_SESSION_MNG from common.enums import DBType from common.logger import LOG from common.utils.gtime import GTime from crud import alert_crud from crud.job_crud import compute_backoff from services import teams_webhook # 중복 억제 창(분). DEDUPE_WINDOW_MIN_ENV = "ALERT_DEDUPE_WINDOW_MIN" DEFAULT_DEDUPE_WINDOW_MIN = 60 # 재시도 상한. MAX_ATTEMPTS = 5 _DETAIL_MAX_LEN = 2000 # ── 비밀·개인정보 마스킹 ────────────────────────────────────────────────── _RE_QUERY_SECRET = re.compile( r"(?i)([?&](?:key|token|api[_-]?key|secret|access[_-]?token|auth)=)[^\s&]+" ) _RE_BEARER = re.compile(r"(?i)\bBearer\s+[A-Za-z0-9\-_.]{8,}") _RE_KV_SECRET = re.compile(r"(?i)\b(password|passwd|pwd|secret|api[_-]?key)\s*[:=]\s*\S+") _RE_EMAIL = re.compile(r"[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}") def _scrub(text: str) -> str: """저장 전에 반드시 한 번 거친다.""" if not text: return "" out = _RE_QUERY_SECRET.sub(r"\1***", text) out = _RE_BEARER.sub("Bearer ***", out) out = _RE_KV_SECRET.sub(lambda m: f"{m.group(1)}=***", out) out = _RE_EMAIL.sub(lambda m: m.group(0)[:2] + "***@***", out) return out[:_DETAIL_MAX_LEN] def _dedupe_window_min() -> int: try: return int(os.environ.get(DEDUPE_WINDOW_MIN_ENV) or DEFAULT_DEDUPE_WINDOW_MIN) except ValueError: return DEFAULT_DEDUPE_WINDOW_MIN async def send_alert(kind: str, title: str, detail: str = "", dedupe_key: str | None = None) -> None: """알림을 큐에 넣는다(즉시 보내지 않는다 — process_outbox 가 보낸다).""" try: async def _op(session): if dedupe_key: existing = await alert_crud.latest_unresolved(session, dedupe_key) if existing is not None: return # 이미 이 사유로 풀리지 않은 알림이 있다 — 또 만들지 않는다. await alert_crud.insert(session, { "kind": kind[:50], "dedupe_key": dedupe_key[:200] if dedupe_key else None, "title": title[:200], "detail": _scrub(detail), }) await DB_SESSION_MNG.execute_lambda_write(DBType.MAIN.value, _op) except Exception as ex: # noqa: BLE001 — 알림 적재 실패가 원래 하던 일(잡 처리)을 죽이면 안 된다 LOG.w(f"[alert] 적재 실패(무시하고 계속): {type(ex).__name__}: {ex}") async def resolve_alert(dedupe_key: str, title: str, detail: str = "") -> None: """이 dedupe_key 로 안 풀린 알림이 있으면 "복구됨" 을 한 번 알리고 풀린 것으로 남긴다.""" try: async def _op(session): existing = await alert_crud.latest_unresolved(session, dedupe_key) if existing is None: return await alert_crud.mark_resolved(session, existing.alert_id) await alert_crud.insert(session, { "kind": "recovery", "dedupe_key": None, # 복구 알림 자신은 dedupe 대상이 아니다 — 매번 보낸다. "title": title[:200], "detail": _scrub(detail), }) await DB_SESSION_MNG.execute_lambda_write(DBType.MAIN.value, _op) except Exception as ex: # noqa: BLE001 LOG.w(f"[alert] 복구 알림 적재 실패(무시하고 계속): {type(ex).__name__}: {ex}") async def process_outbox(limit: int = 20) -> dict: """PENDING 알림을 실제로 보낸다.""" sent = failed = 0 try: async def _load(session): return await alert_crud.due_pending(session, limit) due = await DB_SESSION_MNG.execute_lambda_write(DBType.MAIN.value, _load) except Exception as ex: # noqa: BLE001 LOG.w(f"[alert] outbox 조회 실패: {type(ex).__name__}: {ex}") return {"sent": 0, "failed": 0} for row in due: ok = await teams_webhook.send(row.title, row.detail or "") async def _update(session, row=row, ok=ok): if ok: await alert_crud.mark_sent(session, row.alert_id) else: attempts = row.attempts + 1 if attempts >= MAX_ATTEMPTS: await alert_crud.mark_exhausted(session, row.alert_id, attempts) else: next_at = GTime.UTC() + timedelta(seconds=compute_backoff(attempts)) await alert_crud.mark_retry(session, row.alert_id, attempts, next_at) try: await DB_SESSION_MNG.execute_lambda_write(DBType.MAIN.value, _update) except Exception as ex: # noqa: BLE001 LOG.w(f"[alert] outbox 갱신 실패 {row.alert_id}: {type(ex).__name__}: {ex}") continue if ok: sent += 1 else: failed += 1 if sent or failed: LOG.i(f"[alert] outbox 스윕 — 전송 {sent}건 · 재시도/소진 {failed}건") return {"sent": sent, "failed": failed}