Update observability docs and task utilities

- Add Observability.md documentation
- Standardize task logging with correlation_id support
- Add log_sanitizer utility for PII masking
- Update Tasks.md tracking
- Update geo_cache tasks and other task modules with correlation_id

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>
This commit is contained in:
2026-05-03 11:52:09 +02:00
parent 7b93499551
commit 0133489920
17 changed files with 582 additions and 124 deletions

View File

@@ -7,10 +7,14 @@ database file and maintains query performance on geo lookups.
When a stale IP is encountered again after purge, it will be re-resolved from
the MaxMind database or ip-api.com (if configured), which is acceptable.
Correlation IDs are propagated through the task using :mod:`app.utils.correlation`
so that task logs can be correlated across runs.
"""
from __future__ import annotations
import uuid
from datetime import UTC, datetime, timedelta
from typing import TYPE_CHECKING
@@ -19,6 +23,7 @@ import structlog
from app.repositories import geo_cache_repo
from app.tasks.db import task_db
from app.tasks.timeout_utils import run_with_timeout
from app.utils.correlation import get_correlation_id, reset_correlation_id, set_correlation_id
from app.utils.runtime_state import get_effective_settings
if TYPE_CHECKING:
@@ -41,7 +46,10 @@ JOB_ID: str = "geo_cache_cleanup"
TASK_TIMEOUT_SECONDS: int = 60
async def _run_cleanup_with_resources(settings: Settings) -> None:
async def _run_cleanup_with_resources(
settings: Settings,
correlation_id: str | None = None,
) -> None:
"""Delete stale entries from the geo cache.
Calculates a cutoff timestamp (now - retention period) and removes all
@@ -49,7 +57,20 @@ async def _run_cleanup_with_resources(settings: Settings) -> None:
Args:
settings: The resolved application settings used for database access.
correlation_id: Optional correlation ID for log correlation.
"""
if correlation_id is None:
correlation_id = str(uuid.uuid4())
token = set_correlation_id(correlation_id)
try:
await _do_cleanup_with_settings(settings)
finally:
reset_correlation_id(token)
async def _do_cleanup_with_settings(settings: Settings) -> None:
"""Inner cleanup logic that runs with correlation context set."""
async def _do_cleanup() -> None:
cutoff_dt = datetime.now(UTC) - timedelta(days=GEO_CACHE_RETENTION_DAYS)
@@ -60,9 +81,19 @@ async def _run_cleanup_with_resources(settings: Settings) -> None:
await db.commit()
if deleted > 0:
log.info("geo_cache_cleanup_ran", deleted=deleted, retention_days=GEO_CACHE_RETENTION_DAYS)
log.info(
"geo_cache_cleanup_ran",
correlation_id=get_correlation_id(),
deleted=deleted,
retention_days=GEO_CACHE_RETENTION_DAYS,
)
else:
log.debug("geo_cache_cleanup_ran", deleted=deleted, retention_days=GEO_CACHE_RETENTION_DAYS)
log.debug(
"geo_cache_cleanup_ran",
correlation_id=get_correlation_id(),
deleted=deleted,
retention_days=GEO_CACHE_RETENTION_DAYS,
)
await run_with_timeout("geo_cache_cleanup", _do_cleanup(), TASK_TIMEOUT_SECONDS)