feat(executors): add postgres retention executor
Deletes rows past a retention window from a table on the shared Postgres server. Written for sysmon's check_history, which grows with every monitoring check and had no retention at all despite the docs promising a 30-day rolling window. Connects with the Scheduler's own credentials and overrides only the database name, so no second set of secrets enters the stack. The target database grants scheduler_user just SELECT and DELETE on the table, so a bug here can drop old rows but cannot corrupt or forge history. Table and column names cannot be bound as query parameters, so both are validated against a strict identifier pattern before interpolation, and a retention window below 1 day is refused rather than silently emptying the table. Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,109 @@
|
||||
"""
|
||||
Postgres Retention Executor
|
||||
|
||||
Deletes rows older than a retention window from a table on the shared Postgres
|
||||
server. Written for sysmon's `check_history`, which grows with every monitoring
|
||||
check and had no retention at all, but the executor is table-agnostic.
|
||||
|
||||
Connects with the Scheduler's own Postgres credentials and only overrides the
|
||||
database name. That keeps a second set of credentials out of the stack; the
|
||||
target database grants `scheduler_user` exactly SELECT and DELETE on the table,
|
||||
so a bug here can drop old rows but cannot corrupt or forge history.
|
||||
|
||||
Config schema:
|
||||
{
|
||||
"database": "sysmon", # defaults to the Scheduler's own database
|
||||
"table": "check_history", # required
|
||||
"timestamp_column": "ts", # required
|
||||
"retention_days": 30, # required, must be >= 1
|
||||
"dry_run": false # count what would go, delete nothing
|
||||
}
|
||||
|
||||
Table and column names cannot be passed as query parameters, so both are
|
||||
validated against a strict identifier pattern before being interpolated.
|
||||
|
||||
Autovacuum reclaims the space afterwards; this deliberately does not VACUUM,
|
||||
which would need table ownership the Scheduler intentionally does not have.
|
||||
"""
|
||||
import asyncio
|
||||
import logging
|
||||
import re
|
||||
from typing import Any, Dict
|
||||
|
||||
import psycopg2
|
||||
|
||||
from src.config import Settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Deliberately strict: unquoted lowercase identifiers only. Anything needing
|
||||
# quoting is out of scope and would be a hole in the interpolation below.
|
||||
IDENTIFIER_RE = re.compile(r"^[a-z_][a-z0-9_]*$")
|
||||
|
||||
MAX_RETENTION_DAYS = 3650
|
||||
|
||||
|
||||
def _validate_identifier(value: str, label: str) -> str:
|
||||
if not isinstance(value, str) or not IDENTIFIER_RE.match(value):
|
||||
raise ValueError(
|
||||
f"invalid {label}: {value!r} (expected an unquoted lowercase identifier)"
|
||||
)
|
||||
return value
|
||||
|
||||
|
||||
def _prune(config: Dict[str, Any], settings: Settings) -> str:
|
||||
table = _validate_identifier(config.get("table", ""), "table")
|
||||
column = _validate_identifier(config.get("timestamp_column", ""), "timestamp_column")
|
||||
database = config.get("database") or settings.postgres_db
|
||||
_validate_identifier(database, "database")
|
||||
|
||||
retention_days = config.get("retention_days")
|
||||
if not isinstance(retention_days, int) or isinstance(retention_days, bool):
|
||||
raise ValueError(f"retention_days must be an integer, got {retention_days!r}")
|
||||
# A zero or negative window would delete everything, including the row the
|
||||
# check just wrote. Refuse rather than quietly wipe the table.
|
||||
if retention_days < 1 or retention_days > MAX_RETENTION_DAYS:
|
||||
raise ValueError(
|
||||
f"retention_days must be between 1 and {MAX_RETENTION_DAYS}, got {retention_days}"
|
||||
)
|
||||
|
||||
dry_run = bool(config.get("dry_run", False))
|
||||
cutoff_sql = f"{column} < now() - make_interval(days => %s)"
|
||||
|
||||
conn = psycopg2.connect(
|
||||
host=settings.postgres_host,
|
||||
port=settings.postgres_port,
|
||||
database=database,
|
||||
user=settings.postgres_user,
|
||||
password=settings.postgres_password,
|
||||
connect_timeout=10,
|
||||
)
|
||||
try:
|
||||
with conn:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(f"SELECT count(*) FROM {table} WHERE {cutoff_sql}", (retention_days,))
|
||||
stale = cur.fetchone()[0]
|
||||
|
||||
if dry_run:
|
||||
logger.info("dry run: %s rows in %s.%s exceed %sd", stale, database, table, retention_days)
|
||||
return f"dry run: {stale} rows older than {retention_days}d in {database}.{table}"
|
||||
|
||||
if stale == 0:
|
||||
return f"nothing to prune in {database}.{table} (retention {retention_days}d)"
|
||||
|
||||
cur.execute(f"DELETE FROM {table} WHERE {cutoff_sql}", (retention_days,))
|
||||
deleted = cur.rowcount
|
||||
|
||||
cur.execute(f"SELECT count(*) FROM {table}")
|
||||
remaining = cur.fetchone()[0]
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
logger.info("pruned %s rows from %s.%s, %s remain", deleted, database, table, remaining)
|
||||
return f"pruned {deleted} rows older than {retention_days}d from {database}.{table}, {remaining} remain"
|
||||
|
||||
|
||||
async def execute(config: dict, settings: Settings) -> str:
|
||||
"""Delete rows past the retention window. Returns a one-line summary."""
|
||||
# psycopg2 is synchronous; keep it off the scheduler's event loop.
|
||||
return await asyncio.to_thread(_prune, config, settings)
|
||||
Reference in New Issue
Block a user