backups: record that a config was checked, not only that it changed
Some checks failed
CI / backend (push) Failing after 8s
CI / naming (push) Successful in 2s
CI / frontend (push) Successful in 10s
CI / migrations-mysql (push) Failing after 7s

The stale-backup card could not be built as designed, and the reason is more
important than the card. Dedup means an unchanged configuration writes no
revision, so collectedat moves only on a CHANGE. A machine stable for six
months has a six-month-old newest revision and is perfectly healthy. Keying a
staleness card on revision age would have flagged most of the fleet - exactly
the noise that makes a board worth ignoring.

Underneath that: ShopDB could not distinguish those cases at all. On a no-op
the server returned "unchanged" and wrote nothing, so "we checked yesterday and
it matched" was discarded. That fact is the one thing a backup system must be
able to prove, and the only record of it was a line in a log file on the PC.

lastseenat records the check rather than the change. Touched on every matching
post including the no-op; set on creation, since a new revision has by
definition just been seen; backfilled from collectedat or createdat so existing
rows start from the last moment the config can be PROVEN current, rather than
from now - claiming a check that never happened would be worse than silence.

The card keys on it, one row per CHAIN rather than per asset: a machine with
two part markers can have one still reporting while the other stopped, and a
per-asset view would report the machine as fine. It stays deliberately silent
about assets never backed up, because whether one SHOULD be is a question only
the manifest can answer, and guessing would list a hundred healthy machines.

The rule lives in services/staleness.py rather than the route, so it is
testable without an auth layer in the way - the same split retention.py uses.

Threshold is backups_staledays, default 3, and 0 disables the card.
This commit is contained in:
cproudlock
2026-08-11 13:52:07 -04:00
parent 1ca8a9b8e8
commit 6c975a107c
8 changed files with 325 additions and 2 deletions

View File

@@ -344,3 +344,19 @@ def _diffprojections(old, new):
'aftertype': None if after is None else after[0],
})
return changes
@backups_bp.route('/dashboard/stale', methods=['GET'])
@jwt_required()
@require_permission('backups.view')
def dashboard_stale():
"""Chains whose backup has stopped running.
Thin: the rule lives in services/staleness.py, where it is testable without
an auth layer in the way. Keyed on the last CONFIRMED check, never on the
last change - dedup means an unchanged config writes no revision, so a card
keyed on revision age would flag most of a healthy fleet.
"""
from ..services.staleness import stalechains
return success_response(stalechains())

View File

@@ -0,0 +1,43 @@
"""backups: record when a configuration was last confirmed unchanged.
Dedup means an unchanged config writes NO revision, so `collectedat` moves only
when something changes. A machine whose settings have been stable for six months
therefore has a six-month-old newest revision and is entirely healthy - and
ShopDB had no way to tell it apart from a machine whose backup stopped running
six months ago. The evidence that a check happened existed only in a log file on
the PC.
`lastseenat` records the check rather than the change. It is touched on every
matching post, including the no-op that writes nothing else.
Backfilled from `collectedat` (falling back to `createdat`) so existing rows
start from the last moment we can actually prove the config was current, rather
than from now - claiming a fresh check that never happened would be worse than
saying nothing.
"""
from alembic import op
import sqlalchemy as sa
# revision identifiers, used by Alembic.
revision = 'backups0002lastseenat'
down_revision = 'backups0001baseline'
branch_labels = None
depends_on = None
def upgrade():
columns = {c['name'] for c in
sa.inspect(op.get_bind()).get_columns('backuprevisions')}
if 'lastseenat' not in columns:
op.add_column('backuprevisions',
sa.Column('lastseenat', sa.DateTime(), nullable=True))
op.execute('UPDATE backuprevisions '
'SET lastseenat = COALESCE(collectedat, createdat)')
def downgrade():
columns = {c['name'] for c in
sa.inspect(op.get_bind()).get_columns('backuprevisions')}
if 'lastseenat' in columns:
op.drop_column('backuprevisions', 'lastseenat')

View File

@@ -89,6 +89,16 @@ class BackupRevision(db.Model):
createdat = db.Column(db.DateTime, nullable=False, default=datetime.utcnow)
# When this configuration was last CONFIRMED still current, which is not the
# same as when it last changed. Dedup means an unchanged config writes no
# revision, so collectedat only ever moves on a change - a machine whose
# settings have been stable for six months has a six-month-old newest
# revision and is perfectly healthy. Without this column ShopDB cannot tell
# that machine from one whose backup stopped running six months ago, which
# is the one question a backup system has to be able to answer. Touched on
# every matching post, including the no-op that writes nothing else.
lastseenat = db.Column(db.DateTime, nullable=True)
__table_args__ = (
db.Index('ixbackuprevisionsassetkind', 'assetid', 'backupkind'),
)
@@ -138,6 +148,10 @@ class BackupRevision(db.Model):
# BROWSER-LOCAL and the timestamp silently shifts by the viewer's
# offset before any site-timezone formatting is applied.
'collectedat': _utciso(self.collectedat),
# When the config was last CONFIRMED current, versus when it last
# changed. The history view needs both or a stable machine looks
# abandoned.
'lastseenat': _utciso(self.lastseenat),
'createdat': _utciso(self.createdat),
}
if includepayload:

View File

@@ -161,6 +161,17 @@ class BackupsPlugin(BasePlugin):
'description': 'UNC root that opaque (non-JSON) backups are '
'written under by the collecting PC. Site-specific.',
},
{
'key': 'backups_staledays',
'value': '3',
'valuetype': 'integer',
'category': 'backups',
'description': 'Days without a CONFIRMED check before a backup '
'is listed as stopped on the dashboard. Measured '
'from the last check, not the last change - an '
'unchanged config writes no revision. 0 disables '
'the card.',
},
{
'key': 'backups_retentioncount',
'value': '50',
@@ -180,6 +191,36 @@ class BackupsPlugin(BasePlugin):
},
]
def get_dashboard_widgets(self) -> List[Dict]:
"""Dashboard card: backups that have stopped running.
Keyed on the last CONFIRMED check, never on the last change. Dedup means
an unchanged config writes no revision, so a card keyed on revision age
would flag most of a healthy fleet.
"""
return [
{
'id': 'backups-stale',
'title': 'Backups stopped',
'endpoint': '/api/backups/dashboard/stale',
'render': 'exceptions',
'severity': 'warning',
'permission': 'backups.view',
'empty': 'hide',
'position': 30,
'map': {
'title': 'assetnumber',
'detail': 'backupkind',
'meta': [
{'key': 'sourcehostname', 'label': 'from'},
{'key': 'quietdays', 'label': 'last checked',
'suffix': ' days ago'},
],
'link': '/backups/asset/{assetid}',
},
},
]
# ---- ADR-006 collector contract -------------------------------------
def get_collector_schema(self) -> Optional[dict]:
@@ -299,6 +340,13 @@ class BackupsPlugin(BasePlugin):
.first())
if latest is not None and latest.contenthash == contenthash:
# Unchanged, so no revision - but RECORD THE CHECK. Without this the
# only evidence a backup still runs is a line in a log on the PC,
# and a config stable for six months is indistinguishable from a
# backup that died six months ago. Cheap: one column on a row that
# already exists, no new history.
latest.lastseenat = datetime.utcnow()
db.session.flush()
return {
'action': 'noop',
'assetid': assetid,
@@ -330,6 +378,7 @@ class BackupsPlugin(BasePlugin):
bytesize=bytesize,
sourcehostname=payload.get('sourcehostname'),
collectedat=collectedat or datetime.utcnow(),
lastseenat=datetime.utcnow(),
)
revision.payload = projection
db.session.add(revision)

View File

@@ -0,0 +1,85 @@
"""Which backup chains have stopped running.
Split from the route for the same reason retention.py is: the rule here is the
interesting part and it should be testable without an auth layer in the way.
THE RULE, and it is not the obvious one. Dedup means an unchanged configuration
writes no revision, so `collectedat` moves only when something CHANGES. A
machine whose settings have been stable for six months has a six-month-old
newest revision and is entirely healthy. Staleness is therefore measured from
`lastseenat` - when the config was last CONFIRMED current, recorded on every
matching post including the no-op that writes nothing else.
Keyed on the CHAIN (asset, kind, source PC), because that is the unit that fails
independently. A machine with two part markers can have one still reporting
while the other stopped, and a per-asset view would report the machine as fine.
"""
from datetime import datetime, timedelta
from shopdb.api import db, Asset
from ..models import BackupRevision
DEFAULTSTALEDAYS = 3
def staledays():
"""Configured threshold; 0 or negative disables the card entirely."""
from shopdb.api import Setting
setting = Setting.query.filter_by(key='backups_staledays').first()
if setting and (setting.value or '').strip():
try:
return int(setting.value)
except (TypeError, ValueError):
pass
return DEFAULTSTALEDAYS
def latestperchain():
"""Newest revision of every (asset, kind, source) chain."""
latest = {}
for revision in (BackupRevision.query
.order_by(BackupRevision.backuprevisionid.asc()).all()):
latest[(revision.assetid, revision.backupkind,
revision.sourcehostname)] = revision
return latest
def stalechains(days=None, limit=50):
"""Chains whose last confirmed check is older than the threshold.
Deliberately silent about assets that have NEVER been backed up. Whether one
SHOULD be is a question only the manifest can answer - most machines carry
no NTLARS at all - and guessing would list a hundred healthy assets, which
is the noise that makes a board worth ignoring. That is the
desired-versus-observed card, which needs the manifest resolver.
"""
days = staledays() if days is None else days
if days <= 0:
return []
now = datetime.utcnow()
cutoff = now - timedelta(days=days)
rows = []
for revision in latestperchain().values():
# Rows written before lastseenat existed fall back to the timestamps
# that do exist, so an old install reports something sane on day one
# rather than every chain at once.
seen = revision.lastseenat or revision.collectedat or revision.createdat
if seen is None or seen >= cutoff:
continue
asset = db.session.get(Asset, revision.assetid)
rows.append({
'assetid': revision.assetid,
'assetnumber': asset.assetnumber if asset else str(revision.assetid),
'backupkind': revision.backupkind,
'sourcehostname': revision.sourcehostname,
'lastseenat': seen.isoformat() + 'Z',
'quietdays': (now - seen).days,
})
rows.sort(key=lambda r: -r['quietdays'])
return rows[:limit]