OBS-006. Trois racines etaient baties sur os.getcwd() : le magasin de documents, les journaux et les sauvegardes. La vague G a corrige la premiere, parce qu'elle bloquait aussi OPS-011, et a laisse les deux autres. C'est la lecon deja consignee deux fois : un motif fautif corrige dans une seule couche reste dans les autres. Le plus serieux n'est pas le motif, c'est l'ecart qu'il a ouvert. backup.py gardait sa propre constante DOCUMENTS_DIR sur os.getcwd(), donc il ignorait DOCUMENTS_ROOT -- la variable que la vague G a introduite et que docs/deployment.md dit maintenant de regler pour sortir les televersements des repertoires de version. Des qu'un exploitant suit cette consigne, le script archive un repertoire ou l'application n'a jamais rien ecrit. Et comme il repond a un repertoire absent par une ligne d'information et un code de sortie 0, une tache planifiee qui surveille le code de sortie voit vert indefiniment. Autrement dit : plus l'exploitant suivait correctement la documentation de deploiement, plus surement ses sauvegardes de contrats etaient vides. Les trois racines viennent desormais d'app/storage.py, resolues a l'appel et non a l'import, et la sauvegarde imprime la source qu'elle a utilisee. Le message d'absence nomme le chemin ou elle a cherche : "No documents directory found" se lisait comme "il n'y a pas de documents" plutot que comme "je regarde au mauvais endroit". Le test qui porte est celui qui ouvre l'archive : un zip vide est un fichier de taille non nulle, donc verifier qu'un fichier a ete produit ne prouvait rien. Verifie par mutation. Co-Authored-By: Claude Opus 5 <[email protected]>
378 lines
13 KiB
Python
378 lines
13 KiB
Python
"""Database and document backup for the Team Tryouts application.
|
|
|
|
Dumps the PostgreSQL database with pg_dump and archives the uploaded
|
|
contract documents. Designed to be run from a scheduled task (Windows Task
|
|
Scheduler) or a cron job.
|
|
|
|
Usage:
|
|
python app/supporting_scripts/backup.py
|
|
python app/supporting_scripts/backup.py --verify-only <archive>
|
|
|
|
Configuration via environment variables:
|
|
DATABASE_URL PostgreSQL connection string (required)
|
|
BACKUP_DIR Where to store backups (default: ./backups)
|
|
BACKUP_RETENTION_DAYS How long to keep them (default: 30)
|
|
PG_DUMP Path to pg_dump if not on PATH
|
|
PG_RESTORE Path to pg_restore if not on PATH
|
|
|
|
A note on what this file used to be
|
|
-----------------------------------
|
|
The previous version targeted **SQLite**: it imported sqlite3, read
|
|
DATABASE_PATH defaulting to instance/team_tryouts.db, and used the sqlite3
|
|
backup API. Production runs on PostgreSQL, so the file never existed, the
|
|
script printed "[WARNING] Database not found... Skipping database backup"
|
|
and — because main() only tracked the verification result — still exited 0.
|
|
It reported success while backing up nothing at all. Any scheduled task
|
|
watching the exit code saw green.
|
|
|
|
Restoring is documented in docs/restauration-base.md. A backup that has
|
|
never been restored is not a backup.
|
|
"""
|
|
|
|
import argparse
|
|
import os
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
from datetime import datetime, timedelta
|
|
from urllib.parse import unquote, urlparse
|
|
|
|
# Run as `python app/supporting_scripts/backup.py`, sys.path[0] is this
|
|
# script's directory, so the application package is not importable. It has
|
|
# to be — see DOCUMENTS_DIR below.
|
|
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
|
|
|
|
from app.storage import backups_root, documents_root # noqa: E402 — needs the path above
|
|
|
|
# Configuration
|
|
BACKUP_DIR = backups_root()
|
|
BACKUP_RETENTION_DAYS = int(os.getenv('BACKUP_RETENTION_DAYS', 30))
|
|
PG_DUMP = os.getenv('PG_DUMP', 'pg_dump')
|
|
PG_RESTORE = os.getenv('PG_RESTORE', 'pg_restore')
|
|
|
|
# There is deliberately no DOCUMENTS_DIR constant any more. It held
|
|
# `os.path.join(os.getcwd(), 'documents')`, which had stopped being true:
|
|
# wave G introduced DOCUMENTS_ROOT so a release-directory deployment could
|
|
# keep uploads outside the releases, and docs/deployment.md now tells the
|
|
# operator to set it — at which point this script archived a directory the
|
|
# application had never written to. It does not fail on a missing directory
|
|
# either; it prints "No documents directory found", skips, and exits 0.
|
|
#
|
|
# So the more correctly an operator followed the deployment documentation,
|
|
# the more certainly their contract backups were empty (OBS-006).
|
|
#
|
|
# backup_documents() now asks app.storage, at call time, the same question
|
|
# the upload path asks. One source of truth, and one that a test can move.
|
|
|
|
|
|
class BackupError(Exception):
|
|
"""Raised when a backup step fails in a way that must stop the run."""
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Connection handling
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def parse_database_url(url):
|
|
"""Split a SQLAlchemy/PostgreSQL URL into pg_dump connection settings.
|
|
|
|
Accepts the dialect suffixes SQLAlchemy uses (postgresql+psycopg://),
|
|
which pg_dump does not understand.
|
|
|
|
Args:
|
|
url: The connection string.
|
|
|
|
Returns:
|
|
dict: host, port, dbname, user, password.
|
|
|
|
Raises:
|
|
BackupError: If the URL is missing or is not a PostgreSQL one.
|
|
"""
|
|
if not url:
|
|
raise BackupError('DATABASE_URL is not set.')
|
|
|
|
parsed = urlparse(url)
|
|
scheme = parsed.scheme.split('+')[0]
|
|
if scheme not in ('postgresql', 'postgres'):
|
|
raise BackupError(
|
|
f'DATABASE_URL is not a PostgreSQL connection string (scheme: {scheme!r}). '
|
|
'This script only backs up PostgreSQL.'
|
|
)
|
|
|
|
dbname = (parsed.path or '').lstrip('/')
|
|
if not dbname:
|
|
raise BackupError('DATABASE_URL does not name a database.')
|
|
|
|
return {
|
|
'host': parsed.hostname or 'localhost',
|
|
'port': str(parsed.port or 5432),
|
|
'dbname': dbname,
|
|
'user': unquote(parsed.username) if parsed.username else '',
|
|
'password': unquote(parsed.password) if parsed.password else '',
|
|
}
|
|
|
|
|
|
def describe_target(conn):
|
|
"""Human-readable target, deliberately without the password."""
|
|
user = f'{conn["user"]}@' if conn['user'] else ''
|
|
return f'{user}{conn["host"]}:{conn["port"]}/{conn["dbname"]}'
|
|
|
|
|
|
def build_dump_command(conn, output_path):
|
|
"""Assemble the pg_dump invocation.
|
|
|
|
--format=custom is compressed and lets pg_restore rebuild selectively;
|
|
plain SQL would be larger and all-or-nothing.
|
|
|
|
The password is never placed on the command line — it would be visible
|
|
to anyone able to list processes. It travels through PGPASSWORD instead,
|
|
which is what pg_dump documents for non-interactive use.
|
|
"""
|
|
return [
|
|
PG_DUMP,
|
|
'--host',
|
|
conn['host'],
|
|
'--port',
|
|
conn['port'],
|
|
'--username',
|
|
conn['user'],
|
|
'--dbname',
|
|
conn['dbname'],
|
|
'--format=custom',
|
|
'--no-owner',
|
|
'--no-privileges',
|
|
'--file',
|
|
output_path,
|
|
]
|
|
|
|
|
|
def dump_environment(conn):
|
|
"""Environment for pg_dump/pg_restore, carrying the password out of argv."""
|
|
env = os.environ.copy()
|
|
if conn['password']:
|
|
env['PGPASSWORD'] = conn['password']
|
|
return env
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Backup steps
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def create_backup_dir():
|
|
"""Create the backup directory if it doesn't exist."""
|
|
os.makedirs(BACKUP_DIR, exist_ok=True)
|
|
|
|
|
|
def backup_database(conn):
|
|
"""Dump the PostgreSQL database.
|
|
|
|
Returns:
|
|
str: Path to the created archive.
|
|
|
|
Raises:
|
|
BackupError: If pg_dump is missing, fails, or produces nothing.
|
|
"""
|
|
timestamp = datetime.now().strftime('%Y%m%d_%H%M%S')
|
|
backup_path = os.path.join(BACKUP_DIR, f'db_backup_{timestamp}.dump')
|
|
|
|
print(f'[INFO] Dumping {describe_target(conn)}')
|
|
|
|
try:
|
|
result = subprocess.run(
|
|
build_dump_command(conn, backup_path),
|
|
env=dump_environment(conn),
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=900,
|
|
)
|
|
except FileNotFoundError as err:
|
|
raise BackupError(
|
|
f'{PG_DUMP} not found. Install the PostgreSQL client tools, or set '
|
|
'PG_DUMP to its full path.'
|
|
) from err
|
|
except subprocess.TimeoutExpired as err:
|
|
raise BackupError('pg_dump timed out after 15 minutes.') from err
|
|
|
|
if result.returncode != 0:
|
|
raise BackupError(f'pg_dump failed: {result.stderr.strip()}')
|
|
|
|
if not os.path.exists(backup_path) or os.path.getsize(backup_path) == 0:
|
|
raise BackupError('pg_dump reported success but produced an empty file.')
|
|
|
|
size_mb = os.path.getsize(backup_path) / (1024 * 1024)
|
|
print(f'[OK] Database backed up to: {backup_path} ({size_mb:.1f} MB)')
|
|
return backup_path
|
|
|
|
|
|
def verify_backup(backup_path):
|
|
"""Check that the archive is readable and actually contains tables.
|
|
|
|
pg_restore --list parses the whole archive without touching any
|
|
database. A dump that cannot be listed cannot be restored, and an
|
|
archive holding no table would mean the dump ran against the wrong
|
|
target — both are silent failures worth catching here rather than
|
|
during an incident.
|
|
|
|
Args:
|
|
backup_path: Path to the archive to verify.
|
|
|
|
Returns:
|
|
bool: True if the archive looks restorable.
|
|
"""
|
|
if not backup_path or not os.path.exists(backup_path):
|
|
print('[ERROR] Nothing to verify.')
|
|
return False
|
|
|
|
try:
|
|
result = subprocess.run(
|
|
[PG_RESTORE, '--list', backup_path],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=300,
|
|
)
|
|
except FileNotFoundError:
|
|
print(f'[WARNING] {PG_RESTORE} not found: archive left unverified.')
|
|
return False
|
|
except subprocess.TimeoutExpired:
|
|
print('[ERROR] pg_restore --list timed out.')
|
|
return False
|
|
|
|
if result.returncode != 0:
|
|
print(f'[ERROR] Archive is not readable: {result.stderr.strip()}')
|
|
return False
|
|
|
|
table_count = sum(1 for line in result.stdout.splitlines() if ' TABLE DATA ' in line)
|
|
if table_count == 0:
|
|
print('[ERROR] Archive contains no table data.')
|
|
return False
|
|
|
|
print(f'[OK] Archive verified: {table_count} table(s) present.')
|
|
return True
|
|
|
|
|
|
def backup_documents():
|
|
"""Archive the uploaded contract documents directory.
|
|
|
|
The directory is resolved through `app.storage.documents_root()` — the
|
|
same function the upload path uses — so that setting DOCUMENTS_ROOT
|
|
moves both together. Resolved here rather than at import, so that what
|
|
is backed up depends on the environment the run has, not on the one the
|
|
module happened to be imported with.
|
|
|
|
Returns:
|
|
str: Path to the created archive, or None if there is nothing to
|
|
archive. Signed contracts live only on disk, so losing this
|
|
directory loses the documents themselves.
|
|
"""
|
|
documents_dir = documents_root()
|
|
|
|
if not os.path.exists(documents_dir):
|
|
# Says where it looked. The previous message named no path, so an
|
|
# operator who had moved the documents read it as "there are no
|
|
# documents" rather than "I am looking in the wrong place".
|
|
print(f'[INFO] No documents directory at {documents_dir}. Skipping document backup.')
|
|
return None
|
|
|
|
timestamp = datetime.now().strftime('%Y%m%d_%H%M%S')
|
|
archive_basename = os.path.join(BACKUP_DIR, f'documents_backup_{timestamp}')
|
|
|
|
try:
|
|
shutil.make_archive(archive_basename, 'zip', documents_dir)
|
|
except Exception as exc: # noqa: BLE001 — a failed document archive must not lose the dump
|
|
# This runs after the database dump has already succeeded. Letting
|
|
# anything through here would abort the script with a traceback and
|
|
# take the one part that worked down with it. Reported to stdout, in
|
|
# the format the rest of this script uses; it has no logger.
|
|
print(f'[ERROR] Document backup failed: {exc}')
|
|
return None
|
|
|
|
zip_path = f'{archive_basename}.zip'
|
|
size_mb = os.path.getsize(zip_path) / (1024 * 1024)
|
|
print(f'[OK] Documents backed up to: {zip_path} ({size_mb:.1f} MB)')
|
|
return zip_path
|
|
|
|
|
|
def cleanup_old_backups():
|
|
"""Remove backup files older than BACKUP_RETENTION_DAYS."""
|
|
if not os.path.exists(BACKUP_DIR):
|
|
return
|
|
|
|
cutoff = datetime.now() - timedelta(days=BACKUP_RETENTION_DAYS)
|
|
removed_count = 0
|
|
|
|
for filename in os.listdir(BACKUP_DIR):
|
|
file_path = os.path.join(BACKUP_DIR, filename)
|
|
if not os.path.isfile(file_path):
|
|
continue
|
|
if datetime.fromtimestamp(os.path.getmtime(file_path)) >= cutoff:
|
|
continue
|
|
try:
|
|
os.remove(file_path)
|
|
removed_count += 1
|
|
print(f'[CLEANUP] Removed old backup: {filename}')
|
|
except OSError as exc:
|
|
print(f'[WARNING] Could not remove {filename}: {exc}')
|
|
|
|
if removed_count:
|
|
print(f'[CLEANUP] Removed {removed_count} old backup(s).')
|
|
else:
|
|
print('[CLEANUP] No old backups to remove.')
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Entry point
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def main(argv=None):
|
|
"""Run the full backup process.
|
|
|
|
Returns:
|
|
int: 0 when the database was dumped AND verified, 1 otherwise. The
|
|
previous version returned 0 even when it had backed up nothing.
|
|
"""
|
|
parser = argparse.ArgumentParser(description='Team Tryouts backup')
|
|
parser.add_argument(
|
|
'--verify-only', metavar='ARCHIVE', help='Verify an existing archive and exit'
|
|
)
|
|
args = parser.parse_args(argv)
|
|
|
|
if args.verify_only:
|
|
return 0 if verify_backup(args.verify_only) else 1
|
|
|
|
print('=== Team Tryouts Backup ===')
|
|
print(f'Started at: {datetime.now().strftime("%Y-%m-%d %H:%M:%S")}')
|
|
print(f'Backup directory: {BACKUP_DIR}')
|
|
# Printed because it is the value that was wrong for months without
|
|
# anyone being able to see it from the output.
|
|
print(f'Document source: {documents_root()}')
|
|
print(f'Retention period: {BACKUP_RETENTION_DAYS} days')
|
|
print()
|
|
|
|
try:
|
|
conn = parse_database_url(os.getenv('DATABASE_URL'))
|
|
create_backup_dir()
|
|
backup_path = backup_database(conn)
|
|
except BackupError as exc:
|
|
print(f'[ERROR] {exc}')
|
|
print('\n=== Backup FAILED — no database backup was produced ===')
|
|
return 1
|
|
|
|
verified = verify_backup(backup_path)
|
|
backup_documents()
|
|
cleanup_old_backups()
|
|
|
|
print()
|
|
if verified:
|
|
print('=== Backup completed successfully ===')
|
|
return 0
|
|
|
|
print('=== Backup FAILED verification — do not rely on this archive ===')
|
|
return 1
|
|
|
|
|
|
if __name__ == '__main__':
|
|
sys.exit(main())
|