fix: prevent server crashes during photo processing
CI / skip-ci-check (pull_request) Successful in 1m49s
CI / lint-and-type-check (pull_request) Successful in 2m28s
CI / python-lint (pull_request) Successful in 2m12s
CI / test-backend (pull_request) Successful in 4m5s
CI / build (pull_request) Successful in 4m53s
CI / secret-scanning (pull_request) Successful in 1m56s
CI / dependency-scan (pull_request) Successful in 1m54s
CI / sast-scan (pull_request) Successful in 3m2s
CI / workflow-summary (pull_request) Successful in 1m48s
CI / skip-ci-check (pull_request) Successful in 1m49s
CI / lint-and-type-check (pull_request) Successful in 2m28s
CI / python-lint (pull_request) Successful in 2m12s
CI / test-backend (pull_request) Successful in 4m5s
CI / build (pull_request) Successful in 4m53s
CI / secret-scanning (pull_request) Successful in 1m56s
CI / dependency-scan (pull_request) Successful in 1m54s
CI / sast-scan (pull_request) Successful in 3m2s
CI / workflow-summary (pull_request) Successful in 1m48s
- Add database connection health checks every 10 photos - Add session refresh logic to recover from connection errors - Improve error handling for database disconnections/timeouts - Add explicit image cleanup to prevent memory leaks - Add connection error detection throughout processing pipeline - Gracefully handle database connection failures instead of crashing Fixes issue where server would crash during long-running photo processing tasks when database connections were lost or timed out.
This commit is contained in:
@@ -119,6 +119,34 @@ def process_faces_task(
|
||||
total_faces_detected = 0
|
||||
total_faces_stored = 0
|
||||
|
||||
def refresh_db_session():
|
||||
"""Refresh database session if it becomes stale or disconnected.
|
||||
|
||||
This prevents crashes when the database connection is lost during long-running
|
||||
processing tasks. Closes the old session and creates a new one.
|
||||
"""
|
||||
nonlocal db
|
||||
try:
|
||||
# Test if the session is still alive by executing a simple query
|
||||
from sqlalchemy import text
|
||||
db.execute(text("SELECT 1"))
|
||||
db.commit() # Ensure transaction is clean
|
||||
except Exception as e:
|
||||
# Session is stale or disconnected - create a new one
|
||||
try:
|
||||
print(f"[Task] Database session disconnected, refreshing... Error: {e}")
|
||||
except (BrokenPipeError, OSError):
|
||||
pass
|
||||
try:
|
||||
db.close()
|
||||
except Exception:
|
||||
pass
|
||||
db = SessionLocal()
|
||||
try:
|
||||
print(f"[Task] Database session refreshed")
|
||||
except (BrokenPipeError, OSError):
|
||||
pass
|
||||
|
||||
try:
|
||||
def update_progress(
|
||||
processed: int,
|
||||
@@ -181,6 +209,9 @@ def process_faces_task(
|
||||
# Process faces
|
||||
# Wrap in try-except to ensure we preserve progress even if process_unprocessed_photos fails
|
||||
try:
|
||||
# Refresh session before starting processing to ensure it's healthy
|
||||
refresh_db_session()
|
||||
|
||||
photos_processed, total_faces_detected, total_faces_stored = (
|
||||
process_unprocessed_photos(
|
||||
db,
|
||||
@@ -191,6 +222,27 @@ def process_faces_task(
|
||||
)
|
||||
)
|
||||
except Exception as e:
|
||||
# Check if it's a database connection error
|
||||
error_str = str(e).lower()
|
||||
is_db_error = any(keyword in error_str for keyword in [
|
||||
'connection', 'disconnect', 'timeout', 'closed', 'lost',
|
||||
'operationalerror', 'database', 'server closed', 'connection reset',
|
||||
'connection pool', 'connection refused', 'session needs refresh'
|
||||
])
|
||||
|
||||
if is_db_error:
|
||||
# Try to refresh the session - this helps if the error is recoverable
|
||||
# but we don't retry the entire batch to avoid reprocessing photos
|
||||
try:
|
||||
print(f"[Task] Database error detected, attempting to refresh session: {e}")
|
||||
refresh_db_session()
|
||||
print(f"[Task] Session refreshed - job will fail gracefully. Restart job to continue processing remaining photos.")
|
||||
except Exception as refresh_error:
|
||||
try:
|
||||
print(f"[Task] Failed to refresh database session: {refresh_error}")
|
||||
except (BrokenPipeError, OSError):
|
||||
pass
|
||||
|
||||
# If process_unprocessed_photos fails, preserve any progress made
|
||||
# and re-raise so the outer handler can log it properly
|
||||
try:
|
||||
|
||||
Reference in New Issue
Block a user