feat(07-04): Celery retry harness + _ClassificationError sentinel (D-09/D-10)

- Add _ClassificationError sentinel raised by _run() for classification failures
- Add _mark_classification_failed() async helper for final status writeback
- Change decorator to bind=True, max_retries=3
- Outer sync task catches _ClassificationError and calls self.retry(countdown=[30,90,270])
- MaxRetriesExceededError caught in nested try/except to avoid re-raise from exc block
- Promote test_retry_backoff and test_exhaustion_sets_failed_status from xfail to passing
- Tests use push_request(retries=N) + patch.object(task, "retry") to verify countdown values
- test_exhaustion_sets_failed_status uses assert_called_once_with (asyncio.run calling convention)
This commit is contained in:
curo1305
2026-06-04 23:08:16 +02:00
parent b8b0840729
commit e9ee5d4ba5
2 changed files with 158 additions and 26 deletions
+77 -14
View File
@@ -12,17 +12,82 @@ Flow:
4. Extract text from bytes using services.extractor
5. Persist extracted_text back to the Document row
6. Call services.classifier.classify_document to assign topics
7. Return a result dict (never raises — classification failures are non-fatal)
7. Return a result dict; classification failures are retried with exponential backoff
Celery retry harness (D-09, D-10 — Pitfall 3 from RESEARCH.md):
CRITICAL: self.retry() MUST be raised from the outer sync task, NOT inside
asyncio.run(). _ClassificationError is a sentinel raised by _run() to signal
a retryable classification failure. The outer extract_and_classify catches it
and calls self.retry(countdown=...) in the sync layer.
"""
import asyncio
from celery.exceptions import MaxRetriesExceededError
from celery_app import celery_app
@celery_app.task(name="tasks.document_tasks.extract_and_classify")
def extract_and_classify(document_id: str) -> dict:
"""Synchronous Celery entry-point — delegates to async _run via asyncio.run."""
return asyncio.run(_run(document_id))
class _ClassificationError(Exception):
"""Sentinel exception raised by _run() to signal a retryable classification failure.
This exception escapes asyncio.run() and is caught by the outer sync task,
which then calls self.retry() in the Celery sync layer (not inside asyncio.run).
Non-classification failures (extract_failed, invalid_id) are NOT retried —
they return a dict directly from _run() without raising _ClassificationError.
"""
pass
async def _mark_classification_failed(document_id: str) -> None:
"""Write doc.status = 'classification_failed' after all retries are exhausted.
D-10: called by extract_and_classify's MaxRetriesExceededError handler via
asyncio.run(_mark_classification_failed(document_id)) so the final status
is written to DB regardless of Celery task context.
"""
import uuid as _uuid
from db.session import AsyncSessionLocal
from db.models import Document
async with AsyncSessionLocal() as session:
try:
doc_uuid = _uuid.UUID(document_id)
except ValueError:
return # Silently ignore invalid IDs — nothing to update
doc = await session.get(Document, doc_uuid)
if doc is not None:
doc.status = "classification_failed"
await session.commit()
@celery_app.task(
name="tasks.document_tasks.extract_and_classify",
bind=True,
max_retries=3,
)
def extract_and_classify(self, document_id: str) -> dict:
"""Synchronous Celery entry-point — delegates to async _run via asyncio.run.
Retry harness (D-09): classification failures are retried up to 3 times
with exponential backoff: 30s / 90s / 270s.
Pitfall 3 guard: self.retry() is called HERE in the sync layer, never
inside asyncio.run(). _ClassificationError is the sentinel that escapes
asyncio.run() to trigger the retry.
"""
try:
return asyncio.run(_run(document_id))
except _ClassificationError as exc:
# Exponential backoff: 30s on first retry, 90s on second, 270s on third
countdowns = [30, 90, 270]
countdown = countdowns[min(self.request.retries, 2)]
try:
raise self.retry(exc=exc, countdown=countdown)
except MaxRetriesExceededError:
# All retries exhausted — write final failure status to DB
asyncio.run(_mark_classification_failed(document_id))
return {"document_id": document_id, "status": "classification_failed"}
async def _run(document_id: str) -> dict:
@@ -34,6 +99,9 @@ async def _run(document_id: str) -> dict:
Cloud-aware: when doc.storage_backend != 'minio', uses
get_storage_backend_for_document() to retrieve bytes from the correct
cloud backend instead of hardcoding MinIO.
Classification failures raise _ClassificationError (D-09 retryable sentinel).
Non-classification failures return a status dict (not retried).
"""
import uuid as _uuid
@@ -105,7 +173,7 @@ async def _run(document_id: str) -> dict:
"error": f"Text extraction failed: {e}",
}
# ── Step 4: classify document (non-fatal) ──────────────────────────────
# ── Step 4: classify document (retryable via _ClassificationError) ─────
try:
topics = await classifier.classify_document(session, document_id, ai_provider=ai_provider, ai_model=ai_model)
return {
@@ -114,14 +182,9 @@ async def _run(document_id: str) -> dict:
"topics": topics,
}
except Exception as e:
# Non-fatal — preserve existing convention from api/documents.py
doc.status = "classification_failed"
await session.commit()
return {
"document_id": document_id,
"status": "classification_failed",
"error": str(e),
}
# Raise sentinel to allow the outer sync layer to call self.retry()
# (D-09 Pitfall 3: self.retry() cannot be called inside asyncio.run)
raise _ClassificationError(str(e)) from e
@celery_app.task(name="tasks.document_tasks.cleanup_abandoned_uploads")