- api/admin/users.py: removed module docstring + 7 WHAT function docstrings - api/admin/quotas.py: removed module docstring + 2 WHAT function docstrings - api/admin/ai.py: removed module docstring + 3 WHAT inline comments; security invariant docstrings preserved - api/admin/shared.py: unchanged (all comments are WHY — constraint notes) - api/documents/upload.py: removed 4 WHAT inline comments; T-03-05/T-03-06 WHY notes preserved - api/documents/content.py: removed WHAT function docstring + WHAT inline comment - api/documents/crud.py: removed 7 WHAT inline comments; D-16 + security constraint docstrings preserved - api/auth/shared.py: removed module docstring + 1 WHAT inline comment - api/auth/tokens.py: removed module docstring + 8 WHAT function docstrings/comments; family-revocation + SEC-02 WHY notes preserved - api/auth/totp.py: removed module docstring + 2 WHAT function docstrings + 2 WHAT inline comments - api/auth/password.py: removed module docstring + 2 WHAT function docstrings + 3 WHAT inline comments - All NO-prefix anchor comments and security-invariant WHY comments preserved - pytest: 413 passed, 1 failed (pre-existing ModuleNotFoundError unrelated to purge)
380 lines
15 KiB
Python
380 lines
15 KiB
Python
"""Document CRUD endpoints — list, get, patch, delete, and re-classify.
|
|
|
|
Endpoints:
|
|
GET "" — list documents with sort, folder filter, and FTS (list_documents)
|
|
GET /{doc_id} — get document metadata (get_document)
|
|
PATCH /{doc_id} — update filename and/or folder_id (patch_document)
|
|
DELETE /{doc_id} — delete document, decrement quota atomically (delete_document)
|
|
POST /{doc_id}/classify — re-queue Celery classification (classify_document, D-08)
|
|
|
|
Sub-router carries NO prefix — prefix="/api/documents" lives in __init__.py (D-04).
|
|
|
|
Security:
|
|
T-03-11: ownership assertion on every resource endpoint — cross-user access returns 404.
|
|
T-05-09-01: get_regular_user dep rejects admins (403) and unauthenticated (401).
|
|
T-05-09-02: response uses storage.get_metadata() whitelist — no credentials_enc, no password_hash.
|
|
T-06.2-03-01: cloud documents skip MinIO quota decrement.
|
|
T-06.2-03-02: cloud delete failure returns {success: false, cloud_delete_failed: true} (HTTP 200).
|
|
T-07-10: classify endpoint — IDOR returns 404 per ownership assertion.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import uuid
|
|
from typing import Optional
|
|
|
|
import structlog as _structlog
|
|
|
|
_log = _structlog.get_logger(__name__)
|
|
|
|
from fastapi import APIRouter, Depends, HTTPException, Query, Request, status
|
|
from fastapi.responses import JSONResponse
|
|
from sqlalchemy import func, select
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
from db.models import Document, Folder, Share, User
|
|
from deps.auth import get_regular_user
|
|
from deps.db import get_db
|
|
from deps.utils import get_client_ip
|
|
from services import classifier, storage
|
|
from services.audit import write_audit_log
|
|
from services.rate_limiting import account_limiter
|
|
from storage import get_storage_backend_for_document as _get_storage_backend_for_document
|
|
from tasks.document_tasks import extract_and_classify
|
|
|
|
from api.documents.shared import DocumentPatch, _CLOUD_PROVIDERS
|
|
|
|
router = APIRouter()
|
|
|
|
|
|
# ── GET /api/documents ────────────────────────────────────────────────────────
|
|
# Route registered on parent router in __init__.py (FastAPI 0.100+ disallows
|
|
# include_router when both the include prefix and route path are empty strings).
|
|
|
|
@account_limiter.limit("100/minute")
|
|
async def list_documents(
|
|
request: Request,
|
|
topic: Optional[str] = Query(None),
|
|
page: int = Query(1, ge=1),
|
|
per_page: int = Query(20, ge=1, le=100),
|
|
sort: str = Query("date"),
|
|
order: str = Query("desc"),
|
|
folder_id: Optional[str] = Query(None),
|
|
q: Optional[str] = Query(None),
|
|
session: AsyncSession = Depends(get_db),
|
|
current_user: User = Depends(get_regular_user),
|
|
):
|
|
"""List documents with optional sort, folder filter, and full-text search.
|
|
|
|
D-16: requires authenticated regular user (get_regular_user rejects admins).
|
|
Returns only documents belonging to the current user.
|
|
|
|
FOLD-05: sort by name|date|size; order asc|desc; folder_id filter;
|
|
q full-text search via plainto_tsquery (PostgreSQL only — silently skipped
|
|
on SQLite when function is unavailable). FTS scope is always scoped to
|
|
current_user.id (T-04-03-02).
|
|
|
|
Backward-compat: when sort/order/folder_id/q are not provided, behaviour
|
|
is identical to the pre-Phase-4 implementation.
|
|
"""
|
|
request.state.current_user = current_user
|
|
# If no new params used, fall through to the legacy storage.list_metadata path
|
|
# to preserve full backward compatibility with topic filtering.
|
|
if folder_id is None and q is None and sort == "date" and order == "desc":
|
|
docs = await storage.list_metadata(session, user_id=current_user.id, topic=topic)
|
|
total = len(docs)
|
|
start = (page - 1) * per_page
|
|
# Add is_shared field (Phase 4 addition)
|
|
shared_result = await session.execute(
|
|
select(Share.document_id).where(Share.owner_id == current_user.id)
|
|
)
|
|
shared_ids = {row[0] for row in shared_result.fetchall()}
|
|
items = []
|
|
for d in docs[start : start + per_page]:
|
|
doc_id_str = d.get("id", "")
|
|
try:
|
|
doc_uuid = uuid.UUID(doc_id_str)
|
|
except (ValueError, AttributeError):
|
|
doc_uuid = None
|
|
d["is_shared"] = doc_uuid in shared_ids if doc_uuid else False
|
|
items.append(d)
|
|
return {"items": items, "total": total, "page": page, "per_page": per_page}
|
|
|
|
from db.models import DocumentTopic, Topic # noqa: PLC0415 (avoid circular at module top)
|
|
|
|
stmt = select(Document).where(Document.user_id == current_user.id)
|
|
|
|
if topic is not None:
|
|
stmt = (
|
|
stmt.join(DocumentTopic, DocumentTopic.document_id == Document.id)
|
|
.join(Topic, Topic.id == DocumentTopic.topic_id)
|
|
.where(Topic.name == topic)
|
|
)
|
|
|
|
if folder_id is not None:
|
|
try:
|
|
folder_uuid = uuid.UUID(folder_id)
|
|
except ValueError:
|
|
raise HTTPException(status_code=404, detail="Folder not found")
|
|
stmt = stmt.where(Document.folder_id == folder_uuid)
|
|
|
|
sort_col = Document.created_at # default: date
|
|
if sort == "name":
|
|
sort_col = Document.filename
|
|
elif sort == "size":
|
|
sort_col = Document.size_bytes
|
|
|
|
order_fn = sort_col.asc if order == "asc" else sort_col.desc
|
|
stmt = stmt.order_by(order_fn())
|
|
|
|
# Full-text search — plainto_tsquery on extracted_text (PostgreSQL only)
|
|
# Falls back to unfiltered if the DB dialect doesn't support @@ (e.g. SQLite in test env)
|
|
fts_requested = q is not None and len(q) >= 2
|
|
if fts_requested:
|
|
fts_stmt = stmt.where(
|
|
func.to_tsvector("english", func.coalesce(Document.extracted_text, "")).op("@@")(
|
|
func.plainto_tsquery("english", q)
|
|
)
|
|
)
|
|
try:
|
|
result = await session.execute(fts_stmt)
|
|
except Exception:
|
|
result = await session.execute(stmt)
|
|
else:
|
|
result = await session.execute(stmt)
|
|
docs_orm = result.scalars().all()
|
|
|
|
shared_result = await session.execute(
|
|
select(Share.document_id).where(Share.owner_id == current_user.id)
|
|
)
|
|
shared_ids = {row[0] for row in shared_result.fetchall()}
|
|
|
|
all_items = []
|
|
for doc in docs_orm:
|
|
from services.storage import _doc_to_dict, _load_topic_names # noqa: PLC0415
|
|
topic_names = await _load_topic_names(session, doc.id)
|
|
d = _doc_to_dict(doc, topic_names)
|
|
d["is_shared"] = doc.id in shared_ids
|
|
all_items.append(d)
|
|
|
|
total = len(all_items)
|
|
start = (page - 1) * per_page
|
|
return {
|
|
"items": all_items[start : start + per_page],
|
|
"total": total,
|
|
"page": page,
|
|
"per_page": per_page,
|
|
}
|
|
|
|
|
|
# ── GET /api/documents/{doc_id} ───────────────────────────────────────────────
|
|
|
|
@router.get("/{doc_id}")
|
|
@account_limiter.limit("100/minute")
|
|
async def get_document(
|
|
request: Request,
|
|
doc_id: str,
|
|
session: AsyncSession = Depends(get_db),
|
|
current_user: User = Depends(get_regular_user),
|
|
):
|
|
"""Return document metadata by ID.
|
|
|
|
D-16: requires authenticated regular user. Asserts ownership — cross-user
|
|
access returns 404 (not 403) to avoid information leakage (T-03-11).
|
|
"""
|
|
request.state.current_user = current_user
|
|
try:
|
|
uid = uuid.UUID(doc_id)
|
|
except ValueError:
|
|
raise HTTPException(404, "Document not found")
|
|
|
|
doc = await session.get(Document, uid)
|
|
if doc is None:
|
|
raise HTTPException(404, "Document not found")
|
|
|
|
is_recipient = False
|
|
if doc.user_id != current_user.id:
|
|
share_result = await session.execute(
|
|
select(Share).where(
|
|
Share.document_id == uid,
|
|
Share.recipient_id == current_user.id,
|
|
)
|
|
)
|
|
if share_result.scalar_one_or_none() is None:
|
|
raise HTTPException(404, "Document not found")
|
|
is_recipient = True
|
|
|
|
meta = await storage.get_metadata(session, doc_id)
|
|
if meta is None:
|
|
raise HTTPException(404, "Document not found")
|
|
# T-04-04-03: recipients get metadata only — extracted_text excluded (consistent with /shares/received)
|
|
if is_recipient:
|
|
meta.pop("extracted_text", None)
|
|
return meta
|
|
|
|
|
|
# ── PATCH /api/documents/{doc_id} ────────────────────────────────────────────
|
|
|
|
@router.patch("/{doc_id}")
|
|
@account_limiter.limit("100/minute")
|
|
async def patch_document(
|
|
request: Request,
|
|
doc_id: str,
|
|
body: DocumentPatch,
|
|
session: AsyncSession = Depends(get_db),
|
|
current_user: User = Depends(get_regular_user),
|
|
):
|
|
"""Update document metadata (filename and/or folder_id).
|
|
|
|
T-05-09-01: get_regular_user dep rejects admins (403) and unauthenticated (401).
|
|
T-05-09-01: ownership check — non-owner gets 404 to avoid leaking document IDs (D-16).
|
|
T-05-09-02: response uses storage.get_metadata() which excludes credentials_enc and
|
|
password_hash via the _doc_to_dict whitelist.
|
|
|
|
At least one field must be provided — empty body returns 422.
|
|
folder_id=null moves the document to the root (no folder).
|
|
"""
|
|
request.state.current_user = current_user
|
|
try:
|
|
uid = uuid.UUID(doc_id)
|
|
except ValueError:
|
|
raise HTTPException(404, "Document not found")
|
|
|
|
doc = await session.get(Document, uid)
|
|
if doc is None or doc.user_id != current_user.id:
|
|
raise HTTPException(404, "Document not found")
|
|
|
|
if not body.model_fields_set:
|
|
raise HTTPException(422, "At least one field (filename, folder_id) must be provided")
|
|
|
|
if "filename" in body.model_fields_set and body.filename is not None:
|
|
doc.filename = body.filename
|
|
|
|
if "folder_id" in body.model_fields_set:
|
|
# folder_id=null → move to root (no folder); folder_id=<uuid> → move to folder
|
|
if body.folder_id is not None:
|
|
target = await session.get(Folder, body.folder_id)
|
|
if target is None or target.user_id != current_user.id:
|
|
raise HTTPException(404, "Folder not found")
|
|
doc.folder_id = body.folder_id
|
|
|
|
await session.commit()
|
|
|
|
meta = await storage.get_metadata(session, doc_id)
|
|
if meta is None:
|
|
raise HTTPException(404, "Document not found")
|
|
return meta
|
|
|
|
|
|
# ── DELETE /api/documents/{doc_id} ───────────────────────────────────────────
|
|
|
|
@router.delete("/{doc_id}")
|
|
@account_limiter.limit("100/minute")
|
|
async def delete_document(
|
|
doc_id: str,
|
|
request: Request,
|
|
remove_only: bool = Query(default=False),
|
|
session: AsyncSession = Depends(get_db),
|
|
current_user: User = Depends(get_regular_user),
|
|
):
|
|
"""Delete a document and decrement quota atomically.
|
|
|
|
For cloud-stored documents:
|
|
- Default path: attempt cloud provider delete first; on failure return
|
|
{success: false, cloud_delete_failed: true} (HTTP 200) so the frontend
|
|
can offer a "Remove from app" fallback (T-06.2-03-02).
|
|
- remove_only=true: skip cloud delete, remove DB row only, skip quota decrement.
|
|
- Cloud docs always use skip_quota=True (never charged MinIO quota, T-06.2-03-01).
|
|
|
|
D-16: requires authenticated regular user. Asserts ownership — cross-user
|
|
delete returns 404 (not 403) to avoid information leakage (T-03-11).
|
|
"""
|
|
request.state.current_user = current_user
|
|
try:
|
|
uid = uuid.UUID(doc_id)
|
|
except ValueError:
|
|
raise HTTPException(404, "Document not found")
|
|
|
|
doc = await session.get(Document, uid)
|
|
if doc is None or doc.user_id != current_user.id:
|
|
raise HTTPException(404, "Document not found")
|
|
|
|
is_cloud = doc.storage_backend != "minio"
|
|
_doc_size = doc.size_bytes
|
|
_doc_id = doc.id
|
|
_ip = get_client_ip(request)
|
|
|
|
if is_cloud and not remove_only:
|
|
try:
|
|
import api.documents as _doc_pkg # late import allows test monkeypatching via api.documents
|
|
_gsb = _doc_pkg.get_storage_backend_for_document
|
|
cloud_backend = await _gsb(doc, current_user, session)
|
|
await cloud_backend.delete_object(doc.object_key)
|
|
except Exception as exc:
|
|
_log.warning("cloud_delete_failed", provider=doc.storage_backend, error=str(exc))
|
|
return JSONResponse(
|
|
status_code=200,
|
|
content={
|
|
"success": False,
|
|
"cloud_delete_failed": True,
|
|
"detail": "Cloud provider delete failed. You can remove from app only.",
|
|
},
|
|
)
|
|
|
|
# auto_commit=False defers the commit so the audit log write below happens
|
|
# in the same transaction — avoids the split-transaction gap (WR-08).
|
|
ok = await storage.delete_document(session, doc_id, skip_quota=is_cloud, auto_commit=False)
|
|
if not ok:
|
|
raise HTTPException(404, "Document not found")
|
|
|
|
# D-13: document deleted event — written in the same transaction as the delete (WR-08).
|
|
await write_audit_log(
|
|
session,
|
|
event_type="document.deleted",
|
|
user_id=current_user.id,
|
|
actor_id=current_user.id,
|
|
resource_id=_doc_id,
|
|
ip_address=_ip,
|
|
metadata_={"size_bytes": _doc_size},
|
|
)
|
|
await session.commit()
|
|
|
|
return {"success": True}
|
|
|
|
|
|
# ── POST /api/documents/{doc_id}/classify ────────────────────────────────────
|
|
|
|
@router.post("/{doc_id}/classify")
|
|
@account_limiter.limit("100/minute")
|
|
async def classify_document(
|
|
request: Request,
|
|
doc_id: str,
|
|
session: AsyncSession = Depends(get_db),
|
|
current_user: User = Depends(get_regular_user),
|
|
):
|
|
"""Re-queue a document for classification via Celery (D-11).
|
|
|
|
Sets doc.status='processing', commits, dispatches extract_and_classify.delay(),
|
|
and returns {'document_id': str, 'status': 'processing'}.
|
|
|
|
D-16: requires authenticated regular user. Asserts ownership — cross-user
|
|
classify returns 404 (not 403) to avoid information leakage (T-03-11).
|
|
T-07-10: ownership enforced here; IDOR returns 404 per STATE.md policy.
|
|
Placed in crud.py per D-08: same ownership-check pattern as get/patch/delete.
|
|
"""
|
|
request.state.current_user = current_user
|
|
try:
|
|
uid = uuid.UUID(doc_id)
|
|
except ValueError:
|
|
raise HTTPException(404, "Document not found")
|
|
|
|
doc = await session.get(Document, uid)
|
|
if doc is None or doc.user_id != current_user.id:
|
|
raise HTTPException(404, "Document not found")
|
|
|
|
doc.status = "processing"
|
|
await session.commit()
|
|
|
|
extract_and_classify.delay(str(doc.id))
|
|
|
|
return {"document_id": str(doc.id), "status": "processing"}
|