DocDoeAI / app /core /path_guard.py
asnannp's picture
Deploy backend cd4237ff: support routes + rate limit + exam_date nullable + upload 413 fix
7c6ffa6
Raw
History Blame Contribute Delete
2.28 kB
"""Path guard — validate and sanitise resource identifiers.
Prevents directory-traversal attacks and internal path leakage by
ensuring document / resource IDs are opaque alphanumeric tokens before
they are passed to DB queries or file-system operations.
Usage::
from app.core.path_guard import safe_document_id
from fastapi import HTTPException, status
doc_id = safe_document_id(raw_id)
if doc_id is None:
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST,
detail="Invalid document identifier.")
document = db.get(Document, doc_id)
"""
from __future__ import annotations
import re
# Allowed characters for opaque resource IDs (e.g. "doc_01abcd…", UUID form)
_SAFE_ID_RE: re.Pattern[str] = re.compile(r"^[a-zA-Z0-9_-]{1,128}$")
# Characters that indicate a path or escape sequence rather than an opaque ID
_PATH_CHARS: frozenset[str] = frozenset("./\\~%\x00")
def safe_document_id(document_id: str | None) -> str | None:
"""Return *document_id* only if it is a safe opaque identifier.
Rejects:
- ``None`` or empty strings
- IDs containing path-traversal characters ( ``.`` ``/`` ``\\`` ``~`` ``%``
or the NUL byte)
- IDs that don't match the alphanumeric + hyphen + underscore pattern
Returns ``None`` on rejection so callers can gate on a single falsy check.
"""
if not document_id:
return None
doc_id = document_id.strip()
if not doc_id:
return None
if _PATH_CHARS & set(doc_id):
return None
if not _SAFE_ID_RE.match(doc_id):
return None
return doc_id
def safe_resource_id(resource_id: str | None, *, max_length: int = 128) -> str | None:
"""Generic variant of :func:`safe_document_id` for arbitrary resource IDs.
Same rules as :func:`safe_document_id` but with configurable max length.
Use for job IDs, source IDs, user IDs, or any opaque token that should
never contain path characters.
"""
if not resource_id:
return None
rid = resource_id.strip()
if not rid:
return None
if _PATH_CHARS & set(rid):
return None
pattern = re.compile(rf"^[a-zA-Z0-9_-]{{1,{max_length}}}$")
if not pattern.match(rid):
return None
return rid