Marwan
Restructure + add reverse face search (PimEyes-style)
f5eeb1c
Raw
History Blame Contribute Delete
3.45 kB
"""
Error Level Analysis (ELA) forensics provider.
Re-encodes the image at a known JPEG quality, then compares the
pixel-level differences. Regions that were manipulated (e.g. spliced
in from another image) typically show different ELA values than the
surrounding authentic regions.
Pure OpenCV — no external dependencies, no model downloads.
Algorithm:
1. Encode original to JPEG at quality 90
2. Decode it back
3. Compute absolute difference
4. Mean difference = ELA score (higher = more likely manipulated)
"""
from __future__ import annotations
import cv2
import numpy as np
from config.settings import Settings, settings as _default_settings
from pipeline.feature_extraction import PipelineOutput
from providers.base import BaseProvider, ProviderCapability
class ELAProvider(BaseProvider):
name = "ela"
capability = ProviderCapability.FORENSICS
def __init__(self, settings: Settings | None = None) -> None:
super().__init__(settings=settings or _default_settings)
def is_available(self) -> bool:
return True
def _run(self, pipeline_output: PipelineOutput) -> tuple[dict, dict]:
img: np.ndarray = pipeline_output.image
# Step 1: encode at quality 90
ok, buffer = cv2.imencode(".jpg", img, [cv2.IMWRITE_JPEG_QUALITY, 90])
if not ok:
raise RuntimeError("Could not encode image for ELA")
# Step 2: decode back
reencoded = cv2.imdecode(buffer, cv2.IMREAD_COLOR)
if reencoded is None or reencoded.shape != img.shape:
raise RuntimeError("ELA re-encode shape mismatch")
# Step 3: absolute difference
diff = cv2.absdiff(img, reencoded)
gray_diff = cv2.cvtColor(diff, cv2.COLOR_BGR2GRAY) if diff.ndim == 3 else diff
# Step 4: metrics
mean_diff = float(np.mean(gray_diff))
max_diff = float(np.max(gray_diff))
std_diff = float(np.std(gray_diff))
# Threshold regions with high difference
_, thresholded = cv2.threshold(gray_diff, 15, 255, cv2.THRESH_BINARY)
suspicious_pixels = float(np.count_nonzero(thresholded) / gray_diff.size)
# ELA score 0-1: higher = more likely manipulated
# Heuristic: mean_diff > 5 is suspicious, > 15 is very suspicious
ela_score = min(1.0, mean_diff / 20.0)
manipulation_indicators: list[str] = []
if mean_diff > 10:
manipulation_indicators.append(f"High ELA mean difference ({mean_diff:.2f})")
if suspicious_pixels > 0.05:
manipulation_indicators.append(
f"{suspicious_pixels * 100:.1f}% of pixels show re-encoding artifacts"
)
raw = {
"mean_diff": mean_diff,
"max_diff": max_diff,
"std_diff": std_diff,
"suspicious_pixel_ratio": suspicious_pixels,
"ela_score": ela_score,
}
normalized = {
"integrity_score": None,
"manipulation_indicators": manipulation_indicators,
"ela_score": round(ela_score, 4),
"noise_inconsistency": None,
"details": {
"mean_diff": round(mean_diff, 4),
"max_diff": round(max_diff, 4),
"std_diff": round(std_diff, 4),
"suspicious_pixel_ratio": round(suspicious_pixels, 4),
"quality_level": 90,
},
}
return raw, normalized