face-intel / cores /metadata /forensic.py
Marwan
Restructure + add reverse face search (PimEyes-style)
f5eeb1c
Raw
History Blame Contribute Delete
8.54 kB
"""
Forensic metadata analysis — ICC profile, thumbnail extraction,
editing history, camera fingerprint, compression analysis,
timestamp normalization.
Extends cores.metadata.extractor with deeper forensic intelligence.
Pure Pillow + NumPy — no external deps.
"""
from __future__ import annotations
import hashlib
import io
from datetime import datetime
from typing import Optional
import numpy as np
from PIL import Image, ExifTags
def extract_icc_profile(image_bytes: bytes) -> Optional[dict]:
"""Extract ICC profile metadata from an image."""
if not image_bytes:
return None
try:
img = Image.open(io.BytesIO(image_bytes))
icc = img.info.get("icc_profile")
if not icc:
return None
# Parse minimal ICC profile info (header only — full parsing
# requires LittleCMS or similar; we extract size + a hash)
return {
"present": True,
"size_bytes": len(icc),
"hash": hashlib.sha256(icc).hexdigest()[:16],
}
except Exception:
return None
def extract_thumbnail(image_bytes: bytes) -> Optional[dict]:
"""Extract embedded EXIF thumbnail if present."""
if not image_bytes:
return None
try:
img = Image.open(io.BytesIO(image_bytes))
exif_info = img.getexif()
if not exif_info:
return None
# Thumbnail is stored at tag 0x0201 (JPEGThumbnail) / 0x0202 (ThumbnailLength)
thumb_data = None
for tag_id, value in exif_info.items():
tag_name = ExifTags.TAGS.get(tag_id, f"Tag_{tag_id}")
if tag_name == "JPEGThumbnail":
thumb_data = value
break
if not thumb_data:
# Try ImageMagick-style embedded thumbnail via img.info
thumb_data = img.info.get("thumbnail")
if not thumb_data:
return None
# Decode the thumbnail
try:
thumb_img = Image.open(io.BytesIO(thumb_data))
return {
"present": True,
"width": thumb_img.width,
"height": thumb_img.height,
"format": thumb_img.format,
"size_bytes": len(thumb_data),
}
except Exception:
return {"present": True, "size_bytes": len(thumb_data)}
except Exception:
return None
def extract_embedded_preview(image_bytes: bytes) -> bool:
"""Check if the image has an embedded preview (e.g. PSD, RAW)."""
if not image_bytes:
return False
try:
img = Image.open(io.BytesIO(image_bytes))
# Some formats embed a preview
return bool(img.info.get("preview") or img.info.get("Preview"))
except Exception:
return False
def extract_editing_history(exif: dict) -> list[str]:
"""Extract editing history from EXIF tags.
Looks for:
- Software tag
- XMP history (if present)
- Photoshop tags
"""
history: list[str] = []
if not exif:
return history
# Software tag
software = exif.get("Software")
if software:
history.append(f"Software: {software}")
# ProcessingSoftware
proc_sw = exif.get("ProcessingSoftware")
if proc_sw:
history.append(f"Processing: {proc_sw}")
# Artist
artist = exif.get("Artist")
if artist:
history.append(f"Artist: {artist}")
# Copyright
copyright_tag = exif.get("Copyright")
if copyright_tag:
history.append(f"Copyright: {copyright_tag}")
# ImageDescription
desc = exif.get("ImageDescription")
if desc:
history.append(f"Description: {desc}")
# UserComment
comment = exif.get("UserComment")
if comment:
history.append(f"Comment: {str(comment)[:200]}")
return history
def camera_fingerprint(image_bytes: bytes, img_shape: tuple[int, int]) -> Optional[str]:
"""Compute a camera fingerprint via PRNU (Photo Response Non-Uniformity).
This is a simplified version: we compute the mean noise pattern from
the high-frequency residual of the image. Cameras have unique sensor
noise patterns that can be used for source identification.
Returns a hash of the noise pattern — NOT the full PRNU (which would
require a reference pattern from the same camera).
"""
if not image_bytes:
return None
try:
import cv2
arr = np.frombuffer(image_bytes, np.uint8)
img = cv2.imdecode(arr, cv2.IMREAD_COLOR)
if img is None:
return None
# Denoise then subtract to get the noise residual
import cv2
denoised = cv2.fastNlMeansDenoisingColored(img, None, 10, 10, 7, 21)
noise = img.astype(np.float32) - denoised.astype(np.float32)
# Downsample the noise pattern for a compact fingerprint
h, w = noise.shape[:2]
target_h, target_w = 64, 64
small = cv2.resize(noise, (target_w, target_h))
# Compute mean per channel, then hash
mean_pattern = small.mean(axis=2) # grayscale
# Quantize to 0-255
mean_pattern = np.clip(mean_pattern + 128, 0, 255).astype(np.uint8)
return hashlib.sha256(mean_pattern.tobytes()).hexdigest()[:32]
except Exception:
return None
def analyze_compression(image_bytes: bytes) -> Optional[dict]:
"""Analyze compression characteristics of an image.
Returns:
{
"format": str,
"size_bytes": int,
"quality_estimate": float | None, # for JPEG
"compression_ratio": float,
}
"""
if not image_bytes:
return None
try:
img = Image.open(io.BytesIO(image_bytes))
fmt = img.format or "UNKNOWN"
size_bytes = len(image_bytes)
width, height = img.size
pixels = width * height
# Raw uncompressed size (3 bytes per pixel for RGB)
raw_size = pixels * 3
ratio = raw_size / size_bytes if size_bytes > 0 else 0.0
# JPEG quality estimate — re-encode at multiple qualities and find the
# closest match (simplified heuristic)
quality_estimate = None
if fmt in ("JPEG", "JPG"):
quality_estimate = _estimate_jpeg_quality(img, size_bytes)
return {
"format": fmt,
"size_bytes": size_bytes,
"quality_estimate": quality_estimate,
"compression_ratio": round(ratio, 2),
"dimensions": {"width": width, "height": height},
}
except Exception:
return None
def _estimate_jpeg_quality(img: Image.Image, original_size: int) -> Optional[float]:
"""Estimate JPEG quality by re-encoding at different qualities."""
try:
best_q = 75
best_diff = float("inf")
for q in [50, 60, 70, 75, 80, 85, 90, 95]:
buf = io.BytesIO()
img.save(buf, format="JPEG", quality=q)
size = buf.tell()
diff = abs(size - original_size)
if diff < best_diff:
best_diff = diff
best_q = float(q)
return best_q
except Exception:
return None
def normalize_timestamp(capture_time: str) -> Optional[dict]:
"""Normalize EXIF timestamp to ISO 8601 + estimate timezone.
EXIF timestamps are typically in format "YYYY:MM:DD HH:MM:SS" with
no timezone. We normalize to ISO 8601 and note that the timezone
is unknown (the camera clock may not have been set to local time).
"""
if not capture_time:
return None
try:
# EXIF format: "2023:01:15 14:30:45"
s = str(capture_time).strip()
# Try common formats
for fmt in ("%Y:%m:%d %H:%M:%S", "%Y-%m-%d %H:%M:%S", "%Y:%m:%d %H:%M"):
try:
dt = datetime.strptime(s, fmt)
return {
"iso": dt.isoformat() + "Z", # assume UTC since we don't know
"timezone": "unknown",
"original": s,
}
except ValueError:
continue
return {"iso": None, "timezone": "unknown", "original": s}
except Exception:
return None
def extract_lens_info(exif: dict) -> Optional[str]:
"""Extract lens model from EXIF tags."""
if not exif:
return None
# Try various lens tags
for key in ("LensModel", "LensSpecification", "LensSerialNumber", "Lens"):
val = exif.get(key)
if val:
return str(val)
return None