feat: add forensic sample analysis
This commit is contained in:
@@ -0,0 +1,271 @@
|
||||
import json
|
||||
import re
|
||||
from collections import Counter
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, Iterable, List, Optional, Tuple
|
||||
|
||||
from app.models.report import ForensicReport
|
||||
|
||||
|
||||
AUTH_RESULT_PATTERN = re.compile(r"\b(dkim|spf|dmarc)=([a-zA-Z0-9_-]+)", re.IGNORECASE)
|
||||
HEADER_DOMAIN_PATTERN = re.compile(r"\bheader\.d=([^;\s]+)", re.IGNORECASE)
|
||||
MAILFROM_DOMAIN_PATTERN = re.compile(r"\bsmtp\.mailfrom=([^;\s]+)", re.IGNORECASE)
|
||||
PRIORITY_ORDER = {"high": 3, "medium": 2, "low": 1}
|
||||
|
||||
|
||||
def _clean_value(value: Any) -> str:
|
||||
return str(value or "").strip()
|
||||
|
||||
|
||||
def _normalize(value: Any) -> str:
|
||||
return _clean_value(value).lower()
|
||||
|
||||
|
||||
def _feedback_headers(row: ForensicReport) -> Dict[str, Any]:
|
||||
if not row.feedback_headers:
|
||||
return {}
|
||||
try:
|
||||
parsed = json.loads(row.feedback_headers)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
return {}
|
||||
return parsed if isinstance(parsed, dict) else {}
|
||||
|
||||
|
||||
def _parse_authentication_results(value: str) -> Dict[str, str]:
|
||||
results: Dict[str, str] = {}
|
||||
for mechanism, result in AUTH_RESULT_PATTERN.findall(value or ""):
|
||||
results[mechanism.lower()] = result.lower()
|
||||
return results
|
||||
|
||||
|
||||
def _first_match(pattern: re.Pattern[str], value: str) -> str:
|
||||
match = pattern.search(value or "")
|
||||
return match.group(1).lower().strip(".,") if match else ""
|
||||
|
||||
|
||||
def _failure_kind(row: ForensicReport, auth_results: Dict[str, str]) -> str:
|
||||
reported = _normalize(row.auth_failure)
|
||||
if reported in {"dkim", "spf", "dmarc", "both"}:
|
||||
return reported
|
||||
failed = {name for name, result in auth_results.items() if result in {"fail", "softfail"}}
|
||||
if {"dkim", "spf"}.issubset(failed):
|
||||
return "both"
|
||||
for mechanism in ("dmarc", "dkim", "spf"):
|
||||
if mechanism in failed:
|
||||
return mechanism
|
||||
return reported or "unknown"
|
||||
|
||||
|
||||
def _priority(row: ForensicReport, failure_kind: str) -> str:
|
||||
delivery = _normalize(row.delivery_result)
|
||||
if delivery in {"reject", "quarantine"}:
|
||||
return "high"
|
||||
if failure_kind in {"both", "dmarc"}:
|
||||
return "high"
|
||||
if failure_kind in {"dkim", "spf"}:
|
||||
return "medium"
|
||||
return "low"
|
||||
|
||||
|
||||
def _diagnosis(failure_kind: str, auth_results: Dict[str, str], delivery_result: str) -> str:
|
||||
delivery = _normalize(delivery_result)
|
||||
rejected = delivery in {"reject", "quarantine"}
|
||||
suffix = " The receiver enforced the failure." if rejected else ""
|
||||
if failure_kind == "both":
|
||||
return "Both DKIM and SPF failed, so DMARC could not find an aligned pass." + suffix
|
||||
if failure_kind == "dmarc":
|
||||
return "DMARC failed after the receiver evaluated DKIM and SPF alignment." + suffix
|
||||
if failure_kind == "dkim":
|
||||
if auth_results.get("spf") == "pass":
|
||||
return "DKIM failed while SPF passed; focus on DKIM signing and alignment." + suffix
|
||||
return "DKIM failed for the reported message sample." + suffix
|
||||
if failure_kind == "spf":
|
||||
if auth_results.get("dkim") == "pass":
|
||||
return (
|
||||
"SPF failed while DKIM passed; focus on SPF authorization and alignment." + suffix
|
||||
)
|
||||
return "SPF failed for the reported message sample." + suffix
|
||||
return "The receiver reported an authentication failure, but did not include a clear mechanism."
|
||||
|
||||
|
||||
def _recommendations(
|
||||
failure_kind: str,
|
||||
auth_results: Dict[str, str],
|
||||
source_ip: str,
|
||||
reported_domain: str,
|
||||
) -> List[str]:
|
||||
actions: List[str] = []
|
||||
if failure_kind in {"dkim", "both", "dmarc"}:
|
||||
actions.append(
|
||||
"Confirm the sending system signs mail with a DKIM domain aligned to the visible From domain."
|
||||
)
|
||||
actions.append(
|
||||
"Check recent DKIM key, selector, and canonicalization changes for this sender."
|
||||
)
|
||||
if failure_kind in {"spf", "both", "dmarc"}:
|
||||
actions.append(
|
||||
"Verify the source IP or provider include is authorized in the domain SPF record."
|
||||
)
|
||||
actions.append(
|
||||
"Review forwarding paths, because forwarding commonly breaks SPF while preserving DKIM."
|
||||
)
|
||||
if auth_results.get("spf") == "pass" and failure_kind == "dkim":
|
||||
actions.append(
|
||||
"If SPF is aligned and passing, this may be a DKIM-only repair rather than a sender authorization issue."
|
||||
)
|
||||
if auth_results.get("dkim") == "pass" and failure_kind == "spf":
|
||||
actions.append(
|
||||
"If DKIM is aligned and passing, treat SPF repair as lower risk before changing DMARC policy."
|
||||
)
|
||||
if source_ip:
|
||||
actions.append(
|
||||
f"Compare {source_ip} with known mail sources for {reported_domain or 'this domain'}."
|
||||
)
|
||||
actions.append(
|
||||
"Keep using redacted forensic metadata; do not import or retain message bodies for this investigation."
|
||||
)
|
||||
return actions
|
||||
|
||||
|
||||
def _signals(
|
||||
row: ForensicReport,
|
||||
feedback_headers: Dict[str, Any],
|
||||
auth_results: Dict[str, str],
|
||||
header_domain: str,
|
||||
mailfrom_domain: str,
|
||||
) -> List[str]:
|
||||
signals = []
|
||||
if row.source_ip:
|
||||
signals.append(f"Source IP: {row.source_ip}")
|
||||
if row.reported_domain:
|
||||
signals.append(f"Reported domain: {row.reported_domain}")
|
||||
if row.auth_failure:
|
||||
signals.append(f"Failure: {row.auth_failure}")
|
||||
if row.delivery_result:
|
||||
signals.append(f"Delivery result: {row.delivery_result}")
|
||||
if header_domain:
|
||||
signals.append(f"DKIM header domain: {header_domain}")
|
||||
if mailfrom_domain:
|
||||
signals.append(f"SPF mail-from domain: {mailfrom_domain}")
|
||||
identity_alignment = _clean_value(feedback_headers.get("identity_alignment"))
|
||||
if identity_alignment:
|
||||
signals.append(f"Identity alignment: {identity_alignment}")
|
||||
for mechanism, result in sorted(auth_results.items()):
|
||||
signals.append(f"{mechanism.upper()} result: {result}")
|
||||
return signals
|
||||
|
||||
|
||||
def analyze_forensic_report(row: ForensicReport) -> Dict[str, Any]:
|
||||
"""Build a privacy-preserving operator analysis for one forensic sample."""
|
||||
feedback_headers = _feedback_headers(row)
|
||||
auth_results = _parse_authentication_results(row.authentication_results or "")
|
||||
header_domain = _first_match(
|
||||
HEADER_DOMAIN_PATTERN, row.authentication_results or ""
|
||||
) or _normalize(feedback_headers.get("dkim_domain"))
|
||||
mailfrom_domain = _first_match(MAILFROM_DOMAIN_PATTERN, row.authentication_results or "")
|
||||
failure_kind = _failure_kind(row, auth_results)
|
||||
priority = _priority(row, failure_kind)
|
||||
reported_domain = _clean_value(row.reported_domain or (row.domain.name if row.domain else ""))
|
||||
source_ip = _clean_value(row.source_ip)
|
||||
|
||||
return {
|
||||
"id": row.id,
|
||||
"report_id": row.report_id,
|
||||
"domain": reported_domain,
|
||||
"source_ip": source_ip,
|
||||
"auth_failure": failure_kind,
|
||||
"delivery_result": _clean_value(row.delivery_result),
|
||||
"priority": priority,
|
||||
"diagnosis": _diagnosis(failure_kind, auth_results, row.delivery_result or ""),
|
||||
"recommendations": _recommendations(
|
||||
failure_kind,
|
||||
auth_results,
|
||||
source_ip,
|
||||
reported_domain,
|
||||
),
|
||||
"signals": _signals(row, feedback_headers, auth_results, header_domain, mailfrom_domain),
|
||||
"authentication_results": auth_results,
|
||||
"dkim_domain": header_domain,
|
||||
"mail_from_domain": mailfrom_domain,
|
||||
"privacy_note": "Analysis uses redacted headers and metadata only; message bodies are not stored.",
|
||||
}
|
||||
|
||||
|
||||
def _group_key(row: ForensicReport) -> Tuple[str, str, str, str]:
|
||||
return (
|
||||
_clean_value(row.reported_domain or (row.domain.name if row.domain else "")) or "unknown",
|
||||
_clean_value(row.source_ip) or "unknown",
|
||||
_normalize(row.auth_failure) or "unknown",
|
||||
_normalize(row.delivery_result) or "unknown",
|
||||
)
|
||||
|
||||
|
||||
def _latest(left: Optional[datetime], right: Optional[datetime]) -> Optional[datetime]:
|
||||
if left is None:
|
||||
return right
|
||||
if right is None:
|
||||
return left
|
||||
return max(left, right)
|
||||
|
||||
|
||||
def summarize_forensic_samples(rows: Iterable[ForensicReport]) -> Dict[str, Any]:
|
||||
"""Summarize forensic samples into investigation groups and top examples."""
|
||||
reports = list(rows)
|
||||
analyses = [analyze_forensic_report(row) for row in reports]
|
||||
priority_counts = Counter(item["priority"] for item in analyses)
|
||||
failure_counts = Counter(item["auth_failure"] for item in analyses)
|
||||
result_counts = Counter(_normalize(row.delivery_result) or "unknown" for row in reports)
|
||||
grouped: Dict[Tuple[str, str, str, str], Dict[str, Any]] = {}
|
||||
|
||||
for row, analysis in zip(reports, analyses):
|
||||
key = _group_key(row)
|
||||
group = grouped.setdefault(
|
||||
key,
|
||||
{
|
||||
"key": "|".join(key),
|
||||
"domain": key[0],
|
||||
"source_ip": key[1],
|
||||
"auth_failure": analysis["auth_failure"],
|
||||
"delivery_result": key[3],
|
||||
"count": 0,
|
||||
"priority": analysis["priority"],
|
||||
"latest_arrival": None,
|
||||
"diagnosis": analysis["diagnosis"],
|
||||
"recommendations": analysis["recommendations"][:3],
|
||||
},
|
||||
)
|
||||
group["count"] += 1
|
||||
group["latest_arrival"] = _latest(
|
||||
group["latest_arrival"], row.arrival_date or row.processed_at
|
||||
)
|
||||
if PRIORITY_ORDER[analysis["priority"]] > PRIORITY_ORDER[group["priority"]]:
|
||||
group["priority"] = analysis["priority"]
|
||||
group["diagnosis"] = analysis["diagnosis"]
|
||||
group["recommendations"] = analysis["recommendations"][:3]
|
||||
|
||||
groups = sorted(
|
||||
grouped.values(),
|
||||
key=lambda item: (
|
||||
PRIORITY_ORDER[item["priority"]],
|
||||
item["count"],
|
||||
item["latest_arrival"] or datetime.min,
|
||||
),
|
||||
reverse=True,
|
||||
)
|
||||
for group in groups:
|
||||
if group["latest_arrival"] is not None:
|
||||
group["latest_arrival"] = group["latest_arrival"].isoformat()
|
||||
|
||||
samples = sorted(
|
||||
analyses,
|
||||
key=lambda item: (PRIORITY_ORDER[item["priority"]], item["id"] or 0),
|
||||
reverse=True,
|
||||
)
|
||||
return {
|
||||
"total": len(reports),
|
||||
"priority_counts": dict(priority_counts),
|
||||
"failure_counts": dict(failure_counts),
|
||||
"result_counts": dict(result_counts),
|
||||
"groups": groups,
|
||||
"samples": samples,
|
||||
}
|
||||
Reference in New Issue
Block a user