Spaces:
Running
Running
adidtiya commited on
Commit ·
91491b1
1
Parent(s): 0a9dbb2
Deploy Deepfake Shield Backend v2.0
Browse files- .gitattributes +0 -35
- Dockerfile +29 -0
- README.md +0 -10
- app.py +220 -0
- audio_detector.py +203 -0
- detector.py +238 -0
- model_loader.py +185 -0
- requirements.txt +32 -0
- utils/__init__.py +3 -0
- utils/audio_processor.py +86 -0
- utils/frame_processor.py +124 -0
.gitattributes
DELETED
|
@@ -1,35 +0,0 @@
|
|
| 1 |
-
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
-
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
-
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
-
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
-
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
-
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
-
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
-
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
-
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
-
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
-
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
-
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
-
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
-
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
-
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
-
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
-
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
-
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
-
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
-
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
-
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
-
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
-
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
-
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
-
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
-
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
-
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
-
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
-
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
-
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
-
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
-
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
-
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
-
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
-
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
Dockerfile
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
FROM python:3.10-slim
|
| 2 |
+
|
| 3 |
+
WORKDIR /app
|
| 4 |
+
|
| 5 |
+
# Install system dependencies untuk OpenCV dan MediaPipe
|
| 6 |
+
RUN apt-get update && apt-get install -y \
|
| 7 |
+
libglib2.0-0 \
|
| 8 |
+
libsm6 \
|
| 9 |
+
libxext6 \
|
| 10 |
+
libxrender-dev \
|
| 11 |
+
libgomp1 \
|
| 12 |
+
ffmpeg \
|
| 13 |
+
&& rm -rf /var/lib/apt/lists/*
|
| 14 |
+
|
| 15 |
+
# Copy requirements dan install
|
| 16 |
+
COPY requirements.txt .
|
| 17 |
+
RUN pip install --no-cache-dir -r requirements.txt
|
| 18 |
+
|
| 19 |
+
# Copy source code
|
| 20 |
+
COPY . .
|
| 21 |
+
|
| 22 |
+
# Buat direktori cache untuk HuggingFace models
|
| 23 |
+
RUN mkdir -p models/hf_cache models/torch_cache
|
| 24 |
+
|
| 25 |
+
# Port yang digunakan HF Spaces
|
| 26 |
+
EXPOSE 7860
|
| 27 |
+
|
| 28 |
+
# Jalankan server
|
| 29 |
+
CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "7860"]
|
README.md
DELETED
|
@@ -1,10 +0,0 @@
|
|
| 1 |
-
---
|
| 2 |
-
title: Deepfake Shield Api
|
| 3 |
-
emoji: 📚
|
| 4 |
-
colorFrom: indigo
|
| 5 |
-
colorTo: yellow
|
| 6 |
-
sdk: docker
|
| 7 |
-
pinned: false
|
| 8 |
-
---
|
| 9 |
-
|
| 10 |
-
Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
app.py
ADDED
|
@@ -0,0 +1,220 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
app.py - Entry point untuk Hugging Face Spaces
|
| 3 |
+
===============================================
|
| 4 |
+
Versi backend yang dioptimalkan untuk HF Spaces:
|
| 5 |
+
- Tanpa WebSocket (diganti HTTP polling dari frontend)
|
| 6 |
+
- CORS terbuka untuk domain Vercel
|
| 7 |
+
- Graceful fallback jika model gagal dimuat
|
| 8 |
+
|
| 9 |
+
Deploy: Upload folder backend/ ke HF Spaces sebagai Gradio/Docker App.
|
| 10 |
+
"""
|
| 11 |
+
|
| 12 |
+
import cv2
|
| 13 |
+
import json
|
| 14 |
+
import base64
|
| 15 |
+
import logging
|
| 16 |
+
import asyncio
|
| 17 |
+
import numpy as np
|
| 18 |
+
import os
|
| 19 |
+
from contextlib import asynccontextmanager
|
| 20 |
+
from typing import Optional
|
| 21 |
+
|
| 22 |
+
from fastapi import FastAPI, File, UploadFile, HTTPException
|
| 23 |
+
from fastapi.middleware.cors import CORSMiddleware
|
| 24 |
+
from fastapi.responses import JSONResponse
|
| 25 |
+
import torch
|
| 26 |
+
|
| 27 |
+
# Import modul internal
|
| 28 |
+
from model_loader import model_loader
|
| 29 |
+
from detector import FaceDeepfakeDetector
|
| 30 |
+
from audio_detector import AudioDeepfakeDetector
|
| 31 |
+
|
| 32 |
+
logging.basicConfig(
|
| 33 |
+
level=logging.INFO,
|
| 34 |
+
format="%(asctime)s [%(levelname)s] %(name)s: %(message)s"
|
| 35 |
+
)
|
| 36 |
+
logger = logging.getLogger(__name__)
|
| 37 |
+
|
| 38 |
+
face_detector: Optional[FaceDeepfakeDetector] = None
|
| 39 |
+
audio_detector: Optional[AudioDeepfakeDetector] = None
|
| 40 |
+
startup_error: Optional[str] = None
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
@asynccontextmanager
|
| 44 |
+
async def lifespan(app: FastAPI):
|
| 45 |
+
global face_detector, audio_detector, startup_error
|
| 46 |
+
|
| 47 |
+
logger.info("🚀 Deepfake Shield Backend (HF Spaces) starting...")
|
| 48 |
+
|
| 49 |
+
try:
|
| 50 |
+
model_loader.initialize()
|
| 51 |
+
face_detector = FaceDeepfakeDetector()
|
| 52 |
+
audio_detector = AudioDeepfakeDetector()
|
| 53 |
+
logger.info("✅ Semua model berhasil dimuat!")
|
| 54 |
+
except Exception as e:
|
| 55 |
+
startup_error = str(e)
|
| 56 |
+
logger.error(f"❌ Gagal memuat model: {e}")
|
| 57 |
+
|
| 58 |
+
yield
|
| 59 |
+
|
| 60 |
+
if torch.cuda.is_available():
|
| 61 |
+
torch.cuda.empty_cache()
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
app = FastAPI(
|
| 65 |
+
title="Deepfake Shield API",
|
| 66 |
+
description="Real-time deepfake detection — hosted on Hugging Face Spaces",
|
| 67 |
+
version="2.0.0",
|
| 68 |
+
lifespan=lifespan
|
| 69 |
+
)
|
| 70 |
+
|
| 71 |
+
# CORS: Izinkan Vercel + localhost
|
| 72 |
+
app.add_middleware(
|
| 73 |
+
CORSMiddleware,
|
| 74 |
+
allow_origins=["*"], # Vercel generates random preview URLs, so allow all
|
| 75 |
+
allow_credentials=False,
|
| 76 |
+
allow_methods=["*"],
|
| 77 |
+
allow_headers=["*"],
|
| 78 |
+
)
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
# ─── Health Check ───────────────────────────────────────────────────────────
|
| 82 |
+
|
| 83 |
+
@app.get("/api/health")
|
| 84 |
+
async def health_check():
|
| 85 |
+
if startup_error:
|
| 86 |
+
return JSONResponse({
|
| 87 |
+
"status" : "error",
|
| 88 |
+
"error" : startup_error,
|
| 89 |
+
"models_loaded" : False,
|
| 90 |
+
}, status_code=500)
|
| 91 |
+
|
| 92 |
+
return JSONResponse({
|
| 93 |
+
"status" : "online",
|
| 94 |
+
"version" : "2.0.0",
|
| 95 |
+
"models_loaded" : face_detector is not None,
|
| 96 |
+
"cuda_available": torch.cuda.is_available(),
|
| 97 |
+
"device" : str(model_loader.device),
|
| 98 |
+
})
|
| 99 |
+
|
| 100 |
+
|
| 101 |
+
# ─── Analisis Gambar / Frame ─────────────────────────────────────────────────
|
| 102 |
+
|
| 103 |
+
@app.post("/api/analyze/image")
|
| 104 |
+
async def analyze_image(file: UploadFile = File(...)):
|
| 105 |
+
"""
|
| 106 |
+
Analisis gambar untuk deteksi deepfake.
|
| 107 |
+
Frontend mengirim frame kamera atau foto upload ke endpoint ini.
|
| 108 |
+
"""
|
| 109 |
+
import time
|
| 110 |
+
|
| 111 |
+
if face_detector is None:
|
| 112 |
+
raise HTTPException(
|
| 113 |
+
status_code=503,
|
| 114 |
+
detail=startup_error or "Face detector belum siap. Coba lagi dalam beberapa detik."
|
| 115 |
+
)
|
| 116 |
+
|
| 117 |
+
image_bytes = await file.read()
|
| 118 |
+
if len(image_bytes) == 0:
|
| 119 |
+
raise HTTPException(status_code=400, detail="File gambar kosong")
|
| 120 |
+
|
| 121 |
+
frame_array = np.frombuffer(image_bytes, dtype=np.uint8)
|
| 122 |
+
frame_bgr = cv2.imdecode(frame_array, cv2.IMREAD_COLOR)
|
| 123 |
+
|
| 124 |
+
if frame_bgr is None:
|
| 125 |
+
raise HTTPException(status_code=400, detail="Format gambar tidak valid (gunakan JPG/PNG/WEBP)")
|
| 126 |
+
|
| 127 |
+
start = time.perf_counter()
|
| 128 |
+
|
| 129 |
+
result = await asyncio.get_event_loop().run_in_executor(
|
| 130 |
+
None, face_detector.process_frame, frame_bgr
|
| 131 |
+
)
|
| 132 |
+
|
| 133 |
+
result["processing_time_ms"] = round((time.perf_counter() - start) * 1000, 2)
|
| 134 |
+
result["filename"] = file.filename or "frame.jpg"
|
| 135 |
+
|
| 136 |
+
return JSONResponse(result)
|
| 137 |
+
|
| 138 |
+
|
| 139 |
+
@app.post("/api/analyze/frame")
|
| 140 |
+
async def analyze_frame_base64(request: dict):
|
| 141 |
+
"""
|
| 142 |
+
Analisis frame kamera yang dikirim sebagai base64 JSON.
|
| 143 |
+
Digunakan sebagai pengganti WebSocket untuk polling dari frontend.
|
| 144 |
+
|
| 145 |
+
Body: { "frame": "data:image/jpeg;base64,/9j/..." }
|
| 146 |
+
"""
|
| 147 |
+
import time
|
| 148 |
+
|
| 149 |
+
if face_detector is None:
|
| 150 |
+
raise HTTPException(status_code=503, detail="Detektor belum siap")
|
| 151 |
+
|
| 152 |
+
frame_data_url = request.get("frame", "")
|
| 153 |
+
if not frame_data_url:
|
| 154 |
+
raise HTTPException(status_code=400, detail="Key 'frame' tidak ditemukan")
|
| 155 |
+
|
| 156 |
+
# Hapus prefix data URL
|
| 157 |
+
if "," in frame_data_url:
|
| 158 |
+
frame_b64 = frame_data_url.split(",", 1)[1]
|
| 159 |
+
else:
|
| 160 |
+
frame_b64 = frame_data_url
|
| 161 |
+
|
| 162 |
+
try:
|
| 163 |
+
frame_bytes = base64.b64decode(frame_b64)
|
| 164 |
+
frame_array = np.frombuffer(frame_bytes, dtype=np.uint8)
|
| 165 |
+
frame_bgr = cv2.imdecode(frame_array, cv2.IMREAD_COLOR)
|
| 166 |
+
except Exception:
|
| 167 |
+
raise HTTPException(status_code=400, detail="Gagal decode frame base64")
|
| 168 |
+
|
| 169 |
+
if frame_bgr is None:
|
| 170 |
+
raise HTTPException(status_code=400, detail="Gagal membaca gambar dari frame")
|
| 171 |
+
|
| 172 |
+
start = time.perf_counter()
|
| 173 |
+
result = await asyncio.get_event_loop().run_in_executor(
|
| 174 |
+
None, face_detector.process_frame, frame_bgr
|
| 175 |
+
)
|
| 176 |
+
result["processing_time_ms"] = round((time.perf_counter() - start) * 1000, 2)
|
| 177 |
+
|
| 178 |
+
return JSONResponse(result)
|
| 179 |
+
|
| 180 |
+
|
| 181 |
+
# ─── Audio Analysis ──────────────────────────────────────────────────────────
|
| 182 |
+
|
| 183 |
+
@app.post("/api/analyze/audio")
|
| 184 |
+
async def analyze_audio(file: UploadFile = File(...)):
|
| 185 |
+
if audio_detector is None:
|
| 186 |
+
raise HTTPException(status_code=503, detail="Audio detector belum siap")
|
| 187 |
+
|
| 188 |
+
audio_bytes = await file.read()
|
| 189 |
+
if len(audio_bytes) == 0:
|
| 190 |
+
raise HTTPException(status_code=400, detail="File audio kosong")
|
| 191 |
+
|
| 192 |
+
result = audio_detector.analyze_bytes(audio_bytes)
|
| 193 |
+
return JSONResponse(result)
|
| 194 |
+
|
| 195 |
+
|
| 196 |
+
# ─── Reset Buffer ─────────────────────────────────────────────────────────────
|
| 197 |
+
|
| 198 |
+
@app.get("/api/reset")
|
| 199 |
+
async def reset():
|
| 200 |
+
if face_detector:
|
| 201 |
+
face_detector.reset_temporal_buffer()
|
| 202 |
+
return {"status": "reset"}
|
| 203 |
+
|
| 204 |
+
|
| 205 |
+
# ─── Root ─────────────────────────────────────────────────────────────────────
|
| 206 |
+
|
| 207 |
+
@app.get("/")
|
| 208 |
+
async def root():
|
| 209 |
+
return {
|
| 210 |
+
"name" : "Deepfake Shield API",
|
| 211 |
+
"status" : "online",
|
| 212 |
+
"docs" : "/docs",
|
| 213 |
+
"health" : "/api/health",
|
| 214 |
+
}
|
| 215 |
+
|
| 216 |
+
|
| 217 |
+
if __name__ == "__main__":
|
| 218 |
+
import uvicorn
|
| 219 |
+
port = int(os.environ.get("PORT", 7860)) # HF Spaces default port
|
| 220 |
+
uvicorn.run("app:app", host="0.0.0.0", port=port, log_level="info")
|
audio_detector.py
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
audio_detector.py
|
| 3 |
+
=================
|
| 4 |
+
Modul deteksi deepfake audio (voice forgery detection).
|
| 5 |
+
Pipeline:
|
| 6 |
+
1. Terima chunk audio PCM (numpy array)
|
| 7 |
+
2. Librosa → Ekstrak mel-spectrogram (representasi visual suara)
|
| 8 |
+
3. MobileNetV3 → Klasifikasikan REAL vs FAKE voice
|
| 9 |
+
|
| 10 |
+
Mel-spectrogram mengubah sinyal audio menjadi "gambar" frekuensi
|
| 11 |
+
yang bisa diklasifikasikan oleh CNN, sama seperti klasifikasi gambar biasa.
|
| 12 |
+
"""
|
| 13 |
+
|
| 14 |
+
import torch
|
| 15 |
+
import numpy as np
|
| 16 |
+
import librosa
|
| 17 |
+
import logging
|
| 18 |
+
from PIL import Image
|
| 19 |
+
import torchvision.transforms as transforms
|
| 20 |
+
|
| 21 |
+
from model_loader import model_loader, DEVICE
|
| 22 |
+
|
| 23 |
+
logger = logging.getLogger(__name__)
|
| 24 |
+
|
| 25 |
+
# --- Konfigurasi Mel-Spectrogram ---
|
| 26 |
+
SAMPLE_RATE = 16000 # Sample rate audio (Hz) - standar untuk speech models
|
| 27 |
+
N_MELS = 128 # Jumlah filter mel bank (resolusi frekuensi)
|
| 28 |
+
HOP_LENGTH = 512 # Jumlah sample antara dua kolom spectrogram
|
| 29 |
+
N_FFT = 2048 # Ukuran window FFT
|
| 30 |
+
DURATION = 2.0 # Durasi chunk audio yang dianalisis (detik)
|
| 31 |
+
MAX_SAMPLES = int(SAMPLE_RATE * DURATION) # 32000 samples per chunk
|
| 32 |
+
|
| 33 |
+
# --- Transformasi untuk input MobileNetV3 ---
|
| 34 |
+
SPECTROGRAM_TRANSFORM = transforms.Compose([
|
| 35 |
+
transforms.Resize((224, 224)), # Resize spectrogram ke input size MobileNetV3
|
| 36 |
+
transforms.ToTensor(), # Konversi ke tensor [1, H, W] (grayscale)
|
| 37 |
+
transforms.Normalize(
|
| 38 |
+
mean=[0.5], # Normalisasi single channel
|
| 39 |
+
std=[0.5]
|
| 40 |
+
),
|
| 41 |
+
])
|
| 42 |
+
|
| 43 |
+
|
| 44 |
+
class AudioDeepfakeDetector:
|
| 45 |
+
"""
|
| 46 |
+
Detektor deepfake audio menggunakan Mel-Spectrogram + MobileNetV3.
|
| 47 |
+
|
| 48 |
+
Cara kerja:
|
| 49 |
+
- Audio chunk (PCM float32) → Mel-Spectrogram (2D matrix)
|
| 50 |
+
- Spectrogram diperlakukan sebagai gambar grayscale
|
| 51 |
+
- MobileNetV3 mengklasifikasikan gambar tersebut sebagai REAL/FAKE
|
| 52 |
+
"""
|
| 53 |
+
|
| 54 |
+
def __init__(self):
|
| 55 |
+
logger.info("✅ AudioDeepfakeDetector siap digunakan")
|
| 56 |
+
|
| 57 |
+
def analyze_chunk(self, audio_data: np.ndarray, sample_rate: int = SAMPLE_RATE) -> dict:
|
| 58 |
+
"""
|
| 59 |
+
Analisis chunk audio dan kembalikan skor keaslian.
|
| 60 |
+
|
| 61 |
+
Args:
|
| 62 |
+
audio_data : numpy array PCM float32, shape [N_samples]
|
| 63 |
+
sample_rate: Sample rate audio (default 16000 Hz)
|
| 64 |
+
|
| 65 |
+
Returns:
|
| 66 |
+
dict dengan keys:
|
| 67 |
+
- 'authenticity_score': float 0.0 (FAKE) - 1.0 (REAL)
|
| 68 |
+
- 'label' : str 'REAL' / 'FAKE' / 'UNCERTAIN'
|
| 69 |
+
- 'confidence' : float
|
| 70 |
+
- 'spectrogram_shape' : tuple, dimensi spectrogram yang dihasilkan
|
| 71 |
+
"""
|
| 72 |
+
# Resample jika sample rate berbeda dari yang diharapkan
|
| 73 |
+
if sample_rate != SAMPLE_RATE:
|
| 74 |
+
audio_data = librosa.resample(
|
| 75 |
+
audio_data,
|
| 76 |
+
orig_sr=sample_rate,
|
| 77 |
+
target_sr=SAMPLE_RATE
|
| 78 |
+
)
|
| 79 |
+
|
| 80 |
+
# Pastikan panjang audio sesuai (padding/truncation)
|
| 81 |
+
audio_data = self._normalize_length(audio_data)
|
| 82 |
+
|
| 83 |
+
# Ekstrak mel-spectrogram dari audio
|
| 84 |
+
spectrogram = self._extract_mel_spectrogram(audio_data)
|
| 85 |
+
|
| 86 |
+
# Klasifikasikan spectrogram menggunakan MobileNetV3
|
| 87 |
+
score = self._classify_spectrogram(spectrogram)
|
| 88 |
+
|
| 89 |
+
return {
|
| 90 |
+
"authenticity_score": round(score, 4),
|
| 91 |
+
"label" : self._score_to_label(score),
|
| 92 |
+
"confidence" : round(max(score, 1 - score), 4),
|
| 93 |
+
"spectrogram_shape" : spectrogram.shape,
|
| 94 |
+
}
|
| 95 |
+
|
| 96 |
+
def _normalize_length(self, audio: np.ndarray) -> np.ndarray:
|
| 97 |
+
"""
|
| 98 |
+
Normalisasi panjang audio chunk ke durasi tetap (2 detik).
|
| 99 |
+
- Jika terlalu panjang: truncate
|
| 100 |
+
- Jika terlalu pendek: padding dengan nol (silence)
|
| 101 |
+
"""
|
| 102 |
+
if len(audio) > MAX_SAMPLES:
|
| 103 |
+
# Truncate dari tengah untuk mengurangi efek silence di awal/akhir
|
| 104 |
+
start = (len(audio) - MAX_SAMPLES) // 2
|
| 105 |
+
return audio[start:start + MAX_SAMPLES]
|
| 106 |
+
elif len(audio) < MAX_SAMPLES:
|
| 107 |
+
# Padding nol di kanan
|
| 108 |
+
pad_length = MAX_SAMPLES - len(audio)
|
| 109 |
+
return np.pad(audio, (0, pad_length), mode='constant')
|
| 110 |
+
return audio
|
| 111 |
+
|
| 112 |
+
def _extract_mel_spectrogram(self, audio: np.ndarray) -> np.ndarray:
|
| 113 |
+
"""
|
| 114 |
+
Konversi sinyal audio ke mel-spectrogram.
|
| 115 |
+
|
| 116 |
+
Mel-spectrogram adalah representasi visual dari sinyal audio
|
| 117 |
+
yang menunjukkan distribusi energi pada frekuensi tertentu seiring waktu.
|
| 118 |
+
Deepfake voice sering meninggalkan artefak pada frekuensi tertentu.
|
| 119 |
+
|
| 120 |
+
Returns:
|
| 121 |
+
numpy array shape [N_MELS, T], nilai dalam dB
|
| 122 |
+
"""
|
| 123 |
+
# Hitung mel-spectrogram menggunakan librosa
|
| 124 |
+
mel_spec = librosa.feature.melspectrogram(
|
| 125 |
+
y=audio,
|
| 126 |
+
sr=SAMPLE_RATE,
|
| 127 |
+
n_mels=N_MELS, # 128 filter bank
|
| 128 |
+
hop_length=HOP_LENGTH, # Step antar frame
|
| 129 |
+
n_fft=N_FFT, # FFT window size
|
| 130 |
+
fmin=20, # Frekuensi minimum (Hz) - batas pendengaran manusia
|
| 131 |
+
fmax=8000 # Frekuensi maksimum (Hz) - cukup untuk speech
|
| 132 |
+
)
|
| 133 |
+
|
| 134 |
+
# Konversi ke decibel scale (lebih intuitif dan stabil untuk training)
|
| 135 |
+
mel_db = librosa.power_to_db(mel_spec, ref=np.max)
|
| 136 |
+
|
| 137 |
+
return mel_db
|
| 138 |
+
|
| 139 |
+
def _classify_spectrogram(self, spectrogram: np.ndarray) -> float:
|
| 140 |
+
"""
|
| 141 |
+
Klasifikasikan mel-spectrogram menggunakan MobileNetV3.
|
| 142 |
+
|
| 143 |
+
Args:
|
| 144 |
+
spectrogram: numpy array [N_MELS, T] dalam dB scale
|
| 145 |
+
|
| 146 |
+
Returns:
|
| 147 |
+
float: Skor keaslian (0.0 = FAKE, 1.0 = REAL)
|
| 148 |
+
"""
|
| 149 |
+
# Normalisasi nilai ke range [0, 255] untuk konversi ke PIL Image
|
| 150 |
+
spec_normalized = ((spectrogram - spectrogram.min()) /
|
| 151 |
+
(spectrogram.max() - spectrogram.min() + 1e-8) * 255).astype(np.uint8)
|
| 152 |
+
|
| 153 |
+
# Konversi ke PIL Image grayscale
|
| 154 |
+
pil_image = Image.fromarray(spec_normalized, mode='L')
|
| 155 |
+
|
| 156 |
+
# Terapkan transformasi
|
| 157 |
+
tensor = SPECTROGRAM_TRANSFORM(pil_image) # Shape: [1, 224, 224]
|
| 158 |
+
|
| 159 |
+
# Tambahkan dimensi batch
|
| 160 |
+
tensor = tensor.unsqueeze(0).to(DEVICE) # Shape: [1, 1, 224, 224]
|
| 161 |
+
|
| 162 |
+
# Konversi ke FP16 jika pakai GPU
|
| 163 |
+
if DEVICE.type == "cuda":
|
| 164 |
+
tensor = tensor.half()
|
| 165 |
+
|
| 166 |
+
# Inferensi
|
| 167 |
+
with torch.no_grad():
|
| 168 |
+
logits = model_loader.audio_model(tensor) # Shape: [1, 2]
|
| 169 |
+
probs = torch.softmax(logits, dim=1) # Probabilitas
|
| 170 |
+
|
| 171 |
+
# Index 0 = FAKE, Index 1 = REAL
|
| 172 |
+
real_prob = probs[0, 1].item()
|
| 173 |
+
return real_prob
|
| 174 |
+
|
| 175 |
+
def _score_to_label(self, score: float) -> str:
|
| 176 |
+
"""Konversi skor ke label teks."""
|
| 177 |
+
if score >= 0.75:
|
| 178 |
+
return "REAL"
|
| 179 |
+
elif score <= 0.35:
|
| 180 |
+
return "FAKE"
|
| 181 |
+
else:
|
| 182 |
+
return "UNCERTAIN"
|
| 183 |
+
|
| 184 |
+
def analyze_bytes(self, audio_bytes: bytes, sample_rate: int = SAMPLE_RATE) -> dict:
|
| 185 |
+
"""
|
| 186 |
+
Analisis audio dari raw bytes (dari upload file).
|
| 187 |
+
|
| 188 |
+
Args:
|
| 189 |
+
audio_bytes: Audio dalam format bytes (WAV/MP3/etc)
|
| 190 |
+
sample_rate: Sample rate (jika diketahui)
|
| 191 |
+
"""
|
| 192 |
+
import io
|
| 193 |
+
import soundfile as sf
|
| 194 |
+
|
| 195 |
+
# Baca audio dari bytes menggunakan soundfile
|
| 196 |
+
audio_buffer = io.BytesIO(audio_bytes)
|
| 197 |
+
audio_data, sr = sf.read(audio_buffer, dtype='float32')
|
| 198 |
+
|
| 199 |
+
# Jika stereo, konversi ke mono dengan rata-rata kedua channel
|
| 200 |
+
if audio_data.ndim == 2:
|
| 201 |
+
audio_data = np.mean(audio_data, axis=1)
|
| 202 |
+
|
| 203 |
+
return self.analyze_chunk(audio_data, sample_rate=sr)
|
detector.py
ADDED
|
@@ -0,0 +1,238 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
detector.py
|
| 3 |
+
===========
|
| 4 |
+
Core engine deteksi deepfake wajah secara real-time.
|
| 5 |
+
Menggunakan pipeline 2 tahap:
|
| 6 |
+
1. MediaPipe Face Mesh → Deteksi & crop wajah dari frame
|
| 7 |
+
2. EfficientNet-B4 → Klasifikasi REAL vs FAKE per wajah
|
| 8 |
+
|
| 9 |
+
Mendukung temporal smoothing untuk mengurangi false positive
|
| 10 |
+
pada stream video yang bergerak cepat.
|
| 11 |
+
"""
|
| 12 |
+
|
| 13 |
+
import cv2
|
| 14 |
+
import torch
|
| 15 |
+
import numpy as np
|
| 16 |
+
import mediapipe as mp
|
| 17 |
+
import logging
|
| 18 |
+
from collections import deque
|
| 19 |
+
from typing import Optional
|
| 20 |
+
from PIL import Image
|
| 21 |
+
import torchvision.transforms as transforms
|
| 22 |
+
|
| 23 |
+
from model_loader import model_loader, DEVICE
|
| 24 |
+
|
| 25 |
+
logger = logging.getLogger(__name__)
|
| 26 |
+
|
| 27 |
+
# --- Konfigurasi MediaPipe Face Mesh ---
|
| 28 |
+
MP_FACE_MESH = mp.solutions.face_mesh
|
| 29 |
+
MP_DRAWING = mp.solutions.drawing_utils
|
| 30 |
+
MP_STYLES = mp.solutions.drawing_styles
|
| 31 |
+
|
| 32 |
+
# --- Transformasi gambar untuk input EfficientNet ---
|
| 33 |
+
# ImageNet normalization + resize ke 224x224
|
| 34 |
+
FACE_TRANSFORM = transforms.Compose([
|
| 35 |
+
transforms.Resize((224, 224)), # EfficientNet-B4 input size
|
| 36 |
+
transforms.ToTensor(), # [H,W,C] uint8 → [C,H,W] float32
|
| 37 |
+
transforms.Normalize( # Normalisasi ImageNet
|
| 38 |
+
mean=[0.485, 0.456, 0.406],
|
| 39 |
+
std=[0.229, 0.224, 0.225]
|
| 40 |
+
),
|
| 41 |
+
])
|
| 42 |
+
|
| 43 |
+
# Padding (%) di sekitar bounding box wajah agar konteks wajah lebih lengkap
|
| 44 |
+
FACE_PADDING_RATIO = 0.20
|
| 45 |
+
|
| 46 |
+
# Jumlah frame untuk temporal smoothing (rolling average)
|
| 47 |
+
TEMPORAL_WINDOW = 10
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
class FaceDeepfakeDetector:
|
| 51 |
+
"""
|
| 52 |
+
Detektor deepfake wajah berbasis MediaPipe + EfficientNet.
|
| 53 |
+
|
| 54 |
+
Cara kerja:
|
| 55 |
+
- Terima frame BGR dari OpenCV
|
| 56 |
+
- MediaPipe deteksi wajah & ekstrak bounding box
|
| 57 |
+
- EfficientNet klasifikasikan setiap wajah sebagai REAL/FAKE
|
| 58 |
+
- Temporal smoothing untuk stabilitas score
|
| 59 |
+
"""
|
| 60 |
+
|
| 61 |
+
def __init__(self):
|
| 62 |
+
# Inisialisasi MediaPipe Face Detection (lebih cepat dari Face Mesh untuk bounding box)
|
| 63 |
+
self.face_detection = mp.solutions.face_detection.FaceDetection(
|
| 64 |
+
model_selection=1, # Model 1 = full-range (jarak jauh), 0 = short-range
|
| 65 |
+
min_detection_confidence=0.6
|
| 66 |
+
)
|
| 67 |
+
|
| 68 |
+
# Buffer rolling untuk temporal smoothing score
|
| 69 |
+
self._score_buffer: deque = deque(maxlen=TEMPORAL_WINDOW)
|
| 70 |
+
|
| 71 |
+
# Score terakhir yang sudah di-smooth
|
| 72 |
+
self.last_smoothed_score: float = 0.5
|
| 73 |
+
|
| 74 |
+
logger.info("✅ FaceDeepfakeDetector siap digunakan")
|
| 75 |
+
|
| 76 |
+
def process_frame(self, frame_bgr: np.ndarray) -> dict:
|
| 77 |
+
"""
|
| 78 |
+
Proses satu frame video dan kembalikan hasil deteksi.
|
| 79 |
+
|
| 80 |
+
Args:
|
| 81 |
+
frame_bgr: Frame dari OpenCV dalam format BGR, shape [H, W, 3]
|
| 82 |
+
|
| 83 |
+
Returns:
|
| 84 |
+
dict dengan keys:
|
| 85 |
+
- 'authenticity_score': float 0.0 (FAKE) - 1.0 (REAL)
|
| 86 |
+
- 'smoothed_score' : float, hasil temporal smoothing
|
| 87 |
+
- 'faces_detected' : int, jumlah wajah terdeteksi
|
| 88 |
+
- 'face_boxes' : list of [x, y, w, h] dalam pixel
|
| 89 |
+
- 'label' : str, 'REAL' / 'FAKE' / 'UNCERTAIN'
|
| 90 |
+
- 'confidence' : float, confidence tertinggi
|
| 91 |
+
"""
|
| 92 |
+
h, w = frame_bgr.shape[:2]
|
| 93 |
+
|
| 94 |
+
# Konversi BGR → RGB (MediaPipe butuh RGB)
|
| 95 |
+
frame_rgb = cv2.cvtColor(frame_bgr, cv2.COLOR_BGR2RGB)
|
| 96 |
+
|
| 97 |
+
# --- Step 1: Deteksi Wajah dengan MediaPipe ---
|
| 98 |
+
results = self.face_detection.process(frame_rgb)
|
| 99 |
+
|
| 100 |
+
# Jika tidak ada wajah terdeteksi, kembalikan nilai netral
|
| 101 |
+
if not results.detections:
|
| 102 |
+
return self._empty_result()
|
| 103 |
+
|
| 104 |
+
face_boxes = []
|
| 105 |
+
face_scores = []
|
| 106 |
+
|
| 107 |
+
# --- Step 2: Proses Setiap Wajah yang Terdeteksi ---
|
| 108 |
+
for detection in results.detections:
|
| 109 |
+
# Ekstrak bounding box relatif (0.0 - 1.0) dari MediaPipe
|
| 110 |
+
bbox = detection.location_data.relative_bounding_box
|
| 111 |
+
|
| 112 |
+
# Konversi ke koordinat piksel
|
| 113 |
+
x = int(bbox.xmin * w)
|
| 114 |
+
y = int(bbox.ymin * h)
|
| 115 |
+
bw = int(bbox.width * w)
|
| 116 |
+
bh = int(bbox.height * h)
|
| 117 |
+
|
| 118 |
+
# Tambahkan padding di sekitar wajah
|
| 119 |
+
pad_x = int(bw * FACE_PADDING_RATIO)
|
| 120 |
+
pad_y = int(bh * FACE_PADDING_RATIO)
|
| 121 |
+
|
| 122 |
+
# Pastikan koordinat tidak keluar dari batas frame
|
| 123 |
+
x1 = max(0, x - pad_x)
|
| 124 |
+
y1 = max(0, y - pad_y)
|
| 125 |
+
x2 = min(w, x + bw + pad_x)
|
| 126 |
+
y2 = min(h, y + bh + pad_y)
|
| 127 |
+
|
| 128 |
+
# Crop region wajah dari frame RGB
|
| 129 |
+
face_crop = frame_rgb[y1:y2, x1:x2]
|
| 130 |
+
|
| 131 |
+
# Skip jika crop kosong (edge case)
|
| 132 |
+
if face_crop.size == 0:
|
| 133 |
+
continue
|
| 134 |
+
|
| 135 |
+
# Jalankan EfficientNet untuk klasifikasi
|
| 136 |
+
score = self._classify_face(face_crop)
|
| 137 |
+
face_scores.append(score)
|
| 138 |
+
face_boxes.append([x1, y1, x2 - x1, y2 - y1])
|
| 139 |
+
|
| 140 |
+
# Jika tidak ada wajah valid yang berhasil di-crop
|
| 141 |
+
if not face_scores:
|
| 142 |
+
return self._empty_result()
|
| 143 |
+
|
| 144 |
+
# Ambil score terendah (wajah paling mencurigakan) sebagai output utama
|
| 145 |
+
min_score = min(face_scores)
|
| 146 |
+
|
| 147 |
+
# Tambahkan ke buffer temporal smoothing
|
| 148 |
+
self._score_buffer.append(min_score)
|
| 149 |
+
|
| 150 |
+
# Hitung rata-rata dari buffer (temporal smoothing)
|
| 151 |
+
smoothed = float(np.mean(self._score_buffer))
|
| 152 |
+
self.last_smoothed_score = smoothed
|
| 153 |
+
|
| 154 |
+
# Jika buffer hanya berisi 1 data (misal: langsung setelah reset = mode foto),
|
| 155 |
+
# gunakan raw score agar tidak ada efek smoothing dari data lama
|
| 156 |
+
effective_score = min_score if len(self._score_buffer) == 1 else smoothed
|
| 157 |
+
|
| 158 |
+
return {
|
| 159 |
+
"authenticity_score": round(min_score, 4),
|
| 160 |
+
"smoothed_score" : round(effective_score, 4),
|
| 161 |
+
"faces_detected" : len(face_scores),
|
| 162 |
+
"face_boxes" : face_boxes,
|
| 163 |
+
"label" : self._score_to_label(effective_score),
|
| 164 |
+
"confidence" : round(max(effective_score, 1 - effective_score), 4),
|
| 165 |
+
}
|
| 166 |
+
|
| 167 |
+
def _classify_face(self, face_rgb: np.ndarray) -> float:
|
| 168 |
+
"""
|
| 169 |
+
Klasifikasikan crop wajah menggunakan HuggingFace Pipeline.
|
| 170 |
+
|
| 171 |
+
Args:
|
| 172 |
+
face_rgb: Crop wajah dalam format RGB numpy array
|
| 173 |
+
|
| 174 |
+
Returns:
|
| 175 |
+
float: Skor keaslian (0.0 = FAKE, 1.0 = REAL)
|
| 176 |
+
"""
|
| 177 |
+
# Konversi numpy array → PIL Image
|
| 178 |
+
pil_image = Image.fromarray(face_rgb)
|
| 179 |
+
|
| 180 |
+
try:
|
| 181 |
+
# Jalankan inferensi menggunakan pipeline transformers
|
| 182 |
+
results = model_loader.face_model(pil_image)
|
| 183 |
+
|
| 184 |
+
# Log output mentah model untuk debugging
|
| 185 |
+
logger.debug(f"🔍 Raw model output: {results}")
|
| 186 |
+
logger.info(f"📊 Model output: {results}")
|
| 187 |
+
|
| 188 |
+
# Parsing label model prithivMLmods/deepfake-detector-model-v1
|
| 189 |
+
# Label: Class 0 = 'fake', Class 1 = 'real'
|
| 190 |
+
real_prob = 0.5
|
| 191 |
+
fake_prob = 0.5
|
| 192 |
+
|
| 193 |
+
for res in results:
|
| 194 |
+
lbl = str(res.get('label', '')).strip().lower()
|
| 195 |
+
score = float(res.get('score', 0.5))
|
| 196 |
+
|
| 197 |
+
# Match label 'real' atau 'Real' atau '1'
|
| 198 |
+
if lbl in ('real', '1', 'label_1') or 'real' in lbl:
|
| 199 |
+
real_prob = score
|
| 200 |
+
# Match label 'fake' atau 'Fake' atau '0'
|
| 201 |
+
elif lbl in ('fake', '0', 'label_0') or 'fake' in lbl or 'ai' in lbl or 'manipulated' in lbl or 'artificial' in lbl:
|
| 202 |
+
fake_prob = score
|
| 203 |
+
real_prob = 1.0 - score
|
| 204 |
+
|
| 205 |
+
logger.info(f"✅ real_prob={real_prob:.3f}, fake_prob={fake_prob:.3f}")
|
| 206 |
+
return float(real_prob)
|
| 207 |
+
|
| 208 |
+
except Exception as e:
|
| 209 |
+
logger.error(f"Error inferensi pipeline: {e}")
|
| 210 |
+
return 0.5
|
| 211 |
+
|
| 212 |
+
def _score_to_label(self, score: float) -> str:
|
| 213 |
+
"""Konversi skor numerik menjadi label teks."""
|
| 214 |
+
# score = probabilitas REAL (0.0 = pasti FAKE, 1.0 = pasti REAL)
|
| 215 |
+
# Zona UNCERTAIN dipersempit: 0.45 - 0.55 (lebih decisive)
|
| 216 |
+
if score >= 0.55:
|
| 217 |
+
return "REAL"
|
| 218 |
+
elif score <= 0.45:
|
| 219 |
+
return "FAKE"
|
| 220 |
+
else:
|
| 221 |
+
return "UNCERTAIN"
|
| 222 |
+
|
| 223 |
+
def _empty_result(self) -> dict:
|
| 224 |
+
"""Kembalikan hasil kosong ketika tidak ada wajah terdeteksi."""
|
| 225 |
+
return {
|
| 226 |
+
"authenticity_score": 0.5,
|
| 227 |
+
"smoothed_score" : self.last_smoothed_score,
|
| 228 |
+
"faces_detected" : 0,
|
| 229 |
+
"face_boxes" : [],
|
| 230 |
+
"label" : "NO_FACE",
|
| 231 |
+
"confidence" : 0.0,
|
| 232 |
+
}
|
| 233 |
+
|
| 234 |
+
def reset_temporal_buffer(self):
|
| 235 |
+
"""Reset buffer temporal smoothing. Berguna saat sumber video berganti."""
|
| 236 |
+
self._score_buffer.clear()
|
| 237 |
+
self.last_smoothed_score = 0.5
|
| 238 |
+
logger.info("Buffer temporal smoothing direset.")
|
model_loader.py
ADDED
|
@@ -0,0 +1,185 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
model_loader.py
|
| 3 |
+
===============
|
| 4 |
+
Singleton pattern untuk memuat semua model AI sekali saat startup.
|
| 5 |
+
Mencegah OOM (Out of Memory) pada VRAM 4GB RTX 3050 Ti dengan
|
| 6 |
+
manajemen memori yang ketat dan Mixed Precision (FP16).
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
# ============================================================
|
| 10 |
+
# PENTING: Set cache directory SEBELUM import timm/huggingface
|
| 11 |
+
# Ini menghindari PermissionError di Windows saat timm mencoba
|
| 12 |
+
# menulis ke C:\Users\..\.cache\huggingface yang diblokir ACL.
|
| 13 |
+
# ============================================================
|
| 14 |
+
import os
|
| 15 |
+
from pathlib import Path
|
| 16 |
+
|
| 17 |
+
# Gunakan folder 'models/hf_cache' di dalam project sebagai cache
|
| 18 |
+
_PROJECT_ROOT = Path(__file__).parent
|
| 19 |
+
_HF_CACHE_DIR = str(_PROJECT_ROOT / "models" / "hf_cache")
|
| 20 |
+
|
| 21 |
+
# Set env var SEBELUM huggingface_hub / timm diinisialisasi
|
| 22 |
+
os.environ["HF_HOME"] = _HF_CACHE_DIR
|
| 23 |
+
os.environ["HUGGINGFACE_HUB_CACHE"] = _HF_CACHE_DIR
|
| 24 |
+
os.environ["TORCH_HOME"] = str(_PROJECT_ROOT / "models" / "torch_cache")
|
| 25 |
+
|
| 26 |
+
# Pastikan direktori cache ada
|
| 27 |
+
Path(_HF_CACHE_DIR).mkdir(parents=True, exist_ok=True)
|
| 28 |
+
Path(os.environ["TORCH_HOME"]).mkdir(parents=True, exist_ok=True)
|
| 29 |
+
# ============================================================
|
| 30 |
+
|
| 31 |
+
import torch
|
| 32 |
+
import timm
|
| 33 |
+
from transformers import pipeline
|
| 34 |
+
import logging
|
| 35 |
+
|
| 36 |
+
# Konfigurasi logging
|
| 37 |
+
logging.basicConfig(level=logging.INFO)
|
| 38 |
+
logger = logging.getLogger(__name__)
|
| 39 |
+
|
| 40 |
+
# --- Konfigurasi Global ---
|
| 41 |
+
DEVICE = torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
| 42 |
+
MODEL_DIR = Path(__file__).parent / "models"
|
| 43 |
+
MODEL_DIR.mkdir(exist_ok=True)
|
| 44 |
+
|
| 45 |
+
# Nama model EfficientNet yang digunakan (bisa diganti ke efficientnet_b3 jika VRAM kurang)
|
| 46 |
+
EFFICIENTNET_VARIANT = "efficientnet_b4"
|
| 47 |
+
NUM_CLASSES = 2 # [REAL, FAKE]
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
class ModelLoader:
|
| 51 |
+
"""
|
| 52 |
+
Singleton class untuk mengelola semua model AI.
|
| 53 |
+
Memastikan model hanya dimuat satu kali ke VRAM.
|
| 54 |
+
"""
|
| 55 |
+
_instance = None # Referensi singleton
|
| 56 |
+
_face_model = None # Model klasifikasi wajah deepfake
|
| 57 |
+
_audio_model = None # Model klasifikasi suara deepfake
|
| 58 |
+
_is_initialized = False # Flag apakah sudah diinisialisasi
|
| 59 |
+
|
| 60 |
+
def __new__(cls):
|
| 61 |
+
"""Implementasi singleton: hanya buat satu instance."""
|
| 62 |
+
if cls._instance is None:
|
| 63 |
+
cls._instance = super().__new__(cls)
|
| 64 |
+
return cls._instance
|
| 65 |
+
|
| 66 |
+
def initialize(self):
|
| 67 |
+
"""
|
| 68 |
+
Muat semua model ke device (GPU/CPU).
|
| 69 |
+
Dipanggil sekali saat startup FastAPI.
|
| 70 |
+
"""
|
| 71 |
+
if self._is_initialized:
|
| 72 |
+
logger.info("Models sudah diinisialisasi, skip loading.")
|
| 73 |
+
return
|
| 74 |
+
|
| 75 |
+
logger.info(f"🚀 Menginisialisasi ModelLoader pada device: {DEVICE}")
|
| 76 |
+
self._log_device_info()
|
| 77 |
+
|
| 78 |
+
# Muat model klasifikasi wajah
|
| 79 |
+
self._face_model = self._load_face_model()
|
| 80 |
+
|
| 81 |
+
# Muat model klasifikasi audio (CNN sederhana)
|
| 82 |
+
self._audio_model = self._load_audio_model()
|
| 83 |
+
|
| 84 |
+
self._is_initialized = True
|
| 85 |
+
logger.info("✅ Semua model berhasil dimuat!")
|
| 86 |
+
|
| 87 |
+
def _log_device_info(self):
|
| 88 |
+
"""Tampilkan informasi GPU/CUDA untuk debugging."""
|
| 89 |
+
if torch.cuda.is_available():
|
| 90 |
+
gpu_name = torch.cuda.get_device_name(0)
|
| 91 |
+
vram_total = torch.cuda.get_device_properties(0).total_memory / 1024**3
|
| 92 |
+
logger.info(f"🎮 GPU Terdeteksi: {gpu_name}")
|
| 93 |
+
logger.info(f"💾 VRAM Total: {vram_total:.1f} GB")
|
| 94 |
+
else:
|
| 95 |
+
logger.warning("⚠️ CUDA tidak tersedia. Menggunakan CPU (performa lebih lambat).")
|
| 96 |
+
|
| 97 |
+
def _load_face_model(self):
|
| 98 |
+
"""
|
| 99 |
+
Muat model pre-trained Deepfake Detection dari HuggingFace (prithivMLmods/Deep-Fake-Detector-Model).
|
| 100 |
+
Model ini (ViT) sudah dilatih khusus untuk membedakan Real vs AI-Generated/Deepfake.
|
| 101 |
+
"""
|
| 102 |
+
logger.info(f"📦 Memuat model wajah dari HuggingFace: prithivMLmods/Deep-Fake-Detector-Model...")
|
| 103 |
+
|
| 104 |
+
try:
|
| 105 |
+
# Gunakan transformers pipeline untuk image-classification
|
| 106 |
+
device_id = 0 if DEVICE.type == "cuda" else -1
|
| 107 |
+
model_pipeline = pipeline(
|
| 108 |
+
"image-classification",
|
| 109 |
+
model="prithivMLmods/Deep-Fake-Detector-Model",
|
| 110 |
+
device=device_id
|
| 111 |
+
)
|
| 112 |
+
logger.info("✅ Model wajah HuggingFace siap digunakan!")
|
| 113 |
+
return model_pipeline
|
| 114 |
+
except Exception as e:
|
| 115 |
+
logger.error(f"❌ Gagal memuat model HuggingFace: {e}")
|
| 116 |
+
raise e
|
| 117 |
+
|
| 118 |
+
def _load_audio_model(self) -> torch.nn.Module:
|
| 119 |
+
"""
|
| 120 |
+
Muat model CNN sederhana untuk analisis spektrogram audio.
|
| 121 |
+
Menggunakan MobileNetV3-Small untuk efisiensi VRAM.
|
| 122 |
+
"""
|
| 123 |
+
logger.info("📦 Memuat model audio: mobilenetv3_small_100...")
|
| 124 |
+
|
| 125 |
+
# MobileNetV3 lebih ringan dari EfficientNet, cocok untuk audio spectrogram
|
| 126 |
+
model = timm.create_model(
|
| 127 |
+
"mobilenetv3_small_100",
|
| 128 |
+
pretrained=True,
|
| 129 |
+
num_classes=NUM_CLASSES,
|
| 130 |
+
in_chans=1 # Spectrogram adalah grayscale (1 channel)
|
| 131 |
+
)
|
| 132 |
+
|
| 133 |
+
model.eval()
|
| 134 |
+
model = model.to(DEVICE)
|
| 135 |
+
|
| 136 |
+
if DEVICE.type == "cuda":
|
| 137 |
+
model = model.half()
|
| 138 |
+
logger.info("✅ Model audio: FP16 (half-precision) diaktifkan")
|
| 139 |
+
|
| 140 |
+
logger.info("✅ Model audio 'MobileNetV3-Small' siap digunakan")
|
| 141 |
+
return model
|
| 142 |
+
|
| 143 |
+
@property
|
| 144 |
+
def face_model(self):
|
| 145 |
+
"""Akses model wajah yang sudah dimuat."""
|
| 146 |
+
if not self._is_initialized:
|
| 147 |
+
raise RuntimeError("ModelLoader belum diinisialisasi! Panggil .initialize() dulu.")
|
| 148 |
+
return self._face_model
|
| 149 |
+
|
| 150 |
+
@property
|
| 151 |
+
def audio_model(self) -> torch.nn.Module:
|
| 152 |
+
"""Akses model audio yang sudah dimuat."""
|
| 153 |
+
if not self._is_initialized:
|
| 154 |
+
raise RuntimeError("ModelLoader belum diinisialisasi! Panggil .initialize() dulu.")
|
| 155 |
+
return self._audio_model
|
| 156 |
+
|
| 157 |
+
@property
|
| 158 |
+
def device(self) -> torch.device:
|
| 159 |
+
"""Kembalikan device yang sedang digunakan (cuda/cpu)."""
|
| 160 |
+
return DEVICE
|
| 161 |
+
|
| 162 |
+
def get_vram_usage(self) -> dict:
|
| 163 |
+
"""
|
| 164 |
+
Kembalikan informasi penggunaan VRAM saat ini.
|
| 165 |
+
Berguna untuk monitoring di endpoint /api/health.
|
| 166 |
+
"""
|
| 167 |
+
if DEVICE.type != "cuda":
|
| 168 |
+
return {"available": False, "reason": "CUDA not available"}
|
| 169 |
+
|
| 170 |
+
allocated = torch.cuda.memory_allocated(0) / 1024**3 # GB
|
| 171 |
+
reserved = torch.cuda.memory_reserved(0) / 1024**3 # GB
|
| 172 |
+
total = torch.cuda.get_device_properties(0).total_memory / 1024**3
|
| 173 |
+
|
| 174 |
+
return {
|
| 175 |
+
"available": True,
|
| 176 |
+
"gpu_name": torch.cuda.get_device_name(0),
|
| 177 |
+
"vram_total_gb": round(total, 2),
|
| 178 |
+
"vram_allocated_gb": round(allocated, 3),
|
| 179 |
+
"vram_reserved_gb": round(reserved, 3),
|
| 180 |
+
"vram_free_gb": round(total - reserved, 3),
|
| 181 |
+
}
|
| 182 |
+
|
| 183 |
+
|
| 184 |
+
# Instance singleton global - diimport oleh modul lain
|
| 185 |
+
model_loader = ModelLoader()
|
requirements.txt
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Deepfake Shield - Backend untuk Hugging Face Spaces
|
| 2 |
+
# Requirements yang ringan (tanpa CUDA, pakai CPU di HF Spaces tier gratis)
|
| 3 |
+
|
| 4 |
+
# --- Web Framework ---
|
| 5 |
+
fastapi==0.111.0
|
| 6 |
+
uvicorn[standard]==0.29.0
|
| 7 |
+
python-multipart==0.0.9
|
| 8 |
+
|
| 9 |
+
# --- Deep Learning (CPU version untuk HF Spaces gratis) ---
|
| 10 |
+
torch==2.2.2
|
| 11 |
+
torchvision==0.17.2
|
| 12 |
+
timm==0.9.16
|
| 13 |
+
transformers==4.40.0
|
| 14 |
+
|
| 15 |
+
# --- Computer Vision ---
|
| 16 |
+
opencv-python-headless==4.9.0.80
|
| 17 |
+
mediapipe==0.10.14
|
| 18 |
+
Pillow==10.3.0
|
| 19 |
+
|
| 20 |
+
# --- Audio Processing ---
|
| 21 |
+
librosa==0.10.2
|
| 22 |
+
soundfile==0.12.1
|
| 23 |
+
scipy==1.13.0
|
| 24 |
+
|
| 25 |
+
# --- Data Processing ---
|
| 26 |
+
numpy==1.26.4
|
| 27 |
+
|
| 28 |
+
# --- Utilities ---
|
| 29 |
+
python-dotenv==1.0.1
|
| 30 |
+
aiofiles==23.2.1
|
| 31 |
+
pydantic==2.7.1
|
| 32 |
+
httpx==0.27.0
|
utils/__init__.py
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
__init__.py untuk package utils
|
| 3 |
+
"""
|
utils/audio_processor.py
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
audio_processor.py
|
| 3 |
+
==================
|
| 4 |
+
Utilitas untuk preprocessing audio chunk sebelum dianalisis.
|
| 5 |
+
"""
|
| 6 |
+
|
| 7 |
+
import numpy as np
|
| 8 |
+
import librosa
|
| 9 |
+
import io
|
| 10 |
+
import soundfile as sf
|
| 11 |
+
from typing import Tuple, Optional
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
def bytes_to_audio(audio_bytes: bytes) -> Tuple[np.ndarray, int]:
|
| 15 |
+
"""
|
| 16 |
+
Konversi bytes audio ke numpy array PCM.
|
| 17 |
+
Mendukung format: WAV, MP3, OGG, FLAC.
|
| 18 |
+
|
| 19 |
+
Returns:
|
| 20 |
+
Tuple (audio_array float32 mono, sample_rate)
|
| 21 |
+
"""
|
| 22 |
+
buffer = io.BytesIO(audio_bytes)
|
| 23 |
+
|
| 24 |
+
# Coba baca dengan soundfile terlebih dahulu
|
| 25 |
+
try:
|
| 26 |
+
audio, sr = sf.read(buffer, dtype='float32')
|
| 27 |
+
except Exception:
|
| 28 |
+
# Fallback ke librosa (lebih lambat tapi mendukung lebih banyak format)
|
| 29 |
+
buffer.seek(0)
|
| 30 |
+
audio, sr = librosa.load(buffer, sr=None, mono=True)
|
| 31 |
+
return audio, sr
|
| 32 |
+
|
| 33 |
+
# Konversi stereo ke mono jika diperlukan
|
| 34 |
+
if audio.ndim == 2:
|
| 35 |
+
audio = np.mean(audio, axis=1)
|
| 36 |
+
|
| 37 |
+
return audio, sr
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
def normalize_audio(audio: np.ndarray) -> np.ndarray:
|
| 41 |
+
"""
|
| 42 |
+
Normalisasi amplitudo audio ke range [-1.0, 1.0].
|
| 43 |
+
Mencegah saturasi dan memastikan konsistensi input model.
|
| 44 |
+
"""
|
| 45 |
+
max_val = np.max(np.abs(audio))
|
| 46 |
+
if max_val > 0:
|
| 47 |
+
audio = audio / max_val
|
| 48 |
+
return audio
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
def split_audio_chunks(
|
| 52 |
+
audio: np.ndarray,
|
| 53 |
+
sample_rate: int,
|
| 54 |
+
chunk_duration: float = 2.0,
|
| 55 |
+
overlap: float = 0.5
|
| 56 |
+
) -> list:
|
| 57 |
+
"""
|
| 58 |
+
Bagi audio panjang menjadi chunks kecil dengan overlap.
|
| 59 |
+
|
| 60 |
+
Args:
|
| 61 |
+
audio : Array audio PCM
|
| 62 |
+
sample_rate : Sample rate (Hz)
|
| 63 |
+
chunk_duration: Durasi setiap chunk (detik)
|
| 64 |
+
overlap : Overlap antar chunk (detik)
|
| 65 |
+
|
| 66 |
+
Returns:
|
| 67 |
+
List of numpy arrays, masing-masing adalah satu chunk
|
| 68 |
+
"""
|
| 69 |
+
chunk_samples = int(chunk_duration * sample_rate)
|
| 70 |
+
hop_samples = int((chunk_duration - overlap) * sample_rate)
|
| 71 |
+
|
| 72 |
+
chunks = []
|
| 73 |
+
start = 0
|
| 74 |
+
|
| 75 |
+
while start + chunk_samples <= len(audio):
|
| 76 |
+
chunk = audio[start:start + chunk_samples]
|
| 77 |
+
chunks.append(chunk)
|
| 78 |
+
start += hop_samples
|
| 79 |
+
|
| 80 |
+
# Tambahkan sisa audio jika belum masuk (dengan padding)
|
| 81 |
+
if start < len(audio):
|
| 82 |
+
remainder = audio[start:]
|
| 83 |
+
padded = np.pad(remainder, (0, chunk_samples - len(remainder)))
|
| 84 |
+
chunks.append(padded)
|
| 85 |
+
|
| 86 |
+
return chunks
|
utils/frame_processor.py
ADDED
|
@@ -0,0 +1,124 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
frame_processor.py
|
| 3 |
+
==================
|
| 4 |
+
Utilitas untuk preprocessing frame video sebelum dimasukkan ke model.
|
| 5 |
+
Termasuk: resize, quality enhancement, dan konversi format.
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
import cv2
|
| 9 |
+
import numpy as np
|
| 10 |
+
from typing import Tuple, Optional
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
def decode_base64_frame(b64_string: str) -> Optional[np.ndarray]:
|
| 14 |
+
"""
|
| 15 |
+
Decode string base64 menjadi frame OpenCV BGR.
|
| 16 |
+
|
| 17 |
+
Args:
|
| 18 |
+
b64_string: String base64 (dengan atau tanpa data URL prefix)
|
| 19 |
+
|
| 20 |
+
Returns:
|
| 21 |
+
numpy array [H, W, 3] BGR, atau None jika gagal
|
| 22 |
+
"""
|
| 23 |
+
import base64
|
| 24 |
+
|
| 25 |
+
# Hapus prefix data URL jika ada
|
| 26 |
+
if "," in b64_string:
|
| 27 |
+
b64_string = b64_string.split(",", 1)[1]
|
| 28 |
+
|
| 29 |
+
try:
|
| 30 |
+
# Decode base64 → bytes → numpy
|
| 31 |
+
raw_bytes = base64.b64decode(b64_string)
|
| 32 |
+
arr = np.frombuffer(raw_bytes, dtype=np.uint8)
|
| 33 |
+
frame = cv2.imdecode(arr, cv2.IMREAD_COLOR)
|
| 34 |
+
return frame
|
| 35 |
+
except Exception:
|
| 36 |
+
return None
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def preprocess_frame(
|
| 40 |
+
frame: np.ndarray,
|
| 41 |
+
target_size: Tuple[int, int] = (640, 480),
|
| 42 |
+
enhance_quality: bool = True
|
| 43 |
+
) -> np.ndarray:
|
| 44 |
+
"""
|
| 45 |
+
Preprocess frame video untuk optimasi deteksi.
|
| 46 |
+
|
| 47 |
+
Args:
|
| 48 |
+
frame : Frame BGR dari OpenCV
|
| 49 |
+
target_size : (width, height) target untuk resize
|
| 50 |
+
enhance_quality: Terapkan CLAHE untuk perbaikan kontras
|
| 51 |
+
|
| 52 |
+
Returns:
|
| 53 |
+
Frame BGR yang sudah dipreprocess
|
| 54 |
+
"""
|
| 55 |
+
# Resize frame jika terlalu besar (menghemat waktu proses)
|
| 56 |
+
h, w = frame.shape[:2]
|
| 57 |
+
target_w, target_h = target_size
|
| 58 |
+
|
| 59 |
+
# Hanya resize jika lebih besar dari target
|
| 60 |
+
if w > target_w or h > target_h:
|
| 61 |
+
frame = cv2.resize(frame, target_size, interpolation=cv2.INTER_LINEAR)
|
| 62 |
+
|
| 63 |
+
# CLAHE (Contrast Limited Adaptive Histogram Equalization)
|
| 64 |
+
# Meningkatkan kontras lokal untuk membantu deteksi wajah
|
| 65 |
+
if enhance_quality:
|
| 66 |
+
lab = cv2.cvtColor(frame, cv2.COLOR_BGR2LAB) # Konversi ke LAB color space
|
| 67 |
+
l_channel, a, b = cv2.split(lab)
|
| 68 |
+
|
| 69 |
+
# Terapkan CLAHE hanya pada channel L (luminance/kecerahan)
|
| 70 |
+
clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))
|
| 71 |
+
l_enhanced = clahe.apply(l_channel)
|
| 72 |
+
|
| 73 |
+
# Gabungkan kembali channel
|
| 74 |
+
enhanced_lab = cv2.merge([l_enhanced, a, b])
|
| 75 |
+
frame = cv2.cvtColor(enhanced_lab, cv2.COLOR_LAB2BGR)
|
| 76 |
+
|
| 77 |
+
return frame
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
def draw_detection_overlay(
|
| 81 |
+
frame: np.ndarray,
|
| 82 |
+
face_boxes: list,
|
| 83 |
+
label: str,
|
| 84 |
+
score: float
|
| 85 |
+
) -> np.ndarray:
|
| 86 |
+
"""
|
| 87 |
+
Gambar bounding box dan label deteksi di atas frame.
|
| 88 |
+
|
| 89 |
+
Args:
|
| 90 |
+
frame : Frame BGR
|
| 91 |
+
face_boxes: List of [x, y, w, h] dalam piksel
|
| 92 |
+
label : Label deteksi ('REAL', 'FAKE', 'UNCERTAIN')
|
| 93 |
+
score : Skor keaslian (0.0 - 1.0)
|
| 94 |
+
|
| 95 |
+
Returns:
|
| 96 |
+
Frame dengan overlay
|
| 97 |
+
"""
|
| 98 |
+
# Pilih warna berdasarkan label
|
| 99 |
+
color_map = {
|
| 100 |
+
"REAL" : (0, 255, 100), # Hijau neon
|
| 101 |
+
"FAKE" : (0, 50, 255), # Merah
|
| 102 |
+
"UNCERTAIN": (0, 165, 255), # Oranye
|
| 103 |
+
"NO_FACE" : (128, 128, 128), # Abu-abu
|
| 104 |
+
}
|
| 105 |
+
color = color_map.get(label, (255, 255, 255))
|
| 106 |
+
|
| 107 |
+
# Gambar bounding box untuk setiap wajah
|
| 108 |
+
for (x, y, w, h) in face_boxes:
|
| 109 |
+
# Bounding box utama
|
| 110 |
+
cv2.rectangle(frame, (x, y), (x + w, y + h), color, 2)
|
| 111 |
+
|
| 112 |
+
# Label teks di atas bounding box
|
| 113 |
+
label_text = f"{label} {score:.0%}"
|
| 114 |
+
cv2.putText(
|
| 115 |
+
frame, label_text,
|
| 116 |
+
(x, max(y - 10, 10)), # Posisi: di atas bbox, minimal y=10
|
| 117 |
+
cv2.FONT_HERSHEY_SIMPLEX, # Font
|
| 118 |
+
0.7, # Scale
|
| 119 |
+
color, # Warna
|
| 120 |
+
2, # Tebal
|
| 121 |
+
cv2.LINE_AA # Anti-aliasing
|
| 122 |
+
)
|
| 123 |
+
|
| 124 |
+
return frame
|