pavankumarvk's picture
Update pipeline.py
d9a982f verified
Raw
History Blame
3.5 kB
import os
import cv2
import torch
import zipfile
import librosa
import numpy as np
import tensorflow as tf
from facenet_pytorch import MTCNN
from rawnet import RawNet
tf.random.set_seed(42)
if not os.path.exists("efficientnet-b0"):
local_zip = "./efficientnet-b0.zip"
if os.path.exists(local_zip):
zip_ref = zipfile.ZipFile(local_zip, 'r')
zip_ref.extractall()
zip_ref.close()
print("Model extracted successfully!")
model = tf.keras.models.load_model("efficientnet-b0/", compile=False)
class DetectionPipeline:
def __init__(self, n_frames=None, batch_size=60, resize=None, input_modality='video'):
self.n_frames = n_frames
self.batch_size = batch_size
self.resize = resize
self.input_modality = input_modality
def __call__(self, filename):
if self.input_modality == 'video':
v_cap = cv2.VideoCapture(filename)
v_len = int(v_cap.get(cv2.CAP_PROP_FRAME_COUNT))
sample = np.arange(0, v_len) if self.n_frames is None \
else np.linspace(0, v_len-1, self.n_frames).astype(int)
faces = []
frames = []
for j in range(v_len):
success = v_cap.grab()
if j in sample:
success, frame = v_cap.retrieve()
if not success:
continue
frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)
if self.resize is not None:
frame = frame.resize(
[int(d * self.resize) for d in frame.size]
)
frames.append(frame)
if len(frames) % self.batch_size == 0 or j == sample[-1]:
face2 = cv2.resize(frame, (224, 224))
faces.append(face2)
v_cap.release()
return faces
elif self.input_modality == 'image':
image = cv2.cvtColor(filename, cv2.COLOR_BGR2RGB)
image = cv2.resize(image, (224, 224))
return image
elif self.input_modality == 'audio':
x, sr = librosa.load(filename)
x_pt = torch.Tensor(x)
x_pt = torch.unsqueeze(x_pt, dim=0)
return x_pt
else:
raise ValueError("Invalid modality")
detection_video_pipeline = DetectionPipeline(n_frames=5, batch_size=1, input_modality='video')
detection_image_pipeline = DetectionPipeline(batch_size=1, input_modality='image')
def deepfakes_video_predict(input_video):
faces = detection_video_pipeline(input_video)
real_res, fake_res = [], []
for face in faces:
face2 = face / 255
pred = model.predict(np.expand_dims(face2, axis=0))[0]
real, fake = pred[0], pred[1]
real_res.append(real)
fake_res.append(fake)
real_mean = np.mean(real_res)
fake_mean = np.mean(fake_res)
if real_mean >= 0.5:
return "The video is REAL. Confidence: " + str(round(100 - real_mean*100, 3)) + "%"
else:
return "The video is FAKE. Confidence: " + str(round(fake_mean*100, 3)) + "%"
def deepfakes_image_predict(input_image):
face = detection_image_pipeline(input_image)
face2 = face / 255
pred = model.predict(np.expand_dims(face2, axis=0))[0]
real, fake = pred[0], pred[1]
if real > 0.5:
return "The image is REAL."
else:
return "The image is FAKE."