DEEP-FAKE-DETECTOR / fake_audio.py
aneela-pervez's picture
Update fake_audio.py
43132d8 verified
Raw
History Blame
7.1 kB
# -*- coding: utf-8 -*-
"""
Cleaned fake vs forged vs real audio module (Gradio UI & Notebook Magics removed)
"""
# Note: In pure Python files, install requirements via terminal, not in the code.
# pip install torch librosa numpy laion-clap groq openai-whisper transformers
# apt-get install -y ffmpeg
from huggingface_hub import login
# Aapka provide kiya gaya token use ho raha hai
# login(token="YOUR_TOKEN_HERE")
print("Login Successful! Ab aap models load kar sakti hain.")
import os
# Agar aap locally run kar rahi hain, toh drive mount ki zaroorat nahi hai.
# Agar Colab mein hain toh isay un-comment kar lein:
# from google.colab import drive
# drive.mount('/content/drive')
import torch
import librosa
import numpy as np
import numpy
# Naye PyTorch ko batana ke audio model ki weights safe hain
torch.serialization.add_safe_globals([numpy.core.multiarray.scalar])
import laion_clap
import whisper
from sklearn.metrics.pairwise import cosine_similarity, euclidean_distances
from sklearn.preprocessing import normalize
from transformers import AutoFeatureExtractor, AutoModelForAudioClassification, AutoModelForAudioFrameClassification
from groq import Groq
# --- SETUP & CONFIG ---
GROQ_API_KEY = "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"
client = Groq(api_key=GROQ_API_KEY)
DEVICE = "cuda" if torch.cuda.is_available() else "cpu"
AST_MODEL_PATH = "aneela-pervez/FAKE-AUDIO"
# --- 🏷️ CATEGORIZED FORENSIC LABELS ---
FORENSIC_CATEGORIES = {
"AUTHENTIC": ["Natural human speech", "Human voice with natural breathing", "Consistent room ambiance"],
"AI_SYNTHETIC": ["AI voice clone robotic smoothness", "Synthetic neural vocoder", "Deepfake metallic resonance"],
"MANIPULATED": ["Manually spliced audio", "Inconsistent stitching artifacts", "Micro-cuts and clicks"],
"DISGUISED": ["Artificial pitch shifting", "Muffled voice mask identity", "Time-stretched artifacts"]
}
ALL_LABELS = [item for sublist in FORENSIC_CATEGORIES.values() for item in sublist]
# --- MODEL LOADING ---
print("🔄 Initializing Fuzzy Forensic Engine v3.9...")
whisper_model = whisper.load_model("base", device=DEVICE)
id_extractor = AutoFeatureExtractor.from_pretrained("facebook/wav2vec2-base-960h")
id_model = AutoModelForAudioFrameClassification.from_pretrained("facebook/wav2vec2-base-960h").to(DEVICE).eval()
try:
ast_extractor = AutoFeatureExtractor.from_pretrained(AST_MODEL_PATH)
ast_model = AutoModelForAudioClassification.from_pretrained(AST_MODEL_PATH).to(DEVICE).eval()
except Exception as e:
print(f"⚠️ AST Model issue (using default logic if fails): {e}")
clap_model = laion_clap.CLAP_Module(enable_fusion=False, amodel='HTSAT-tiny').to(DEVICE)
clap_model.load_ckpt(model_id=1)
def extract_dna_embeddings(y):
inputs = id_extractor(y, sampling_rate=16000, return_tensors="pt").to(DEVICE)
with torch.no_grad():
logits = id_model(**inputs).logits
emb = torch.mean(logits, dim=1).cpu().numpy()
return normalize(emb)
def analyze_voice_forensics(audio_path):
try:
y, sr = librosa.load(audio_path, sr=16000)
y_48, _ = librosa.load(audio_path, sr=48000)
y = y / (np.max(np.abs(y)) + 1e-9)
# 1. 🛡️ SEGMENT-WISE CROSS VALIDATION
segments = np.array_split(y, 3)
seg_embs = [extract_dna_embeddings(s) for s in segments]
cos_sim = cosine_similarity(seg_embs[0], seg_embs[-1])[0][0]
euc_dist = euclidean_distances(seg_embs[0], seg_embs[-1])[0][0]
# 2. FEATURE EXTRACTION
transcription = whisper_model.transcribe(audio_path)["text"]
inputs = ast_extractor(y, sampling_rate=16000, return_tensors="pt").to(DEVICE)
with torch.no_grad():
ast_prob = torch.nn.functional.softmax(ast_model(**inputs).logits, dim=-1).cpu().numpy()[0][1]
ast_score = ast_prob * 100
c_sims = torch.nn.functional.cosine_similarity(torch.from_numpy(clap_model.get_audio_embedding_from_data(x=[y_48])).to(DEVICE),
torch.from_numpy(clap_model.get_text_embedding(ALL_LABELS)).to(DEVICE))
top_idx = torch.argmax(c_sims).item()
clap_label = ALL_LABELS[top_idx]
category = "UNKNOWN"
for cat, labels in FORENSIC_CATEGORIES.items():
if clap_label in labels:
category = cat
break
num_edits = len(np.where(np.diff(librosa.onset.onset_strength(y=y, sr=16000)) > 6.5)[0])
# --- 🧠 STEP 3: FUZZY LOGIC MEMBERSHIP SCORING ---
# AI Membership: Degrees of synthetic texture
mu_ai = min(1.0, max(0.0, (ast_score - 40) / 45))
# Splicing Membership: Degrees of manual tampering
mu_spliced = min(1.0, max(0.0, (num_edits - 8) / 10))
# Authenticity Membership: DNA + Category Confidence
dna_conf = (cos_sim - 0.85) / 0.10
cat_conf = 1.0 if category == "AUTHENTIC" else 0.0
mu_auth = min(1.0, max(0.0, (dna_conf + cat_conf) / 2))
# --- 🛡️ STEP 4: FUZZY DECISION ENGINE ---
# --- 🛡️ BALANCED FUZZY DECISION ENGINE (V4.0) ---
# 1. AUTHENTIC HUMAN PRIORITY (Identity Override)
# Agar DNA bohot strong hy aur cuts moderate hain, toh texture ko secondary rakho
if mu_auth > 0.80 and mu_spliced < 0.75:
verdict = "✅ AUTHENTIC HUMAN VOICE"
category_override = "AUTHENTIC"
audit_focus = f"Identity DNA Dominance (Auth: {mu_auth:.2f})"
# 2. AI CLONE (High Texture + Lower Identity Confidence)
elif mu_ai > 0.85 and mu_auth < 0.80:
verdict = "🚨 AI CLONE DETECTED"
category_override = "AI_SYNTHETIC"
audit_focus = f"Synthetic Texture Supremacy (AI: {mu_ai:.2f})"
# 3. MANIPULATED REAL VOICE (Spliced)
else:
verdict = "⚠️ MANIPULATED REAL VOICE (Spliced)"
category_override = "MANIPULATED"
audit_focus = f"Manual Splicing Flux (Spliced: {mu_spliced:.2f})"
# --- 🔥 ALIGNED PROFESSIONAL REASONING ---
audit_context = (f"Verdict: {verdict}. Category: {category_override}. Metrics: ID {cos_sim:.2f}, "
f"Texture {ast_score:.1f}%, Cuts {num_edits}. "
f"Fuzzy Scores: AI={mu_ai:.2f}, Spliced={mu_spliced:.2f}, Auth={mu_auth:.2f}")
prompt = (f"Act as a Senior Forensic Auditor. {audit_context} "
"Instruction: Justify the verdict. If identity stability is high (Auth > 0.8), "
"emphasize that DNA integrity confirms human origin despite high spectral texture. "
"Give a 2-line direct forensic conclusion.")
res = client.chat.completions.create(model="llama-3.3-70b-versatile", messages=[{"role": "user", "content": prompt}])
reasoning = res.choices[0].message.content
evidence = f"Category: {category} | ID: {cos_sim:.2f} | Texture: {ast_score:.1f}% | Cuts: {num_edits}"
return verdict, evidence, reasoning
except Exception as e:
return "Error", str(e), "N/A"