# -*- coding: utf-8 -*- """ Cleaned fake vs forged vs real audio module (Gradio UI & Notebook Magics removed) """ # Note: In pure Python files, install requirements via terminal, not in the code. # pip install torch librosa numpy laion-clap groq openai-whisper transformers # apt-get install -y ffmpeg from huggingface_hub import login # Aapka provide kiya gaya token use ho raha hai # login(token="YOUR_TOKEN_HERE") print("Login Successful! Ab aap models load kar sakti hain.") import os # Agar aap locally run kar rahi hain, toh drive mount ki zaroorat nahi hai. # Agar Colab mein hain toh isay un-comment kar lein: # from google.colab import drive # drive.mount('/content/drive') import torch import librosa import numpy as np import numpy # Naye PyTorch ko batana ke audio model ki weights safe hain torch.serialization.add_safe_globals([numpy.core.multiarray.scalar]) import laion_clap import whisper from sklearn.metrics.pairwise import cosine_similarity, euclidean_distances from sklearn.preprocessing import normalize from transformers import AutoFeatureExtractor, AutoModelForAudioClassification, AutoModelForAudioFrameClassification from groq import Groq # --- SETUP & CONFIG --- GROQ_API_KEY = "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" client = Groq(api_key=GROQ_API_KEY) DEVICE = "cuda" if torch.cuda.is_available() else "cpu" AST_MODEL_PATH = "./Final_Audio_Model" # --- 🏷️ CATEGORIZED FORENSIC LABELS --- FORENSIC_CATEGORIES = { "AUTHENTIC": ["Natural human speech", "Human voice with natural breathing", "Consistent room ambiance"], "AI_SYNTHETIC": ["AI voice clone robotic smoothness", "Synthetic neural vocoder", "Deepfake metallic resonance"], "MANIPULATED": ["Manually spliced audio", "Inconsistent stitching artifacts", "Micro-cuts and clicks"], "DISGUISED": ["Artificial pitch shifting", "Muffled voice mask identity", "Time-stretched artifacts"] } ALL_LABELS = [item for sublist in FORENSIC_CATEGORIES.values() for item in sublist] # --- MODEL LOADING --- print("🔄 Initializing Fuzzy Forensic Engine v3.9...") whisper_model = whisper.load_model("base", device=DEVICE) id_extractor = AutoFeatureExtractor.from_pretrained("facebook/wav2vec2-base-960h") id_model = AutoModelForAudioFrameClassification.from_pretrained("facebook/wav2vec2-base-960h").to(DEVICE).eval() try: ast_extractor = AutoFeatureExtractor.from_pretrained(AST_MODEL_PATH, local_files_only=True) ast_model = AutoModelForAudioClassification.from_pretrained(AST_MODEL_PATH, local_files_only=True).to(DEVICE).eval() except Exception as e: print(f"⚠️ AST Model issue (using default logic if fails): {e}") clap_model = laion_clap.CLAP_Module(enable_fusion=False, amodel='HTSAT-tiny').to(DEVICE) clap_model.load_ckpt(model_id=1) def extract_dna_embeddings(y): inputs = id_extractor(y, sampling_rate=16000, return_tensors="pt").to(DEVICE) with torch.no_grad(): logits = id_model(**inputs).logits emb = torch.mean(logits, dim=1).cpu().numpy() return normalize(emb) def analyze_voice_forensics(audio_path): try: y, sr = librosa.load(audio_path, sr=16000) y_48, _ = librosa.load(audio_path, sr=48000) y = y / (np.max(np.abs(y)) + 1e-9) # 1. 🛡️ SEGMENT-WISE CROSS VALIDATION segments = np.array_split(y, 3) seg_embs = [extract_dna_embeddings(s) for s in segments] cos_sim = cosine_similarity(seg_embs[0], seg_embs[-1])[0][0] euc_dist = euclidean_distances(seg_embs[0], seg_embs[-1])[0][0] # 2. FEATURE EXTRACTION transcription = whisper_model.transcribe(audio_path)["text"] inputs = ast_extractor(y, sampling_rate=16000, return_tensors="pt").to(DEVICE) with torch.no_grad(): ast_prob = torch.nn.functional.softmax(ast_model(**inputs).logits, dim=-1).cpu().numpy()[0][1] ast_score = ast_prob * 100 c_sims = torch.nn.functional.cosine_similarity(torch.from_numpy(clap_model.get_audio_embedding_from_data(x=[y_48])).to(DEVICE), torch.from_numpy(clap_model.get_text_embedding(ALL_LABELS)).to(DEVICE)) top_idx = torch.argmax(c_sims).item() clap_label = ALL_LABELS[top_idx] category = "UNKNOWN" for cat, labels in FORENSIC_CATEGORIES.items(): if clap_label in labels: category = cat break num_edits = len(np.where(np.diff(librosa.onset.onset_strength(y=y, sr=16000)) > 6.5)[0]) # --- 🧠 STEP 3: FUZZY LOGIC MEMBERSHIP SCORING --- # AI Membership: Degrees of synthetic texture mu_ai = min(1.0, max(0.0, (ast_score - 40) / 45)) # Splicing Membership: Degrees of manual tampering mu_spliced = min(1.0, max(0.0, (num_edits - 8) / 10)) # Authenticity Membership: DNA + Category Confidence dna_conf = (cos_sim - 0.85) / 0.10 cat_conf = 1.0 if category == "AUTHENTIC" else 0.0 mu_auth = min(1.0, max(0.0, (dna_conf + cat_conf) / 2)) # --- 🛡️ STEP 4: FUZZY DECISION ENGINE --- # --- 🛡️ BALANCED FUZZY DECISION ENGINE (V4.0) --- # 1. AUTHENTIC HUMAN PRIORITY (Identity Override) # Agar DNA bohot strong hy aur cuts moderate hain, toh texture ko secondary rakho if mu_auth > 0.80 and mu_spliced < 0.75: verdict = "✅ AUTHENTIC HUMAN VOICE" category_override = "AUTHENTIC" audit_focus = f"Identity DNA Dominance (Auth: {mu_auth:.2f})" # 2. AI CLONE (High Texture + Lower Identity Confidence) elif mu_ai > 0.85 and mu_auth < 0.80: verdict = "🚨 AI CLONE DETECTED" category_override = "AI_SYNTHETIC" audit_focus = f"Synthetic Texture Supremacy (AI: {mu_ai:.2f})" # 3. MANIPULATED REAL VOICE (Spliced) else: verdict = "⚠️ MANIPULATED REAL VOICE (Spliced)" category_override = "MANIPULATED" audit_focus = f"Manual Splicing Flux (Spliced: {mu_spliced:.2f})" # --- 🔥 ALIGNED PROFESSIONAL REASONING --- audit_context = (f"Verdict: {verdict}. Category: {category_override}. Metrics: ID {cos_sim:.2f}, " f"Texture {ast_score:.1f}%, Cuts {num_edits}. " f"Fuzzy Scores: AI={mu_ai:.2f}, Spliced={mu_spliced:.2f}, Auth={mu_auth:.2f}") prompt = (f"Act as a Senior Forensic Auditor. {audit_context} " "Instruction: Justify the verdict. If identity stability is high (Auth > 0.8), " "emphasize that DNA integrity confirms human origin despite high spectral texture. " "Give a 2-line direct forensic conclusion.") res = client.chat.completions.create(model="llama-3.3-70b-versatile", messages=[{"role": "user", "content": prompt}]) reasoning = res.choices[0].message.content evidence = f"Category: {category} | ID: {cos_sim:.2f} | Texture: {ast_score:.1f}% | Cuts: {num_edits}" return verdict, evidence, reasoning except Exception as e: return "Error", str(e), "N/A"