detector-ai / detector.py
Slovand's picture
Update detector.py
3f92e11 verified
Raw History Blame Contribute Delete
4.57 kB
# detector.py - IndoBERT untuk Deteksi Teks AI
import torch
from transformers import AutoTokenizer, AutoModelForSequenceClassification
import re
import pdfplumber
import docx
import os
class AIDetector:
def __init__(self):
self.model = None
self.tokenizer = None
self.device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
self.load_model()
def load_model(self):
try:
model_name = "indobenchmark/indobert-base-p2"
print("📥 Loading IndoBERT model...")
self.tokenizer = AutoTokenizer.from_pretrained(model_name)
self.model = AutoModelForSequenceClassification.from_pretrained(
model_name,
num_labels=2
)
self.model.to(self.device)
self.model.eval()
print(f"✅ Model loaded on {self.device}")
return True
except Exception as e:
print(f"❌ Error loading model: {e}")
return False
def clean_text(self, text):
if not isinstance(text, str):
return ""
text = text.replace('\n', ' ').replace('\r', ' ')
text = re.sub(r'http\S+|www.\S+', '', text)
text = re.sub(r'[^a-zA-Z\s\.\,\!\?\-]', '', text)
text = re.sub(r'\s+', ' ', text).strip()
return text
def detect(self, text):
if not text or text.strip() == "":
return {
"is_ai": False,
"ai_probability": 0,
"human_probability": 0,
"confidence": 0,
"error": "Text is empty"
}
if self.model is None:
return {
"is_ai": False,
"ai_probability": 0,
"human_probability": 0,
"confidence": 0,
"error": "Model not loaded"
}
cleaned = self.clean_text(text)
if len(cleaned.split()) < 5:
return {
"is_ai": False,
"ai_probability": 0,
"human_probability": 0,
"confidence": 0,
"error": "Text terlalu pendek (minimal 5 kata)"
}
inputs = self.tokenizer(
cleaned,
return_tensors="pt",
truncation=True,
max_length=512,
padding=True
)
inputs = {k: v.to(self.device) for k, v in inputs.items()}
with torch.no_grad():
outputs = self.model(**inputs)
probs = torch.nn.functional.softmax(outputs.logits, dim=-1)
pred = torch.argmax(probs, dim=-1).item()
ai_prob = probs[0][1].item() * 100
human_prob = probs[0][0].item() * 100
return {
"is_ai": bool(pred == 1),
"ai_probability": round(ai_prob, 2),
"human_probability": round(human_prob, 2),
"confidence": round(max(ai_prob, human_prob), 2),
"error": None
}
def detect_file(self, file_path, file_type):
if file_type == "pdf":
text = self.extract_from_pdf(file_path)
elif file_type == "txt":
text = self.extract_from_txt(file_path)
elif file_type == "docx":
text = self.extract_from_docx(file_path)
else:
return {"error": f"Unsupported file type: {file_type}"}
if not text:
return {"error": "Could not extract text from file"}
result = self.detect(text)
result["extracted_text_length"] = len(text)
result["preview"] = text[:500] + "..." if len(text) > 500 else text
return result
def extract_from_pdf(self, file_path):
text = ""
try:
with pdfplumber.open(file_path) as pdf:
for page in pdf.pages:
page_text = page.extract_text()
if page_text:
text += page_text + "\n"
except:
pass
return text
def extract_from_txt(self, file_path):
try:
with open(file_path, 'r', encoding='utf-8') as f:
return f.read()
except:
return ""
def extract_from_docx(self, file_path):
text = ""
try:
doc = docx.Document(file_path)
for para in doc.paragraphs:
text += para.text + "\n"
except:
pass
return text