Spaces:
Sleeping
Sleeping
Download detector.py from Slovand/detector-ai: direct link, hf CLI and curl.
- Browser
- Download file 4.57 kB
-
https://huggingface.co/spaces/Slovand/detector-ai/resolve/main/detector.py
- Command line
-
hf download hf://spaces/Slovand/detector-ai/detector.py
-
curl -L -o detector.py https://huggingface.co/spaces/Slovand/detector-ai/resolve/main/detector.py
4.57 kB
| # detector.py - IndoBERT untuk Deteksi Teks AI | |
| import torch | |
| from transformers import AutoTokenizer, AutoModelForSequenceClassification | |
| import re | |
| import pdfplumber | |
| import docx | |
| import os | |
| class AIDetector: | |
| def __init__(self): | |
| self.model = None | |
| self.tokenizer = None | |
| self.device = torch.device('cuda' if torch.cuda.is_available() else 'cpu') | |
| self.load_model() | |
| def load_model(self): | |
| try: | |
| model_name = "indobenchmark/indobert-base-p2" | |
| print("📥 Loading IndoBERT model...") | |
| self.tokenizer = AutoTokenizer.from_pretrained(model_name) | |
| self.model = AutoModelForSequenceClassification.from_pretrained( | |
| model_name, | |
| num_labels=2 | |
| ) | |
| self.model.to(self.device) | |
| self.model.eval() | |
| print(f"✅ Model loaded on {self.device}") | |
| return True | |
| except Exception as e: | |
| print(f"❌ Error loading model: {e}") | |
| return False | |
| def clean_text(self, text): | |
| if not isinstance(text, str): | |
| return "" | |
| text = text.replace('\n', ' ').replace('\r', ' ') | |
| text = re.sub(r'http\S+|www.\S+', '', text) | |
| text = re.sub(r'[^a-zA-Z\s\.\,\!\?\-]', '', text) | |
| text = re.sub(r'\s+', ' ', text).strip() | |
| return text | |
| def detect(self, text): | |
| if not text or text.strip() == "": | |
| return { | |
| "is_ai": False, | |
| "ai_probability": 0, | |
| "human_probability": 0, | |
| "confidence": 0, | |
| "error": "Text is empty" | |
| } | |
| if self.model is None: | |
| return { | |
| "is_ai": False, | |
| "ai_probability": 0, | |
| "human_probability": 0, | |
| "confidence": 0, | |
| "error": "Model not loaded" | |
| } | |
| cleaned = self.clean_text(text) | |
| if len(cleaned.split()) < 5: | |
| return { | |
| "is_ai": False, | |
| "ai_probability": 0, | |
| "human_probability": 0, | |
| "confidence": 0, | |
| "error": "Text terlalu pendek (minimal 5 kata)" | |
| } | |
| inputs = self.tokenizer( | |
| cleaned, | |
| return_tensors="pt", | |
| truncation=True, | |
| max_length=512, | |
| padding=True | |
| ) | |
| inputs = {k: v.to(self.device) for k, v in inputs.items()} | |
| with torch.no_grad(): | |
| outputs = self.model(**inputs) | |
| probs = torch.nn.functional.softmax(outputs.logits, dim=-1) | |
| pred = torch.argmax(probs, dim=-1).item() | |
| ai_prob = probs[0][1].item() * 100 | |
| human_prob = probs[0][0].item() * 100 | |
| return { | |
| "is_ai": bool(pred == 1), | |
| "ai_probability": round(ai_prob, 2), | |
| "human_probability": round(human_prob, 2), | |
| "confidence": round(max(ai_prob, human_prob), 2), | |
| "error": None | |
| } | |
| def detect_file(self, file_path, file_type): | |
| if file_type == "pdf": | |
| text = self.extract_from_pdf(file_path) | |
| elif file_type == "txt": | |
| text = self.extract_from_txt(file_path) | |
| elif file_type == "docx": | |
| text = self.extract_from_docx(file_path) | |
| else: | |
| return {"error": f"Unsupported file type: {file_type}"} | |
| if not text: | |
| return {"error": "Could not extract text from file"} | |
| result = self.detect(text) | |
| result["extracted_text_length"] = len(text) | |
| result["preview"] = text[:500] + "..." if len(text) > 500 else text | |
| return result | |
| def extract_from_pdf(self, file_path): | |
| text = "" | |
| try: | |
| with pdfplumber.open(file_path) as pdf: | |
| for page in pdf.pages: | |
| page_text = page.extract_text() | |
| if page_text: | |
| text += page_text + "\n" | |
| except: | |
| pass | |
| return text | |
| def extract_from_txt(self, file_path): | |
| try: | |
| with open(file_path, 'r', encoding='utf-8') as f: | |
| return f.read() | |
| except: | |
| return "" | |
| def extract_from_docx(self, file_path): | |
| text = "" | |
| try: | |
| doc = docx.Document(file_path) | |
| for para in doc.paragraphs: | |
| text += para.text + "\n" | |
| except: | |
| pass | |
| return text |