Spaces:
Running
Running
Download train_finetune.py from Slovand/detector-ai: direct link, hf CLI and curl.
- Browser
- Download file 6.1 kB
-
https://huggingface.co/spaces/Slovand/detector-ai/resolve/main/train_finetune.py
- Command line
-
hf download hf://spaces/Slovand/detector-ai/train_finetune.py
-
curl -L -o train_finetune.py https://huggingface.co/spaces/Slovand/detector-ai/resolve/main/train_finetune.py
6.1 kB
| # train_finetune.py - Fine-Tuning IndoBERT untuk Deteksi Teks AI | |
| import pandas as pd | |
| import numpy as np | |
| from sklearn.model_selection import train_test_split | |
| from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score | |
| from transformers import AutoTokenizer, AutoModelForSequenceClassification, Trainer, TrainingArguments | |
| from datasets import Dataset | |
| import torch | |
| import os | |
| print("=" * 60) | |
| print("FINE-TUNING INDOBERT UNTUK DETEKSI TEKS AI") | |
| print("Lingkungan Perguruan Tinggi") | |
| print("=" * 60) | |
| # ============================================================ | |
| # 1. LOAD DATASET HC3 (pakai library Hugging Face langsung) | |
| # ============================================================ | |
| print("\n📖 Loading HC3 dataset from Hugging Face...") | |
| from datasets import load_dataset | |
| # Load dataset HC3 (Human ChatGPT Comparison Corpus) | |
| dataset = load_dataset("Hello-SimpleAI/HC3", "all", split="train") | |
| print(f"✅ Dataset loaded: {len(dataset)} samples") | |
| # Extract text and labels | |
| texts = [] | |
| labels = [] | |
| for item in dataset: | |
| # Human answers (label 0) | |
| for human_answer in item['human_answers']: | |
| if human_answer and len(human_answer.split()) > 20: | |
| texts.append(human_answer) | |
| labels.append(0) | |
| # ChatGPT answers (label 1) | |
| for chatgpt_answer in item['chatgpt_answers']: | |
| if chatgpt_answer and len(chatgpt_answer.split()) > 20: | |
| texts.append(chatgpt_answer) | |
| labels.append(1) | |
| print(f" Human (0): {labels.count(0)}") | |
| print(f" AI (1): {labels.count(1)}") | |
| # Convert to DataFrame | |
| df = pd.DataFrame({'text': texts, 'label': labels}) | |
| # Balance dataset (ambil 3000 masing-masing) | |
| df_human = df[df['label'] == 0].sample(n=3000, random_state=42) | |
| df_ai = df[df['label'] == 1].sample(n=3000, random_state=42) | |
| df = pd.concat([df_human, df_ai]).reset_index(drop=True) | |
| print(f" Balanced: {len(df)} samples (Human: {len(df[df['label']==0])}, AI: {len(df[df['label']==1])})") | |
| # Split data | |
| X_train, X_test, y_train, y_test = train_test_split( | |
| df['text'].values, df['label'].values, test_size=0.2, random_state=42, stratify=df['label'].values | |
| ) | |
| print(f" Train: {len(X_train)}, Test: {len(X_test)}") | |
| # ============================================================ | |
| # 2. LOAD INDOBERT MODEL | |
| # ============================================================ | |
| print("\n🔧 Loading IndoBERT model...") | |
| model_name = "indobenchmark/indobert-base-p2" | |
| tokenizer = AutoTokenizer.from_pretrained(model_name) | |
| model = AutoModelForSequenceClassification.from_pretrained(model_name, num_labels=2) | |
| # ============================================================ | |
| # 3. TOKENIZE | |
| # ============================================================ | |
| print("\n📝 Tokenizing data...") | |
| def tokenize_function(examples): | |
| return tokenizer(examples['text'], padding='max_length', truncation=True, max_length=512) | |
| train_dataset = Dataset.from_dict({'text': X_train.tolist(), 'label': y_train.tolist()}) | |
| test_dataset = Dataset.from_dict({'text': X_test.tolist(), 'label': y_test.tolist()}) | |
| train_dataset = train_dataset.map(tokenize_function, batched=True) | |
| test_dataset = test_dataset.map(tokenize_function, batched=True) | |
| train_dataset.set_format('torch', columns=['input_ids', 'attention_mask', 'label']) | |
| test_dataset.set_format('torch', columns=['input_ids', 'attention_mask', 'label']) | |
| # ============================================================ | |
| # 4. TRAINING CONFIGURATION | |
| # ============================================================ | |
| print("\n⚙️ Setting up training...") | |
| training_args = TrainingArguments( | |
| output_dir='./finetuned-indobert', | |
| num_train_epochs=3, | |
| per_device_train_batch_size=8, | |
| per_device_eval_batch_size=8, | |
| learning_rate=2e-5, | |
| warmup_steps=100, | |
| weight_decay=0.01, | |
| eval_strategy="epoch", | |
| save_strategy="epoch", | |
| load_best_model_at_end=True, | |
| metric_for_best_model="accuracy", | |
| save_total_limit=2, | |
| fp16=False, | |
| report_to="none" | |
| ) | |
| def compute_metrics(eval_pred): | |
| logits, labels = eval_pred | |
| predictions = np.argmax(logits, axis=-1) | |
| accuracy = accuracy_score(labels, predictions) | |
| precision = precision_score(labels, predictions, average='binary', zero_division=0) | |
| recall = recall_score(labels, predictions, average='binary', zero_division=0) | |
| f1 = f1_score(labels, predictions, average='binary', zero_division=0) | |
| return { | |
| 'accuracy': accuracy, | |
| 'precision': precision, | |
| 'recall': recall, | |
| 'f1': f1 | |
| } | |
| # ============================================================ | |
| # 5. START TRAINING | |
| # ============================================================ | |
| print("\n🤖 Starting fine-tuning (this will take 10-20 minutes)...") | |
| trainer = Trainer( | |
| model=model, | |
| args=training_args, | |
| train_dataset=train_dataset, | |
| eval_dataset=test_dataset, | |
| compute_metrics=compute_metrics, | |
| ) | |
| trainer.train() | |
| # ============================================================ | |
| # 6. EVALUATE | |
| # ============================================================ | |
| print("\n📈 Evaluating model...") | |
| eval_results = trainer.evaluate() | |
| print("\n" + "=" * 60) | |
| print("HASIL FINE-TUNING") | |
| print("=" * 60) | |
| print(f" Accuracy: {eval_results['eval_accuracy']:.4f} ({eval_results['eval_accuracy']*100:.2f}%)") | |
| print(f" Precision: {eval_results['eval_precision']:.4f} ({eval_results['eval_precision']*100:.2f}%)") | |
| print(f" Recall: {eval_results['eval_recall']:.4f} ({eval_results['eval_recall']*100:.2f}%)") | |
| print(f" F1-Score: {eval_results['eval_f1']:.4f} ({eval_results['eval_f1']*100:.2f}%)") | |
| # ============================================================ | |
| # 7. SAVE MODEL | |
| # ============================================================ | |
| print("\n💾 Saving fine-tuned model...") | |
| os.makedirs("models/indobert-finetuned", exist_ok=True) | |
| model.save_pretrained("models/indobert-finetuned") | |
| tokenizer.save_pretrained("models/indobert-finetuned") | |
| print("✅ Model saved to models/indobert-finetuned") | |
| print("\n" + "=" * 60) | |
| print("FINE-TUNING SELESAI!") | |
| print("Model siap digunakan untuk deteksi teks AI") | |
| print("=" * 60) |