import sys import os import torch import torch.nn as nn from transformers import AutoTokenizer class SourceCodeAuthorCheck(nn.Module): def __init__(self, vocab_size=50257, d_model=128, nhead=8, num_layers=4, dim_feedforward=512): super().__init__() self.embedding = nn.Embedding(vocab_size, d_model) self.pos_encoder = nn.Parameter(torch.zeros(1, 1024, d_model)) encoder_layers = nn.TransformerEncoderLayer( d_model=d_model, nhead=nhead, dim_feedforward=dim_feedforward, batch_first=True ) self.transformer = nn.TransformerEncoder(encoder_layers, num_layers=num_layers) self.fc = nn.Linear(d_model, 1) def forward(self, input_ids, attention_mask): seq_len = input_ids.size(1) x = self.embedding(input_ids) + self.pos_encoder[:, :seq_len, :] src_key_padding_mask = ~attention_mask.bool() x = self.transformer(x, src_key_padding_mask=src_key_padding_mask) mask_expanded = attention_mask.unsqueeze(-1).float() sum_embeddings = torch.sum(x * mask_expanded, 1) sum_mask = torch.clamp(mask_expanded.sum(1), min=1e-9) pooled = sum_embeddings / sum_mask return self.fc(pooled) def predict(code_snippet, model, tokenizer, device): inputs = tokenizer( code_snippet, return_tensors="pt", truncation=True, padding="max_length", max_length=1024 ).to(device) with torch.no_grad(): with torch.autocast(device_type='cuda', dtype=torch.bfloat16): logits = model(inputs['input_ids'], inputs['attention_mask']) prob = torch.sigmoid(logits).item() return prob def main(): if len(sys.argv) < 2: print("Usage: python 4-inference_file.py ") sys.exit(1) target_file = sys.argv[1] if not os.path.exists(target_file): print(f"Error: File '{target_file}' not found.") sys.exit(1) with open(target_file, 'r', encoding='utf-8', errors='ignore') as f: code_content = f.read() if not code_content.strip(): print(f"Error: File '{target_file}' is empty.") sys.exit(1) device = torch.device("cuda" if torch.cuda.is_available() else "cpu") tokenizer = AutoTokenizer.from_pretrained("gpt2") tokenizer.pad_token = tokenizer.eos_token model = SourceCodeAuthorCheck().to(device) model.load_state_dict(torch.load("source_code_classifier.pth", map_location=device, weights_only=True)) model.eval() print(f"\n--- Testing File: {target_file} ---") prob = predict(code_content, model, tokenizer, device) score = round(prob * 100, 2) verdict = "AI Generated" if prob > 0.5 else "Human Written" print(f"Verdict: {verdict} (AI Probability: {score}%)") print(f"Preview: {code_content[:150].strip()}...\n") if __name__ == "__main__": main()