from functools import lru_cache from threading import Thread import gradio as gr import torch from transformers import AutoModelForCausalLM, AutoTokenizer, StoppingCriteria, StoppingCriteriaList, TextIteratorStreamer SPACE_TITLE = "⚖️ Vidhik AI: Sovereign Legal SLM" DEFAULT_REPO_ID = "Bhishaj/Vidhik-Llama-1B-GGU" DEFAULT_FILENAME = "Llama-3.2-1B-Instruct.Q4_K_M.gguf" TEMPERATURE = 0.0 MAX_NEW_TOKENS = 768 STOP_SEQUENCES = ["Question:", "Choose the", "(A)", "END DRAFT"] SYSTEM_PROMPT = ( "You are a Senior Advocate specializing in Indian legal notice drafting. " "You only produce the final document draft in polished formal language. " "You never add notes, internal reasoning, multiple-choice options, or follow-up questions. " "You never explain your drafting choices. " "TASK: Output the final drafted legal document in a professional 'Babu-speak' format. " "FORMAT: Use placeholders like [Name], [Date], and [Amount]. " "Begin immediately with 'LEGAL NOTICE' at the top whenever the request is for a notice." ) class StopOnSequences(StoppingCriteria): def __init__(self, tokenizer, prompt_length, stop_sequences): self.tokenizer = tokenizer self.prompt_length = prompt_length self.stop_sequences = tuple(stop_sequences) def __call__(self, input_ids, scores, **kwargs): generated_ids = input_ids[0][self.prompt_length:] generated_text = self.tokenizer.decode(generated_ids, skip_special_tokens=True) return any(sequence in generated_text for sequence in self.stop_sequences) def _prompt_to_messages(message): return [ {"role": "system", "content": SYSTEM_PROMPT}, {"role": "user", "content": message}, ] def _build_prompt(tokenizer, messages): if hasattr(tokenizer, "apply_chat_template") and tokenizer.chat_template: return tokenizer.apply_chat_template( messages, tokenize=False, add_generation_prompt=True, ) prompt_parts = [] for item in messages: role = item["role"].upper() prompt_parts.append(f"{role}: {item['content']}") prompt_parts.append("ASSISTANT:") return "\n\n".join(prompt_parts) @lru_cache(maxsize=1) def load_pipeline(): tokenizer = AutoTokenizer.from_pretrained( DEFAULT_REPO_ID, gguf_file=DEFAULT_FILENAME, ) if tokenizer.pad_token_id is None: tokenizer.pad_token = tokenizer.eos_token model = AutoModelForCausalLM.from_pretrained( DEFAULT_REPO_ID, gguf_file=DEFAULT_FILENAME, torch_dtype=torch.float32, device_map="cpu", low_cpu_mem_usage=True, ) model.eval() return tokenizer, model def _trim_stop_sequences(text): end_positions = [text.find(sequence) for sequence in STOP_SEQUENCES if sequence in text] if not end_positions: return text.strip() return text[: min(end_positions)].strip() def generate_notice(message): tokenizer, model = load_pipeline() prompt = _build_prompt(tokenizer, _prompt_to_messages(message)) inputs = tokenizer(prompt, return_tensors="pt") prompt_length = inputs["input_ids"].shape[-1] stopping_criteria = StoppingCriteriaList( [StopOnSequences(tokenizer, prompt_length, STOP_SEQUENCES)] ) streamer = TextIteratorStreamer( tokenizer, skip_prompt=True, skip_special_tokens=True, ) generation_kwargs = dict( **inputs, streamer=streamer, max_new_tokens=512, temperature=0.0, do_sample=False, stopping_criteria=stopping_criteria, pad_token_id=tokenizer.pad_token_id, eos_token_id=tokenizer.eos_token_id, ) def _run_generation(): with torch.inference_mode(): model.generate(**generation_kwargs) thread = Thread(target=_run_generation) thread.start() output_text = "" for new_text in streamer: output_text += new_text trimmed_text = _trim_stop_sequences(output_text) yield trimmed_text if trimmed_text != output_text: break custom_css = """ body { background: radial-gradient(circle at top, rgba(245, 219, 161, 0.22), transparent 40%), linear-gradient(135deg, #f7f1e3 0%, #efe2c1 100%); } #vidhik-shell { max-width: 980px; margin: 0 auto; } #vidhik-hero { padding: 1.2rem 0 0.4rem 0; } #vidhik-hero h1 { font-size: clamp(2rem, 3vw, 3rem); margin-bottom: 0.2rem; } #vidhik-hero p { font-size: 1rem; color: #43331f; } footer { visibility: hidden; } """ with gr.Blocks(css=custom_css) as demo: with gr.Column(elem_id="vidhik-shell"): gr.Markdown( """
Indo-specialized legal notice drafting assistant powered by a compact GGUF model served with Transformers.