import streamlit as st from huggingface_hub import hf_hub_download import os # --- CONFIGURATION --- MODEL_REPO = "bharatgenai/Param2-17B-A2.4B-Thinking" # Note: Replace this with the actual GGUF repo once it is confirmed live GGUF_REPO = "DarkWolfX/Param2-17B-A2.4B-Thinking-GGUF" GGUF_FILE = "param2-thinking-q4_k_m.gguf" # --- UI SETUP --- st.set_page_config(page_title="Param2 Multilingual Chat", layout="centered") st.title("🇮🇳 Param2-17B Thinking Chatbot") # Language Selection Dropdown languages = [ "English", "Hindi", "Assamese", "Bengali", "Bodo", "Dogri", "Gujarati", "Kannada", "Konkani", "Kashmiri", "Maithili", "Malayalam", "Manipuri", "Marathi", "Nepali", "Oriya", "Punjabi", "Sanskrit", "Santali", "Sindhi", "Tamil", "Telugu", "Urdu" ] selected_lang = st.selectbox("Select Response Language:", languages) # --- MODEL LOADING LOGIC --- @st.cache_resource def load_model(): try: # 1. Check if GGUF exists and download model_path = hf_hub_download(repo_id=GGUF_REPO, filename=GGUF_FILE) # 2. Initialize llama-cpp (Optimized for CPU) from llama_cpp import Llama llm = Llama( model_path=model_path, n_ctx=4096, n_threads=8, # Optimized for your 8 vCPU Space ) return llm except Exception as e: return f"Error: GGUF model not found or incompatible. {str(e)}" # Attempt to load llm = load_model() if isinstance(llm, str): st.error(llm) st.info("The GGUF version of this model might not be available yet. Please check back later!") else: # --- CHAT INTERFACE --- if "messages" not in st.session_state: st.session_state.messages = [] for message in st.session_state.messages: with st.chat_message(message["role"]): st.markdown(message["content"]) if prompt := st.chat_input("Ask something..."): st.session_state.messages.append({"role": "user", "content": prompt}) with st.chat_message("user"): st.markdown(prompt) with st.chat_message("assistant"): # System Prompt Injection for Language system_instruction = f"You are a helpful assistant. You must respond ONLY in {selected_lang}." full_prompt = f"<|system|>\n{system_instruction}\n<|user|>\n{prompt}\n<|assistant|>\n" # Generate response response_container = st.empty() full_response = "" # Stream the response for a better UI feel for chunk in llm(full_prompt, max_tokens=1024, stream=True): text = chunk["choices"][0]["text"] full_response += text response_container.markdown(full_response + "▌") response_container.markdown(full_response) st.session_state.messages.append({"role": "assistant", "content": full_response})