import requests from bs4 import BeautifulSoup import os import logging from datetime import datetime import fitz from docx import Document from docx.shared import Pt, Inches from docx.enum.text import WD_ALIGN_PARAGRAPH import re from flask import Flask, request, jsonify, send_from_directory, send_file, Response, stream_with_context from werkzeug.utils import secure_filename from docx.oxml.ns import qn import subprocess import threading import json from urllib3.util.retry import Retry from requests.adapters import HTTPAdapter from http.client import HTTPConnection import markdown from bs4 import BeautifulSoup from weasyprint import HTML # Your helpers from gpt_helpers import ask_gpt41_mini from prompt_builder import build_grok_prompt, build_editor_prompt from flask_cors import CORS # ===== HTTP debug ===== HTTPConnection.debuglevel = 1 app = Flask(__name__) CORS( app, resources={r"/api/*": {"origins": [ "https://your-imatrix-domain.com", # replace with your iMatrix origin "https://www.your-imatrix-domain.com" ]}}, supports_credentials=False ) os.environ["HF_HOME"] = "/data/.huggingface" # ===== Logging ===== logging.basicConfig(level=logging.DEBUG) logger = logging.getLogger("app") logger.debug("✅ Logging initialized. Starting app setup.") # ===== File storage ===== UPLOAD_DIR = '/data/uploads' try: os.makedirs(UPLOAD_DIR, exist_ok=True, mode=0o777) os.chmod(UPLOAD_DIR, 0o777) logger.debug(f"Created/verified upload directory: {UPLOAD_DIR}") except Exception as e: logger.error(f"Failed to create upload directory {UPLOAD_DIR}: {str(e)}") uploaded_files = [] file_lock = threading.Lock() # ===== Grok API (always used) ===== GROK_API_URL = "https://api.x.ai/v1/chat/completions" GROK_API_TOKEN = "xai-5FWFmCOKosDriT1VTmh5EaQBxg00v3zOS3W10LrprJnGxMQwKQGp7TtR0pRX0ouAlUwSPv2AKznS5GLb" # Optional OpenAI key for your GPT-4.1-mini draft pass OPENAI_API_KEY = os.getenv("OPENAI_API_KEY") or os.getenv("OPENAI_API_KEY_VERDICTAI") if not OPENAI_API_KEY: logger.error("OPENAI_API_KEY or OPENAI_API_KEY_VERDICTAI not set in environment variables.") logger.debug("✅ Grok and OpenAI API endpoints and tokens set.") # ===== States ===== STATES = { "KY": "Kentucky", "SC": "South Carolina", "IN": "Indiana", "OH": "Ohio", "WV": "West Virginia", "TN": "Tennessee", # Add more as needed } # ===== Citation regex ===== CITATION_RE = re.compile( r'([A-Z][A-Za-z&.\- ]+ v\. [A-Z][A-Za-z&.\- ]+, \d+ [A-Z][A-Za-z.]*\d* \d+ \([A-Za-z. ]+ \d{4}\))' ) def extract_citations(text: str): return CITATION_RE.findall(text or "") # ===== File text extraction ===== def extract_text_from_file(file_path): try: if not os.path.exists(file_path): logger.error(f"File not found: {file_path}") return f"Error: File not found - {file_path}" if file_path.lower().endswith('.pdf'): try: doc = fitz.open(file_path) text = "" for page in doc: text += page.get_text() + "\n\n" # Add double newline for page separation doc.close() return text if text else "Error: No text extracted from PDF." except Exception as e: logging.error(f"Error extracting text from {file_path}: {str(e)}") return f"Error: Failed to extract text from {file_path} - {str(e)}" elif file_path.lower().endswith('.docx'): doc = Document(file_path) structured = [] for para in doc.paragraphs: if para.text.strip(): if para.style.name.startswith('Heading'): level = int(para.style.name[-1]) if para.style.name[-1].isdigit() else 1 structured.append(f"{'#' * level} {para.text}") else: structured.append(para.text) text = "\n".join(structured) return text if text else "Error: No text extracted from DOCX." else: try: with open(file_path, 'r', encoding='utf-8', errors='ignore') as f: t = f.read() return t if t else "Error: No text extracted from file." except Exception: return f"Error: Unsupported file type for {file_path}." except Exception as e: logger.error(f"Error extracting text from {file_path}: {str(e)}") return f"Error: Failed to extract text from {file_path} - {str(e)}" # ===== Prompt classification ===== def classify_prompt(prompt): p = (prompt or "").lower() if "irac" in p: return "irac" if any(k in p for k in ["create", "draft", "prepare", "compose", "generate", "order", "motion"]): return "document_creation" if "shepard" in p or "shepardize" in p or ("citations" in p and "check" in p): return "citation_check" if "brief" in p and any(x in p for x in ["irac", "issue", "rule", "application", "conclusion"]): return "brief_builder" if "deposition" in p or "depo" in p: return "deposition_qa" if "checklist" in p or "red flag" in p or "red flags" in p: return "checklist" if "compare" in p and any(x in p for x in ["indiana", "ohio", "federal"]): return "comparative_law" if "plain english" in p or "client summary" in p or "explain like" in p: return "client_summary" if "skeleton" in p or ("motion to" in p and "skeleton" in p): return "motion_skeleton" if "oppose" in p or "opposition" in p or "counter-argument" in p or "counterargument" in p: return "opposing_argument" if "extract cases" in p or "extract citations" in p: return "case_extractor" if "judge" in p or "county" in p or "judge-style" in p: return "judge_style" if "case law" in p or "precedent" in p: return "case_law" if "statute" in p or "regulation" in p: return "statute" if "strategy" in p or "plan" in p: return "legal_strategy" if any(x in p for x in ["summarize", "what does", "analyze", "analysis of", "research"]): return "document_analysis" return "general_qa" def _is_researchy(prompt: str, task_type: str) -> bool: p = (prompt or "").lower() research_modes = { "document_analysis", "brief_builder", "comparative_law", "citation_check", "case_law", "statute", "legal_strategy", "opposing_argument" } keywords = ["research", "case law", "precedent", "authority", "statute", "regulation", "holding", "cite", "citation"] return (task_type in research_modes) or any(k in p for k in keywords) # ===== Guardrails for research / analysis ===== def research_guardrails(): return ( "Hallucination control / sourcing rules:\n" "• Every material legal proposition must include (a) a Bluebook citation with pinpoint and (b) a working URL to a public source such as CourtListener, law.cornell.edu, or an official judiciary .gov page.\n" "• Prefer primary sources; otherwise use authoritative secondary (e.g., LII). Do not cite paywalled or unverifiable sources.\n" "• If no on-point authority is found, explicitly write: “No on-point authority found after reasonable search.”\n" "• Quote key language where relevant and attribute with precise pincites.\n" "• Do not fabricate citations, pincites, or URLs. If uncertain, say so plainly.\n" ) # ===== Grok call (resilient) ===== def ask_grok(messages, stream=False, deep_search=False, timeout=360, max_tokens=262144, retries_total=5, backoff_factor=2): try: headers = { "Accept": "application/json", "Content-Type": "application/json", "Authorization": f"Bearer {GROK_API_TOKEN}" } search_params = { "mode": "on", "deepSearch": deep_search, "trustedSources": [ "courtlistener.com", "scholar.google.com", "law.cornell.edu", "supremecourt.gov" ] } if deep_search else {} payload = { "messages": messages, "model": "grok-4-0709", "stream": stream, "temperature": 0.1, "max_tokens": max_tokens, "search_parameters": search_params } logger.debug(f"Sending Grok payload: {json.dumps(payload, indent=2)[:2000]} ...") session = requests.Session() retries = Retry( total=retries_total, backoff_factor=backoff_factor, status_forcelist=[429, 500, 502, 503, 504], allowed_methods=["POST"] ) session.mount('https://', HTTPAdapter(max_retries=retries)) response = session.post(GROK_API_URL, headers=headers, json=payload, stream=stream, timeout=timeout) logger.debug(f"Grok response status: {response.status_code}") response.raise_for_status() if stream: def stream_gen(): logger.debug("Starting Grok stream...") for raw_chunk in response.iter_lines(): if not raw_chunk: continue chunk = raw_chunk.decode("utf-8", errors="ignore").strip() if not chunk: continue if chunk == "data: [DONE]" or chunk == "[DONE]": yield "data: [DONE]\n\n" break if chunk.startswith("data: "): chunk_data = chunk[6:] else: chunk_data = chunk try: result = json.loads(chunk_data) delta = result.get("choices", [{}])[0].get("delta", {}) content = delta.get("content", "") if content: yield f'data: {{"chunk": {json.dumps(content)}}}\n\n' except Exception as e: logger.warning(f"Grok JSON parse error: {e} | raw: {chunk[:200]}") yield f'data: {{"chunk": {json.dumps(chunk)}}}\n\n' logger.debug("Stream ended.") return stream_gen() else: result = response.json() logger.debug(f"Grok non-stream result: {json.dumps(result, indent=2)[:2000]} ...") if "choices" in result and result["choices"]: msg = result["choices"][0].get("message") or {} content = msg.get("content", "") if len(content) > 65536: content = content[:65536] + "... [Truncated]" return content.strip() if content else "[No response]" logger.warning("No valid content in Grok response") return "[No response]" except requests.exceptions.HTTPError as http_err: error_msg = f"Grok HTTP error: {http_err}, Response: {response.text if 'response' in locals() else 'N/A'}" logger.error(error_msg) if stream: def error_gen(): yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return error_gen() return "[Grok Error] " + str(http_err) except Exception as e: error_msg = f"Grok general error: {type(e).__name__}: {str(e)}" logger.error(error_msg) if stream: def error_gen(): yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return error_gen() return "[Grok Error] " + str(e) # ===== DOCX building ===== def _set_doc_defaults(doc): style = doc.styles['Normal'] style.font.name = 'Times New Roman' style._element.rPr.rFonts.set(qn('w:eastAsia'), 'Times New Roman') style.font.size = Pt(12) for sec in doc.sections: sec.top_margin = Inches(1) sec.bottom_margin = Inches(1) sec.left_margin = Inches(1.25) sec.right_margin = Inches(1) def _add_caption_block(doc, state_name, action_no, petitioner="Petitioner", respondent="Respondent", title="ORDER"): p = doc.add_paragraph() p.alignment = WD_ALIGN_PARAGRAPH.CENTER lines = [ f"{state_name.upper()} CIRCUIT COURT", "", f"ACTION NO. {action_no or 'Unknown'}", "", f"{petitioner.upper()},", "PETITIONER,", "", "v.", "", f"{respondent.upper()},", "RESPONDENT.", "", title.upper(), "" ] for line in lines: run = p.add_run(line + "\n") run.bold = True if line and line.isupper() and line not in ("PETITIONER,", "RESPONDENT,", "v.") else False def create_legal_docx(content, jurisdiction, filename, task_type, state_name="Kentucky"): try: # Split content at commentary separator if '--- Commentary ---' in content: main_content, commentary = content.split('--- Commentary ---', 1) content = main_content.strip() else: commentary = "" doc = Document() _set_doc_defaults(doc) if task_type == "document_creation": motion_title = "ORDER" if "maintenance" in (content or "").lower(): motion_title = "ORDER REGARDING MAINTENANCE" _add_caption_block(doc, state_name, action_no="Unknown", petitioner="Petitioner", respondent="Respondent", title=motion_title) # Parse markdown to HTML, then to DOCX elements html = markdown.markdown(content) soup = BeautifulSoup(html, 'html.parser') for element in soup.find_all(recursive=False): if element.name == 'h1': heading = doc.add_heading(element.text, level=1) heading.alignment = WD_ALIGN_PARAGRAPH.CENTER elif element.name == 'h2': doc.add_heading(element.text, level=2) elif element.name == 'p': p = doc.add_paragraph(element.text) p.alignment = WD_ALIGN_PARAGRAPH.JUSTIFY elif element.name == 'ul' or element.name == 'ol': list_style = 'List Bullet' if element.name == 'ul' else 'List Number' for li in element.find_all('li'): doc.add_paragraph(li.text, style=list_style) elif element.name == 'blockquote': p = doc.add_paragraph(element.text) p.alignment = WD_ALIGN_PARAGRAPH.LEFT p.paragraph_format.left_indent = Inches(0.5) else: p = doc.add_paragraph(element.text or "") p.alignment = WD_ALIGN_PARAGRAPH.JUSTIFY if task_type != "general_qa": doc.add_paragraph("\nRespectfully submitted,") doc.add_paragraph("[Attorney Name]\n[Bar Number]\n[Firm]\n[Address]\n[Phone]\n[Email]") doc.add_heading("CERTIFICATE OF SERVICE", level=1) doc.add_paragraph( "I hereby certify that on [Date], a true and correct copy of the foregoing was served upon all parties via [Method].\n\n[Attorney Name]" ) else: doc.add_paragraph("\nSincerely,\n[Attorney Name]\n[Firm]\n[Address]\n[Phone]\n[Email]") doc.save(filename) logger.debug(f"Created DOCX: {filename}") return content # Return main content for preview except Exception as e: logger.error(f"Error creating DOCX {filename}: {str(e)}") return f"Error creating document: {str(e)}" # ===== Mini web check (heuristic) ===== def web_search(q: str) -> str: try: url = f"https://www.google.com/search?q={requests.utils.quote(q)}" headers = { "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.114 Safari/537.36" } response = requests.get(url, headers=headers, timeout=12) response.raise_for_status() soup = BeautifulSoup(response.text, "html.parser") results = [] for result in soup.select(".g"): title = result.select_one("h3") title = title.text if title else None snippet = result.select_one(".VwiC3b") snippet = snippet.text if snippet else "" link = result.select_one("a") link = link["href"] if link else None if title and link: results.append(f"{title}\n{link}\n{snippet}\n") logger.debug(f"Web search results for '{q}': {len(results)} results") return "\n".join(results) except Exception as e: logger.error(f"Web search error: {str(e)}") return "" # ===== Mode builders with stricter guardrails ===== def build_prompt_for_mode(mode, user_prompt, jurisdiction, state_name, combined_text): uploaded_structure = combined_text if mode == "document_creation" else "" # Use for inference base_context = ( f"You are Verdict AI, a legal AI tool. Jurisdiction: {state_name} ({jurisdiction}). " f"Act as a highly capable junior associate. Follow the user's assignment exactly. Output only the document content in markdown." ) text_note = f"\n\n### Attached document excerpts (truncated):\n{combined_text[:10000]}" if combined_text else "" mode = (mode or "").lower() researchy_modes = { "document_analysis", "brief_builder", "comparative_law", "citation_check", "case_law", "statute", "legal_strategy", "opposing_argument" } guard = ("\n\n" + research_guardrails()) if mode in researchy_modes else "" if mode == "document_creation": sys = base_context + guard + " Produce a ready-made legal document in markdown format, matching any uploaded structure." usr = f"Draft the document based on: {text_note}\n\nPrompt: {user_prompt}" return sys, usr # ... (keep other modes similar, adding uploaded_structure to sys if relevant) # At the end: sys = build_grok_prompt(user_prompt, mode or "general_qa", jurisdiction, "", uploaded_structure) + guard usr = f"{user_prompt}{text_note}" return sys, usr # ===== Pipeline ===== def route_model(messages, task_type, files, deep_search, jurisdiction, explicit_mode=None): logger.debug(f"Starting route_model with task_type: {task_type}, jurisdiction: {jurisdiction}, deep_search: {deep_search}, files: {files}, explicit_mode: {explicit_mode}") try: jurisdiction = jurisdiction if jurisdiction in STATES else "KY" state_name = STATES.get(jurisdiction, "Kentucky") rag_context = "" file_text_combined = "\n".join([extract_text_from_file(p) for p in files if p]) grounding_block = "" if file_text_combined and not file_text_combined.startswith("Error"): grounding_block = f"\n\n### Attached document excerpts (truncated):\n{file_text_combined[:15000]}" messages[-1]['content'] = f"{messages[-1]['content']}{rag_context}{grounding_block}" user_prompt_text = messages[-1]['content'] researchy = _is_researchy(user_prompt_text, task_type) auto_deep = deep_search or researchy logger.debug(f"Auto deepSearch? {auto_deep} (researchy={researchy}, ui_toggle={deep_search})") yield f'data: {{"chunk": "Step 1: Understanding the assignment..."}}\n\n' mode = explicit_mode or task_type if task_type in ["document_creation", "irac", "case_law", "statute", "legal_strategy", "general_qa"] and mode == task_type: yield f'data: {{"chunk": "Step 2: Drafting initial version..."}}\n\n' gpt_response = ask_gpt41_mini(messages[-1]['content'], jurisdiction) logger.debug(f"GPT-4.1-mini response length: {len(gpt_response)} | Content snippet: {gpt_response[:200]}...") if not gpt_response.strip() or gpt_response.startswith("[GPT-4.1-mini Error"): error_msg = f"Empty draft from GPT-4.1-mini or API error: {gpt_response}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return MAX_GPT_LEN = 32768 if len(gpt_response) > MAX_GPT_LEN: gpt_response = gpt_response[:MAX_GPT_LEN] + "\n[Truncated]" logger.warning(f"Truncated GPT response to {MAX_GPT_LEN} chars.") yield f'data: {{"chunk": "Step 3: Polishing with Grok..."}}\n\n' editor_prompt = build_editor_prompt(messages[-1]['content'], task_type, jurisdiction, gpt_response, rag_context, uploaded_structure=file_text_combined if task_type == "document_creation" else "") editor_prompt = f"{editor_prompt}{grounding_block}" system_content = build_grok_prompt(messages[-1]['content'], task_type, jurisdiction, rag_context, uploaded_structure=file_text_combined if task_type == "document_creation" else "") editor_messages = [ {'role': 'system', 'content': system_content}, {'role': 'user', 'content': messages[-1]['content']}, {'role': 'assistant', 'content': gpt_response}, {'role': 'user', 'content': editor_prompt} ] full_grok_response = ask_grok( editor_messages, stream=False, deep_search=auto_deep, timeout=360, max_tokens=262144, retries_total=5, backoff_factor=2 ) full_response = full_grok_response if (isinstance(full_grok_response, str) and not full_grok_response.startswith("[Grok Error")) else gpt_response elif mode in [ "citation_check", "brief_builder", "deposition_qa", "checklist", "comparative_law", "client_summary", "motion_skeleton", "opposing_argument", "case_extractor", "judge_style", "document_analysis" ]: yield f'data: {{"chunk": "Step 2: Researching & drafting with Grok..."}}\n\n' sys_msg, usr_msg = build_prompt_for_mode(mode, user_prompt_text, jurisdiction, state_name, file_text_combined or "") messages_mode = [{'role': 'system', 'content': sys_msg}, {'role': 'user', 'content': usr_msg}] full_response = ask_grok( messages_mode, stream=False, deep_search=True if mode in ["citation_check", "comparative_law", "document_analysis"] else auto_deep, timeout=360, max_tokens=262144, retries_total=5, backoff_factor=2 ) if isinstance(full_response, str) and full_response.startswith("[Grok Error"): error_msg = f"Grok failed: {full_response}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return else: yield f'data: {{"chunk": "Step 2: Processing with Grok..."}}\n\n' system_content = build_grok_prompt(user_prompt_text, task_type, jurisdiction, rag_context, uploaded_structure=file_text_combined if task_type == "document_creation" else "") if researchy: system_content += "\n\n" + research_guardrails() grok_messages = [{'role': 'system', 'content': system_content}] + messages full_response = ask_grok( grok_messages, stream=False, deep_search=auto_deep, timeout=360, max_tokens=262144, retries_total=5, backoff_factor=2 ) if isinstance(full_response, str) and full_response.startswith("[Grok Error"): error_msg = f"Grok failed: {full_response}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return yield f'data: {{"chunk": "Step 4: Verifying citations..."}}\n\n' citations = extract_citations(full_response or "") verified = True for citation in citations[:5]: verification_query = f"verify legal citation: {citation} full case name and details" verification_result = web_search(verification_query) if citation not in verification_result: verified = False break if researchy and (not citations or not verified): yield f'data: {{"chunk": "Step 5: Enforcing no-cite-no-claim..."}}\n\n' enforce_system = ( "You are Verdict AI. Enforce strict sourcing:\n" + research_guardrails() + "\nRewrite the draft so that every material proposition includes a Bluebook citation with pinpoint and a working public URL to an accepted source. " "If no authority is found, insert the sentence exactly: “No on-point authority found after reasonable search.” " "Do not invent citations or URLs. Preserve structure and improve clarity." ) enforce_user = f"--- DRAFT TO FIX ---\n{full_response}\n\n{grounding_block}" enforce_messages = [ {'role': 'system', 'content': enforce_system}, {'role': 'user', 'content': enforce_user} ] enforced = ask_grok( enforce_messages, stream=False, deep_search=True, timeout=360, max_tokens=262144, retries_total=5, backoff_factor=2 ) if isinstance(enforced, str) and not enforced.startswith("[Grok Error]"): full_response = enforced timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") pdf_filename = f"/tmp/legal_doc_{timestamp}.pdf" docx_filename = f"/tmp/legal_doc_{timestamp}.docx" if task_type == "document_creation" and mode == "document_creation": yield f'data: {{"chunk": "Step 6: Generating document & PDF preview..."}}\n\n' main_content = create_legal_docx(full_response, jurisdiction, docx_filename, task_type, state_name=state_name) if isinstance(main_content, str) and main_content.startswith("Error creating document"): yield f'data: {{"error": {json.dumps(main_content)}}}\n\n' yield "data: [DONE]\n\n" return try: full_docx_path = os.path.abspath(docx_filename) full_pdf_path = os.path.abspath(pdf_filename) # Use WeasyPrint for PDF conversion html_content = markdown.markdown(main_content) HTML(string=html_content).write_pdf(full_pdf_path) logger.debug(f"PDF generated using WeasyPrint: {full_pdf_path}") if os.path.exists(full_pdf_path) and os.path.getsize(full_pdf_path) > 1000: yield f'data: {{"pdf_download_url": "/download/" + os.path.basename(pdf_filename)}}\n\n' yield f'data: {{"download_url": "/download/" + os.path.basename(docx_filename)}}\n\n' else: logger.warning("PDF generation failed or empty; falling back to markdown preview.") # Fallback: Send markdown as chunk yield f'data: {{"chunk": {json.dumps(main_content)}}}\n\n' yield f'data: {{"download_url": "/download/" + os.path.basename(docx_filename)}}\n\n' except Exception as e: error_msg = f"PDF conversion failed: {str(e)}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' # Still provide DOCX download yield f'data: {{"download_url": "/download/" + os.path.basename(docx_filename)}}\n\n' # Fallback to markdown yield f'data: {{"chunk": {json.dumps(main_content)}}}\n\n' else: main_content = full_response yield f'data: {{"chunk": "Step 7: Finalizing..."}}\n\n' chunks = [main_content[i:i+200] for i in range(0, len(main_content or ""), 200)] for part in chunks: yield f'data: {{"chunk": {json.dumps(part)}}}\n\n' if task_type == "document_creation" and mode == "document_creation" and os.path.exists(pdf_filename): yield f'data: {{"download_url": "/download/legal_doc_{timestamp}.docx", "pdf_download_url": "/download/legal_doc_{timestamp}.pdf"}}\n\n' yield "data: [DONE]\n\n" except Exception as e: error_msg = f"Error in route_model: {str(e)}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" # ===== Utility flows ===== def summarize_document(files): def gen(): try: yield f'data: {{"chunk": "Step 1: Extracting text from files..."}}\n\n' texts = [extract_text_from_file(f) for f in files if f] text = "\n".join(texts) if text and not text.startswith("Error"): yield f'data: {{"chunk": "Step 2: Generating summary with Grok..."}}\n\n' summary = ask_grok([{"role": "user", "content": f"Summarize the following document(s): {text[:10000]}"}], stream=False, timeout=360) if isinstance(summary, str) and summary.startswith("[Grok Error]"): error_msg = f"Summary failed: {summary}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return full_response = f"Summary: {summary}" chunks = [full_response[i:i+200] for i in range(0, len(full_response), 200)] for part in chunks: yield f'data: {{"chunk": {json.dumps(part)}}}\n\n' else: yield f'data: {{"chunk": "No text extracted from file."}}\n\n' yield "data: [DONE]\n\n" except Exception as e: error_msg = f"Error in summarize_document: {str(e)}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return gen() def analyze_document(files): def gen(): try: yield f'data: {{"chunk": "Step 1: Extracting text from files..."}}\n\n' texts = [extract_text_from_file(f) for f in files if f] text = "\n".join(texts) if text and not text.startswith("Error"): yield f'data: {{"chunk": "Step 2: Analyzing with Grok..."}}\n\n' analysis = ask_grok([{"role": "user", "content": f"Analyze the following document(s) for legal issues, risks, or key clauses: {text[:10000]}"}], stream=False, timeout=360) if isinstance(analysis, str) and analysis.startswith("[Grok Error]"): error_msg = f"Analysis failed: {analysis}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return full_response = f"Analysis: {analysis}" chunks = [full_response[i:i+200] for i in range(0, len(full_response), 200)] for part in chunks: yield f'data: {{"chunk": {json.dumps(part)}}}\n\n' else: yield f'data: {{"chunk": "No text extracted from file."}}\n\n' yield "data: [DONE]\n\n" except Exception as e: error_msg = f"Error in analyze_document: {str(e)}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return gen() def check_issues(files): def gen(): try: yield f'data: {{"chunk": "Step 1: Extracting text from files..."}}\n\n' texts = [extract_text_from_file(f) for f in files if f] text = "\n".join(texts) if text and not text.startswith("Error"): yield f'data: {{"chunk": "Step 2: Checking issues with Grok..."}}\n\n' issues = ask_grok([{"role": "user", "content": f"Check for red flags, unusual clauses, or potential issues in this legal document(s) and highlight them: {text[:10000]}"}], stream=False, timeout=360) if isinstance(issues, str) and issues.startswith("[Grok Error]"): error_msg = f"Issues check failed: {issues}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return full_response = f"Highlighted Issues: {issues}" chunks = [full_response[i:i+200] for i in range(0, len(full_response), 200)] for part in chunks: yield f'data: {{"chunk": {json.dumps(part)}}}\n\n' else: yield f'data: {{"chunk": "No text extracted from file."}}\n\n' yield "data: [DONE]\n\n" except Exception as e: error_msg = f"Error in check_issues: {str(e)}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return gen() # ===== Flask routes ===== @app.route('/') def index(): logger.debug("Serving index.html") return send_from_directory('.', 'index.html') @app.route('/api/chat', methods=['POST']) def api_chat(): logger.info("Received request to /api/chat") def generate(): try: data = request.get_json() logger.debug(f"Received JSON payload: {json.dumps(data, indent=2)}") if not data: yield f'data: {{"error": "No JSON payload in request"}}\n\n' yield "data: [DONE]\n\n" return message = data.get('message', '') if not message: yield f'data: {{"error": "No message provided"}}\n\n' yield "data: [DONE]\n\n" return explicit_mode = data.get('mode') # optional jurisdiction = data.get('jurisdiction', 'KY') irac_mode = data.get('irac', False) deep_search = data.get('deepSearch', False) files_filenames = data.get('files', []) temp_paths = [] with file_lock: for f in uploaded_files: if f['filename'] in files_filenames: temp_paths.append(f['path']) yield f'data: {{"uploaded_files": {json.dumps(files_filenames)}}}\n\n' prompt_lower = message.lower() task_type = "irac" if irac_mode else classify_prompt(message) logger.info(f"Task type: {task_type}, Explicit mode: {explicit_mode}, Jurisdiction: {jurisdiction}, Deep search (UI): {deep_search}, Files: {files_filenames}") if "summarize" in prompt_lower and temp_paths: for chunk in summarize_document(temp_paths): yield chunk elif (("analyze" in prompt_lower) or ("what does" in prompt_lower)) and temp_paths: for chunk in analyze_document(temp_paths): yield chunk elif (("check" in prompt_lower) or ("issues" in prompt_lower) or ("highlight" in prompt_lower)) and temp_paths: for chunk in check_issues(temp_paths): yield chunk else: messages = [{'role': 'user', 'content': message}] for line in route_model(messages, task_type, temp_paths, deep_search, jurisdiction, explicit_mode=explicit_mode): yield line logger.info("Response streamed successfully.") except Exception as e: error_msg = f"Error in /api/chat: {str(e)}" logger.error(error_msg) yield f'data: {{"error": {json.dumps(error_msg)}}}\n\n' yield "data: [DONE]\n\n" return Response( stream_with_context(generate()), mimetype='text/event-stream', headers={'Cache-Control': 'no-cache', 'Connection': 'keep-alive', 'X-Accel-Buffering': 'no'} ) @app.route('/api/files', methods=['POST']) def upload_files(): logger.info("Received request to /api/files POST") try: uploaded = [] for file in request.files.getlist('files'): if file and file.filename: filename = secure_filename(file.filename) path = os.path.join(UPLOAD_DIR, filename) file.save(path) try: os.chmod(path, 0o666) logger.debug(f"Saved file: {path}") with file_lock: uploaded_files.append({ 'filename': filename, 'upload_time': datetime.now().strftime("%Y-%m-%d %H:%M:%S"), 'status': 'saved', 'path': path }) uploaded.append(filename) except Exception as e: logger.error(f"Error saving file {filename}: {str(e)}") return jsonify({'error': f"Failed to save file {filename}: {str(e)}"}), 500 logger.debug(f"Uploaded files: {uploaded}") return jsonify({'filenames': uploaded}), 200 except Exception as e: logger.error(f"Error in upload_files: {str(e)}") return jsonify({'error': str(e)}), 500 @app.route('/api/files', methods=['GET']) def list_files(): logger.info("Received request to /api/files GET") try: with file_lock: files_list = [{ 'filename': f['filename'], 'upload_time': f['upload_time'], 'status': f['status'] } for f in uploaded_files] logger.debug(f"Returning files: {files_list}") return jsonify({'files': files_list}), 200 except Exception as e: logger.error(f"Error in list_files: {str(e)}") return jsonify({'error': str(e)}), 500 @app.route('/api/files/', methods=['DELETE']) def delete_file(filename): logger.info(f"Received request to delete file: {filename}") try: with file_lock: for i, f in enumerate(uploaded_files): if f['filename'] == filename: try: os.remove(f['path']) del uploaded_files[i] logger.debug(f"Deleted file: {filename}") return jsonify({'message': f'File {filename} deleted'}), 200 except Exception as e: logger.error(f"Error deleting file {filename}: {str(e)}") return jsonify({'error': str(e)}), 500 return jsonify({'error': 'File not found'}), 404 except Exception as e: logger.error(f"Error in delete_file: {str(e)}") return jsonify({'error': str(e)}), 500 @app.route('/download/', methods=['GET']) def download(filename): """Serve PDFs inline for preview, DOCX as attachment for download.""" logger.info(f"Received request to download/view: {filename}") try: file_path = os.path.join('/tmp', filename) if not os.path.exists(file_path): logger.error(f"Download file not found: {file_path}") return jsonify({'error': f'File {filename} not found'}), 404 is_pdf = filename.lower().endswith('.pdf') # Inline preview for PDF (fixes blank issue); attachment for others if is_pdf: # Optional: guard tiny PDFs (which often means empty) size = os.path.getsize(file_path) if size < 1000: logger.warning(f"PDF seems too small for preview ({size} bytes): {filename}") return send_file( file_path, as_attachment=False, mimetype='application/pdf', download_name=filename, conditional=True, max_age=0 ) else: return send_file( file_path, as_attachment=True, download_name=filename, conditional=True, max_age=0 ) except Exception as e: logger.error(f"Error in download: {str(e)}") return jsonify({'error': str(e)}), 500 @app.route('/health', methods=['GET']) def health(): logger.debug("Health check requested") return jsonify({'status': 'healthy'}), 200 if __name__ == '__main__': app.run(host='0.0.0.0', port=7860, debug=True)