Text Generation
Transformers
Safetensors
Trellis
English
Chinese
glm_moe_dsa
glm
exl3
vllm
blackwell
mixture-of-experts
conversational
modelopt
Instructions to use brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw") model = AutoModelForCausalLM.from_pretrained("brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=256) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Trellis
How to use brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw with Trellis:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw
- SGLang
How to use brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw with Docker Model Runner:
docker model run hf.co/brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw
Download independent-eval/rerun_errors.py from brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw: direct link, hf CLI and curl.
- Browser
- Download file 3.44 kB
-
https://huggingface.co/brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw/resolve/main/independent-eval/rerun_errors.py
- Command line
-
hf download hf://brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw/independent-eval/rerun_errors.py
-
curl -L -o rerun_errors.py https://huggingface.co/brandonmusic/GLM-5.2-EXL3-TR3-3.0bpw/resolve/main/independent-eval/rerun_errors.py
3.44 kB
| #!/usr/bin/env python3 | |
| """Re-run samples that died with exhausted retries (vLLM crash window) and merge | |
| corrected records + summary. Works for both mathbench and gpqa samples files.""" | |
| import argparse | |
| import asyncio | |
| import json | |
| import sys | |
| from pathlib import Path | |
| sys.path.insert(0, str(Path(__file__).parent)) | |
| async def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--samples", required=True, help="path to *_samples.jsonl") | |
| ap.add_argument("--kind", choices=["math", "gpqa"], required=True) | |
| ap.add_argument("--dataset", help="HF dataset name (math kind)", default=None) | |
| ap.add_argument("--concurrency", type=int, default=6) | |
| args = ap.parse_args() | |
| path = Path(args.samples) | |
| records = [json.loads(l) for l in path.open()] | |
| bad = [r for r in records if str(r.get("finish_reason", "")).startswith("error")] | |
| print(f"{path.name}: {len(records)} records, {len(bad)} error records to re-run") | |
| if not bad: | |
| return | |
| import time | |
| from openai import AsyncOpenAI | |
| client = AsyncOpenAI(base_url="http://localhost:8000/v1", api_key="dummy", timeout=10800.0, max_retries=0) | |
| sem = asyncio.Semaphore(args.concurrency) | |
| results = [] | |
| t0 = time.monotonic() | |
| if args.kind == "math": | |
| from datasets import load_dataset | |
| import mathbench as mb | |
| ds = load_dataset(args.dataset) | |
| items = {str(i["problem_idx"]): i for i in ds[list(ds.keys())[0]]} | |
| total = len(bad) | |
| await asyncio.gather(*[ | |
| mb.run_one(client, sem, "GLM-5.2", items[str(r["problem_idx"])], r["repeat"], 163840, results, t0, total) | |
| for r in bad | |
| ]) | |
| key = lambda r: (str(r["problem_idx"]), r["repeat"]) | |
| else: | |
| from datasets import load_dataset | |
| import gpqa_bench as gb | |
| ds = load_dataset("Idavidrein/gpqa", "gpqa_diamond") | |
| items = list(ds[list(ds.keys())[0]]) | |
| total = len(bad) | |
| await asyncio.gather(*[ | |
| gb.run_one(client, sem, "GLM-5.2", items[r["idx"]], r["idx"], r["repeat"], 131072, results, t0, total) | |
| for r in bad | |
| ]) | |
| key = lambda r: (r["idx"], r["repeat"]) | |
| fixed = {key(r): r for r in results} | |
| merged = [fixed.get(key(r), r) for r in records] | |
| still_bad = sum(1 for r in merged if str(r.get("finish_reason", "")).startswith("error")) | |
| backup = path.with_suffix(".jsonl.pre-fixup") | |
| path.rename(backup) | |
| with path.open("w") as f: | |
| for r in merged: | |
| f.write(json.dumps(r) + "\n") | |
| n = len(merged) | |
| acc = sum(r["correct"] for r in merged) / n | |
| print(f"MERGED: {n} records, accuracy_pass_at_1={acc:.4f}, still_errored={still_bad}") | |
| # patch summary file if present | |
| sp = path.parent / path.name.replace("_samples.jsonl", "_summary.json") | |
| if sp.exists(): | |
| s = json.loads(sp.read_text()) | |
| s["accuracy_pass_at_1"] = round(acc, 4) | |
| s["errors"] = still_bad | |
| s["crash_fixup"] = f"re-ran {len(bad)} samples killed by vLLM crash" | |
| per_q = {} | |
| idx_field = "problem_idx" if args.kind == "math" else "idx" | |
| for r in merged: | |
| per_q.setdefault(str(r[idx_field]), []).append(r["correct"]) | |
| s["per_question_correct_rate"] = {k: round(sum(v) / len(v), 3) for k, v in sorted(per_q.items())} | |
| sp.write_text(json.dumps(s, indent=2)) | |
| print(f"summary updated: {sp}") | |
| if __name__ == "__main__": | |
| asyncio.run(main()) | |