Spaces:
Running on Zero
Running on Zero
Upload folder using huggingface_hub
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +1 -0
- app.py +163 -0
- backend/__pycache__/main.cpython-313.pyc +0 -0
- backend/dataset_generator.py +97 -0
- backend/dataset_generator_distilled.py +112 -0
- backend/dataset_generator_prod.py +134 -0
- backend/llama.cpp/.clang-format +171 -0
- backend/llama.cpp/.clang-tidy +28 -0
- backend/llama.cpp/.devops/cann.Dockerfile +163 -0
- backend/llama.cpp/.devops/cpu.Dockerfile +124 -0
- backend/llama.cpp/.devops/cuda.Dockerfile +133 -0
- backend/llama.cpp/.devops/intel.Dockerfile +162 -0
- backend/llama.cpp/.devops/llama-cli-cann.Dockerfile +62 -0
- backend/llama.cpp/.devops/llama-cpp-cuda.srpm.spec +85 -0
- backend/llama.cpp/.devops/llama-cpp.srpm.spec +87 -0
- backend/llama.cpp/.devops/musa.Dockerfile +135 -0
- backend/llama.cpp/.devops/nix/apps.nix +21 -0
- backend/llama.cpp/.devops/nix/devshells.nix +52 -0
- backend/llama.cpp/.devops/nix/docker.nix +37 -0
- backend/llama.cpp/.devops/nix/jetson-support.nix +39 -0
- backend/llama.cpp/.devops/nix/nixpkgs-instances.nix +45 -0
- backend/llama.cpp/.devops/nix/package-gguf-py.nix +38 -0
- backend/llama.cpp/.devops/nix/package.nix +275 -0
- backend/llama.cpp/.devops/nix/python-scripts.nix +66 -0
- backend/llama.cpp/.devops/nix/scope.nix +35 -0
- backend/llama.cpp/.devops/nix/sif.nix +27 -0
- backend/llama.cpp/.devops/openvino.Dockerfile +234 -0
- backend/llama.cpp/.devops/rocm.Dockerfile +147 -0
- backend/llama.cpp/.devops/s390x.Dockerfile +145 -0
- backend/llama.cpp/.devops/tools.sh +53 -0
- backend/llama.cpp/.devops/vulkan.Dockerfile +127 -0
- backend/llama.cpp/.devops/zendnn.Dockerfile +117 -0
- backend/llama.cpp/.dockerignore +22 -0
- backend/llama.cpp/.ecrc +6 -0
- backend/llama.cpp/.editorconfig +62 -0
- backend/llama.cpp/.flake8 +18 -0
- backend/llama.cpp/.gemini/settings.json +1 -0
- backend/llama.cpp/.github/ISSUE_TEMPLATE/010-bug-compilation.yml +90 -0
- backend/llama.cpp/.github/ISSUE_TEMPLATE/011-bug-results.yml +117 -0
- backend/llama.cpp/.github/ISSUE_TEMPLATE/019-bug-misc.yml +105 -0
- backend/llama.cpp/.github/ISSUE_TEMPLATE/020-enhancement.yml +53 -0
- backend/llama.cpp/.github/ISSUE_TEMPLATE/030-research.yml +54 -0
- backend/llama.cpp/.github/ISSUE_TEMPLATE/040-refactor.yml +30 -0
- backend/llama.cpp/.github/ISSUE_TEMPLATE/config.yml +11 -0
- backend/llama.cpp/.github/actions/ccache-clear/action.yml +22 -0
- backend/llama.cpp/.github/actions/get-tag-name/action.yml +22 -0
- backend/llama.cpp/.github/actions/install-exe/action.yml +36 -0
- backend/llama.cpp/.github/actions/linux-setup-openvino/action.yml +25 -0
- backend/llama.cpp/.github/actions/linux-setup-spacemit/action.yml +20 -0
- backend/llama.cpp/.github/actions/linux-setup-vulkan/action.yml +20 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
backend/llama.cpp/conversion/__pycache__/base.cpython-313.pyc filter=lfs diff=lfs merge=lfs -text
|
app.py
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from fastapi import FastAPI, HTTPException
|
| 2 |
+
from fastapi.middleware.cors import CORSMiddleware
|
| 3 |
+
from pydantic import BaseModel
|
| 4 |
+
from fastapi.responses import FileResponse
|
| 5 |
+
from fastapi.staticfiles import StaticFiles
|
| 6 |
+
import os
|
| 7 |
+
import re
|
| 8 |
+
|
| 9 |
+
try:
|
| 10 |
+
from llama_cpp import Llama
|
| 11 |
+
USE_LLAMA_CPP = True
|
| 12 |
+
print("llama-cpp-python is installed. Will use GGUF inference.")
|
| 13 |
+
except ImportError:
|
| 14 |
+
USE_LLAMA_CPP = False
|
| 15 |
+
import torch
|
| 16 |
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
| 17 |
+
print("llama-cpp-python is not installed. Falling back to transformers inference.")
|
| 18 |
+
|
| 19 |
+
app = FastAPI()
|
| 20 |
+
|
| 21 |
+
# CORS configuration to allow the frontend to communicate with this API
|
| 22 |
+
app.add_middleware(
|
| 23 |
+
CORSMiddleware,
|
| 24 |
+
allow_origins=["*"], # Allows all origins for local development
|
| 25 |
+
allow_credentials=True,
|
| 26 |
+
allow_methods=["*"], # Allows all methods
|
| 27 |
+
allow_headers=["*"], # Allows all headers
|
| 28 |
+
)
|
| 29 |
+
|
| 30 |
+
class TextRequest(BaseModel):
|
| 31 |
+
text: str
|
| 32 |
+
|
| 33 |
+
class MapResponse(BaseModel):
|
| 34 |
+
markdown: str
|
| 35 |
+
|
| 36 |
+
# ---------------------------------------------------------
|
| 37 |
+
# AI Model Initialization (Phase 5)
|
| 38 |
+
# ---------------------------------------------------------
|
| 39 |
+
|
| 40 |
+
if USE_LLAMA_CPP:
|
| 41 |
+
GGUF_PATH = "./production_mindmap_model.gguf"
|
| 42 |
+
print("Loading AI Model (GGUF) into memory...")
|
| 43 |
+
model = Llama(
|
| 44 |
+
model_path=GGUF_PATH,
|
| 45 |
+
n_ctx=2048,
|
| 46 |
+
n_gpu_layers=-1 # Use all GPU layers if available
|
| 47 |
+
)
|
| 48 |
+
print("AIモデル(GGUF)の起動完了!")
|
| 49 |
+
else:
|
| 50 |
+
MERGED_MODEL_PATH = "./production_mindmap_model_merged"
|
| 51 |
+
print("Loading AI Model (Transformers FP16) into memory...")
|
| 52 |
+
tokenizer = AutoTokenizer.from_pretrained(MERGED_MODEL_PATH)
|
| 53 |
+
if tokenizer.pad_token is None:
|
| 54 |
+
tokenizer.pad_token = tokenizer.eos_token
|
| 55 |
+
|
| 56 |
+
model = AutoModelForCausalLM.from_pretrained(
|
| 57 |
+
MERGED_MODEL_PATH,
|
| 58 |
+
torch_dtype=torch.float16,
|
| 59 |
+
device_map="auto"
|
| 60 |
+
)
|
| 61 |
+
model.eval()
|
| 62 |
+
print("AIモデル(Transformers)の起動完了!")
|
| 63 |
+
|
| 64 |
+
STRICT_SYSTEM_PROMPT = """あなたは極めて優秀で厳密な情報抽出アシスタントです。入力文章の論理構造を正確に読み取り、Markdown形式の目次(マインドマップ)を出力してください。
|
| 65 |
+
|
| 66 |
+
【厳格なルール】
|
| 67 |
+
1. 企業名、人物名、数値などの固有名詞や事実関係は、入力文章から「正確に」抽出すること。絶対に捏造したり、他の主語と混同したりしてはなりません。
|
| 68 |
+
2. ただし、情報を階層化して分かりやすく整理するための「一般的なカテゴリ名(例:背景、特徴、課題、概要など)」を親見出しとして補足することは許可します。
|
| 69 |
+
3. あなたの推測や外部知識を一切混ぜず、入力文章内の事実のみを構造化してください。"""
|
| 70 |
+
|
| 71 |
+
@app.post("/generate", response_model=MapResponse)
|
| 72 |
+
def generate_mindmap(request: TextRequest):
|
| 73 |
+
input_text = request.text.strip()
|
| 74 |
+
if not input_text:
|
| 75 |
+
raise HTTPException(status_code=400, detail="Text is empty")
|
| 76 |
+
|
| 77 |
+
print(f"APIリクエストを受信しました(文字数: {len(input_text)}文字)")
|
| 78 |
+
print("AIが推論(Markdown構造)を生成中...")
|
| 79 |
+
|
| 80 |
+
# Qwen用のチャットプロンプト構築
|
| 81 |
+
USER_PROMPT = f"""以下の文章から論理構造を抽出し、Markdown形式の目次(マインドマップ)を出力してください。
|
| 82 |
+
|
| 83 |
+
【出力時の厳守ルール(違反厳禁)】
|
| 84 |
+
1. 否定表現の厳守:「〜しない」「過度に依存しない」などの否定表現を絶対に見落とさず、意味を逆転させないこと。
|
| 85 |
+
2. 創作の禁止:記事に明記されていない具体的な行動や予定(例:「〜への参加」「〜の強化を目指す」など)を勝手に推測して付け足さないこと。
|
| 86 |
+
3. 事実の完全一致:抽出した内容が、元の文章の事実と完全に一致していることのみを出力すること。
|
| 87 |
+
|
| 88 |
+
入力文章:
|
| 89 |
+
{input_text}"""
|
| 90 |
+
messages = [
|
| 91 |
+
{"role": "system", "content": STRICT_SYSTEM_PROMPT},
|
| 92 |
+
{"role": "user", "content": USER_PROMPT}
|
| 93 |
+
]
|
| 94 |
+
|
| 95 |
+
if USE_LLAMA_CPP:
|
| 96 |
+
response = model.create_chat_completion(
|
| 97 |
+
messages=messages,
|
| 98 |
+
max_tokens=1024,
|
| 99 |
+
temperature=0.0, # Greedy Decoding (推測・創造を完全に排除)
|
| 100 |
+
repeat_penalty=1.1
|
| 101 |
+
)
|
| 102 |
+
generated_markdown = response['choices'][0]['message']['content'].strip()
|
| 103 |
+
else:
|
| 104 |
+
# トークナイザーでプロンプト化
|
| 105 |
+
prompt = tokenizer.apply_chat_template(
|
| 106 |
+
messages,
|
| 107 |
+
tokenize=False,
|
| 108 |
+
add_generation_prompt=True
|
| 109 |
+
)
|
| 110 |
+
|
| 111 |
+
inputs = tokenizer(prompt, return_tensors="pt").to(model.device)
|
| 112 |
+
|
| 113 |
+
with torch.no_grad():
|
| 114 |
+
outputs = model.generate(
|
| 115 |
+
**inputs,
|
| 116 |
+
max_new_tokens=1024, # 新しく生成するトークン数の上限
|
| 117 |
+
do_sample=False, # Greedy Decoding を有��化
|
| 118 |
+
repetition_penalty=1.1,
|
| 119 |
+
pad_token_id=tokenizer.pad_token_id,
|
| 120 |
+
eos_token_id=tokenizer.eos_token_id
|
| 121 |
+
)
|
| 122 |
+
|
| 123 |
+
# 4. 入力プロンプト部分を切り落として、新しく生成された部分だけを取り出す
|
| 124 |
+
input_length = inputs["input_ids"].shape[1]
|
| 125 |
+
generated_tokens = outputs[0][input_length:]
|
| 126 |
+
|
| 127 |
+
# 5. 数値(トークン)を人間が読めるMarkdownテキストにデコード
|
| 128 |
+
generated_markdown = tokenizer.decode(generated_tokens, skip_special_tokens=True).strip()
|
| 129 |
+
|
| 130 |
+
# 【超重要】T5トークナイザーが改行(\n)を消してしまう問題の対策
|
| 131 |
+
# 見出し記号(#)の前に強制的に改行を挿入して、Markmapが認識できるようにする
|
| 132 |
+
generated_markdown = re.sub(r'\s*(#+ )', r'\n\1', generated_markdown).strip()
|
| 133 |
+
|
| 134 |
+
# Markmapは必ず「#(大見出し)」から始まる必要があるため、存在しない場合は強制付与
|
| 135 |
+
if not generated_markdown.startswith('#'):
|
| 136 |
+
if '##' in generated_markdown or '###' in generated_markdown:
|
| 137 |
+
generated_markdown = "# マインドマップ\n" + generated_markdown
|
| 138 |
+
else:
|
| 139 |
+
generated_markdown = "# マインドマップ\n## 抽出結果\n- " + generated_markdown.replace('\n', '\n- ')
|
| 140 |
+
|
| 141 |
+
print("生成が完了しました! (コンソール出力時の文字化けエラーを防ぐため内容の表示を省略します)")
|
| 142 |
+
|
| 143 |
+
return MapResponse(markdown=generated_markdown)
|
| 144 |
+
|
| 145 |
+
# フロントエンドの静的ファイルをマウント
|
| 146 |
+
app.mount("/assets", StaticFiles(directory="frontend"), name="assets")
|
| 147 |
+
|
| 148 |
+
@app.get("/")
|
| 149 |
+
async def root():
|
| 150 |
+
# ルートURLにアクセスされたらフロントエンドのindex.htmlを返す
|
| 151 |
+
return FileResponse("frontend/index.html")
|
| 152 |
+
|
| 153 |
+
@app.get("/{filename}")
|
| 154 |
+
async def get_frontend_file(filename: str):
|
| 155 |
+
# app.js や style.css などのファイルを返す
|
| 156 |
+
file_path = os.path.join("frontend", filename)
|
| 157 |
+
if os.path.exists(file_path):
|
| 158 |
+
return FileResponse(file_path)
|
| 159 |
+
raise HTTPException(status_code=404, detail="File not found")
|
| 160 |
+
|
| 161 |
+
if __name__ == "__main__":
|
| 162 |
+
import uvicorn
|
| 163 |
+
uvicorn.run(app, host="0.0.0.0", port=7860)
|
backend/__pycache__/main.cpython-313.pyc
ADDED
|
Binary file (7.01 kB). View file
|
|
|
backend/dataset_generator.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import wikipediaapi
|
| 2 |
+
import json
|
| 3 |
+
import time
|
| 4 |
+
from tqdm import tqdm
|
| 5 |
+
|
| 6 |
+
def extract_structure(sections, level=1):
|
| 7 |
+
"""
|
| 8 |
+
再帰的にWikipediaのセクション(目次とテキスト)を抽出し、
|
| 9 |
+
正解データ(Markdown)と入力データ(ベタ打ち長文)のペアを作成する。
|
| 10 |
+
"""
|
| 11 |
+
markdown_output = ""
|
| 12 |
+
flat_input = ""
|
| 13 |
+
|
| 14 |
+
for s in sections:
|
| 15 |
+
# 見出しを除外する設定などもあるが、基本的に見出し自体は不要(役に立たないもの)もあるため除外も検討できる
|
| 16 |
+
# 参考文献や関連項目などは除外
|
| 17 |
+
if s.title in ["参考文献", "関連項目", "外部リンク", "脚注", "出典"]:
|
| 18 |
+
continue
|
| 19 |
+
|
| 20 |
+
# Markdownの構造データ(Output用)
|
| 21 |
+
# H1, H2, H3 などの見出しレベルを付与
|
| 22 |
+
markdown_output += f"{'#' * level} {s.title}\n"
|
| 23 |
+
|
| 24 |
+
# 概要を箇条書きで追加(各セクションの最初の1文などを要約として入れることも可能)
|
| 25 |
+
# ここではシンプルに見出しのみ、または少しのテキストを入れる
|
| 26 |
+
if s.text:
|
| 27 |
+
first_sentence = s.text.split("。")[0] + "。"
|
| 28 |
+
if len(first_sentence) > 5:
|
| 29 |
+
markdown_output += f"- {first_sentence}\n"
|
| 30 |
+
|
| 31 |
+
# ベタ打ち長文(Input用)
|
| 32 |
+
# 見出し(s.title)を隠して、テキストだけを連結する
|
| 33 |
+
if s.text:
|
| 34 |
+
flat_input += s.text + "\n"
|
| 35 |
+
|
| 36 |
+
# サブセクションを再帰的に処理
|
| 37 |
+
child_md, child_flat = extract_structure(s.sections, level + 1)
|
| 38 |
+
markdown_output += child_md
|
| 39 |
+
flat_input += child_flat
|
| 40 |
+
|
| 41 |
+
return markdown_output, flat_input
|
| 42 |
+
|
| 43 |
+
def generate_dataset(num_pages=10, output_file="mindmap_dataset.jsonl"):
|
| 44 |
+
# Wikipedia APIの設定 (日本語)
|
| 45 |
+
wiki = wikipediaapi.Wikipedia('MindMapStudio/1.0', 'ja')
|
| 46 |
+
|
| 47 |
+
print(f"Generating dataset from {num_pages} random Wikipedia articles...")
|
| 48 |
+
|
| 49 |
+
# 完全にランダムにページを取得するためのハック
|
| 50 |
+
# Wikipediaの「おまかせ表示」APIを利用してランダムなタイトルを取得できるが、
|
| 51 |
+
# ここではテスト用に特定のカテゴリやキーワードからページを取得するか、有名な記事を指定する。
|
| 52 |
+
sample_topics = [
|
| 53 |
+
"量子力学", "人工知能", "日本", "宇宙", "徳川家康",
|
| 54 |
+
"相対性理論", "コンピュータ", "インターネット", "哲学", "経済学",
|
| 55 |
+
"ブラックホール", "細胞", "DNA", "地球", "太陽系"
|
| 56 |
+
]
|
| 57 |
+
|
| 58 |
+
dataset = []
|
| 59 |
+
|
| 60 |
+
for topic in tqdm(sample_topics[:num_pages]):
|
| 61 |
+
page = wiki.page(topic)
|
| 62 |
+
if not page.exists():
|
| 63 |
+
continue
|
| 64 |
+
|
| 65 |
+
# ページの最初の要約部分(サマリー)
|
| 66 |
+
summary_md = f"# {page.title}\n- {page.summary.split('。')[0]}。\n"
|
| 67 |
+
summary_flat = page.summary + "\n"
|
| 68 |
+
|
| 69 |
+
# セクション構造の抽出
|
| 70 |
+
body_md, body_flat = extract_structure(page.sections, level=2)
|
| 71 |
+
|
| 72 |
+
final_markdown = summary_md + body_md
|
| 73 |
+
final_flat = summary_flat + body_flat
|
| 74 |
+
|
| 75 |
+
# テキストが短すぎる場合はスキップ
|
| 76 |
+
if len(final_flat) < 100:
|
| 77 |
+
continue
|
| 78 |
+
|
| 79 |
+
# JSONLフォーマットに整形
|
| 80 |
+
data_row = {
|
| 81 |
+
"instruction": "以下の長文から論理構造を抽出し、Markdown形式の目次(マインドマップ)を出力してください。",
|
| 82 |
+
"input": final_flat.strip(),
|
| 83 |
+
"output": final_markdown.strip()
|
| 84 |
+
}
|
| 85 |
+
dataset.append(data_row)
|
| 86 |
+
time.sleep(1) # APIへの負荷軽減
|
| 87 |
+
|
| 88 |
+
# JSONLファイルとして保存
|
| 89 |
+
with open(output_file, "w", encoding="utf-8") as f:
|
| 90 |
+
for row in dataset:
|
| 91 |
+
f.write(json.dumps(row, ensure_ascii=False) + "\n")
|
| 92 |
+
|
| 93 |
+
print(f"\nSuccessfully generated {len(dataset)} examples!")
|
| 94 |
+
print(f"Saved to {output_file}")
|
| 95 |
+
|
| 96 |
+
if __name__ == "__main__":
|
| 97 |
+
generate_dataset(num_pages=5)
|
backend/dataset_generator_distilled.py
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
import json
|
| 3 |
+
import time
|
| 4 |
+
import requests
|
| 5 |
+
import wikipediaapi
|
| 6 |
+
from tqdm import tqdm
|
| 7 |
+
|
| 8 |
+
# LM Studioのローカルサーバー設定
|
| 9 |
+
LM_STUDIO_URL = "http://127.0.0.1:1234/v1/chat/completions"
|
| 10 |
+
|
| 11 |
+
# 教師モデルへのプロンプト(システム命令)
|
| 12 |
+
SYSTEM_PROMPT = """あなたは超優秀な情報整理AI(マインドマップ職人)です。
|
| 13 |
+
ユーザーから提供された長文テキストを深く読み解き、内容を論理的に構造化して、マインドマップ用のMarkdownを出力してください。
|
| 14 |
+
|
| 15 |
+
【厳格なルール】
|
| 16 |
+
1. 出力は必ずMarkdownの階層構造(#、##、###、および箇条書き - )のみとすること。
|
| 17 |
+
2. 一番上の階層(#)は、そのテキストの主題(メインテーマ)を1つだけ記述すること。
|
| 18 |
+
3. 下位の階層(##、###)で、大項目・中項目を整理すること。
|
| 19 |
+
4. 具体的な要素や詳細な情報は、箇条書き(- )を用いて末端の枝として表現すること。
|
| 20 |
+
5. Markdown以外の説明文、挨拶、感想、「以下に示します」などの前置きは一切出力してはならない(厳禁)。
|
| 21 |
+
6. 「概要」や「利用状況」など、テキストにない見出しを適当にでっち上げないこと。テキストに沿った見出しを作ること。
|
| 22 |
+
"""
|
| 23 |
+
|
| 24 |
+
USER_INSTRUCTION = "以下の長文から論理構造を抽出し、Markdown形式の目次(マインドマップ)を出力してください。"
|
| 25 |
+
|
| 26 |
+
def generate_teacher_output(text):
|
| 27 |
+
"""LM StudioのAPIを叩いて、教師モデルにマインドマップを作らせる"""
|
| 28 |
+
payload = {
|
| 29 |
+
"messages": [
|
| 30 |
+
{"role": "system", "content": SYSTEM_PROMPT},
|
| 31 |
+
{"role": "user", "content": f"{USER_INSTRUCTION}\n\n【対象テキスト】\n{text}"}
|
| 32 |
+
],
|
| 33 |
+
"temperature": 0.2, # 創造性よりも正確性を重視
|
| 34 |
+
"max_tokens": -1
|
| 35 |
+
}
|
| 36 |
+
|
| 37 |
+
try:
|
| 38 |
+
response = requests.post(LM_STUDIO_URL, json=payload, timeout=300)
|
| 39 |
+
if response.status_code == 200:
|
| 40 |
+
result = response.json()
|
| 41 |
+
return result['choices'][0]['message']['content'].strip()
|
| 42 |
+
else:
|
| 43 |
+
print(f"API Error: {response.status_code}")
|
| 44 |
+
return None
|
| 45 |
+
except Exception as e:
|
| 46 |
+
print(f"Connection Error: {e}")
|
| 47 |
+
return None
|
| 48 |
+
|
| 49 |
+
def fetch_random_wikipedia_text(count=10):
|
| 50 |
+
"""Wikipediaからランダムな記事のテキストを取得する(見出し構造は使わない)"""
|
| 51 |
+
wiki_wiki = wikipediaapi.Wikipedia(
|
| 52 |
+
user_agent='MindMapStudio_DistillationGen/1.0',
|
| 53 |
+
language='ja',
|
| 54 |
+
extract_format=wikipediaapi.ExtractFormat.WIKI
|
| 55 |
+
)
|
| 56 |
+
|
| 57 |
+
print(f"Wikipediaからランダムに {count} 件の記事タイトルを取得中...")
|
| 58 |
+
titles = set()
|
| 59 |
+
while len(titles) < count:
|
| 60 |
+
limit = min(50, count - len(titles))
|
| 61 |
+
url = f"https://ja.wikipedia.org/w/api.php?action=query&list=random&rnnamespace=0&rnlimit={limit}&format=json"
|
| 62 |
+
headers = {'User-Agent': 'MindMapStudio_DistillationGen/1.0'}
|
| 63 |
+
try:
|
| 64 |
+
res = requests.get(url, headers=headers).json()
|
| 65 |
+
for q in res.get("query", {}).get("random", []):
|
| 66 |
+
titles.add(q["title"])
|
| 67 |
+
except Exception:
|
| 68 |
+
time.sleep(1)
|
| 69 |
+
|
| 70 |
+
print("記事のテキストをダウンロード中...")
|
| 71 |
+
texts = []
|
| 72 |
+
for title in tqdm(list(titles)):
|
| 73 |
+
try:
|
| 74 |
+
page = wiki_wiki.page(title)
|
| 75 |
+
if page.exists() and len(page.text) > 500:
|
| 76 |
+
# 最初の2000文字程度を学習用テキストとする(長すぎるとAPIが遅くなるため)
|
| 77 |
+
text = page.text[:2000]
|
| 78 |
+
texts.append(text)
|
| 79 |
+
except Exception:
|
| 80 |
+
pass
|
| 81 |
+
|
| 82 |
+
return texts
|
| 83 |
+
|
| 84 |
+
def generate_distilled_dataset(num_samples=10, output_file="mindmap_dataset_distilled.jsonl"):
|
| 85 |
+
print("=== 教師モデルによるデータセット蒸留(Knowledge Distillation)を開始 ===")
|
| 86 |
+
|
| 87 |
+
texts = fetch_random_wikipedia_text(num_samples)
|
| 88 |
+
print(f"合計 {len(texts)} 件の有効なテキストを取得しました。")
|
| 89 |
+
print("LM Studio (Teacher) に推論させています...")
|
| 90 |
+
|
| 91 |
+
with open(output_file, 'w', encoding='utf-8') as f:
|
| 92 |
+
for text in tqdm(texts):
|
| 93 |
+
teacher_md = generate_teacher_output(text)
|
| 94 |
+
if teacher_md:
|
| 95 |
+
# AIがMarkdownのコードブロック(```markdown)を含めてしまった場合は除去
|
| 96 |
+
teacher_md = teacher_md.replace("```markdown\n", "").replace("```\n", "").replace("```", "").strip()
|
| 97 |
+
|
| 98 |
+
data = {
|
| 99 |
+
"instruction": USER_INSTRUCTION,
|
| 100 |
+
"input": text,
|
| 101 |
+
"output": teacher_md
|
| 102 |
+
}
|
| 103 |
+
f.write(json.dumps(data, ensure_ascii=False) + '\n')
|
| 104 |
+
f.flush()
|
| 105 |
+
else:
|
| 106 |
+
print("\n警告: APIは成功しましたが、出力(content)が空でした。")
|
| 107 |
+
|
| 108 |
+
print(f"完了!蒸留データセットを {output_file} に保存しました。")
|
| 109 |
+
|
| 110 |
+
if __name__ == "__main__":
|
| 111 |
+
# 本番用データセットの生成(500件)
|
| 112 |
+
generate_distilled_dataset(500)
|
backend/dataset_generator_prod.py
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import wikipediaapi
|
| 2 |
+
import json
|
| 3 |
+
import concurrent.futures
|
| 4 |
+
from tqdm import tqdm
|
| 5 |
+
import re
|
| 6 |
+
import os
|
| 7 |
+
import requests
|
| 8 |
+
import time
|
| 9 |
+
|
| 10 |
+
# ==========================================
|
| 11 |
+
# 本番用 大規模データ生成スクリプト
|
| 12 |
+
# ==========================================
|
| 13 |
+
NUM_ARTICLES_TO_FETCH = 5000 # 目標とする高品質データ件数
|
| 14 |
+
OUTPUT_FILE = "mindmap_dataset_production.jsonl"
|
| 15 |
+
MAX_WORKERS = 10 # スレッド数(多すぎるとAPI制限に引っかかるため10程度)
|
| 16 |
+
|
| 17 |
+
# ユーザーエージェントを明記(Wikipedia APIのルール)
|
| 18 |
+
wiki_wiki = wikipediaapi.Wikipedia(
|
| 19 |
+
user_agent='MindMapStudio_ProductionGen/1.0',
|
| 20 |
+
language='ja',
|
| 21 |
+
extract_format=wikipediaapi.ExtractFormat.WIKI
|
| 22 |
+
)
|
| 23 |
+
|
| 24 |
+
def get_random_titles(count):
|
| 25 |
+
"""Wikipedia APIからランダムな記事タイトルを一括取得する"""
|
| 26 |
+
titles = set()
|
| 27 |
+
print(f"ランダムな記事のタイトルを {count} 件収集中...")
|
| 28 |
+
pbar = tqdm(total=count)
|
| 29 |
+
while len(titles) < count:
|
| 30 |
+
# 一度に最大50件まで取得可能
|
| 31 |
+
limit = min(50, count - len(titles))
|
| 32 |
+
url = f"https://ja.wikipedia.org/w/api.php?action=query&list=random&rnnamespace=0&rnlimit={limit}&format=json"
|
| 33 |
+
headers = {'User-Agent': 'MindMapStudio_ProductionGen/1.0 (contact@example.com)'}
|
| 34 |
+
try:
|
| 35 |
+
res = requests.get(url, headers=headers).json()
|
| 36 |
+
for q in res.get("query", {}).get("random", []):
|
| 37 |
+
if q["title"] not in titles:
|
| 38 |
+
titles.add(q["title"])
|
| 39 |
+
pbar.update(1)
|
| 40 |
+
except Exception as e:
|
| 41 |
+
print(f"\nAPI Fetch Error: {e}")
|
| 42 |
+
time.sleep(1) # エラー時は少し待機
|
| 43 |
+
pbar.close()
|
| 44 |
+
return list(titles)
|
| 45 |
+
|
| 46 |
+
def is_valid_article(page):
|
| 47 |
+
"""品質チェック(ノイズを除外)"""
|
| 48 |
+
if not page.exists(): return False
|
| 49 |
+
text = page.text
|
| 50 |
+
if len(text) < 1000: return False # 文字数が少なすぎる(スタブ記事)
|
| 51 |
+
|
| 52 |
+
sections = page.sections
|
| 53 |
+
if len(sections) < 4: return False # 見出しが少なすぎる(構造がない)
|
| 54 |
+
|
| 55 |
+
return True
|
| 56 |
+
|
| 57 |
+
def process_section(section, level=1):
|
| 58 |
+
"""再帰的にセクションを処理し、ノイズセクションを除去"""
|
| 59 |
+
title = section.title.strip()
|
| 60 |
+
|
| 61 |
+
# AIの学習に不要なメタセクションを完全除外
|
| 62 |
+
exclude_keywords = ["脚注", "出典", "参考文献", "関連項目", "外部リンク", "注釈", "ギャラリー", "一覧"]
|
| 63 |
+
if any(ex in title for ex in exclude_keywords):
|
| 64 |
+
return "", ""
|
| 65 |
+
|
| 66 |
+
markdown = f"{'#' * level} {title}\n"
|
| 67 |
+
plain_text = section.text.strip() + "\n" if section.text.strip() else ""
|
| 68 |
+
|
| 69 |
+
for sub_section in section.sections:
|
| 70 |
+
sub_md, sub_pt = process_section(sub_section, level + 1)
|
| 71 |
+
markdown += sub_md
|
| 72 |
+
plain_text += sub_pt
|
| 73 |
+
|
| 74 |
+
return markdown, plain_text
|
| 75 |
+
|
| 76 |
+
def fetch_and_process(title):
|
| 77 |
+
try:
|
| 78 |
+
page = wiki_wiki.page(title)
|
| 79 |
+
if not is_valid_article(page):
|
| 80 |
+
return None
|
| 81 |
+
|
| 82 |
+
markdown_output = f"# {page.title}\n"
|
| 83 |
+
plain_input = page.summary + "\n"
|
| 84 |
+
|
| 85 |
+
for section in page.sections:
|
| 86 |
+
sec_md, sec_pt = process_section(section, level=2)
|
| 87 |
+
markdown_output += sec_md
|
| 88 |
+
plain_input += sec_pt
|
| 89 |
+
|
| 90 |
+
# 改行などのクリーニング
|
| 91 |
+
plain_input = re.sub(r'\n+', '\n', plain_input).strip()
|
| 92 |
+
markdown_output = markdown_output.strip()
|
| 93 |
+
|
| 94 |
+
# 最終チェック: 構造が浅すぎるものは除外
|
| 95 |
+
if len(plain_input) < 500 or len(markdown_output.split('\n')) < 5:
|
| 96 |
+
return None
|
| 97 |
+
|
| 98 |
+
return {
|
| 99 |
+
"instruction": "以下の長文から論理構造を抽出し、Markdown形式の目次(マインドマップ)を出力してください。",
|
| 100 |
+
"input": plain_input,
|
| 101 |
+
"output": markdown_output
|
| 102 |
+
}
|
| 103 |
+
except Exception:
|
| 104 |
+
return None
|
| 105 |
+
|
| 106 |
+
def main():
|
| 107 |
+
print("--- 本番環境用 大規模データ生成パイプライン起動 ---")
|
| 108 |
+
|
| 109 |
+
# 既存のファイルがあれば削除(やり直し用)
|
| 110 |
+
if os.path.exists(OUTPUT_FILE):
|
| 111 |
+
os.remove(OUTPUT_FILE)
|
| 112 |
+
|
| 113 |
+
titles = get_random_titles(NUM_ARTICLES_TO_FETCH * 2) # 除外されることを見越して多めに取得
|
| 114 |
+
|
| 115 |
+
success_count = 0
|
| 116 |
+
print("\n記事のダウンロードと解析(マルチスレッド処理)を開始します...")
|
| 117 |
+
|
| 118 |
+
with open(OUTPUT_FILE, 'a', encoding='utf-8') as f:
|
| 119 |
+
with concurrent.futures.ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
|
| 120 |
+
# 進行状況バーの表示
|
| 121 |
+
results = list(tqdm(executor.map(fetch_and_process, titles), total=len(titles)))
|
| 122 |
+
|
| 123 |
+
for res in results:
|
| 124 |
+
if res is not None:
|
| 125 |
+
f.write(json.dumps(res, ensure_ascii=False) + '\n')
|
| 126 |
+
success_count += 1
|
| 127 |
+
|
| 128 |
+
if success_count >= NUM_ARTICLES_TO_FETCH:
|
| 129 |
+
break
|
| 130 |
+
|
| 131 |
+
print(f"\n完了! 超高品質なデータセット {success_count} 件を {OUTPUT_FILE} に保存しました。")
|
| 132 |
+
|
| 133 |
+
if __name__ == "__main__":
|
| 134 |
+
main()
|
backend/llama.cpp/.clang-format
ADDED
|
@@ -0,0 +1,171 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
Language: Cpp
|
| 3 |
+
AlignAfterOpenBracket: Align
|
| 4 |
+
AlignArrayOfStructures: Left
|
| 5 |
+
AlignConsecutiveAssignments: AcrossComments
|
| 6 |
+
AlignConsecutiveBitFields: AcrossComments
|
| 7 |
+
AlignConsecutiveDeclarations: AcrossComments
|
| 8 |
+
AlignConsecutiveMacros: AcrossComments
|
| 9 |
+
# AlignConsecutiveShortCaseStatements: AcrossComments
|
| 10 |
+
AlignEscapedNewlines: Left # LeftWithLastLine
|
| 11 |
+
AlignOperands: Align
|
| 12 |
+
AlignTrailingComments:
|
| 13 |
+
Kind: Always
|
| 14 |
+
OverEmptyLines: 1
|
| 15 |
+
AllowAllArgumentsOnNextLine: true
|
| 16 |
+
AllowAllParametersOfDeclarationOnNextLine: false
|
| 17 |
+
# AllowBreakBeforeNoexceptSpecifier: OnlyWithParen
|
| 18 |
+
AllowShortBlocksOnASingleLine: Never
|
| 19 |
+
AllowShortCaseLabelsOnASingleLine: false
|
| 20 |
+
AllowShortFunctionsOnASingleLine: Inline
|
| 21 |
+
AllowShortIfStatementsOnASingleLine: Never
|
| 22 |
+
AllowShortLambdasOnASingleLine: Inline
|
| 23 |
+
AllowShortLoopsOnASingleLine: false
|
| 24 |
+
AlwaysBreakBeforeMultilineStrings: true
|
| 25 |
+
# Treat CUDA keywords/attributes as "attribute macros" and avoid breaking lines inside them
|
| 26 |
+
AttributeMacros:
|
| 27 |
+
- __host__
|
| 28 |
+
- __device__
|
| 29 |
+
- __global__
|
| 30 |
+
- __forceinline__
|
| 31 |
+
- __launch_bounds__
|
| 32 |
+
BinPackArguments: true
|
| 33 |
+
BinPackParameters: false # OnePerLine
|
| 34 |
+
BitFieldColonSpacing: Both
|
| 35 |
+
BreakBeforeBraces: Custom # Attach
|
| 36 |
+
BraceWrapping:
|
| 37 |
+
AfterCaseLabel: true
|
| 38 |
+
AfterClass: false
|
| 39 |
+
AfterControlStatement: false
|
| 40 |
+
AfterEnum: false
|
| 41 |
+
AfterFunction: false
|
| 42 |
+
AfterNamespace: false
|
| 43 |
+
AfterObjCDeclaration: false
|
| 44 |
+
AfterStruct: false
|
| 45 |
+
AfterUnion: false
|
| 46 |
+
AfterExternBlock: false
|
| 47 |
+
BeforeCatch: false
|
| 48 |
+
BeforeElse: false
|
| 49 |
+
BeforeLambdaBody: false
|
| 50 |
+
BeforeWhile: false
|
| 51 |
+
IndentBraces: false
|
| 52 |
+
SplitEmptyFunction: false
|
| 53 |
+
SplitEmptyRecord: false
|
| 54 |
+
SplitEmptyNamespace: false
|
| 55 |
+
# BreakAdjacentStringLiterals: true
|
| 56 |
+
BreakAfterAttributes: Never
|
| 57 |
+
BreakBeforeBinaryOperators: None
|
| 58 |
+
BreakBeforeInlineASMColon: OnlyMultiline
|
| 59 |
+
BreakBeforeTernaryOperators: false
|
| 60 |
+
# BreakBinaryOperations: Never
|
| 61 |
+
BreakConstructorInitializers: AfterColon
|
| 62 |
+
# BreakFunctionDefinitionParameters: false
|
| 63 |
+
BreakInheritanceList: AfterComma
|
| 64 |
+
BreakStringLiterals: true
|
| 65 |
+
# BreakTemplateDeclarations: Yes
|
| 66 |
+
ColumnLimit: 120
|
| 67 |
+
CommentPragmas: '^ IWYU pragma:'
|
| 68 |
+
CompactNamespaces: false
|
| 69 |
+
ConstructorInitializerIndentWidth: 4
|
| 70 |
+
ContinuationIndentWidth: 4
|
| 71 |
+
Cpp11BracedListStyle: false
|
| 72 |
+
DerivePointerAlignment: false
|
| 73 |
+
DisableFormat: false
|
| 74 |
+
EmptyLineBeforeAccessModifier: Leave
|
| 75 |
+
EmptyLineAfterAccessModifier: Never
|
| 76 |
+
ExperimentalAutoDetectBinPacking: false
|
| 77 |
+
FixNamespaceComments: true
|
| 78 |
+
IncludeBlocks: Regroup
|
| 79 |
+
IncludeCategories:
|
| 80 |
+
- Regex: '".*"'
|
| 81 |
+
Priority: 1
|
| 82 |
+
SortPriority: 0
|
| 83 |
+
- Regex: '^<.*\.h>'
|
| 84 |
+
Priority: 2
|
| 85 |
+
SortPriority: 0
|
| 86 |
+
- Regex: '^<.*'
|
| 87 |
+
Priority: 3
|
| 88 |
+
SortPriority: 0
|
| 89 |
+
- Regex: '.*'
|
| 90 |
+
Priority: 4
|
| 91 |
+
SortPriority: 0
|
| 92 |
+
IncludeIsMainRegex: '([-_](test|unittest))?$'
|
| 93 |
+
IncludeIsMainSourceRegex: ''
|
| 94 |
+
IndentAccessModifiers: false
|
| 95 |
+
IndentCaseBlocks: true
|
| 96 |
+
IndentCaseLabels: true
|
| 97 |
+
IndentExternBlock: NoIndent
|
| 98 |
+
IndentGotoLabels: false
|
| 99 |
+
IndentPPDirectives: AfterHash
|
| 100 |
+
IndentWidth: 4
|
| 101 |
+
IndentWrappedFunctionNames: false
|
| 102 |
+
InsertBraces: true # NOTE: may lead to incorrect formatting
|
| 103 |
+
InsertNewlineAtEOF: true
|
| 104 |
+
JavaScriptQuotes: Leave
|
| 105 |
+
JavaScriptWrapImports: true
|
| 106 |
+
KeepEmptyLinesAtTheStartOfBlocks: false
|
| 107 |
+
LambdaBodyIndentation: Signature
|
| 108 |
+
LineEnding: LF
|
| 109 |
+
MacroBlockBegin: ''
|
| 110 |
+
MacroBlockEnd: ''
|
| 111 |
+
MaxEmptyLinesToKeep: 1
|
| 112 |
+
NamespaceIndentation: None
|
| 113 |
+
ObjCBinPackProtocolList: Auto
|
| 114 |
+
ObjCBlockIndentWidth: 4
|
| 115 |
+
ObjCSpaceAfterProperty: true
|
| 116 |
+
ObjCSpaceBeforeProtocolList: true
|
| 117 |
+
PPIndentWidth: -1
|
| 118 |
+
PackConstructorInitializers: CurrentLine
|
| 119 |
+
PenaltyBreakAssignment: 2
|
| 120 |
+
PenaltyBreakBeforeFirstCallParameter: 1
|
| 121 |
+
PenaltyBreakComment: 300
|
| 122 |
+
PenaltyBreakFirstLessLess: 120
|
| 123 |
+
PenaltyBreakString: 1000
|
| 124 |
+
PenaltyBreakTemplateDeclaration: 10
|
| 125 |
+
PenaltyExcessCharacter: 1000000
|
| 126 |
+
PenaltyReturnTypeOnItsOwnLine: 200
|
| 127 |
+
PointerAlignment: Middle
|
| 128 |
+
QualifierAlignment: Left
|
| 129 |
+
#QualifierOrder: ['static', 'inline', 'friend', 'constexpr', 'const', 'volatile', 'type', 'restrict']
|
| 130 |
+
RawStringFormats:
|
| 131 |
+
- Language: Cpp
|
| 132 |
+
Delimiters:
|
| 133 |
+
- cc
|
| 134 |
+
- CC
|
| 135 |
+
- cpp
|
| 136 |
+
- Cpp
|
| 137 |
+
- CPP
|
| 138 |
+
- 'c++'
|
| 139 |
+
- 'C++'
|
| 140 |
+
CanonicalDelimiter: ''
|
| 141 |
+
ReferenceAlignment: Middle
|
| 142 |
+
ReflowComments: false # IndentOnly
|
| 143 |
+
SeparateDefinitionBlocks: Always
|
| 144 |
+
SortIncludes: CaseInsensitive
|
| 145 |
+
SortUsingDeclarations: LexicographicNumeric
|
| 146 |
+
SpaceAfterCStyleCast: true
|
| 147 |
+
SpaceAfterLogicalNot: false
|
| 148 |
+
SpaceAfterTemplateKeyword: true
|
| 149 |
+
SpaceBeforeAssignmentOperators: true
|
| 150 |
+
SpaceBeforeCpp11BracedList: false
|
| 151 |
+
SpaceBeforeCtorInitializerColon: true
|
| 152 |
+
SpaceBeforeInheritanceColon: true
|
| 153 |
+
SpaceBeforeParens: ControlStatements
|
| 154 |
+
SpaceBeforeRangeBasedForLoopColon: true
|
| 155 |
+
SpaceInEmptyBlock: false
|
| 156 |
+
SpaceInEmptyParentheses: false
|
| 157 |
+
SpacesBeforeTrailingComments: 2
|
| 158 |
+
SpacesInAngles: Never
|
| 159 |
+
SpacesInContainerLiterals: true
|
| 160 |
+
SpacesInLineCommentPrefix:
|
| 161 |
+
Minimum: 1
|
| 162 |
+
Maximum: -1
|
| 163 |
+
SpacesInParentheses: false
|
| 164 |
+
SpacesInSquareBrackets: false
|
| 165 |
+
SpaceBeforeSquareBrackets: false
|
| 166 |
+
Standard: c++17
|
| 167 |
+
TabWidth: 4
|
| 168 |
+
UseTab: Never
|
| 169 |
+
WhitespaceSensitiveMacros: ['STRINGIZE']
|
| 170 |
+
...
|
| 171 |
+
|
backend/llama.cpp/.clang-tidy
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
Checks: >
|
| 3 |
+
bugprone-*,
|
| 4 |
+
-bugprone-easily-swappable-parameters,
|
| 5 |
+
-bugprone-implicit-widening-of-multiplication-result,
|
| 6 |
+
-bugprone-misplaced-widening-cast,
|
| 7 |
+
-bugprone-narrowing-conversions,
|
| 8 |
+
readability-*,
|
| 9 |
+
-readability-avoid-unconditional-preprocessor-if,
|
| 10 |
+
-readability-function-cognitive-complexity,
|
| 11 |
+
-readability-identifier-length,
|
| 12 |
+
-readability-implicit-bool-conversion,
|
| 13 |
+
-readability-magic-numbers,
|
| 14 |
+
-readability-uppercase-literal-suffix,
|
| 15 |
+
-readability-simplify-boolean-expr,
|
| 16 |
+
-readability-math-missing-parentheses,
|
| 17 |
+
clang-analyzer-*,
|
| 18 |
+
-clang-analyzer-security.insecureAPI.DeprecatedOrUnsafeBufferHandling,
|
| 19 |
+
performance-*,
|
| 20 |
+
-performance-enum-size,
|
| 21 |
+
portability-*,
|
| 22 |
+
-portability-simd-intrinsics,
|
| 23 |
+
misc-*,
|
| 24 |
+
-misc-const-correctness,
|
| 25 |
+
-misc-non-private-member-variables-in-classes,
|
| 26 |
+
-misc-no-recursion,
|
| 27 |
+
-misc-use-anonymous-namespace,
|
| 28 |
+
FormatStyle: none
|
backend/llama.cpp/.devops/cann.Dockerfile
ADDED
|
@@ -0,0 +1,163 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# ==============================================================================
|
| 2 |
+
# ARGUMENTS
|
| 3 |
+
# ==============================================================================
|
| 4 |
+
|
| 5 |
+
# Define the CANN base image for easier version updates later
|
| 6 |
+
ARG CHIP_TYPE=910b
|
| 7 |
+
ARG CANN_BASE_IMAGE=quay.io/ascend/cann:8.5.0-${CHIP_TYPE}-openeuler24.03-py3.11
|
| 8 |
+
ARG BUILD_DATE=N/A
|
| 9 |
+
ARG APP_VERSION=N/A
|
| 10 |
+
ARG APP_REVISION=N/A
|
| 11 |
+
|
| 12 |
+
# ==============================================================================
|
| 13 |
+
# BUILD STAGE
|
| 14 |
+
# Compile all binary files and libraries
|
| 15 |
+
# ==============================================================================
|
| 16 |
+
ARG NODE_VERSION=24
|
| 17 |
+
|
| 18 |
+
FROM docker.io/node:$NODE_VERSION AS web
|
| 19 |
+
|
| 20 |
+
ARG APP_VERSION
|
| 21 |
+
|
| 22 |
+
WORKDIR /app/tools/ui
|
| 23 |
+
|
| 24 |
+
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
| 25 |
+
RUN npm ci
|
| 26 |
+
|
| 27 |
+
COPY tools/ui/ ./
|
| 28 |
+
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
| 29 |
+
|
| 30 |
+
FROM ${CANN_BASE_IMAGE} AS build
|
| 31 |
+
|
| 32 |
+
# -- Install build dependencies --
|
| 33 |
+
RUN yum install -y gcc g++ cmake make git openssl-devel python3 python3-pip && \
|
| 34 |
+
yum clean all && \
|
| 35 |
+
rm -rf /var/cache/yum
|
| 36 |
+
|
| 37 |
+
# -- Set the working directory --
|
| 38 |
+
WORKDIR /app
|
| 39 |
+
|
| 40 |
+
# -- Copy project files --
|
| 41 |
+
COPY . .
|
| 42 |
+
|
| 43 |
+
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
| 44 |
+
|
| 45 |
+
# -- Set CANN environment variables (required for compilation) --
|
| 46 |
+
# Using ENV instead of `source` allows environment variables to persist across the entire image layer
|
| 47 |
+
ENV ASCEND_TOOLKIT_HOME=/usr/local/Ascend/ascend-toolkit/latest
|
| 48 |
+
ENV LD_LIBRARY_PATH=${ASCEND_TOOLKIT_HOME}/lib64:${LD_LIBRARY_PATH}
|
| 49 |
+
ENV PATH=${ASCEND_TOOLKIT_HOME}/bin:${PATH}
|
| 50 |
+
ENV ASCEND_OPP_PATH=${ASCEND_TOOLKIT_HOME}/opp
|
| 51 |
+
ENV LD_LIBRARY_PATH=${ASCEND_TOOLKIT_HOME}/runtime/lib64/stub:$LD_LIBRARY_PATH
|
| 52 |
+
# ... You can add other environment variables from the original file as needed ...
|
| 53 |
+
# For brevity, only core variables are listed here. You can paste the original ENV list here.
|
| 54 |
+
|
| 55 |
+
# -- Build llama.cpp --
|
| 56 |
+
# Use the passed CHIP_TYPE argument and add general build options
|
| 57 |
+
ARG CHIP_TYPE
|
| 58 |
+
RUN source /usr/local/Ascend/ascend-toolkit/set_env.sh --force \
|
| 59 |
+
&& \
|
| 60 |
+
cmake -B build \
|
| 61 |
+
-DGGML_CANN=ON \
|
| 62 |
+
-DCMAKE_BUILD_TYPE=Release \
|
| 63 |
+
-DSOC_TYPE=ascend${CHIP_TYPE} \
|
| 64 |
+
-DUSE_ACL_GRAPH=ON \
|
| 65 |
+
. && \
|
| 66 |
+
cmake --build build --config Release -j$(nproc)
|
| 67 |
+
|
| 68 |
+
# -- Organize build artifacts for copying in later stages --
|
| 69 |
+
# Create a lib directory to store all .so files
|
| 70 |
+
RUN mkdir -p /app/lib && \
|
| 71 |
+
find build -name "*.so*" -exec cp -P {} /app/lib \;
|
| 72 |
+
|
| 73 |
+
# Create a full directory to store all executables and Python scripts
|
| 74 |
+
RUN mkdir -p /app/full && \
|
| 75 |
+
cp build/bin/* /app/full/ && \
|
| 76 |
+
cp *.py /app/full/ && \
|
| 77 |
+
cp -r conversion /app/full/ && \
|
| 78 |
+
cp -r gguf-py /app/full/ && \
|
| 79 |
+
cp -r requirements /app/full/ && \
|
| 80 |
+
cp requirements.txt /app/full/
|
| 81 |
+
# If you have a tools.sh script, make sure it is copied here
|
| 82 |
+
# cp .devops/tools.sh /app/full/tools.sh
|
| 83 |
+
|
| 84 |
+
# ==============================================================================
|
| 85 |
+
# BASE STAGE
|
| 86 |
+
# Create a minimal base image with CANN runtime and common libraries
|
| 87 |
+
# ==============================================================================
|
| 88 |
+
FROM ${CANN_BASE_IMAGE} AS base
|
| 89 |
+
|
| 90 |
+
ARG BUILD_DATE=N/A
|
| 91 |
+
ARG APP_VERSION=N/A
|
| 92 |
+
ARG APP_REVISION=N/A
|
| 93 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 94 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 95 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 96 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 97 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 98 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 99 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 100 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 101 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 102 |
+
|
| 103 |
+
# -- Install runtime dependencies --
|
| 104 |
+
RUN yum install -y libgomp curl && \
|
| 105 |
+
yum clean all && \
|
| 106 |
+
rm -rf /var/cache/yum
|
| 107 |
+
|
| 108 |
+
# -- Set CANN environment variables (required for runtime) --
|
| 109 |
+
ENV ASCEND_TOOLKIT_HOME=/usr/local/Ascend/ascend-toolkit/latest
|
| 110 |
+
ENV LD_LIBRARY_PATH=/app:${ASCEND_TOOLKIT_HOME}/lib64:${LD_LIBRARY_PATH}
|
| 111 |
+
ENV PATH=${ASCEND_TOOLKIT_HOME}/bin:${PATH}
|
| 112 |
+
ENV ASCEND_OPP_PATH=${ASCEND_TOOLKIT_HOME}/opp
|
| 113 |
+
# ... You can add other environment variables from the original file as needed ...
|
| 114 |
+
|
| 115 |
+
WORKDIR /app
|
| 116 |
+
|
| 117 |
+
# Copy compiled .so files from the build stage
|
| 118 |
+
COPY --from=build /app/lib/ /app
|
| 119 |
+
|
| 120 |
+
# ==============================================================================
|
| 121 |
+
# FINAL STAGES (TARGETS)
|
| 122 |
+
# ==============================================================================
|
| 123 |
+
|
| 124 |
+
### Target: full
|
| 125 |
+
# Complete image with all tools, Python bindings, and dependencies
|
| 126 |
+
# ==============================================================================
|
| 127 |
+
FROM base AS full
|
| 128 |
+
|
| 129 |
+
COPY --from=build /app/full /app
|
| 130 |
+
|
| 131 |
+
# Install Python dependencies
|
| 132 |
+
RUN yum install -y git python3 python3-pip && \
|
| 133 |
+
pip3 install --no-cache-dir --upgrade pip setuptools wheel && \
|
| 134 |
+
pip3 install --no-cache-dir -r requirements.txt && \
|
| 135 |
+
yum clean all && \
|
| 136 |
+
rm -rf /var/cache/yum
|
| 137 |
+
|
| 138 |
+
# You need to provide a tools.sh script as the entrypoint
|
| 139 |
+
ENTRYPOINT ["/app/tools.sh"]
|
| 140 |
+
# If there is no tools.sh, you can set the default to start the server
|
| 141 |
+
# ENTRYPOINT ["/app/llama-server"]
|
| 142 |
+
|
| 143 |
+
### Target: light
|
| 144 |
+
# Lightweight image containing only llama-cli and llama-completion
|
| 145 |
+
# ==============================================================================
|
| 146 |
+
FROM base AS light
|
| 147 |
+
|
| 148 |
+
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app
|
| 149 |
+
|
| 150 |
+
ENTRYPOINT [ "/app/llama-cli" ]
|
| 151 |
+
|
| 152 |
+
### Target: server
|
| 153 |
+
# Dedicated server image containing only llama-server
|
| 154 |
+
# ==============================================================================
|
| 155 |
+
FROM base AS server
|
| 156 |
+
|
| 157 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 158 |
+
|
| 159 |
+
COPY --from=build /app/full/llama /app/full/llama-server /app
|
| 160 |
+
|
| 161 |
+
HEALTHCHECK --interval=5m CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
| 162 |
+
|
| 163 |
+
ENTRYPOINT [ "/app/llama-server" ]
|
backend/llama.cpp/.devops/cpu.Dockerfile
ADDED
|
@@ -0,0 +1,124 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG UBUNTU_VERSION=24.04
|
| 2 |
+
ARG BUILD_DATE=N/A
|
| 3 |
+
ARG APP_VERSION=N/A
|
| 4 |
+
ARG APP_REVISION=N/A
|
| 5 |
+
|
| 6 |
+
ARG NODE_VERSION=24
|
| 7 |
+
|
| 8 |
+
FROM docker.io/node:$NODE_VERSION AS web
|
| 9 |
+
|
| 10 |
+
ARG APP_VERSION
|
| 11 |
+
|
| 12 |
+
WORKDIR /app/tools/ui
|
| 13 |
+
|
| 14 |
+
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
| 15 |
+
RUN npm ci
|
| 16 |
+
|
| 17 |
+
COPY tools/ui/ ./
|
| 18 |
+
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
| 19 |
+
|
| 20 |
+
FROM docker.io/ubuntu:$UBUNTU_VERSION AS build
|
| 21 |
+
|
| 22 |
+
ARG TARGETARCH
|
| 23 |
+
|
| 24 |
+
RUN apt-get update && \
|
| 25 |
+
apt-get install -y gcc-14 g++-14 build-essential git cmake libssl-dev
|
| 26 |
+
|
| 27 |
+
ENV CC=gcc-14 CXX=g++-14
|
| 28 |
+
|
| 29 |
+
WORKDIR /app
|
| 30 |
+
|
| 31 |
+
COPY . .
|
| 32 |
+
|
| 33 |
+
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
| 34 |
+
|
| 35 |
+
RUN if [ "$TARGETARCH" = "amd64" ] || [ "$TARGETARCH" = "arm64" ]; then \
|
| 36 |
+
cmake -S . -B build -DCMAKE_BUILD_TYPE=Release -DGGML_NATIVE=OFF -DLLAMA_BUILD_TESTS=OFF -DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON; \
|
| 37 |
+
else \
|
| 38 |
+
echo "Unsupported architecture"; \
|
| 39 |
+
exit 1; \
|
| 40 |
+
fi && \
|
| 41 |
+
cmake --build build -j $(nproc)
|
| 42 |
+
|
| 43 |
+
RUN mkdir -p /app/lib && \
|
| 44 |
+
find build -name "*.so*" -exec cp -P {} /app/lib \;
|
| 45 |
+
|
| 46 |
+
RUN mkdir -p /app/full \
|
| 47 |
+
&& cp build/bin/* /app/full \
|
| 48 |
+
&& cp *.py /app/full \
|
| 49 |
+
&& cp -r conversion /app/full \
|
| 50 |
+
&& cp -r gguf-py /app/full \
|
| 51 |
+
&& cp -r requirements /app/full \
|
| 52 |
+
&& cp requirements.txt /app/full \
|
| 53 |
+
&& cp .devops/tools.sh /app/full/tools.sh
|
| 54 |
+
|
| 55 |
+
## Base image
|
| 56 |
+
FROM docker.io/ubuntu:$UBUNTU_VERSION AS base
|
| 57 |
+
|
| 58 |
+
ARG BUILD_DATE=N/A
|
| 59 |
+
ARG APP_VERSION=N/A
|
| 60 |
+
ARG APP_REVISION=N/A
|
| 61 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 62 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 63 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 64 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 65 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 66 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 67 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 68 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 69 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 70 |
+
|
| 71 |
+
RUN apt-get update \
|
| 72 |
+
&& apt-get install -y libgomp1 curl ffmpeg \
|
| 73 |
+
&& apt autoremove -y \
|
| 74 |
+
&& apt clean -y \
|
| 75 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 76 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 77 |
+
&& find /var/cache -type f -delete
|
| 78 |
+
|
| 79 |
+
COPY --from=build /app/lib/ /app
|
| 80 |
+
|
| 81 |
+
### Full
|
| 82 |
+
FROM base AS full
|
| 83 |
+
|
| 84 |
+
COPY --from=build /app/full /app
|
| 85 |
+
|
| 86 |
+
WORKDIR /app
|
| 87 |
+
|
| 88 |
+
RUN apt-get update \
|
| 89 |
+
&& apt-get install -y \
|
| 90 |
+
git \
|
| 91 |
+
python3 \
|
| 92 |
+
python3-pip \
|
| 93 |
+
python3-wheel \
|
| 94 |
+
&& pip install --break-system-packages --upgrade setuptools \
|
| 95 |
+
&& pip install --break-system-packages -r requirements.txt \
|
| 96 |
+
&& apt autoremove -y \
|
| 97 |
+
&& apt clean -y \
|
| 98 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 99 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 100 |
+
&& find /var/cache -type f -delete
|
| 101 |
+
|
| 102 |
+
ENTRYPOINT ["/app/tools.sh"]
|
| 103 |
+
|
| 104 |
+
### Light, CLI only
|
| 105 |
+
FROM base AS light
|
| 106 |
+
|
| 107 |
+
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app
|
| 108 |
+
|
| 109 |
+
WORKDIR /app
|
| 110 |
+
|
| 111 |
+
ENTRYPOINT [ "/app/llama-cli" ]
|
| 112 |
+
|
| 113 |
+
### Server, Server only
|
| 114 |
+
FROM base AS server
|
| 115 |
+
|
| 116 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 117 |
+
|
| 118 |
+
COPY --from=build /app/full/llama /app/full/llama-server /app
|
| 119 |
+
|
| 120 |
+
WORKDIR /app
|
| 121 |
+
|
| 122 |
+
HEALTHCHECK CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
| 123 |
+
|
| 124 |
+
ENTRYPOINT [ "/app/llama-server" ]
|
backend/llama.cpp/.devops/cuda.Dockerfile
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG UBUNTU_VERSION=24.04
|
| 2 |
+
# This needs to generally match the container host's environment.
|
| 3 |
+
ARG CUDA_VERSION=12.8.1
|
| 4 |
+
ARG GCC_VERSION=14
|
| 5 |
+
# Target the CUDA build image
|
| 6 |
+
ARG BASE_CUDA_DEV_CONTAINER=docker.io/nvidia/cuda:${CUDA_VERSION}-devel-ubuntu${UBUNTU_VERSION}
|
| 7 |
+
|
| 8 |
+
ARG BASE_CUDA_RUN_CONTAINER=docker.io/nvidia/cuda:${CUDA_VERSION}-runtime-ubuntu${UBUNTU_VERSION}
|
| 9 |
+
|
| 10 |
+
ARG BUILD_DATE=N/A
|
| 11 |
+
ARG APP_VERSION=N/A
|
| 12 |
+
ARG APP_REVISION=N/A
|
| 13 |
+
|
| 14 |
+
ARG NODE_VERSION=24
|
| 15 |
+
|
| 16 |
+
FROM docker.io/node:$NODE_VERSION AS web
|
| 17 |
+
|
| 18 |
+
ARG APP_VERSION
|
| 19 |
+
|
| 20 |
+
WORKDIR /app/tools/ui
|
| 21 |
+
|
| 22 |
+
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
| 23 |
+
RUN npm ci
|
| 24 |
+
|
| 25 |
+
COPY tools/ui/ ./
|
| 26 |
+
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
| 27 |
+
|
| 28 |
+
FROM ${BASE_CUDA_DEV_CONTAINER} AS build
|
| 29 |
+
|
| 30 |
+
ARG GCC_VERSION
|
| 31 |
+
# CUDA architecture to build for (defaults to all supported archs)
|
| 32 |
+
ARG CUDA_DOCKER_ARCH=default
|
| 33 |
+
|
| 34 |
+
RUN apt-get update && \
|
| 35 |
+
apt-get install -y gcc-${GCC_VERSION} g++-${GCC_VERSION} build-essential cmake python3 python3-pip git libssl-dev libgomp1
|
| 36 |
+
|
| 37 |
+
ENV CC=gcc-${GCC_VERSION} CXX=g++-${GCC_VERSION} CUDAHOSTCXX=g++-${GCC_VERSION}
|
| 38 |
+
|
| 39 |
+
WORKDIR /app
|
| 40 |
+
|
| 41 |
+
COPY . .
|
| 42 |
+
|
| 43 |
+
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
| 44 |
+
|
| 45 |
+
RUN if [ "${CUDA_DOCKER_ARCH}" != "default" ]; then \
|
| 46 |
+
export CMAKE_ARGS="-DCMAKE_CUDA_ARCHITECTURES=${CUDA_DOCKER_ARCH}"; \
|
| 47 |
+
fi && \
|
| 48 |
+
cmake -B build -DGGML_NATIVE=OFF -DGGML_CUDA=ON -DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON -DLLAMA_BUILD_TESTS=OFF ${CMAKE_ARGS} -DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined . && \
|
| 49 |
+
cmake --build build --config Release -j$(nproc)
|
| 50 |
+
|
| 51 |
+
RUN mkdir -p /app/lib && \
|
| 52 |
+
find build -name "*.so*" -exec cp -P {} /app/lib \;
|
| 53 |
+
|
| 54 |
+
RUN mkdir -p /app/full \
|
| 55 |
+
&& cp build/bin/* /app/full \
|
| 56 |
+
&& cp *.py /app/full \
|
| 57 |
+
&& cp -r conversion /app/full \
|
| 58 |
+
&& cp -r gguf-py /app/full \
|
| 59 |
+
&& cp -r requirements /app/full \
|
| 60 |
+
&& cp requirements.txt /app/full \
|
| 61 |
+
&& cp .devops/tools.sh /app/full/tools.sh
|
| 62 |
+
|
| 63 |
+
## Base image
|
| 64 |
+
FROM ${BASE_CUDA_RUN_CONTAINER} AS base
|
| 65 |
+
|
| 66 |
+
ARG BUILD_DATE=N/A
|
| 67 |
+
ARG APP_VERSION=N/A
|
| 68 |
+
ARG APP_REVISION=N/A
|
| 69 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 70 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 71 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 72 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 73 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 74 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 75 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 76 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 77 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 78 |
+
|
| 79 |
+
RUN apt-get update \
|
| 80 |
+
&& apt-get install -y libgomp1 curl ffmpeg \
|
| 81 |
+
&& apt autoremove -y \
|
| 82 |
+
&& apt clean -y \
|
| 83 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 84 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 85 |
+
&& find /var/cache -type f -delete
|
| 86 |
+
|
| 87 |
+
COPY --from=build /app/lib/ /app
|
| 88 |
+
|
| 89 |
+
### Full
|
| 90 |
+
FROM base AS full
|
| 91 |
+
|
| 92 |
+
COPY --from=build /app/full /app
|
| 93 |
+
|
| 94 |
+
WORKDIR /app
|
| 95 |
+
|
| 96 |
+
RUN apt-get update \
|
| 97 |
+
&& apt-get install -y \
|
| 98 |
+
git \
|
| 99 |
+
python3 \
|
| 100 |
+
python3-pip \
|
| 101 |
+
python3-wheel \
|
| 102 |
+
&& pip install --break-system-packages --upgrade setuptools \
|
| 103 |
+
&& pip install --break-system-packages -r requirements.txt \
|
| 104 |
+
&& apt autoremove -y \
|
| 105 |
+
&& apt clean -y \
|
| 106 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 107 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 108 |
+
&& find /var/cache -type f -delete
|
| 109 |
+
|
| 110 |
+
|
| 111 |
+
ENTRYPOINT ["/app/tools.sh"]
|
| 112 |
+
|
| 113 |
+
### Light, CLI only
|
| 114 |
+
FROM base AS light
|
| 115 |
+
|
| 116 |
+
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app
|
| 117 |
+
|
| 118 |
+
WORKDIR /app
|
| 119 |
+
|
| 120 |
+
ENTRYPOINT [ "/app/llama-cli" ]
|
| 121 |
+
|
| 122 |
+
### Server, Server only
|
| 123 |
+
FROM base AS server
|
| 124 |
+
|
| 125 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 126 |
+
|
| 127 |
+
COPY --from=build /app/full/llama /app/full/llama-server /app
|
| 128 |
+
|
| 129 |
+
WORKDIR /app
|
| 130 |
+
|
| 131 |
+
HEALTHCHECK CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
| 132 |
+
|
| 133 |
+
ENTRYPOINT [ "/app/llama-server" ]
|
backend/llama.cpp/.devops/intel.Dockerfile
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG ONEAPI_VERSION=2025.3.3-0-devel-ubuntu24.04
|
| 2 |
+
ARG BUILD_DATE=N/A
|
| 3 |
+
ARG APP_VERSION=N/A
|
| 4 |
+
ARG APP_REVISION=N/A
|
| 5 |
+
|
| 6 |
+
## Build Image
|
| 7 |
+
|
| 8 |
+
ARG NODE_VERSION=24
|
| 9 |
+
|
| 10 |
+
FROM docker.io/node:$NODE_VERSION AS web
|
| 11 |
+
|
| 12 |
+
ARG APP_VERSION
|
| 13 |
+
|
| 14 |
+
WORKDIR /app/tools/ui
|
| 15 |
+
|
| 16 |
+
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
| 17 |
+
RUN npm ci
|
| 18 |
+
|
| 19 |
+
COPY tools/ui/ ./
|
| 20 |
+
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
| 21 |
+
|
| 22 |
+
FROM docker.io/intel/deep-learning-essentials:$ONEAPI_VERSION AS build
|
| 23 |
+
|
| 24 |
+
ARG GGML_SYCL_F16=ON
|
| 25 |
+
ARG LEVEL_ZERO_VERSION=1.28.2
|
| 26 |
+
ARG LEVEL_ZERO_UBUNTU_VERSION=u24.04
|
| 27 |
+
RUN apt-get update && \
|
| 28 |
+
apt-get install -y git libssl-dev wget ca-certificates && \
|
| 29 |
+
cd /tmp && \
|
| 30 |
+
wget -q "https://github.com/oneapi-src/level-zero/releases/download/v${LEVEL_ZERO_VERSION}/level-zero_${LEVEL_ZERO_VERSION}%2B${LEVEL_ZERO_UBUNTU_VERSION}_amd64.deb" -O level-zero.deb && \
|
| 31 |
+
wget -q "https://github.com/oneapi-src/level-zero/releases/download/v${LEVEL_ZERO_VERSION}/level-zero-devel_${LEVEL_ZERO_VERSION}%2B${LEVEL_ZERO_UBUNTU_VERSION}_amd64.deb" -O level-zero-devel.deb && \
|
| 32 |
+
apt-get -o Dpkg::Options::="--force-overwrite" install -y ./level-zero.deb ./level-zero-devel.deb && \
|
| 33 |
+
rm -f /tmp/level-zero.deb /tmp/level-zero-devel.deb
|
| 34 |
+
|
| 35 |
+
WORKDIR /app
|
| 36 |
+
|
| 37 |
+
COPY . .
|
| 38 |
+
|
| 39 |
+
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
| 40 |
+
|
| 41 |
+
RUN if [ "${GGML_SYCL_F16}" = "ON" ]; then \
|
| 42 |
+
echo "GGML_SYCL_F16 is set" \
|
| 43 |
+
&& export OPT_SYCL_F16="-DGGML_SYCL_F16=ON" \
|
| 44 |
+
&& export SYCL_PROGRAM_COMPILE_OPTIONS="-cl-fp32-correctly-rounded-divide-sqrt"; \
|
| 45 |
+
fi && \
|
| 46 |
+
echo "Building with dynamic libs" && \
|
| 47 |
+
cmake -B build -DGGML_NATIVE=OFF -DGGML_SYCL=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON -DLLAMA_BUILD_TESTS=OFF ${OPT_SYCL_F16} && \
|
| 48 |
+
cmake --build build --config Release -j$(nproc)
|
| 49 |
+
|
| 50 |
+
RUN mkdir -p /app/lib && \
|
| 51 |
+
find build -name "*.so*" -exec cp -P {} /app/lib \;
|
| 52 |
+
|
| 53 |
+
RUN mkdir -p /app/full \
|
| 54 |
+
&& cp build/bin/* /app/full \
|
| 55 |
+
&& cp *.py /app/full \
|
| 56 |
+
&& cp -r conversion /app/full \
|
| 57 |
+
&& cp -r gguf-py /app/full \
|
| 58 |
+
&& cp -r requirements /app/full \
|
| 59 |
+
&& cp requirements.txt /app/full \
|
| 60 |
+
&& cp .devops/tools.sh /app/full/tools.sh
|
| 61 |
+
|
| 62 |
+
FROM docker.io/intel/deep-learning-essentials:$ONEAPI_VERSION AS base
|
| 63 |
+
|
| 64 |
+
ARG BUILD_DATE=N/A
|
| 65 |
+
ARG APP_VERSION=N/A
|
| 66 |
+
ARG APP_REVISION=N/A
|
| 67 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 68 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 69 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 70 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 71 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 72 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 73 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 74 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 75 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 76 |
+
|
| 77 |
+
#Following versions are for multiple GPUs, since 26.x has known issue:
|
| 78 |
+
# https://github.com/ggml-org/llama.cpp/issues/21747,
|
| 79 |
+
# https://github.com/intel/compute-runtime/issues/921.
|
| 80 |
+
#ARG IGC_VERSION=v2.20.5
|
| 81 |
+
#ARG IGC_VERSION_FULL=2_2.20.5+19972
|
| 82 |
+
#ARG COMPUTE_RUNTIME_VERSION=25.40.35563.10
|
| 83 |
+
#ARG COMPUTE_RUNTIME_VERSION_FULL=25.40.35563.10-0
|
| 84 |
+
#ARG IGDGMM_VERSION=22.8.2
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
ARG IGC_VERSION=v2.34.4
|
| 88 |
+
ARG IGC_VERSION_FULL=2_2.34.4+21428
|
| 89 |
+
ARG COMPUTE_RUNTIME_VERSION=26.18.38308.1
|
| 90 |
+
ARG COMPUTE_RUNTIME_VERSION_FULL=26.18.38308.1-0
|
| 91 |
+
ARG IGDGMM_VERSION=22.10.0
|
| 92 |
+
RUN mkdir /tmp/neo/ && cd /tmp/neo/ \
|
| 93 |
+
&& wget https://github.com/intel/intel-graphics-compiler/releases/download/$IGC_VERSION/intel-igc-core-${IGC_VERSION_FULL}_amd64.deb \
|
| 94 |
+
&& wget https://github.com/intel/intel-graphics-compiler/releases/download/$IGC_VERSION/intel-igc-opencl-${IGC_VERSION_FULL}_amd64.deb \
|
| 95 |
+
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/intel-ocloc-dbgsym_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.ddeb \
|
| 96 |
+
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/intel-ocloc_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.deb \
|
| 97 |
+
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/intel-opencl-icd-dbgsym_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.ddeb \
|
| 98 |
+
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/intel-opencl-icd_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.deb \
|
| 99 |
+
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/libigdgmm12_${IGDGMM_VERSION}_amd64.deb \
|
| 100 |
+
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/libze-intel-gpu1-dbgsym_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.ddeb \
|
| 101 |
+
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/libze-intel-gpu1_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.deb \
|
| 102 |
+
&& dpkg --install *.deb
|
| 103 |
+
|
| 104 |
+
RUN apt-get update \
|
| 105 |
+
&& apt-get install -y libgomp1 curl ffmpeg \
|
| 106 |
+
&& apt autoremove -y \
|
| 107 |
+
&& apt clean -y \
|
| 108 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 109 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 110 |
+
&& find /var/cache -type f -delete
|
| 111 |
+
|
| 112 |
+
### Full
|
| 113 |
+
FROM base AS full
|
| 114 |
+
|
| 115 |
+
COPY --from=build /app/lib/ /app
|
| 116 |
+
COPY --from=build /app/full /app
|
| 117 |
+
|
| 118 |
+
WORKDIR /app
|
| 119 |
+
|
| 120 |
+
RUN apt-get update && \
|
| 121 |
+
apt-get install -y \
|
| 122 |
+
git \
|
| 123 |
+
python3 \
|
| 124 |
+
python3-pip \
|
| 125 |
+
python3-venv && \
|
| 126 |
+
python3 -m venv /opt/venv && \
|
| 127 |
+
. /opt/venv/bin/activate && \
|
| 128 |
+
pip install --upgrade pip setuptools wheel && \
|
| 129 |
+
pip install -r requirements.txt && \
|
| 130 |
+
apt autoremove -y && \
|
| 131 |
+
apt clean -y && \
|
| 132 |
+
rm -rf /tmp/* /var/tmp/* && \
|
| 133 |
+
find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete && \
|
| 134 |
+
find /var/cache -type f -delete
|
| 135 |
+
|
| 136 |
+
ENV PATH="/opt/venv/bin:$PATH"
|
| 137 |
+
|
| 138 |
+
ENTRYPOINT ["/app/tools.sh"]
|
| 139 |
+
|
| 140 |
+
### Light, CLI only
|
| 141 |
+
FROM base AS light
|
| 142 |
+
|
| 143 |
+
COPY --from=build /app/lib/ /app
|
| 144 |
+
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app
|
| 145 |
+
|
| 146 |
+
WORKDIR /app
|
| 147 |
+
|
| 148 |
+
ENTRYPOINT [ "/app/llama-cli" ]
|
| 149 |
+
|
| 150 |
+
### Server, Server only
|
| 151 |
+
FROM base AS server
|
| 152 |
+
|
| 153 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 154 |
+
|
| 155 |
+
COPY --from=build /app/lib/ /app
|
| 156 |
+
COPY --from=build /app/full/llama /app/full/llama-server /app
|
| 157 |
+
|
| 158 |
+
WORKDIR /app
|
| 159 |
+
|
| 160 |
+
HEALTHCHECK CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
| 161 |
+
|
| 162 |
+
ENTRYPOINT [ "/app/llama-server" ]
|
backend/llama.cpp/.devops/llama-cli-cann.Dockerfile
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG ASCEND_VERSION=8.5.0-910b-openeuler22.03-py3.10
|
| 2 |
+
ARG BUILD_DATE=N/A
|
| 3 |
+
ARG APP_VERSION=N/A
|
| 4 |
+
ARG APP_REVISION=N/A
|
| 5 |
+
|
| 6 |
+
FROM docker.io/ascendai/cann:$ASCEND_VERSION AS build
|
| 7 |
+
|
| 8 |
+
WORKDIR /app
|
| 9 |
+
|
| 10 |
+
COPY . .
|
| 11 |
+
|
| 12 |
+
RUN yum install -y gcc g++ cmake make openssl-devel
|
| 13 |
+
ENV ASCEND_TOOLKIT_HOME=/usr/local/Ascend/ascend-toolkit/latest
|
| 14 |
+
ENV LIBRARY_PATH=${ASCEND_TOOLKIT_HOME}/lib64:$LIBRARY_PATH
|
| 15 |
+
ENV LD_LIBRARY_PATH=${ASCEND_TOOLKIT_HOME}/lib64:${ASCEND_TOOLKIT_HOME}/lib64/plugin/opskernel:${ASCEND_TOOLKIT_HOME}/lib64/plugin/nnengine:${ASCEND_TOOLKIT_HOME}/opp/built-in/op_impl/ai_core/tbe/op_tiling:${LD_LIBRARY_PATH}
|
| 16 |
+
ENV PYTHONPATH=${ASCEND_TOOLKIT_HOME}/python/site-packages:${ASCEND_TOOLKIT_HOME}/opp/built-in/op_impl/ai_core/tbe:${PYTHONPATH}
|
| 17 |
+
ENV PATH=${ASCEND_TOOLKIT_HOME}/bin:${ASCEND_TOOLKIT_HOME}/compiler/ccec_compiler/bin:${PATH}
|
| 18 |
+
ENV ASCEND_AICPU_PATH=${ASCEND_TOOLKIT_HOME}
|
| 19 |
+
ENV ASCEND_OPP_PATH=${ASCEND_TOOLKIT_HOME}/opp
|
| 20 |
+
ENV TOOLCHAIN_HOME=${ASCEND_TOOLKIT_HOME}/toolkit
|
| 21 |
+
ENV ASCEND_HOME_PATH=${ASCEND_TOOLKIT_HOME}
|
| 22 |
+
|
| 23 |
+
# find libascend_hal.so, because the drive hasn`t been mounted.
|
| 24 |
+
ENV LD_LIBRARY_PATH=${ASCEND_TOOLKIT_HOME}/runtime/lib64/stub:$LD_LIBRARY_PATH
|
| 25 |
+
|
| 26 |
+
RUN echo "Building with static libs" && \
|
| 27 |
+
source /usr/local/Ascend/ascend-toolkit/set_env.sh --force && \
|
| 28 |
+
cmake -B build -DGGML_NATIVE=OFF -DGGML_CANN=ON -DBUILD_SHARED_LIBS=OFF -DLLAMA_BUILD_TESTS=OFF && \
|
| 29 |
+
cmake --build build --config Release --target llama-cli && \
|
| 30 |
+
cmake --build build --config Release --target llama-completion
|
| 31 |
+
|
| 32 |
+
# TODO: use image with NNRT
|
| 33 |
+
FROM docker.io/ascendai/cann:$ASCEND_VERSION AS runtime
|
| 34 |
+
|
| 35 |
+
ARG BUILD_DATE=N/A
|
| 36 |
+
ARG APP_VERSION=N/A
|
| 37 |
+
ARG APP_REVISION=N/A
|
| 38 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 39 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 40 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 41 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 42 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 43 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 44 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 45 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 46 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 47 |
+
|
| 48 |
+
COPY --from=build /app/build/bin/llama-cli /app/build/bin/llama-completion /
|
| 49 |
+
|
| 50 |
+
ENV LC_ALL=C.utf8
|
| 51 |
+
|
| 52 |
+
ENV ASCEND_TOOLKIT_HOME=/usr/local/Ascend/ascend-toolkit/latest
|
| 53 |
+
ENV LIBRARY_PATH=${ASCEND_TOOLKIT_HOME}/lib64:$LIBRARY_PATH
|
| 54 |
+
ENV LD_LIBRARY_PATH=${ASCEND_TOOLKIT_HOME}/lib64:${ASCEND_TOOLKIT_HOME}/lib64/plugin/opskernel:${ASCEND_TOOLKIT_HOME}/lib64/plugin/nnengine:${ASCEND_TOOLKIT_HOME}/opp/built-in/op_impl/ai_core/tbe/op_tiling:${LD_LIBRARY_PATH}
|
| 55 |
+
ENV PYTHONPATH=${ASCEND_TOOLKIT_HOME}/python/site-packages:${ASCEND_TOOLKIT_HOME}/opp/built-in/op_impl/ai_core/tbe:${PYTHONPATH}
|
| 56 |
+
ENV PATH=${ASCEND_TOOLKIT_HOME}/bin:${ASCEND_TOOLKIT_HOME}/compiler/ccec_compiler/bin:${PATH}
|
| 57 |
+
ENV ASCEND_AICPU_PATH=${ASCEND_TOOLKIT_HOME}
|
| 58 |
+
ENV ASCEND_OPP_PATH=${ASCEND_TOOLKIT_HOME}/opp
|
| 59 |
+
ENV TOOLCHAIN_HOME=${ASCEND_TOOLKIT_HOME}/toolkit
|
| 60 |
+
ENV ASCEND_HOME_PATH=${ASCEND_TOOLKIT_HOME}
|
| 61 |
+
|
| 62 |
+
ENTRYPOINT ["/llama-cli" ]
|
backend/llama.cpp/.devops/llama-cpp-cuda.srpm.spec
ADDED
|
@@ -0,0 +1,85 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# SRPM for building from source and packaging an RPM for RPM-based distros.
|
| 2 |
+
# https://docs.fedoraproject.org/en-US/quick-docs/creating-rpm-packages
|
| 3 |
+
# Built and maintained by John Boero - boeroboy@gmail.com
|
| 4 |
+
# In honor of Seth Vidal https://www.redhat.com/it/blog/thank-you-seth-vidal
|
| 5 |
+
|
| 6 |
+
# Notes for llama.cpp:
|
| 7 |
+
# 1. Tags are currently based on hash - which will not sort asciibetically.
|
| 8 |
+
# We need to declare standard versioning if people want to sort latest releases.
|
| 9 |
+
# 2. Builds for CUDA/OpenCL support are separate, with different depenedencies.
|
| 10 |
+
# 3. NVidia's developer repo must be enabled with nvcc, cublas, clblas, etc installed.
|
| 11 |
+
# Example: https://developer.download.nvidia.com/compute/cuda/repos/fedora37/x86_64/cuda-fedora37.repo
|
| 12 |
+
# 4. OpenCL/CLBLAST support simply requires the ICD loader and basic opencl libraries.
|
| 13 |
+
# It is up to the user to install the correct vendor-specific support.
|
| 14 |
+
|
| 15 |
+
Name: llama.cpp-cuda
|
| 16 |
+
Version: %( date "+%%Y%%m%%d" )
|
| 17 |
+
Release: 1%{?dist}
|
| 18 |
+
Summary: CPU Inference of LLaMA model in pure C/C++ (no CUDA/OpenCL)
|
| 19 |
+
License: MIT
|
| 20 |
+
Source0: https://github.com/ggml-org/llama.cpp/archive/refs/heads/master.tar.gz
|
| 21 |
+
BuildRequires: coreutils make gcc-c++ git cuda-toolkit
|
| 22 |
+
Requires: cuda-toolkit
|
| 23 |
+
URL: https://github.com/ggml-org/llama.cpp
|
| 24 |
+
|
| 25 |
+
%define debug_package %{nil}
|
| 26 |
+
%define source_date_epoch_from_changelog 0
|
| 27 |
+
|
| 28 |
+
%description
|
| 29 |
+
CPU inference for Meta's Lllama2 models using default options.
|
| 30 |
+
|
| 31 |
+
%prep
|
| 32 |
+
%setup -n llama.cpp-master
|
| 33 |
+
|
| 34 |
+
%build
|
| 35 |
+
make -j GGML_CUDA=1
|
| 36 |
+
|
| 37 |
+
%install
|
| 38 |
+
mkdir -p %{buildroot}%{_bindir}/
|
| 39 |
+
cp -p llama-cli %{buildroot}%{_bindir}/llama-cuda-cli
|
| 40 |
+
cp -p llama-completion %{buildroot}%{_bindir}/llama-cuda-completion
|
| 41 |
+
cp -p llama-server %{buildroot}%{_bindir}/llama-cuda-server
|
| 42 |
+
cp -p llama-simple %{buildroot}%{_bindir}/llama-cuda-simple
|
| 43 |
+
|
| 44 |
+
mkdir -p %{buildroot}/usr/lib/systemd/system
|
| 45 |
+
%{__cat} <<EOF > %{buildroot}/usr/lib/systemd/system/llamacuda.service
|
| 46 |
+
[Unit]
|
| 47 |
+
Description=Llama.cpp server, CPU only (no GPU support in this build).
|
| 48 |
+
After=syslog.target network.target local-fs.target remote-fs.target nss-lookup.target
|
| 49 |
+
|
| 50 |
+
[Service]
|
| 51 |
+
Type=simple
|
| 52 |
+
EnvironmentFile=/etc/sysconfig/llama
|
| 53 |
+
ExecStart=/usr/bin/llama-cuda-server $LLAMA_ARGS
|
| 54 |
+
ExecReload=/bin/kill -s HUP $MAINPID
|
| 55 |
+
Restart=never
|
| 56 |
+
|
| 57 |
+
[Install]
|
| 58 |
+
WantedBy=default.target
|
| 59 |
+
EOF
|
| 60 |
+
|
| 61 |
+
mkdir -p %{buildroot}/etc/sysconfig
|
| 62 |
+
%{__cat} <<EOF > %{buildroot}/etc/sysconfig/llama
|
| 63 |
+
LLAMA_ARGS="-m /opt/llama2/ggml-model-f32.bin"
|
| 64 |
+
EOF
|
| 65 |
+
|
| 66 |
+
%clean
|
| 67 |
+
rm -rf %{buildroot}
|
| 68 |
+
rm -rf %{_builddir}/*
|
| 69 |
+
|
| 70 |
+
%files
|
| 71 |
+
%{_bindir}/llama-cuda-cli
|
| 72 |
+
%{_bindir}/llama-cuda-completion
|
| 73 |
+
%{_bindir}/llama-cuda-server
|
| 74 |
+
%{_bindir}/llama-cuda-simple
|
| 75 |
+
/usr/lib/systemd/system/llamacuda.service
|
| 76 |
+
%config /etc/sysconfig/llama
|
| 77 |
+
|
| 78 |
+
%pre
|
| 79 |
+
|
| 80 |
+
%post
|
| 81 |
+
|
| 82 |
+
%preun
|
| 83 |
+
%postun
|
| 84 |
+
|
| 85 |
+
%changelog
|
backend/llama.cpp/.devops/llama-cpp.srpm.spec
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# SRPM for building from source and packaging an RPM for RPM-based distros.
|
| 2 |
+
# https://docs.fedoraproject.org/en-US/quick-docs/creating-rpm-packages
|
| 3 |
+
# Built and maintained by John Boero - boeroboy@gmail.com
|
| 4 |
+
# In honor of Seth Vidal https://www.redhat.com/it/blog/thank-you-seth-vidal
|
| 5 |
+
|
| 6 |
+
# Notes for llama.cpp:
|
| 7 |
+
# 1. Tags are currently based on hash - which will not sort asciibetically.
|
| 8 |
+
# We need to declare standard versioning if people want to sort latest releases.
|
| 9 |
+
# In the meantime, YYYYMMDD format will be used.
|
| 10 |
+
# 2. Builds for CUDA/OpenCL support are separate, with different depenedencies.
|
| 11 |
+
# 3. NVidia's developer repo must be enabled with nvcc, cublas, clblas, etc installed.
|
| 12 |
+
# Example: https://developer.download.nvidia.com/compute/cuda/repos/fedora37/x86_64/cuda-fedora37.repo
|
| 13 |
+
# 4. OpenCL/CLBLAST support simply requires the ICD loader and basic opencl libraries.
|
| 14 |
+
# It is up to the user to install the correct vendor-specific support.
|
| 15 |
+
|
| 16 |
+
Name: llama.cpp
|
| 17 |
+
Version: %( date "+%%Y%%m%%d" )
|
| 18 |
+
Release: 1%{?dist}
|
| 19 |
+
Summary: CPU Inference of LLaMA model in pure C/C++ (no CUDA/OpenCL)
|
| 20 |
+
License: MIT
|
| 21 |
+
Source0: https://github.com/ggml-org/llama.cpp/archive/refs/heads/master.tar.gz
|
| 22 |
+
BuildRequires: coreutils make gcc-c++ git libstdc++-devel
|
| 23 |
+
Requires: libstdc++
|
| 24 |
+
URL: https://github.com/ggml-org/llama.cpp
|
| 25 |
+
|
| 26 |
+
%define debug_package %{nil}
|
| 27 |
+
%define source_date_epoch_from_changelog 0
|
| 28 |
+
|
| 29 |
+
%description
|
| 30 |
+
CPU inference for Meta's Lllama2 models using default options.
|
| 31 |
+
Models are not included in this package and must be downloaded separately.
|
| 32 |
+
|
| 33 |
+
%prep
|
| 34 |
+
%setup -n llama.cpp-master
|
| 35 |
+
|
| 36 |
+
%build
|
| 37 |
+
make -j
|
| 38 |
+
|
| 39 |
+
%install
|
| 40 |
+
mkdir -p %{buildroot}%{_bindir}/
|
| 41 |
+
cp -p llama-cli %{buildroot}%{_bindir}/llama-cli
|
| 42 |
+
cp -p llama-completion %{buildroot}%{_bindir}/llama-completion
|
| 43 |
+
cp -p llama-server %{buildroot}%{_bindir}/llama-server
|
| 44 |
+
cp -p llama-simple %{buildroot}%{_bindir}/llama-simple
|
| 45 |
+
|
| 46 |
+
mkdir -p %{buildroot}/usr/lib/systemd/system
|
| 47 |
+
%{__cat} <<EOF > %{buildroot}/usr/lib/systemd/system/llama.service
|
| 48 |
+
[Unit]
|
| 49 |
+
Description=Llama.cpp server, CPU only (no GPU support in this build).
|
| 50 |
+
After=syslog.target network.target local-fs.target remote-fs.target nss-lookup.target
|
| 51 |
+
|
| 52 |
+
[Service]
|
| 53 |
+
Type=simple
|
| 54 |
+
EnvironmentFile=/etc/sysconfig/llama
|
| 55 |
+
ExecStart=/usr/bin/llama-server $LLAMA_ARGS
|
| 56 |
+
ExecReload=/bin/kill -s HUP $MAINPID
|
| 57 |
+
Restart=never
|
| 58 |
+
|
| 59 |
+
[Install]
|
| 60 |
+
WantedBy=default.target
|
| 61 |
+
EOF
|
| 62 |
+
|
| 63 |
+
mkdir -p %{buildroot}/etc/sysconfig
|
| 64 |
+
%{__cat} <<EOF > %{buildroot}/etc/sysconfig/llama
|
| 65 |
+
LLAMA_ARGS="-m /opt/llama2/ggml-model-f32.bin"
|
| 66 |
+
EOF
|
| 67 |
+
|
| 68 |
+
%clean
|
| 69 |
+
rm -rf %{buildroot}
|
| 70 |
+
rm -rf %{_builddir}/*
|
| 71 |
+
|
| 72 |
+
%files
|
| 73 |
+
%{_bindir}/llama-cli
|
| 74 |
+
%{_bindir}/llama-completion
|
| 75 |
+
%{_bindir}/llama-server
|
| 76 |
+
%{_bindir}/llama-simple
|
| 77 |
+
/usr/lib/systemd/system/llama.service
|
| 78 |
+
%config /etc/sysconfig/llama
|
| 79 |
+
|
| 80 |
+
%pre
|
| 81 |
+
|
| 82 |
+
%post
|
| 83 |
+
|
| 84 |
+
%preun
|
| 85 |
+
%postun
|
| 86 |
+
|
| 87 |
+
%changelog
|
backend/llama.cpp/.devops/musa.Dockerfile
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG UBUNTU_VERSION=22.04
|
| 2 |
+
# This needs to generally match the container host's environment.
|
| 3 |
+
ARG MUSA_VERSION=rc4.3.0
|
| 4 |
+
# Target the MUSA build image
|
| 5 |
+
ARG BASE_MUSA_DEV_CONTAINER=docker.io/mthreads/musa:${MUSA_VERSION}-devel-ubuntu${UBUNTU_VERSION}-amd64
|
| 6 |
+
|
| 7 |
+
ARG BASE_MUSA_RUN_CONTAINER=docker.io/mthreads/musa:${MUSA_VERSION}-runtime-ubuntu${UBUNTU_VERSION}-amd64
|
| 8 |
+
|
| 9 |
+
ARG BUILD_DATE=N/A
|
| 10 |
+
ARG APP_VERSION=N/A
|
| 11 |
+
ARG APP_REVISION=N/A
|
| 12 |
+
|
| 13 |
+
ARG NODE_VERSION=24
|
| 14 |
+
|
| 15 |
+
FROM docker.io/node:$NODE_VERSION AS web
|
| 16 |
+
|
| 17 |
+
ARG APP_VERSION
|
| 18 |
+
|
| 19 |
+
WORKDIR /app/tools/ui
|
| 20 |
+
|
| 21 |
+
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
| 22 |
+
RUN npm ci
|
| 23 |
+
|
| 24 |
+
COPY tools/ui/ ./
|
| 25 |
+
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
| 26 |
+
|
| 27 |
+
FROM ${BASE_MUSA_DEV_CONTAINER} AS build
|
| 28 |
+
|
| 29 |
+
# MUSA architecture to build for (defaults to all supported archs)
|
| 30 |
+
ARG MUSA_DOCKER_ARCH=default
|
| 31 |
+
|
| 32 |
+
RUN apt-get update && \
|
| 33 |
+
apt-get install -y \
|
| 34 |
+
build-essential \
|
| 35 |
+
cmake \
|
| 36 |
+
python3 \
|
| 37 |
+
python3-pip \
|
| 38 |
+
git \
|
| 39 |
+
libssl-dev \
|
| 40 |
+
libgomp1
|
| 41 |
+
|
| 42 |
+
WORKDIR /app
|
| 43 |
+
|
| 44 |
+
COPY . .
|
| 45 |
+
|
| 46 |
+
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
| 47 |
+
|
| 48 |
+
RUN if [ "${MUSA_DOCKER_ARCH}" != "default" ]; then \
|
| 49 |
+
export CMAKE_ARGS="-DMUSA_ARCHITECTURES=${MUSA_DOCKER_ARCH}"; \
|
| 50 |
+
fi && \
|
| 51 |
+
cmake -B build -DGGML_NATIVE=OFF -DGGML_MUSA=ON -DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON -DLLAMA_BUILD_TESTS=OFF ${CMAKE_ARGS} -DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined . && \
|
| 52 |
+
cmake --build build --config Release -j$(nproc)
|
| 53 |
+
|
| 54 |
+
RUN mkdir -p /app/lib && \
|
| 55 |
+
find build -name "*.so*" -exec cp -P {} /app/lib \;
|
| 56 |
+
|
| 57 |
+
RUN mkdir -p /app/full \
|
| 58 |
+
&& cp build/bin/* /app/full \
|
| 59 |
+
&& cp *.py /app/full \
|
| 60 |
+
&& cp -r conversion /app/full \
|
| 61 |
+
&& cp -r gguf-py /app/full \
|
| 62 |
+
&& cp -r requirements /app/full \
|
| 63 |
+
&& cp requirements.txt /app/full \
|
| 64 |
+
&& cp .devops/tools.sh /app/full/tools.sh
|
| 65 |
+
|
| 66 |
+
## Base image
|
| 67 |
+
FROM ${BASE_MUSA_RUN_CONTAINER} AS base
|
| 68 |
+
|
| 69 |
+
ARG BUILD_DATE=N/A
|
| 70 |
+
ARG APP_VERSION=N/A
|
| 71 |
+
ARG APP_REVISION=N/A
|
| 72 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 73 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 74 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 75 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 76 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 77 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 78 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 79 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 80 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 81 |
+
|
| 82 |
+
RUN apt-get update \
|
| 83 |
+
&& apt-get install -y libgomp1 curl ffmpeg \
|
| 84 |
+
&& apt autoremove -y \
|
| 85 |
+
&& apt clean -y \
|
| 86 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 87 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 88 |
+
&& find /var/cache -type f -delete
|
| 89 |
+
|
| 90 |
+
COPY --from=build /app/lib/ /app
|
| 91 |
+
|
| 92 |
+
### Full
|
| 93 |
+
FROM base AS full
|
| 94 |
+
|
| 95 |
+
COPY --from=build /app/full /app
|
| 96 |
+
|
| 97 |
+
WORKDIR /app
|
| 98 |
+
|
| 99 |
+
RUN apt-get update \
|
| 100 |
+
&& apt-get install -y \
|
| 101 |
+
git \
|
| 102 |
+
python3 \
|
| 103 |
+
python3-pip \
|
| 104 |
+
&& pip install --upgrade pip setuptools wheel \
|
| 105 |
+
&& pip install -r requirements.txt \
|
| 106 |
+
&& apt autoremove -y \
|
| 107 |
+
&& apt clean -y \
|
| 108 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 109 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 110 |
+
&& find /var/cache -type f -delete
|
| 111 |
+
|
| 112 |
+
|
| 113 |
+
ENTRYPOINT ["/app/tools.sh"]
|
| 114 |
+
|
| 115 |
+
### Light, CLI only
|
| 116 |
+
FROM base AS light
|
| 117 |
+
|
| 118 |
+
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app
|
| 119 |
+
|
| 120 |
+
WORKDIR /app
|
| 121 |
+
|
| 122 |
+
ENTRYPOINT [ "/app/llama-cli" ]
|
| 123 |
+
|
| 124 |
+
### Server, Server only
|
| 125 |
+
FROM base AS server
|
| 126 |
+
|
| 127 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 128 |
+
|
| 129 |
+
COPY --from=build /app/full/llama /app/full/llama-server /app
|
| 130 |
+
|
| 131 |
+
WORKDIR /app
|
| 132 |
+
|
| 133 |
+
HEALTHCHECK CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
| 134 |
+
|
| 135 |
+
ENTRYPOINT [ "/app/llama-server" ]
|
backend/llama.cpp/.devops/nix/apps.nix
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
perSystem =
|
| 3 |
+
{ config, lib, ... }:
|
| 4 |
+
{
|
| 5 |
+
apps =
|
| 6 |
+
let
|
| 7 |
+
inherit (config.packages) default;
|
| 8 |
+
binaries = [
|
| 9 |
+
"llama-cli"
|
| 10 |
+
"llama-embedding"
|
| 11 |
+
"llama-server"
|
| 12 |
+
"llama-quantize"
|
| 13 |
+
];
|
| 14 |
+
mkApp = name: {
|
| 15 |
+
type = "app";
|
| 16 |
+
program = "${default}/bin/${name}";
|
| 17 |
+
};
|
| 18 |
+
in
|
| 19 |
+
lib.genAttrs binaries mkApp;
|
| 20 |
+
};
|
| 21 |
+
}
|
backend/llama.cpp/.devops/nix/devshells.nix
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{ inputs, ... }:
|
| 2 |
+
|
| 3 |
+
{
|
| 4 |
+
perSystem =
|
| 5 |
+
{
|
| 6 |
+
config,
|
| 7 |
+
lib,
|
| 8 |
+
system,
|
| 9 |
+
...
|
| 10 |
+
}:
|
| 11 |
+
{
|
| 12 |
+
devShells =
|
| 13 |
+
let
|
| 14 |
+
pkgs = import inputs.nixpkgs { inherit system; };
|
| 15 |
+
stdenv = pkgs.stdenv;
|
| 16 |
+
scripts = config.packages.python-scripts;
|
| 17 |
+
in
|
| 18 |
+
lib.pipe (config.packages) [
|
| 19 |
+
(lib.concatMapAttrs (
|
| 20 |
+
name: package: {
|
| 21 |
+
${name} = pkgs.mkShell {
|
| 22 |
+
name = "${name}";
|
| 23 |
+
inputsFrom = [ package ];
|
| 24 |
+
shellHook = ''
|
| 25 |
+
echo "Entering ${name} devShell"
|
| 26 |
+
'';
|
| 27 |
+
};
|
| 28 |
+
"${name}-extra" =
|
| 29 |
+
if (name == "python-scripts") then
|
| 30 |
+
null
|
| 31 |
+
else
|
| 32 |
+
pkgs.mkShell {
|
| 33 |
+
name = "${name}-extra";
|
| 34 |
+
inputsFrom = [
|
| 35 |
+
package
|
| 36 |
+
scripts
|
| 37 |
+
];
|
| 38 |
+
# Extra packages that *may* be used by some scripts
|
| 39 |
+
packages = [
|
| 40 |
+
pkgs.python3Packages.tiktoken
|
| 41 |
+
];
|
| 42 |
+
shellHook = ''
|
| 43 |
+
echo "Entering ${name} devShell"
|
| 44 |
+
addToSearchPath "LD_LIBRARY_PATH" "${lib.getLib stdenv.cc.cc}/lib"
|
| 45 |
+
'';
|
| 46 |
+
};
|
| 47 |
+
}
|
| 48 |
+
))
|
| 49 |
+
(lib.filterAttrs (name: value: value != null))
|
| 50 |
+
];
|
| 51 |
+
};
|
| 52 |
+
}
|
backend/llama.cpp/.devops/nix/docker.nix
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
lib,
|
| 3 |
+
dockerTools,
|
| 4 |
+
buildEnv,
|
| 5 |
+
llama-cpp,
|
| 6 |
+
interactive ? true,
|
| 7 |
+
coreutils,
|
| 8 |
+
}:
|
| 9 |
+
|
| 10 |
+
# A tar that can be fed into `docker load`:
|
| 11 |
+
#
|
| 12 |
+
# $ nix build .#llamaPackages.docker
|
| 13 |
+
# $ docker load < result
|
| 14 |
+
|
| 15 |
+
# For details and variations cf.
|
| 16 |
+
# - https://nixos.org/manual/nixpkgs/unstable/#ssec-pkgs-dockerTools-buildLayeredImage
|
| 17 |
+
# - https://discourse.nixos.org/t/a-faster-dockertools-buildimage-prototype/16922
|
| 18 |
+
# - https://nixery.dev/
|
| 19 |
+
|
| 20 |
+
# Approximate (compressed) sizes, at the time of writing, are:
|
| 21 |
+
#
|
| 22 |
+
# .#llamaPackages.docker: 125M;
|
| 23 |
+
# .#llamaPackagesCuda.docker: 537M;
|
| 24 |
+
# .#legacyPackages.aarch64-linux.llamaPackagesXavier.docker: 415M.
|
| 25 |
+
|
| 26 |
+
dockerTools.buildLayeredImage {
|
| 27 |
+
name = llama-cpp.pname;
|
| 28 |
+
tag = "latest";
|
| 29 |
+
|
| 30 |
+
contents =
|
| 31 |
+
[ llama-cpp ]
|
| 32 |
+
++ lib.optionals interactive [
|
| 33 |
+
coreutils
|
| 34 |
+
dockerTools.binSh
|
| 35 |
+
dockerTools.caCertificates
|
| 36 |
+
];
|
| 37 |
+
}
|
backend/llama.cpp/.devops/nix/jetson-support.nix
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{ inputs, ... }:
|
| 2 |
+
{
|
| 3 |
+
perSystem =
|
| 4 |
+
{
|
| 5 |
+
config,
|
| 6 |
+
system,
|
| 7 |
+
lib,
|
| 8 |
+
pkgsCuda,
|
| 9 |
+
...
|
| 10 |
+
}:
|
| 11 |
+
{
|
| 12 |
+
legacyPackages =
|
| 13 |
+
let
|
| 14 |
+
caps.llamaPackagesXavier = "7.2";
|
| 15 |
+
caps.llamaPackagesOrin = "8.7";
|
| 16 |
+
caps.llamaPackagesTX2 = "6.2";
|
| 17 |
+
caps.llamaPackagesNano = "5.3";
|
| 18 |
+
|
| 19 |
+
pkgsFor =
|
| 20 |
+
cap:
|
| 21 |
+
import inputs.nixpkgs {
|
| 22 |
+
inherit system;
|
| 23 |
+
config = {
|
| 24 |
+
cudaSupport = true;
|
| 25 |
+
cudaCapabilities = [ cap ];
|
| 26 |
+
cudaEnableForwardCompat = false;
|
| 27 |
+
inherit (pkgsCuda.config) allowUnfreePredicate;
|
| 28 |
+
};
|
| 29 |
+
};
|
| 30 |
+
in
|
| 31 |
+
builtins.mapAttrs (name: cap: (pkgsFor cap).callPackage ./scope.nix { }) caps;
|
| 32 |
+
|
| 33 |
+
packages = lib.optionalAttrs (system == "aarch64-linux") {
|
| 34 |
+
jetson-xavier = config.legacyPackages.llamaPackagesXavier.llama-cpp;
|
| 35 |
+
jetson-orin = config.legacyPackages.llamaPackagesOrin.llama-cpp;
|
| 36 |
+
jetson-nano = config.legacyPackages.llamaPackagesNano.llama-cpp;
|
| 37 |
+
};
|
| 38 |
+
};
|
| 39 |
+
}
|
backend/llama.cpp/.devops/nix/nixpkgs-instances.nix
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{ inputs, ... }:
|
| 2 |
+
{
|
| 3 |
+
# The _module.args definitions are passed on to modules as arguments. E.g.
|
| 4 |
+
# the module `{ pkgs ... }: { /* config */ }` implicitly uses
|
| 5 |
+
# `_module.args.pkgs` (defined in this case by flake-parts).
|
| 6 |
+
perSystem =
|
| 7 |
+
{ lib, system, ... }:
|
| 8 |
+
{
|
| 9 |
+
_module.args = {
|
| 10 |
+
# Note: bringing up https://zimbatm.com/notes/1000-instances-of-nixpkgs
|
| 11 |
+
# again, the below creates several nixpkgs instances which the
|
| 12 |
+
# flake-centric CLI will be forced to evaluate e.g. on `nix flake show`.
|
| 13 |
+
#
|
| 14 |
+
# This is currently "slow" and "expensive", on a certain scale.
|
| 15 |
+
# This also isn't "right" in that this hinders dependency injection at
|
| 16 |
+
# the level of flake inputs. This might get removed in the foreseeable
|
| 17 |
+
# future.
|
| 18 |
+
#
|
| 19 |
+
# Note that you can use these expressions without Nix
|
| 20 |
+
# (`pkgs.callPackage ./devops/nix/scope.nix { }` is the entry point).
|
| 21 |
+
|
| 22 |
+
pkgsCuda = import inputs.nixpkgs {
|
| 23 |
+
inherit system;
|
| 24 |
+
# Ensure dependencies use CUDA consistently (e.g. that openmpi, ucc,
|
| 25 |
+
# and ucx are built with CUDA support)
|
| 26 |
+
config.cudaSupport = true;
|
| 27 |
+
config.allowUnfreePredicate =
|
| 28 |
+
p:
|
| 29 |
+
builtins.all (
|
| 30 |
+
license:
|
| 31 |
+
license.free
|
| 32 |
+
|| builtins.elem license.shortName [
|
| 33 |
+
"CUDA EULA"
|
| 34 |
+
"cuDNN EULA"
|
| 35 |
+
]
|
| 36 |
+
) (p.meta.licenses or (lib.toList p.meta.license));
|
| 37 |
+
};
|
| 38 |
+
# Ensure dependencies use ROCm consistently
|
| 39 |
+
pkgsRocm = import inputs.nixpkgs {
|
| 40 |
+
inherit system;
|
| 41 |
+
config.rocmSupport = true;
|
| 42 |
+
};
|
| 43 |
+
};
|
| 44 |
+
};
|
| 45 |
+
}
|
backend/llama.cpp/.devops/nix/package-gguf-py.nix
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
lib,
|
| 3 |
+
llamaVersion,
|
| 4 |
+
numpy,
|
| 5 |
+
tqdm,
|
| 6 |
+
requests,
|
| 7 |
+
sentencepiece,
|
| 8 |
+
pyyaml,
|
| 9 |
+
poetry-core,
|
| 10 |
+
buildPythonPackage,
|
| 11 |
+
pytestCheckHook,
|
| 12 |
+
}:
|
| 13 |
+
|
| 14 |
+
buildPythonPackage {
|
| 15 |
+
pname = "gguf";
|
| 16 |
+
version = llamaVersion;
|
| 17 |
+
pyproject = true;
|
| 18 |
+
nativeBuildInputs = [ poetry-core ];
|
| 19 |
+
propagatedBuildInputs = [
|
| 20 |
+
numpy
|
| 21 |
+
tqdm
|
| 22 |
+
sentencepiece
|
| 23 |
+
pyyaml
|
| 24 |
+
requests
|
| 25 |
+
];
|
| 26 |
+
src = lib.cleanSource ../../gguf-py;
|
| 27 |
+
pythonImportsCheck = [
|
| 28 |
+
"numpy"
|
| 29 |
+
"gguf"
|
| 30 |
+
];
|
| 31 |
+
nativeCheckInputs = [ pytestCheckHook ];
|
| 32 |
+
doCheck = true;
|
| 33 |
+
meta = with lib; {
|
| 34 |
+
description = "Python package for writing binary files in the GGUF format";
|
| 35 |
+
license = licenses.mit;
|
| 36 |
+
maintainers = [ maintainers.ditsuke ];
|
| 37 |
+
};
|
| 38 |
+
}
|
backend/llama.cpp/.devops/nix/package.nix
ADDED
|
@@ -0,0 +1,275 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
lib,
|
| 3 |
+
glibc,
|
| 4 |
+
config,
|
| 5 |
+
stdenv,
|
| 6 |
+
stdenvNoCC,
|
| 7 |
+
runCommand,
|
| 8 |
+
cmake,
|
| 9 |
+
ninja,
|
| 10 |
+
pkg-config,
|
| 11 |
+
git,
|
| 12 |
+
mpi,
|
| 13 |
+
blas,
|
| 14 |
+
cudaPackages,
|
| 15 |
+
autoAddDriverRunpath,
|
| 16 |
+
darwin,
|
| 17 |
+
rocmPackages,
|
| 18 |
+
vulkan-headers,
|
| 19 |
+
vulkan-loader,
|
| 20 |
+
openssl,
|
| 21 |
+
shaderc,
|
| 22 |
+
spirv-headers,
|
| 23 |
+
nodejs,
|
| 24 |
+
importNpmLock,
|
| 25 |
+
useBlas ?
|
| 26 |
+
builtins.all (x: !x) [
|
| 27 |
+
useCuda
|
| 28 |
+
useMetalKit
|
| 29 |
+
useRocm
|
| 30 |
+
useVulkan
|
| 31 |
+
]
|
| 32 |
+
&& blas.meta.available,
|
| 33 |
+
useCuda ? config.cudaSupport,
|
| 34 |
+
useMetalKit ? stdenv.isAarch64 && stdenv.isDarwin,
|
| 35 |
+
# Increases the runtime closure size by ~700M
|
| 36 |
+
useMpi ? false,
|
| 37 |
+
useRocm ? config.rocmSupport,
|
| 38 |
+
rocmGpuTargets ? builtins.concatStringsSep ";" rocmPackages.clr.gpuTargets,
|
| 39 |
+
useVulkan ? false,
|
| 40 |
+
useRpc ? false,
|
| 41 |
+
llamaVersion ? "0.0.0", # Arbitrary version, substituted by the flake
|
| 42 |
+
|
| 43 |
+
# It's necessary to consistently use backendStdenv when building with CUDA support,
|
| 44 |
+
# otherwise we get libstdc++ errors downstream.
|
| 45 |
+
effectiveStdenv ? if useCuda then cudaPackages.backendStdenv else stdenv,
|
| 46 |
+
enableStatic ? effectiveStdenv.hostPlatform.isStatic,
|
| 47 |
+
precompileMetalShaders ? false,
|
| 48 |
+
useWebUi ? true,
|
| 49 |
+
}:
|
| 50 |
+
|
| 51 |
+
let
|
| 52 |
+
inherit (lib)
|
| 53 |
+
cmakeBool
|
| 54 |
+
cmakeFeature
|
| 55 |
+
optionalAttrs
|
| 56 |
+
optionals
|
| 57 |
+
strings
|
| 58 |
+
;
|
| 59 |
+
|
| 60 |
+
stdenv = throw "Use effectiveStdenv instead";
|
| 61 |
+
|
| 62 |
+
suffices =
|
| 63 |
+
lib.optionals useBlas [ "BLAS" ]
|
| 64 |
+
++ lib.optionals useCuda [ "CUDA" ]
|
| 65 |
+
++ lib.optionals useMetalKit [ "MetalKit" ]
|
| 66 |
+
++ lib.optionals useMpi [ "MPI" ]
|
| 67 |
+
++ lib.optionals useRocm [ "ROCm" ]
|
| 68 |
+
++ lib.optionals useVulkan [ "Vulkan" ];
|
| 69 |
+
|
| 70 |
+
pnameSuffix =
|
| 71 |
+
strings.optionalString (suffices != [ ])
|
| 72 |
+
"-${strings.concatMapStringsSep "-" strings.toLower suffices}";
|
| 73 |
+
descriptionSuffix = strings.optionalString (
|
| 74 |
+
suffices != [ ]
|
| 75 |
+
) ", accelerated with ${strings.concatStringsSep ", " suffices}";
|
| 76 |
+
|
| 77 |
+
xcrunHost = runCommand "xcrunHost" { } ''
|
| 78 |
+
mkdir -p $out/bin
|
| 79 |
+
ln -s /usr/bin/xcrun $out/bin
|
| 80 |
+
'';
|
| 81 |
+
|
| 82 |
+
# apple_sdk is supposed to choose sane defaults, no need to handle isAarch64
|
| 83 |
+
# separately
|
| 84 |
+
darwinBuildInputs =
|
| 85 |
+
with darwin.apple_sdk.frameworks;
|
| 86 |
+
[
|
| 87 |
+
Accelerate
|
| 88 |
+
CoreVideo
|
| 89 |
+
CoreGraphics
|
| 90 |
+
]
|
| 91 |
+
++ optionals useMetalKit [ MetalKit ];
|
| 92 |
+
|
| 93 |
+
cudaBuildInputs = with cudaPackages; [
|
| 94 |
+
cuda_cudart
|
| 95 |
+
cuda_cccl # <nv/target>
|
| 96 |
+
libcublas
|
| 97 |
+
];
|
| 98 |
+
|
| 99 |
+
rocmBuildInputs = with rocmPackages; [
|
| 100 |
+
clr
|
| 101 |
+
hipblas
|
| 102 |
+
rocblas
|
| 103 |
+
];
|
| 104 |
+
|
| 105 |
+
vulkanBuildInputs = [
|
| 106 |
+
vulkan-headers
|
| 107 |
+
vulkan-loader
|
| 108 |
+
shaderc
|
| 109 |
+
spirv-headers
|
| 110 |
+
];
|
| 111 |
+
in
|
| 112 |
+
|
| 113 |
+
effectiveStdenv.mkDerivation (finalAttrs: {
|
| 114 |
+
pname = "llama-cpp${pnameSuffix}";
|
| 115 |
+
version = llamaVersion;
|
| 116 |
+
|
| 117 |
+
# Note: none of the files discarded here are visible in the sandbox or
|
| 118 |
+
# affect the output hash. This also means they can be modified without
|
| 119 |
+
# triggering a rebuild.
|
| 120 |
+
src = lib.cleanSourceWith {
|
| 121 |
+
filter =
|
| 122 |
+
name: type:
|
| 123 |
+
let
|
| 124 |
+
noneOf = builtins.all (x: !x);
|
| 125 |
+
baseName = baseNameOf name;
|
| 126 |
+
in
|
| 127 |
+
noneOf [
|
| 128 |
+
(lib.hasSuffix ".nix" name) # Ignore *.nix files when computing outPaths
|
| 129 |
+
(lib.hasSuffix ".md" name) # Ignore *.md changes whe computing outPaths
|
| 130 |
+
(lib.hasPrefix "." baseName) # Skip hidden files and directories
|
| 131 |
+
(baseName == "flake.lock")
|
| 132 |
+
];
|
| 133 |
+
src = lib.cleanSource ../../.;
|
| 134 |
+
};
|
| 135 |
+
|
| 136 |
+
# Builds the webui locally, taking care not to require updating any sha256 hash.
|
| 137 |
+
webui = stdenvNoCC.mkDerivation {
|
| 138 |
+
pname = "webui";
|
| 139 |
+
version = llamaVersion;
|
| 140 |
+
src = lib.cleanSource ../../tools/ui;
|
| 141 |
+
|
| 142 |
+
nativeBuildInputs = [
|
| 143 |
+
nodejs
|
| 144 |
+
importNpmLock.linkNodeModulesHook
|
| 145 |
+
];
|
| 146 |
+
|
| 147 |
+
# no sha256 required when using buildNodeModules
|
| 148 |
+
npmDeps = importNpmLock.buildNodeModules {
|
| 149 |
+
npmRoot = ../../tools/ui;
|
| 150 |
+
inherit nodejs;
|
| 151 |
+
};
|
| 152 |
+
|
| 153 |
+
installPhase = ''
|
| 154 |
+
LLAMA_UI_OUT_DIR=$out npm run build --offline
|
| 155 |
+
'';
|
| 156 |
+
};
|
| 157 |
+
|
| 158 |
+
postPatch = lib.optionalString useWebUi ''
|
| 159 |
+
cp -r ${finalAttrs.webui} tools/ui/dist
|
| 160 |
+
chmod -R u+w tools/ui/dist
|
| 161 |
+
'';
|
| 162 |
+
|
| 163 |
+
# With PR#6015 https://github.com/ggml-org/llama.cpp/pull/6015,
|
| 164 |
+
# `default.metallib` may be compiled with Metal compiler from XCode
|
| 165 |
+
# and we need to escape sandbox on MacOS to access Metal compiler.
|
| 166 |
+
# `xcrun` is used find the path of the Metal compiler, which is varible
|
| 167 |
+
# and not on $PATH
|
| 168 |
+
# see https://github.com/ggml-org/llama.cpp/pull/6118 for discussion
|
| 169 |
+
__noChroot = effectiveStdenv.isDarwin && useMetalKit && precompileMetalShaders;
|
| 170 |
+
|
| 171 |
+
nativeBuildInputs =
|
| 172 |
+
[
|
| 173 |
+
cmake
|
| 174 |
+
ninja
|
| 175 |
+
pkg-config
|
| 176 |
+
git
|
| 177 |
+
]
|
| 178 |
+
++ optionals useCuda [
|
| 179 |
+
cudaPackages.cuda_nvcc
|
| 180 |
+
|
| 181 |
+
autoAddDriverRunpath
|
| 182 |
+
]
|
| 183 |
+
++ optionals (effectiveStdenv.hostPlatform.isGnu && enableStatic) [ glibc.static ]
|
| 184 |
+
++ optionals (effectiveStdenv.isDarwin && useMetalKit && precompileMetalShaders) [ xcrunHost ];
|
| 185 |
+
|
| 186 |
+
buildInputs =
|
| 187 |
+
optionals effectiveStdenv.isDarwin darwinBuildInputs
|
| 188 |
+
++ optionals useCuda cudaBuildInputs
|
| 189 |
+
++ optionals useMpi [ mpi ]
|
| 190 |
+
++ optionals useRocm rocmBuildInputs
|
| 191 |
+
++ optionals useBlas [ blas ]
|
| 192 |
+
++ optionals useVulkan vulkanBuildInputs
|
| 193 |
+
++ [ openssl ];
|
| 194 |
+
|
| 195 |
+
cmakeFlags =
|
| 196 |
+
[
|
| 197 |
+
(cmakeBool "LLAMA_BUILD_SERVER" true)
|
| 198 |
+
(cmakeBool "LLAMA_BUILD_WEBUI" useWebUi)
|
| 199 |
+
(cmakeBool "BUILD_SHARED_LIBS" (!enableStatic))
|
| 200 |
+
(cmakeBool "CMAKE_SKIP_BUILD_RPATH" true)
|
| 201 |
+
(cmakeBool "GGML_NATIVE" false)
|
| 202 |
+
(cmakeBool "GGML_BLAS" useBlas)
|
| 203 |
+
(cmakeBool "GGML_CUDA" useCuda)
|
| 204 |
+
(cmakeBool "GGML_HIP" useRocm)
|
| 205 |
+
(cmakeBool "GGML_METAL" useMetalKit)
|
| 206 |
+
(cmakeBool "GGML_VULKAN" useVulkan)
|
| 207 |
+
(cmakeBool "GGML_STATIC" enableStatic)
|
| 208 |
+
(cmakeBool "GGML_RPC" useRpc)
|
| 209 |
+
]
|
| 210 |
+
++ optionals useCuda [
|
| 211 |
+
(
|
| 212 |
+
with cudaPackages.flags;
|
| 213 |
+
cmakeFeature "CMAKE_CUDA_ARCHITECTURES" (
|
| 214 |
+
builtins.concatStringsSep ";" (map dropDot cudaCapabilities)
|
| 215 |
+
)
|
| 216 |
+
)
|
| 217 |
+
]
|
| 218 |
+
++ optionals useRocm [
|
| 219 |
+
(cmakeFeature "CMAKE_HIP_COMPILER" "${rocmPackages.llvm.clang}/bin/clang")
|
| 220 |
+
(cmakeFeature "CMAKE_HIP_ARCHITECTURES" rocmGpuTargets)
|
| 221 |
+
]
|
| 222 |
+
++ optionals useMetalKit [
|
| 223 |
+
(lib.cmakeFeature "CMAKE_C_FLAGS" "-D__ARM_FEATURE_DOTPROD=1")
|
| 224 |
+
(cmakeBool "GGML_METAL_EMBED_LIBRARY" (!precompileMetalShaders))
|
| 225 |
+
];
|
| 226 |
+
|
| 227 |
+
# Environment variables needed for ROCm
|
| 228 |
+
env = optionalAttrs useRocm {
|
| 229 |
+
ROCM_PATH = "${rocmPackages.clr}";
|
| 230 |
+
HIP_DEVICE_LIB_PATH = "${rocmPackages.rocm-device-libs}/amdgcn/bitcode";
|
| 231 |
+
};
|
| 232 |
+
|
| 233 |
+
# TODO(SomeoneSerge): It's better to add proper install targets at the CMake level,
|
| 234 |
+
# if they haven't been added yet.
|
| 235 |
+
postInstall = ''
|
| 236 |
+
mkdir -p $out/include
|
| 237 |
+
cp $src/include/llama.h $out/include/
|
| 238 |
+
'';
|
| 239 |
+
|
| 240 |
+
meta = {
|
| 241 |
+
# Configurations we don't want even the CI to evaluate. Results in the
|
| 242 |
+
# "unsupported platform" messages. This is mostly a no-op, because
|
| 243 |
+
# cudaPackages would've refused to evaluate anyway.
|
| 244 |
+
badPlatforms = optionals useCuda lib.platforms.darwin;
|
| 245 |
+
|
| 246 |
+
# Configurations that are known to result in build failures. Can be
|
| 247 |
+
# overridden by importing Nixpkgs with `allowBroken = true`.
|
| 248 |
+
broken = (useMetalKit && !effectiveStdenv.isDarwin);
|
| 249 |
+
|
| 250 |
+
description = "Inference of LLaMA model in pure C/C++${descriptionSuffix}";
|
| 251 |
+
homepage = "https://github.com/ggml-org/llama.cpp/";
|
| 252 |
+
license = lib.licenses.mit;
|
| 253 |
+
|
| 254 |
+
# Accommodates `nix run` and `lib.getExe`
|
| 255 |
+
mainProgram = "llama-cli";
|
| 256 |
+
|
| 257 |
+
# These people might respond, on the best effort basis, if you ping them
|
| 258 |
+
# in case of Nix-specific regressions or for reviewing Nix-specific PRs.
|
| 259 |
+
# Consider adding yourself to this list if you want to ensure this flake
|
| 260 |
+
# stays maintained and you're willing to invest your time. Do not add
|
| 261 |
+
# other people without their consent. Consider removing people after
|
| 262 |
+
# they've been unreachable for long periods of time.
|
| 263 |
+
|
| 264 |
+
# Note that lib.maintainers is defined in Nixpkgs, but you may just add
|
| 265 |
+
# an attrset following the same format as in
|
| 266 |
+
# https://github.com/NixOS/nixpkgs/blob/f36a80e54da29775c78d7eff0e628c2b4e34d1d7/maintainers/maintainer-list.nix
|
| 267 |
+
maintainers = with lib.maintainers; [
|
| 268 |
+
philiptaron
|
| 269 |
+
SomeoneSerge
|
| 270 |
+
];
|
| 271 |
+
|
| 272 |
+
# Extend `badPlatforms` instead
|
| 273 |
+
platforms = lib.platforms.all;
|
| 274 |
+
};
|
| 275 |
+
})
|
backend/llama.cpp/.devops/nix/python-scripts.nix
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
lib,
|
| 3 |
+
stdenv,
|
| 4 |
+
buildPythonPackage,
|
| 5 |
+
poetry-core,
|
| 6 |
+
mkShell,
|
| 7 |
+
python3Packages,
|
| 8 |
+
gguf-py,
|
| 9 |
+
}@inputs:
|
| 10 |
+
|
| 11 |
+
let
|
| 12 |
+
llama-python-deps = with python3Packages; [
|
| 13 |
+
numpy
|
| 14 |
+
sentencepiece
|
| 15 |
+
transformers
|
| 16 |
+
protobuf
|
| 17 |
+
torchWithoutCuda
|
| 18 |
+
gguf-py
|
| 19 |
+
tqdm
|
| 20 |
+
|
| 21 |
+
# for scripts/compare-llama-bench.py
|
| 22 |
+
gitpython
|
| 23 |
+
tabulate
|
| 24 |
+
|
| 25 |
+
# for examples/pydantic-models-to-grammar-examples.py
|
| 26 |
+
docstring-parser
|
| 27 |
+
pydantic
|
| 28 |
+
|
| 29 |
+
];
|
| 30 |
+
|
| 31 |
+
llama-python-test-deps = with python3Packages; [
|
| 32 |
+
# Server bench
|
| 33 |
+
matplotlib
|
| 34 |
+
|
| 35 |
+
# server tests
|
| 36 |
+
openai
|
| 37 |
+
pytest
|
| 38 |
+
prometheus-client
|
| 39 |
+
];
|
| 40 |
+
in
|
| 41 |
+
|
| 42 |
+
buildPythonPackage ({
|
| 43 |
+
pname = "llama-scripts";
|
| 44 |
+
version = "0.0.0";
|
| 45 |
+
pyproject = true;
|
| 46 |
+
|
| 47 |
+
# NOTE: The files filtered out here are not visible in the build sandbox, neither
|
| 48 |
+
# do they affect the output hash. They can be modified without triggering a rebuild.
|
| 49 |
+
src = lib.cleanSourceWith {
|
| 50 |
+
filter =
|
| 51 |
+
name: type:
|
| 52 |
+
let
|
| 53 |
+
any = builtins.any (x: x);
|
| 54 |
+
baseName = builtins.baseNameOf name;
|
| 55 |
+
in
|
| 56 |
+
any [
|
| 57 |
+
(lib.hasSuffix ".py" name)
|
| 58 |
+
(baseName == "README.md")
|
| 59 |
+
(baseName == "pyproject.toml")
|
| 60 |
+
];
|
| 61 |
+
src = lib.cleanSource ../../.;
|
| 62 |
+
};
|
| 63 |
+
nativeBuildInputs = [ poetry-core ];
|
| 64 |
+
nativeCheckInputs = llama-python-test-deps;
|
| 65 |
+
dependencies = llama-python-deps;
|
| 66 |
+
})
|
backend/llama.cpp/.devops/nix/scope.nix
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
lib,
|
| 3 |
+
newScope,
|
| 4 |
+
python3,
|
| 5 |
+
llamaVersion ? "0.0.0",
|
| 6 |
+
}:
|
| 7 |
+
|
| 8 |
+
let
|
| 9 |
+
pythonPackages = python3.pkgs;
|
| 10 |
+
in
|
| 11 |
+
|
| 12 |
+
# We're using `makeScope` instead of just writing out an attrset
|
| 13 |
+
# because it allows users to apply overlays later using `overrideScope'`.
|
| 14 |
+
# Cf. https://noogle.dev/f/lib/makeScope
|
| 15 |
+
|
| 16 |
+
lib.makeScope newScope (self: {
|
| 17 |
+
inherit llamaVersion;
|
| 18 |
+
gguf-py = self.callPackage ./package-gguf-py.nix {
|
| 19 |
+
inherit (pythonPackages)
|
| 20 |
+
numpy
|
| 21 |
+
tqdm
|
| 22 |
+
sentencepiece
|
| 23 |
+
pyyaml
|
| 24 |
+
pytestCheckHook
|
| 25 |
+
requests
|
| 26 |
+
buildPythonPackage
|
| 27 |
+
poetry-core
|
| 28 |
+
;
|
| 29 |
+
};
|
| 30 |
+
python-scripts = self.callPackage ./python-scripts.nix { inherit (pythonPackages) buildPythonPackage poetry-core; };
|
| 31 |
+
llama-cpp = self.callPackage ./package.nix { };
|
| 32 |
+
docker = self.callPackage ./docker.nix { };
|
| 33 |
+
docker-min = self.callPackage ./docker.nix { interactive = false; };
|
| 34 |
+
sif = self.callPackage ./sif.nix { };
|
| 35 |
+
})
|
backend/llama.cpp/.devops/nix/sif.nix
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
lib,
|
| 3 |
+
singularity-tools,
|
| 4 |
+
llama-cpp,
|
| 5 |
+
bashInteractive,
|
| 6 |
+
interactive ? false,
|
| 7 |
+
}:
|
| 8 |
+
|
| 9 |
+
let
|
| 10 |
+
optionalInt = cond: x: if cond then x else 0;
|
| 11 |
+
in
|
| 12 |
+
singularity-tools.buildImage rec {
|
| 13 |
+
inherit (llama-cpp) name;
|
| 14 |
+
contents = [ llama-cpp ] ++ lib.optionals interactive [ bashInteractive ];
|
| 15 |
+
|
| 16 |
+
# These are excessive (but safe) for most variants. Building singularity
|
| 17 |
+
# images requires superuser privileges, so we build them inside a VM in a
|
| 18 |
+
# writable image of pre-determined size.
|
| 19 |
+
#
|
| 20 |
+
# ROCm is currently affected by https://github.com/NixOS/nixpkgs/issues/276846
|
| 21 |
+
#
|
| 22 |
+
# Expected image sizes:
|
| 23 |
+
# - cpu/blas: 150M,
|
| 24 |
+
# - cuda, all gencodes: 560M,
|
| 25 |
+
diskSize = 4096 + optionalInt llama-cpp.useRocm 16384;
|
| 26 |
+
memSize = diskSize;
|
| 27 |
+
}
|
backend/llama.cpp/.devops/openvino.Dockerfile
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG OPENVINO_VERSION_MAJOR=2026.2.1
|
| 2 |
+
ARG OPENVINO_VERSION_FULL=2026.2.1.21919.ede283a88e3
|
| 3 |
+
ARG UBUNTU_VERSION=24.04
|
| 4 |
+
|
| 5 |
+
# Intel GPU driver versions. https://github.com/intel/compute-runtime/releases
|
| 6 |
+
ARG IGC_VERSION=v2.36.3
|
| 7 |
+
ARG IGC_VERSION_FULL=2_2.36.3+21719
|
| 8 |
+
ARG COMPUTE_RUNTIME_VERSION=26.22.38646.4
|
| 9 |
+
ARG COMPUTE_RUNTIME_VERSION_FULL=26.22.38646.4-0
|
| 10 |
+
ARG IGDGMM_VERSION=22.10.0
|
| 11 |
+
|
| 12 |
+
# Intel NPU driver versions. https://github.com/intel/linux-npu-driver/releases
|
| 13 |
+
ARG NPU_DRIVER_VERSION=v1.33.0
|
| 14 |
+
ARG NPU_DRIVER_FULL=v1.33.0.20260529-26625960453
|
| 15 |
+
ARG LIBZE1_VERSION=1.27.0-1~24.04~ppa2
|
| 16 |
+
|
| 17 |
+
# Optional proxy build arguments
|
| 18 |
+
ARG http_proxy=
|
| 19 |
+
ARG https_proxy=
|
| 20 |
+
|
| 21 |
+
ARG BUILD_DATE=N/A
|
| 22 |
+
ARG APP_VERSION=N/A
|
| 23 |
+
ARG APP_REVISION=N/A
|
| 24 |
+
|
| 25 |
+
ARG NODE_VERSION=24
|
| 26 |
+
|
| 27 |
+
FROM docker.io/node:$NODE_VERSION AS web
|
| 28 |
+
|
| 29 |
+
ARG APP_VERSION
|
| 30 |
+
|
| 31 |
+
WORKDIR /app/tools/ui
|
| 32 |
+
|
| 33 |
+
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
| 34 |
+
RUN npm ci
|
| 35 |
+
|
| 36 |
+
COPY tools/ui/ ./
|
| 37 |
+
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
| 38 |
+
|
| 39 |
+
## Build Image
|
| 40 |
+
FROM docker.io/ubuntu:${UBUNTU_VERSION} AS build
|
| 41 |
+
|
| 42 |
+
# Pass proxy args to build stage
|
| 43 |
+
ARG http_proxy
|
| 44 |
+
ARG https_proxy
|
| 45 |
+
|
| 46 |
+
RUN apt-get update && \
|
| 47 |
+
apt-get install -y --no-install-recommends \
|
| 48 |
+
ca-certificates \
|
| 49 |
+
gnupg \
|
| 50 |
+
wget \
|
| 51 |
+
git \
|
| 52 |
+
cmake \
|
| 53 |
+
ninja-build \
|
| 54 |
+
build-essential \
|
| 55 |
+
libtbb12 \
|
| 56 |
+
libssl-dev \
|
| 57 |
+
ocl-icd-opencl-dev \
|
| 58 |
+
opencl-headers \
|
| 59 |
+
opencl-clhpp-headers \
|
| 60 |
+
intel-opencl-icd && \
|
| 61 |
+
rm -rf /var/lib/apt/lists/*
|
| 62 |
+
|
| 63 |
+
# OpenVINO toolkit and GPU/NPU drivers are cached via BuildKit cache mounts to avoid re-downloading on rebuilds.
|
| 64 |
+
# Install OpenVINO for Ubuntu 24.04.
|
| 65 |
+
ARG OPENVINO_VERSION_MAJOR
|
| 66 |
+
ARG OPENVINO_VERSION_FULL
|
| 67 |
+
RUN --mount=type=cache,target=/var/cache/openvino,sharing=locked \
|
| 68 |
+
mkdir -p /opt/intel && \
|
| 69 |
+
TGZ=/var/cache/openvino/openvino_toolkit_ubuntu24_${OPENVINO_VERSION_FULL}_x86_64.tgz && \
|
| 70 |
+
if [ ! -f "$TGZ" ]; then \
|
| 71 |
+
wget -O "$TGZ" https://storage.openvinotoolkit.org/repositories/openvino/packages/${OPENVINO_VERSION_MAJOR}/linux/openvino_toolkit_ubuntu24_${OPENVINO_VERSION_FULL}_x86_64.tgz; \
|
| 72 |
+
fi && \
|
| 73 |
+
tar -xf "$TGZ" -C /opt/intel/ && \
|
| 74 |
+
mv /opt/intel/openvino_toolkit_ubuntu24_${OPENVINO_VERSION_FULL}_x86_64 /opt/intel/openvino_${OPENVINO_VERSION_MAJOR} && \
|
| 75 |
+
cd /opt/intel/openvino_${OPENVINO_VERSION_MAJOR} && \
|
| 76 |
+
echo "Y" | ./install_dependencies/install_openvino_dependencies.sh && \
|
| 77 |
+
cd - && \
|
| 78 |
+
ln -s /opt/intel/openvino_${OPENVINO_VERSION_MAJOR} /opt/intel/openvino
|
| 79 |
+
|
| 80 |
+
ENV OpenVINO_DIR=/opt/intel/openvino
|
| 81 |
+
|
| 82 |
+
WORKDIR /app
|
| 83 |
+
|
| 84 |
+
COPY . .
|
| 85 |
+
|
| 86 |
+
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
| 87 |
+
|
| 88 |
+
# Build Stage
|
| 89 |
+
RUN bash -c "source ${OpenVINO_DIR}/setupvars.sh && \
|
| 90 |
+
cmake -B build/ReleaseOV -G Ninja \
|
| 91 |
+
-DCMAKE_BUILD_TYPE=Release \
|
| 92 |
+
-DLLAMA_BUILD_TESTS=OFF \
|
| 93 |
+
-DGGML_OPENVINO=ON && \
|
| 94 |
+
cmake --build build/ReleaseOV --parallel "
|
| 95 |
+
|
| 96 |
+
# Copy all necessary libraries (build outputs + OpenVINO runtime libs)
|
| 97 |
+
RUN mkdir -p /app/lib && \
|
| 98 |
+
find build/ReleaseOV -name '*.so*' -exec cp -P {} /app/lib \; && \
|
| 99 |
+
find "${OpenVINO_DIR}/runtime/lib/intel64" -name '*.so*' -exec cp -P {} /app/lib \;
|
| 100 |
+
|
| 101 |
+
# Create runtime directories and copy binaries
|
| 102 |
+
RUN mkdir -p /app/full \
|
| 103 |
+
&& cp build/ReleaseOV/bin/* /app/full/ \
|
| 104 |
+
&& cp *.py /app/full \
|
| 105 |
+
&& cp -r conversion /app/full \
|
| 106 |
+
&& cp -r gguf-py /app/full \
|
| 107 |
+
&& cp -r requirements /app/full \
|
| 108 |
+
&& cp requirements.txt /app/full \
|
| 109 |
+
&& cp .devops/tools.sh /app/full/tools.sh
|
| 110 |
+
|
| 111 |
+
## Base Runtime Image
|
| 112 |
+
FROM docker.io/ubuntu:${UBUNTU_VERSION} AS base
|
| 113 |
+
|
| 114 |
+
# Pass proxy args to runtime stage
|
| 115 |
+
ARG http_proxy
|
| 116 |
+
ARG https_proxy
|
| 117 |
+
ARG BUILD_DATE=N/A
|
| 118 |
+
ARG APP_VERSION=N/A
|
| 119 |
+
ARG APP_REVISION=N/A
|
| 120 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 121 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 122 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 123 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 124 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 125 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 126 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 127 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 128 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 129 |
+
|
| 130 |
+
RUN apt-get update \
|
| 131 |
+
&& apt-get install -y libgomp1 libtbb12 curl wget ffmpeg ocl-icd-libopencl1 \
|
| 132 |
+
&& apt autoremove -y \
|
| 133 |
+
&& apt clean -y \
|
| 134 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 135 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 136 |
+
&& find /var/cache -type f -delete
|
| 137 |
+
|
| 138 |
+
# Install GPU drivers
|
| 139 |
+
ARG IGC_VERSION
|
| 140 |
+
ARG IGC_VERSION_FULL
|
| 141 |
+
ARG COMPUTE_RUNTIME_VERSION
|
| 142 |
+
ARG COMPUTE_RUNTIME_VERSION_FULL
|
| 143 |
+
ARG IGDGMM_VERSION
|
| 144 |
+
RUN --mount=type=cache,target=/var/cache/intel-gpu,sharing=locked \
|
| 145 |
+
set -eux; \
|
| 146 |
+
cd /var/cache/intel-gpu; \
|
| 147 |
+
for url in \
|
| 148 |
+
https://github.com/intel/intel-graphics-compiler/releases/download/${IGC_VERSION}/intel-igc-core-${IGC_VERSION_FULL}_amd64.deb \
|
| 149 |
+
https://github.com/intel/intel-graphics-compiler/releases/download/${IGC_VERSION}/intel-igc-opencl-${IGC_VERSION_FULL}_amd64.deb \
|
| 150 |
+
https://github.com/intel/compute-runtime/releases/download/${COMPUTE_RUNTIME_VERSION}/intel-ocloc_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.deb \
|
| 151 |
+
https://github.com/intel/compute-runtime/releases/download/${COMPUTE_RUNTIME_VERSION}/intel-opencl-icd_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.deb \
|
| 152 |
+
https://github.com/intel/compute-runtime/releases/download/${COMPUTE_RUNTIME_VERSION}/libigdgmm12_${IGDGMM_VERSION}_amd64.deb \
|
| 153 |
+
https://github.com/intel/compute-runtime/releases/download/${COMPUTE_RUNTIME_VERSION}/libze-intel-gpu1_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.deb ; do \
|
| 154 |
+
f=$(basename "$url"); \
|
| 155 |
+
[ -f "$f" ] || wget -q -O "$f" "$url"; \
|
| 156 |
+
done; \
|
| 157 |
+
apt-get update; \
|
| 158 |
+
apt-get install -y --no-install-recommends ./*.deb; \
|
| 159 |
+
rm -rf /var/lib/apt/lists/*
|
| 160 |
+
|
| 161 |
+
# Install NPU drivers
|
| 162 |
+
ARG NPU_DRIVER_VERSION
|
| 163 |
+
ARG NPU_DRIVER_FULL
|
| 164 |
+
ARG LIBZE1_VERSION
|
| 165 |
+
RUN --mount=type=cache,target=/var/cache/intel-npu,sharing=locked \
|
| 166 |
+
set -eux; \
|
| 167 |
+
TGZ=/var/cache/intel-npu/linux-npu-driver-${NPU_DRIVER_FULL}-ubuntu2404.tar.gz; \
|
| 168 |
+
if [ ! -f "$TGZ" ]; then \
|
| 169 |
+
wget -q -O "$TGZ" https://github.com/intel/linux-npu-driver/releases/download/${NPU_DRIVER_VERSION}/linux-npu-driver-${NPU_DRIVER_FULL}-ubuntu2404.tar.gz; \
|
| 170 |
+
fi; \
|
| 171 |
+
DEB=/var/cache/intel-npu/libze1_${LIBZE1_VERSION}_amd64.deb; \
|
| 172 |
+
if [ ! -f "$DEB" ]; then \
|
| 173 |
+
wget -q -O "$DEB" https://snapshot.ppa.launchpadcontent.net/kobuk-team/intel-graphics/ubuntu/20260324T100000Z/pool/main/l/level-zero-loader/libze1_${LIBZE1_VERSION}_amd64.deb; \
|
| 174 |
+
fi; \
|
| 175 |
+
mkdir /tmp/npu/ && cd /tmp/npu/ && tar -xf "$TGZ" && cp "$DEB" .; \
|
| 176 |
+
apt-get update; \
|
| 177 |
+
apt-get install -y --no-install-recommends ./*.deb; \
|
| 178 |
+
rm -rf /tmp/npu/ /var/lib/apt/lists/*
|
| 179 |
+
|
| 180 |
+
COPY --from=build /app/lib/ /app/
|
| 181 |
+
|
| 182 |
+
### Full (all binaries)
|
| 183 |
+
FROM base AS full
|
| 184 |
+
|
| 185 |
+
ARG http_proxy
|
| 186 |
+
ARG https_proxy
|
| 187 |
+
|
| 188 |
+
COPY --from=build /app/full /app/
|
| 189 |
+
|
| 190 |
+
WORKDIR /app
|
| 191 |
+
|
| 192 |
+
RUN apt-get update && \
|
| 193 |
+
apt-get install -y --no-install-recommends \
|
| 194 |
+
git \
|
| 195 |
+
python3 \
|
| 196 |
+
python3-venv \
|
| 197 |
+
python3-pip && \
|
| 198 |
+
python3 -m venv /openvino-venv && \
|
| 199 |
+
/openvino-venv/bin/pip install --no-cache-dir --upgrade pip setuptools wheel && \
|
| 200 |
+
/openvino-venv/bin/pip install --no-cache-dir -r requirements.txt && \
|
| 201 |
+
apt-get autoremove -y && \
|
| 202 |
+
apt-get clean && \
|
| 203 |
+
rm -rf /tmp/* /var/tmp/* && \
|
| 204 |
+
find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete && \
|
| 205 |
+
find /var/cache -type f -delete
|
| 206 |
+
|
| 207 |
+
# Activate the venv
|
| 208 |
+
ENV VIRTUAL_ENV=/openvino-venv \
|
| 209 |
+
PATH=/openvino-venv/bin:$PATH
|
| 210 |
+
|
| 211 |
+
ENTRYPOINT ["/app/tools.sh"]
|
| 212 |
+
|
| 213 |
+
|
| 214 |
+
### Light, CLI only
|
| 215 |
+
FROM base AS light
|
| 216 |
+
|
| 217 |
+
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app/
|
| 218 |
+
|
| 219 |
+
WORKDIR /app
|
| 220 |
+
|
| 221 |
+
ENTRYPOINT [ "/app/llama-cli" ]
|
| 222 |
+
|
| 223 |
+
### Server, Server only
|
| 224 |
+
FROM base AS server
|
| 225 |
+
|
| 226 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 227 |
+
|
| 228 |
+
COPY --from=build /app/full/llama /app/full/llama-server /app/
|
| 229 |
+
|
| 230 |
+
WORKDIR /app
|
| 231 |
+
|
| 232 |
+
HEALTHCHECK CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
| 233 |
+
|
| 234 |
+
ENTRYPOINT [ "/app/llama-server" ]
|
backend/llama.cpp/.devops/rocm.Dockerfile
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG UBUNTU_VERSION=24.04
|
| 2 |
+
|
| 3 |
+
# This needs to generally match the container host's environment.
|
| 4 |
+
ARG ROCM_VERSION=7.2.1
|
| 5 |
+
ARG AMDGPU_VERSION=7.2.1
|
| 6 |
+
|
| 7 |
+
# Target the ROCm build image
|
| 8 |
+
ARG BASE_ROCM_DEV_CONTAINER=docker.io/rocm/dev-ubuntu-${UBUNTU_VERSION}:${ROCM_VERSION}-complete
|
| 9 |
+
|
| 10 |
+
ARG BUILD_DATE=N/A
|
| 11 |
+
ARG APP_VERSION=N/A
|
| 12 |
+
ARG APP_REVISION=N/A
|
| 13 |
+
|
| 14 |
+
ARG NODE_VERSION=24
|
| 15 |
+
|
| 16 |
+
FROM docker.io/node:$NODE_VERSION AS web
|
| 17 |
+
|
| 18 |
+
ARG APP_VERSION
|
| 19 |
+
|
| 20 |
+
WORKDIR /app/tools/ui
|
| 21 |
+
|
| 22 |
+
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
| 23 |
+
RUN npm ci
|
| 24 |
+
|
| 25 |
+
COPY tools/ui/ ./
|
| 26 |
+
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
| 27 |
+
|
| 28 |
+
### Build image
|
| 29 |
+
FROM ${BASE_ROCM_DEV_CONTAINER} AS build
|
| 30 |
+
|
| 31 |
+
# Unless otherwise specified, we make a fat build.
|
| 32 |
+
# This is mostly tied to rocBLAS supported archs.
|
| 33 |
+
# check https://rocm.docs.amd.com/projects/install-on-linux/en/docs-7.2.1/reference/system-requirements.html
|
| 34 |
+
# check https://rocm.docs.amd.com/projects/radeon-ryzen/en/latest/docs/compatibility/compatibilityrad/native_linux/native_linux_compatibility.html
|
| 35 |
+
# check https://rocm.docs.amd.com/projects/radeon-ryzen/en/latest/docs/compatibility/compatibilityryz/native_linux/native_linux_compatibility.html
|
| 36 |
+
|
| 37 |
+
ARG ROCM_DOCKER_ARCH='gfx908;gfx90a;gfx942;gfx1030;gfx1100;gfx1101;gfx1102;gfx1151;gfx1150;gfx1200;gfx1201'
|
| 38 |
+
|
| 39 |
+
# Set ROCm architectures
|
| 40 |
+
ENV AMDGPU_TARGETS=${ROCM_DOCKER_ARCH}
|
| 41 |
+
|
| 42 |
+
RUN apt-get update \
|
| 43 |
+
&& apt-get install -y \
|
| 44 |
+
build-essential \
|
| 45 |
+
cmake \
|
| 46 |
+
git \
|
| 47 |
+
libssl-dev \
|
| 48 |
+
curl \
|
| 49 |
+
libgomp1
|
| 50 |
+
|
| 51 |
+
WORKDIR /app
|
| 52 |
+
|
| 53 |
+
COPY . .
|
| 54 |
+
|
| 55 |
+
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
| 56 |
+
|
| 57 |
+
RUN HIPCXX="$(hipconfig -l)/clang" HIP_PATH="$(hipconfig -R)" \
|
| 58 |
+
cmake -S . -B build \
|
| 59 |
+
-DGGML_HIP=ON \
|
| 60 |
+
-DGGML_HIP_ROCWMMA_FATTN=ON \
|
| 61 |
+
-DAMDGPU_TARGETS="$ROCM_DOCKER_ARCH" \
|
| 62 |
+
-DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON \
|
| 63 |
+
-DCMAKE_BUILD_TYPE=Release -DLLAMA_BUILD_TESTS=OFF \
|
| 64 |
+
&& cmake --build build --config Release -j$(nproc)
|
| 65 |
+
|
| 66 |
+
RUN mkdir -p /app/lib \
|
| 67 |
+
&& find build -name "*.so*" -exec cp -P {} /app/lib \;
|
| 68 |
+
|
| 69 |
+
RUN mkdir -p /app/full \
|
| 70 |
+
&& cp build/bin/* /app/full \
|
| 71 |
+
&& cp *.py /app/full \
|
| 72 |
+
&& cp -r conversion /app/full \
|
| 73 |
+
&& cp -r gguf-py /app/full \
|
| 74 |
+
&& cp -r requirements /app/full \
|
| 75 |
+
&& cp requirements.txt /app/full \
|
| 76 |
+
&& cp .devops/tools.sh /app/full/tools.sh
|
| 77 |
+
|
| 78 |
+
## Base image
|
| 79 |
+
FROM ${BASE_ROCM_DEV_CONTAINER} AS base
|
| 80 |
+
|
| 81 |
+
ARG BUILD_DATE=N/A
|
| 82 |
+
ARG APP_VERSION=N/A
|
| 83 |
+
ARG APP_REVISION=N/A
|
| 84 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 85 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 86 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 87 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 88 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 89 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 90 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 91 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 92 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 93 |
+
|
| 94 |
+
RUN apt-get update \
|
| 95 |
+
&& apt-get install -y libgomp1 curl ffmpeg \
|
| 96 |
+
&& apt autoremove -y \
|
| 97 |
+
&& apt clean -y \
|
| 98 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 99 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 100 |
+
&& find /var/cache -type f -delete
|
| 101 |
+
|
| 102 |
+
COPY --from=build /app/lib/ /app
|
| 103 |
+
|
| 104 |
+
### Full
|
| 105 |
+
FROM base AS full
|
| 106 |
+
|
| 107 |
+
COPY --from=build /app/full /app
|
| 108 |
+
|
| 109 |
+
WORKDIR /app
|
| 110 |
+
|
| 111 |
+
RUN apt-get update \
|
| 112 |
+
&& apt-get install -y \
|
| 113 |
+
git \
|
| 114 |
+
python3-pip \
|
| 115 |
+
python3 \
|
| 116 |
+
python3-wheel \
|
| 117 |
+
&& pip install --break-system-packages --upgrade setuptools \
|
| 118 |
+
&& pip install --break-system-packages -r requirements.txt \
|
| 119 |
+
&& apt autoremove -y \
|
| 120 |
+
&& apt clean -y \
|
| 121 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 122 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 123 |
+
&& find /var/cache -type f -delete
|
| 124 |
+
|
| 125 |
+
ENTRYPOINT ["/app/tools.sh"]
|
| 126 |
+
|
| 127 |
+
### Light, CLI only
|
| 128 |
+
FROM base AS light
|
| 129 |
+
|
| 130 |
+
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app
|
| 131 |
+
|
| 132 |
+
WORKDIR /app
|
| 133 |
+
|
| 134 |
+
ENTRYPOINT [ "/app/llama-cli" ]
|
| 135 |
+
|
| 136 |
+
### Server, Server only
|
| 137 |
+
FROM base AS server
|
| 138 |
+
|
| 139 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 140 |
+
|
| 141 |
+
COPY --from=build /app/full/llama /app/full/llama-server /app
|
| 142 |
+
|
| 143 |
+
WORKDIR /app
|
| 144 |
+
|
| 145 |
+
HEALTHCHECK CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
| 146 |
+
|
| 147 |
+
ENTRYPOINT [ "/app/llama-server" ]
|
backend/llama.cpp/.devops/s390x.Dockerfile
ADDED
|
@@ -0,0 +1,145 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG GCC_VERSION=15.2.0
|
| 2 |
+
ARG UBUNTU_VERSION=24.04
|
| 3 |
+
ARG BUILD_DATE=N/A
|
| 4 |
+
ARG APP_VERSION=N/A
|
| 5 |
+
ARG APP_REVISION=N/A
|
| 6 |
+
|
| 7 |
+
### Build Llama.cpp stage
|
| 8 |
+
FROM docker.io/gcc:${GCC_VERSION} AS build
|
| 9 |
+
|
| 10 |
+
RUN --mount=type=cache,target=/var/cache/apt,sharing=locked \
|
| 11 |
+
--mount=type=cache,target=/var/lib/apt/lists,sharing=locked \
|
| 12 |
+
apt update -y && \
|
| 13 |
+
apt upgrade -y && \
|
| 14 |
+
apt install -y --no-install-recommends \
|
| 15 |
+
git cmake ccache ninja-build \
|
| 16 |
+
# WARNING: Do not use libopenblas-openmp-dev. libopenblas-dev is faster.
|
| 17 |
+
libopenblas-dev libssl-dev && \
|
| 18 |
+
rm -rf /var/lib/apt/lists/*
|
| 19 |
+
|
| 20 |
+
WORKDIR /app
|
| 21 |
+
COPY . .
|
| 22 |
+
|
| 23 |
+
RUN --mount=type=cache,target=/root/.ccache \
|
| 24 |
+
--mount=type=cache,target=/app/build \
|
| 25 |
+
cmake -S . -B build -G Ninja \
|
| 26 |
+
-DCMAKE_BUILD_TYPE=Release \
|
| 27 |
+
-DCMAKE_C_COMPILER_LAUNCHER=ccache \
|
| 28 |
+
-DCMAKE_CXX_COMPILER_LAUNCHER=ccache \
|
| 29 |
+
-DLLAMA_BUILD_TESTS=OFF \
|
| 30 |
+
-DGGML_NATIVE=OFF \
|
| 31 |
+
-DGGML_BACKEND_DL=ON \
|
| 32 |
+
-DGGML_CPU_ALL_VARIANTS=ON \
|
| 33 |
+
-DGGML_BLAS=ON \
|
| 34 |
+
-DGGML_BLAS_VENDOR=OpenBLAS && \
|
| 35 |
+
cmake --build build --config Release -j $(nproc) && \
|
| 36 |
+
cmake --install build --prefix /opt/llama.cpp
|
| 37 |
+
|
| 38 |
+
COPY *.py /opt/llama.cpp/bin
|
| 39 |
+
COPY .devops/tools.sh /opt/llama.cpp/bin
|
| 40 |
+
COPY conversion /opt/llama.cpp/conversion
|
| 41 |
+
|
| 42 |
+
COPY gguf-py /opt/llama.cpp/gguf-py
|
| 43 |
+
COPY requirements.txt /opt/llama.cpp/gguf-py
|
| 44 |
+
COPY requirements /opt/llama.cpp/gguf-py/requirements
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
### Collect all llama.cpp binaries, libraries and distro libraries
|
| 48 |
+
FROM scratch AS collector
|
| 49 |
+
|
| 50 |
+
# Copy llama.cpp binaries and libraries
|
| 51 |
+
COPY --from=build /opt/llama.cpp/bin /llama.cpp/bin
|
| 52 |
+
COPY --from=build /opt/llama.cpp/lib /llama.cpp/lib
|
| 53 |
+
COPY --from=build /opt/llama.cpp/gguf-py /llama.cpp/gguf-py
|
| 54 |
+
COPY --from=build /opt/llama.cpp/conversion /llama.cpp/conversion
|
| 55 |
+
|
| 56 |
+
|
| 57 |
+
### Base image
|
| 58 |
+
FROM docker.io/ubuntu:${UBUNTU_VERSION} AS base
|
| 59 |
+
|
| 60 |
+
ARG BUILD_DATE=N/A
|
| 61 |
+
ARG APP_VERSION=N/A
|
| 62 |
+
ARG APP_REVISION=N/A
|
| 63 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 64 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 65 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 66 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 67 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 68 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 69 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 70 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 71 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 72 |
+
|
| 73 |
+
RUN --mount=type=cache,target=/var/cache/apt,sharing=locked \
|
| 74 |
+
--mount=type=cache,target=/var/lib/apt/lists,sharing=locked \
|
| 75 |
+
apt update -y && \
|
| 76 |
+
apt install -y --no-install-recommends \
|
| 77 |
+
# WARNING: Do not use libopenblas-openmp-dev. libopenblas-dev is faster.
|
| 78 |
+
# See: https://github.com/ggml-org/llama.cpp/pull/15915#issuecomment-3317166506
|
| 79 |
+
curl libgomp1 libopenblas-dev && \
|
| 80 |
+
apt autoremove -y && \
|
| 81 |
+
apt clean -y && \
|
| 82 |
+
rm -rf /tmp/* /var/tmp/* && \
|
| 83 |
+
find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete && \
|
| 84 |
+
find /var/cache -type f -delete
|
| 85 |
+
|
| 86 |
+
# Copy llama.cpp libraries
|
| 87 |
+
COPY --from=collector /llama.cpp/lib /usr/lib/s390x-linux-gnu
|
| 88 |
+
|
| 89 |
+
|
| 90 |
+
### Full
|
| 91 |
+
FROM base AS full
|
| 92 |
+
|
| 93 |
+
ENV PATH="/root/.cargo/bin:${PATH}"
|
| 94 |
+
WORKDIR /app
|
| 95 |
+
|
| 96 |
+
RUN --mount=type=cache,target=/var/cache/apt,sharing=locked \
|
| 97 |
+
--mount=type=cache,target=/var/lib/apt/lists,sharing=locked \
|
| 98 |
+
apt update -y && \
|
| 99 |
+
apt install -y \
|
| 100 |
+
git cmake libjpeg-dev \
|
| 101 |
+
python3 python3-pip python3-dev && \
|
| 102 |
+
apt autoremove -y && \
|
| 103 |
+
apt clean -y && \
|
| 104 |
+
rm -rf /tmp/* /var/tmp/* && \
|
| 105 |
+
find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete && \
|
| 106 |
+
find /var/cache -type f -delete
|
| 107 |
+
|
| 108 |
+
RUN curl https://sh.rustup.rs -sSf | bash -s -- -y
|
| 109 |
+
|
| 110 |
+
COPY --from=collector /llama.cpp/bin /app
|
| 111 |
+
COPY --from=collector /llama.cpp/gguf-py /app/gguf-py
|
| 112 |
+
COPY --from=collector /llama.cpp/conversion /app/conversion
|
| 113 |
+
|
| 114 |
+
RUN pip install --no-cache-dir --break-system-packages \
|
| 115 |
+
-r /app/gguf-py/requirements.txt
|
| 116 |
+
|
| 117 |
+
ENTRYPOINT [ "/app/tools.sh" ]
|
| 118 |
+
|
| 119 |
+
|
| 120 |
+
### CLI Only
|
| 121 |
+
FROM base AS light
|
| 122 |
+
|
| 123 |
+
WORKDIR /llama.cpp/bin
|
| 124 |
+
|
| 125 |
+
# Copy llama.cpp binaries and libraries
|
| 126 |
+
COPY --from=collector /llama.cpp/bin/*.so /llama.cpp/bin
|
| 127 |
+
COPY --from=collector /llama.cpp/bin/llama /llama.cpp/bin/llama-cli /llama.cpp/bin/llama-completion /llama.cpp/bin
|
| 128 |
+
|
| 129 |
+
ENTRYPOINT [ "/llama.cpp/bin/llama-cli" ]
|
| 130 |
+
|
| 131 |
+
|
| 132 |
+
### Server
|
| 133 |
+
FROM base AS server
|
| 134 |
+
|
| 135 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 136 |
+
|
| 137 |
+
WORKDIR /llama.cpp/bin
|
| 138 |
+
|
| 139 |
+
# Copy llama.cpp binaries and libraries
|
| 140 |
+
COPY --from=collector /llama.cpp/bin/*.so /llama.cpp/bin
|
| 141 |
+
COPY --from=collector /llama.cpp/bin/llama /llama.cpp/bin/llama-server /llama.cpp/bin
|
| 142 |
+
|
| 143 |
+
EXPOSE 8080
|
| 144 |
+
|
| 145 |
+
ENTRYPOINT [ "/llama.cpp/bin/llama-server" ]
|
backend/llama.cpp/.devops/tools.sh
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env bash
|
| 2 |
+
set -e
|
| 3 |
+
|
| 4 |
+
# Read the first argument into a variable
|
| 5 |
+
arg1="$1"
|
| 6 |
+
|
| 7 |
+
# Shift the arguments to remove the first one
|
| 8 |
+
shift
|
| 9 |
+
|
| 10 |
+
if [[ "$arg1" == '--convert' || "$arg1" == '-c' ]]; then
|
| 11 |
+
exec python3 ./convert_hf_to_gguf.py "$@"
|
| 12 |
+
elif [[ "$arg1" == '--quantize' || "$arg1" == '-q' ]]; then
|
| 13 |
+
exec ./llama-quantize "$@"
|
| 14 |
+
elif [[ "$arg1" == '--run' || "$arg1" == '-r' ]]; then
|
| 15 |
+
exec ./llama-cli "$@"
|
| 16 |
+
elif [[ "$arg1" == '--run-legacy' || "$arg1" == '-l' ]]; then
|
| 17 |
+
exec ./llama-completion "$@"
|
| 18 |
+
elif [[ "$arg1" == '--bench' || "$arg1" == '-b' ]]; then
|
| 19 |
+
exec ./llama-bench "$@"
|
| 20 |
+
elif [[ "$arg1" == '--perplexity' || "$arg1" == '-p' ]]; then
|
| 21 |
+
exec ./llama-perplexity "$@"
|
| 22 |
+
elif [[ "$arg1" == '--all-in-one' || "$arg1" == '-a' ]]; then
|
| 23 |
+
echo "Converting PTH to GGML..."
|
| 24 |
+
for i in $(ls $1/$2/ggml-model-f16.bin*); do
|
| 25 |
+
if [ -f "${i/f16/q4_0}" ]; then
|
| 26 |
+
echo "Skip model quantization, it already exists: ${i/f16/q4_0}"
|
| 27 |
+
else
|
| 28 |
+
echo "Converting PTH to GGML: $i into ${i/f16/q4_0}..."
|
| 29 |
+
exec ./llama-quantize "$i" "${i/f16/q4_0}" q4_0
|
| 30 |
+
fi
|
| 31 |
+
done
|
| 32 |
+
elif [[ "$arg1" == '--server' || "$arg1" == '-s' ]]; then
|
| 33 |
+
exec ./llama-server "$@"
|
| 34 |
+
else
|
| 35 |
+
echo "Unknown command: $arg1"
|
| 36 |
+
echo "Available commands: "
|
| 37 |
+
echo " --run (-r): Run a model (chat) previously converted into ggml"
|
| 38 |
+
echo " ex: -m /models/7B/ggml-model-q4_0.bin"
|
| 39 |
+
echo " --run-legacy (-l): Run a model (legacy completion) previously converted into ggml"
|
| 40 |
+
echo " ex: -m /models/7B/ggml-model-q4_0.bin -no-cnv -p \"Building a website can be done in 10 simple steps:\" -n 512"
|
| 41 |
+
echo " --bench (-b): Benchmark the performance of the inference for various parameters."
|
| 42 |
+
echo " ex: -m model.gguf"
|
| 43 |
+
echo " --perplexity (-p): Measure the perplexity of a model over a given text."
|
| 44 |
+
echo " ex: -m model.gguf -f file.txt"
|
| 45 |
+
echo " --convert (-c): Convert a llama model into ggml"
|
| 46 |
+
echo " ex: --outtype f16 \"/models/7B/\" "
|
| 47 |
+
echo " --quantize (-q): Optimize with quantization process ggml"
|
| 48 |
+
echo " ex: \"/models/7B/ggml-model-f16.bin\" \"/models/7B/ggml-model-q4_0.bin\" 2"
|
| 49 |
+
echo " --all-in-one (-a): Execute --convert & --quantize"
|
| 50 |
+
echo " ex: \"/models/\" 7B"
|
| 51 |
+
echo " --server (-s): Run a model on the server"
|
| 52 |
+
echo " ex: -m /models/7B/ggml-model-q4_0.bin -c 2048 -ngl 43 -mg 1 --port 8080"
|
| 53 |
+
fi
|
backend/llama.cpp/.devops/vulkan.Dockerfile
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG UBUNTU_VERSION=26.04
|
| 2 |
+
ARG BUILD_DATE=N/A
|
| 3 |
+
ARG APP_VERSION=N/A
|
| 4 |
+
ARG APP_REVISION=N/A
|
| 5 |
+
|
| 6 |
+
ARG NODE_VERSION=24
|
| 7 |
+
|
| 8 |
+
FROM docker.io/node:$NODE_VERSION AS web
|
| 9 |
+
|
| 10 |
+
ARG APP_VERSION
|
| 11 |
+
|
| 12 |
+
WORKDIR /app/tools/ui
|
| 13 |
+
|
| 14 |
+
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
| 15 |
+
RUN npm ci
|
| 16 |
+
|
| 17 |
+
COPY tools/ui/ ./
|
| 18 |
+
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
| 19 |
+
|
| 20 |
+
FROM docker.io/ubuntu:$UBUNTU_VERSION AS build
|
| 21 |
+
|
| 22 |
+
# Install build tools
|
| 23 |
+
RUN apt update && apt install -y git build-essential cmake wget xz-utils
|
| 24 |
+
|
| 25 |
+
# Install SSL and Vulkan SDK dependencies
|
| 26 |
+
RUN apt install -y libssl-dev curl \
|
| 27 |
+
libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers
|
| 28 |
+
|
| 29 |
+
# Build it
|
| 30 |
+
WORKDIR /app
|
| 31 |
+
|
| 32 |
+
COPY . .
|
| 33 |
+
|
| 34 |
+
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
| 35 |
+
|
| 36 |
+
RUN cmake -B build -DGGML_NATIVE=OFF -DGGML_VULKAN=ON -DLLAMA_BUILD_TESTS=OFF -DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON && \
|
| 37 |
+
cmake --build build --config Release -j$(nproc)
|
| 38 |
+
|
| 39 |
+
RUN mkdir -p /app/lib && \
|
| 40 |
+
find build -name "*.so*" -exec cp -P {} /app/lib \;
|
| 41 |
+
|
| 42 |
+
RUN mkdir -p /app/full \
|
| 43 |
+
&& cp build/bin/* /app/full \
|
| 44 |
+
&& cp *.py /app/full \
|
| 45 |
+
&& cp -r conversion /app/full \
|
| 46 |
+
&& cp -r gguf-py /app/full \
|
| 47 |
+
&& cp -r requirements /app/full \
|
| 48 |
+
&& cp requirements.txt /app/full \
|
| 49 |
+
&& cp .devops/tools.sh /app/full/tools.sh
|
| 50 |
+
|
| 51 |
+
## Base image
|
| 52 |
+
FROM docker.io/ubuntu:$UBUNTU_VERSION AS base
|
| 53 |
+
|
| 54 |
+
ARG BUILD_DATE=N/A
|
| 55 |
+
ARG APP_VERSION=N/A
|
| 56 |
+
ARG APP_REVISION=N/A
|
| 57 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 58 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 59 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 60 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 61 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 62 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 63 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 64 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 65 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 66 |
+
|
| 67 |
+
RUN apt-get update \
|
| 68 |
+
&& apt-get install -y libgomp1 curl ffmpeg libvulkan1 mesa-vulkan-drivers \
|
| 69 |
+
libglvnd0 libgl1 libglx0 libegl1 libgles2 \
|
| 70 |
+
&& apt autoremove -y \
|
| 71 |
+
&& apt clean -y \
|
| 72 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 73 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 74 |
+
&& find /var/cache -type f -delete
|
| 75 |
+
|
| 76 |
+
COPY --from=build /app/lib/ /app
|
| 77 |
+
|
| 78 |
+
### Full
|
| 79 |
+
FROM base AS full
|
| 80 |
+
|
| 81 |
+
COPY --from=build /app/full /app
|
| 82 |
+
|
| 83 |
+
WORKDIR /app
|
| 84 |
+
|
| 85 |
+
ENV PATH="/root/.venv/bin:/root/.local/bin:${PATH}"
|
| 86 |
+
|
| 87 |
+
# Flag for compatibility with pip
|
| 88 |
+
ARG UV_INDEX_STRATEGY="unsafe-best-match"
|
| 89 |
+
RUN apt-get update \
|
| 90 |
+
&& apt-get install -y \
|
| 91 |
+
build-essential \
|
| 92 |
+
curl \
|
| 93 |
+
git \
|
| 94 |
+
ca-certificates \
|
| 95 |
+
&& curl -LsSf https://astral.sh/uv/install.sh | sh \
|
| 96 |
+
&& uv python install 3.13 \
|
| 97 |
+
&& uv venv --python 3.13 /root/.venv \
|
| 98 |
+
&& uv pip install --python /root/.venv/bin/python -r requirements.txt \
|
| 99 |
+
&& apt autoremove -y \
|
| 100 |
+
&& apt clean -y \
|
| 101 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 102 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 103 |
+
&& find /var/cache -type f -delete
|
| 104 |
+
|
| 105 |
+
ENTRYPOINT ["/app/tools.sh"]
|
| 106 |
+
|
| 107 |
+
### Light, CLI only
|
| 108 |
+
FROM base AS light
|
| 109 |
+
|
| 110 |
+
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app
|
| 111 |
+
|
| 112 |
+
WORKDIR /app
|
| 113 |
+
|
| 114 |
+
ENTRYPOINT [ "/app/llama-cli" ]
|
| 115 |
+
|
| 116 |
+
### Server, Server only
|
| 117 |
+
FROM base AS server
|
| 118 |
+
|
| 119 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 120 |
+
|
| 121 |
+
COPY --from=build /app/full/llama /app/full/llama-server /app
|
| 122 |
+
|
| 123 |
+
WORKDIR /app
|
| 124 |
+
|
| 125 |
+
HEALTHCHECK CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
| 126 |
+
|
| 127 |
+
ENTRYPOINT [ "/app/llama-server" ]
|
backend/llama.cpp/.devops/zendnn.Dockerfile
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
ARG UBUNTU_VERSION=24.04
|
| 2 |
+
ARG BUILD_DATE=N/A
|
| 3 |
+
ARG APP_VERSION=N/A
|
| 4 |
+
ARG APP_REVISION=N/A
|
| 5 |
+
|
| 6 |
+
ARG NODE_VERSION=24
|
| 7 |
+
|
| 8 |
+
FROM docker.io/node:$NODE_VERSION AS web
|
| 9 |
+
|
| 10 |
+
ARG APP_VERSION
|
| 11 |
+
|
| 12 |
+
WORKDIR /app/tools/ui
|
| 13 |
+
|
| 14 |
+
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
| 15 |
+
RUN npm ci
|
| 16 |
+
|
| 17 |
+
COPY tools/ui/ ./
|
| 18 |
+
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
| 19 |
+
|
| 20 |
+
FROM docker.io/ubuntu:$UBUNTU_VERSION AS build
|
| 21 |
+
|
| 22 |
+
RUN apt-get update && \
|
| 23 |
+
apt-get install -y gcc-13 g++-13 build-essential git cmake libssl-dev libomp-dev libnuma-dev python3 ca-certificates
|
| 24 |
+
|
| 25 |
+
ENV CC=gcc-13 CXX=g++-13
|
| 26 |
+
|
| 27 |
+
WORKDIR /app
|
| 28 |
+
|
| 29 |
+
COPY . .
|
| 30 |
+
|
| 31 |
+
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
| 32 |
+
|
| 33 |
+
RUN cmake -S . -B build -DCMAKE_BUILD_TYPE=Release -DGGML_NATIVE=OFF -DLLAMA_BUILD_TESTS=OFF -DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON -DGGML_ZENDNN=ON && \
|
| 34 |
+
cmake --build build -j $(nproc)
|
| 35 |
+
|
| 36 |
+
RUN mkdir -p /app/lib && \
|
| 37 |
+
find build -name "*.so*" -exec cp -P {} /app/lib \;
|
| 38 |
+
|
| 39 |
+
RUN mkdir -p /app/full \
|
| 40 |
+
&& cp build/bin/* /app/full \
|
| 41 |
+
&& cp *.py /app/full \
|
| 42 |
+
&& cp -r conversion /app/full \
|
| 43 |
+
&& cp -r gguf-py /app/full \
|
| 44 |
+
&& cp -r requirements /app/full \
|
| 45 |
+
&& cp requirements.txt /app/full \
|
| 46 |
+
&& cp .devops/tools.sh /app/full/tools.sh
|
| 47 |
+
|
| 48 |
+
## Base image
|
| 49 |
+
FROM docker.io/ubuntu:$UBUNTU_VERSION AS base
|
| 50 |
+
|
| 51 |
+
ARG BUILD_DATE=N/A
|
| 52 |
+
ARG APP_VERSION=N/A
|
| 53 |
+
ARG APP_REVISION=N/A
|
| 54 |
+
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
| 55 |
+
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
| 56 |
+
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
| 57 |
+
org.opencontainers.image.version=$APP_VERSION \
|
| 58 |
+
org.opencontainers.image.revision=$APP_REVISION \
|
| 59 |
+
org.opencontainers.image.title="llama.cpp" \
|
| 60 |
+
org.opencontainers.image.description="LLM inference in C/C++" \
|
| 61 |
+
org.opencontainers.image.url=$IMAGE_URL \
|
| 62 |
+
org.opencontainers.image.source=$IMAGE_SOURCE
|
| 63 |
+
|
| 64 |
+
RUN apt-get update \
|
| 65 |
+
&& apt-get install -y libgomp1 libnuma1 curl ffmpeg \
|
| 66 |
+
&& apt autoremove -y \
|
| 67 |
+
&& apt clean -y \
|
| 68 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 69 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 70 |
+
&& find /var/cache -type f -delete
|
| 71 |
+
|
| 72 |
+
COPY --from=build /app/lib/ /app
|
| 73 |
+
|
| 74 |
+
### Full
|
| 75 |
+
FROM base AS full
|
| 76 |
+
|
| 77 |
+
COPY --from=build /app/full /app
|
| 78 |
+
|
| 79 |
+
WORKDIR /app
|
| 80 |
+
|
| 81 |
+
RUN apt-get update \
|
| 82 |
+
&& apt-get install -y \
|
| 83 |
+
git \
|
| 84 |
+
python3 \
|
| 85 |
+
python3-pip \
|
| 86 |
+
python3-wheel \
|
| 87 |
+
&& pip install --break-system-packages --upgrade setuptools \
|
| 88 |
+
&& pip install --break-system-packages -r requirements.txt \
|
| 89 |
+
&& apt autoremove -y \
|
| 90 |
+
&& apt clean -y \
|
| 91 |
+
&& rm -rf /tmp/* /var/tmp/* \
|
| 92 |
+
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
| 93 |
+
&& find /var/cache -type f -delete
|
| 94 |
+
|
| 95 |
+
ENTRYPOINT ["/app/tools.sh"]
|
| 96 |
+
|
| 97 |
+
### Light, CLI only
|
| 98 |
+
FROM base AS light
|
| 99 |
+
|
| 100 |
+
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app
|
| 101 |
+
|
| 102 |
+
WORKDIR /app
|
| 103 |
+
|
| 104 |
+
ENTRYPOINT [ "/app/llama-cli" ]
|
| 105 |
+
|
| 106 |
+
### Server, Server only
|
| 107 |
+
FROM base AS server
|
| 108 |
+
|
| 109 |
+
ENV LLAMA_ARG_HOST=0.0.0.0
|
| 110 |
+
|
| 111 |
+
COPY --from=build /app/full/llama /app/full/llama-server /app
|
| 112 |
+
|
| 113 |
+
WORKDIR /app
|
| 114 |
+
|
| 115 |
+
HEALTHCHECK CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
| 116 |
+
|
| 117 |
+
ENTRYPOINT [ "/app/llama-server" ]
|
backend/llama.cpp/.dockerignore
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
*.o
|
| 2 |
+
*.a
|
| 3 |
+
.cache/
|
| 4 |
+
# Do not ignore .git directory, otherwise the reported build number will always be 0
|
| 5 |
+
.github/
|
| 6 |
+
.gitignore
|
| 7 |
+
.vs/
|
| 8 |
+
.vscode/
|
| 9 |
+
.DS_Store
|
| 10 |
+
|
| 11 |
+
build*/
|
| 12 |
+
|
| 13 |
+
tools/ui/node_modules/
|
| 14 |
+
|
| 15 |
+
models/*
|
| 16 |
+
|
| 17 |
+
/llama-cli
|
| 18 |
+
/llama-quantize
|
| 19 |
+
|
| 20 |
+
arm_neon.h
|
| 21 |
+
compile_commands.json
|
| 22 |
+
Dockerfile
|
backend/llama.cpp/.ecrc
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"Exclude": ["^\\.gitmodules$", "stb_image\\.h"],
|
| 3 |
+
"Disable": {
|
| 4 |
+
"IndentSize": true
|
| 5 |
+
}
|
| 6 |
+
}
|
backend/llama.cpp/.editorconfig
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# https://EditorConfig.org
|
| 2 |
+
|
| 3 |
+
# Top-most EditorConfig file
|
| 4 |
+
root = true
|
| 5 |
+
|
| 6 |
+
# Unix-style newlines with a newline ending every file, utf-8 charset
|
| 7 |
+
[*]
|
| 8 |
+
end_of_line = lf
|
| 9 |
+
insert_final_newline = true
|
| 10 |
+
trim_trailing_whitespace = true
|
| 11 |
+
charset = utf-8
|
| 12 |
+
indent_style = space
|
| 13 |
+
indent_size = 4
|
| 14 |
+
|
| 15 |
+
[Makefile]
|
| 16 |
+
indent_style = tab
|
| 17 |
+
|
| 18 |
+
[scripts/*.mk]
|
| 19 |
+
indent_style = tab
|
| 20 |
+
|
| 21 |
+
[prompts/*.txt]
|
| 22 |
+
insert_final_newline = unset
|
| 23 |
+
|
| 24 |
+
[tools/server/deps_*]
|
| 25 |
+
trim_trailing_whitespace = unset
|
| 26 |
+
indent_style = unset
|
| 27 |
+
indent_size = unset
|
| 28 |
+
|
| 29 |
+
[examples/llama.swiftui/llama.swiftui.xcodeproj/*]
|
| 30 |
+
indent_style = tab
|
| 31 |
+
|
| 32 |
+
[tools/cvector-generator/*.txt]
|
| 33 |
+
trim_trailing_whitespace = unset
|
| 34 |
+
insert_final_newline = unset
|
| 35 |
+
|
| 36 |
+
[models/templates/*.jinja]
|
| 37 |
+
indent_style = unset
|
| 38 |
+
indent_size = unset
|
| 39 |
+
end_of_line = unset
|
| 40 |
+
charset = unset
|
| 41 |
+
trim_trailing_whitespace = unset
|
| 42 |
+
insert_final_newline = unset
|
| 43 |
+
|
| 44 |
+
[vendor/miniaudio/miniaudio.h]
|
| 45 |
+
trim_trailing_whitespace = unset
|
| 46 |
+
insert_final_newline = unset
|
| 47 |
+
|
| 48 |
+
[tools/ui/**]
|
| 49 |
+
indent_style = unset
|
| 50 |
+
indent_size = unset
|
| 51 |
+
end_of_line = unset
|
| 52 |
+
charset = unset
|
| 53 |
+
trim_trailing_whitespace = unset
|
| 54 |
+
insert_final_newline = unset
|
| 55 |
+
|
| 56 |
+
[benches/**]
|
| 57 |
+
indent_style = unset
|
| 58 |
+
indent_size = unset
|
| 59 |
+
end_of_line = unset
|
| 60 |
+
charset = unset
|
| 61 |
+
trim_trailing_whitespace = unset
|
| 62 |
+
insert_final_newline = unset
|
backend/llama.cpp/.flake8
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[flake8]
|
| 2 |
+
max-line-length = 125
|
| 3 |
+
ignore = E203,E211,E221,E225,E231,E241,E251,E261,E266,E501,E701,E704,W503
|
| 4 |
+
exclude =
|
| 5 |
+
# Do not traverse examples and tools
|
| 6 |
+
examples,
|
| 7 |
+
tools,
|
| 8 |
+
# Do not include package initializers
|
| 9 |
+
__init__.py,
|
| 10 |
+
# No need to traverse our git directory
|
| 11 |
+
.git,
|
| 12 |
+
# There's no value in checking cache directories
|
| 13 |
+
__pycache__,
|
| 14 |
+
# No need to include the build path
|
| 15 |
+
build,
|
| 16 |
+
# This contains builds that we don't want to check
|
| 17 |
+
dist # This is generated with `python build .` for package releases
|
| 18 |
+
# max-complexity = 10
|
backend/llama.cpp/.gemini/settings.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{ "contextFileName": "AGENTS.md" }
|
backend/llama.cpp/.github/ISSUE_TEMPLATE/010-bug-compilation.yml
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: Bug (compilation)
|
| 2 |
+
description: Something goes wrong when trying to compile llama.cpp.
|
| 3 |
+
title: "Compile bug: "
|
| 4 |
+
labels: ["bug-unconfirmed", "compilation"]
|
| 5 |
+
body:
|
| 6 |
+
- type: markdown
|
| 7 |
+
attributes:
|
| 8 |
+
value: >
|
| 9 |
+
Thanks for taking the time to fill out this bug report!
|
| 10 |
+
This issue template is intended for bug reports where the compilation of llama.cpp fails.
|
| 11 |
+
Before opening an issue, please confirm that the compilation still fails
|
| 12 |
+
after recreating the CMake build directory and with `-DGGML_CCACHE=OFF`.
|
| 13 |
+
If the compilation succeeds with ccache disabled you should be able to permanently fix the issue
|
| 14 |
+
by clearing `~/.cache/ccache` (on Linux).
|
| 15 |
+
|
| 16 |
+
Please fill out this template yourself, copypasting language model outputs is [strictly prohibited](https://github.com/ggml-org/llama.cpp/blob/master/CONTRIBUTING.md#ai-usage-policy).
|
| 17 |
+
- type: textarea
|
| 18 |
+
id: commit
|
| 19 |
+
attributes:
|
| 20 |
+
label: Git commit
|
| 21 |
+
description: Which commit are you trying to compile?
|
| 22 |
+
placeholder: |
|
| 23 |
+
$git rev-parse HEAD
|
| 24 |
+
84a07a17b1b08cf2b9747c633a2372782848a27f
|
| 25 |
+
validations:
|
| 26 |
+
required: true
|
| 27 |
+
- type: dropdown
|
| 28 |
+
id: operating-system
|
| 29 |
+
attributes:
|
| 30 |
+
label: Operating systems
|
| 31 |
+
description: Which operating systems do you know to be affected?
|
| 32 |
+
multiple: true
|
| 33 |
+
options:
|
| 34 |
+
- Linux
|
| 35 |
+
- Mac
|
| 36 |
+
- Windows
|
| 37 |
+
- BSD
|
| 38 |
+
- Other? (Please let us know in description)
|
| 39 |
+
validations:
|
| 40 |
+
required: true
|
| 41 |
+
- type: dropdown
|
| 42 |
+
id: backends
|
| 43 |
+
attributes:
|
| 44 |
+
label: GGML backends
|
| 45 |
+
description: Which GGML backends do you know to be affected?
|
| 46 |
+
options: [AMX, BLAS, CANN, CPU, CUDA, Hexagon, HIP, Metal, Musa, OpenCL, OpenVINO, RPC, SYCL, VirtGPU, Vulkan, WebGPU, zDNN, ZenDNN]
|
| 47 |
+
multiple: true
|
| 48 |
+
validations:
|
| 49 |
+
required: true
|
| 50 |
+
- type: textarea
|
| 51 |
+
id: info
|
| 52 |
+
attributes:
|
| 53 |
+
label: Problem description & steps to reproduce
|
| 54 |
+
description: >
|
| 55 |
+
Please give us a summary of the problem and tell us how to reproduce it.
|
| 56 |
+
If you can narrow down the bug to specific compile flags, that information would be very much appreciated by us.
|
| 57 |
+
placeholder: >
|
| 58 |
+
I'm trying to compile llama.cpp with CUDA support on a fresh install of Ubuntu and get error XY.
|
| 59 |
+
Here are the exact commands that I used: ...
|
| 60 |
+
validations:
|
| 61 |
+
required: true
|
| 62 |
+
- type: textarea
|
| 63 |
+
id: first_bad_commit
|
| 64 |
+
attributes:
|
| 65 |
+
label: First Bad Commit
|
| 66 |
+
description: >
|
| 67 |
+
If the bug was not present on an earlier version: when did it start appearing?
|
| 68 |
+
If possible, please do a git bisect and identify the exact commit that introduced the bug.
|
| 69 |
+
validations:
|
| 70 |
+
required: false
|
| 71 |
+
- type: textarea
|
| 72 |
+
id: command
|
| 73 |
+
attributes:
|
| 74 |
+
label: Compile command
|
| 75 |
+
description: >
|
| 76 |
+
Please provide the exact command you used to compile llama.cpp. For example: `cmake -B ...`.
|
| 77 |
+
This will be automatically formatted into code, so no need for backticks.
|
| 78 |
+
render: shell
|
| 79 |
+
validations:
|
| 80 |
+
required: true
|
| 81 |
+
- type: textarea
|
| 82 |
+
id: logs
|
| 83 |
+
attributes:
|
| 84 |
+
label: Relevant log output
|
| 85 |
+
description: >
|
| 86 |
+
Please copy and paste any relevant log output, including any generated text.
|
| 87 |
+
This will be automatically formatted into code, so no need for backticks.
|
| 88 |
+
render: shell
|
| 89 |
+
validations:
|
| 90 |
+
required: true
|
backend/llama.cpp/.github/ISSUE_TEMPLATE/011-bug-results.yml
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: Bug (model use)
|
| 2 |
+
description: Something goes wrong when running a model (crashes, garbled outputs, etc.).
|
| 3 |
+
title: "Eval bug: "
|
| 4 |
+
labels: ["bug-unconfirmed", "model evaluation"]
|
| 5 |
+
body:
|
| 6 |
+
- type: markdown
|
| 7 |
+
attributes:
|
| 8 |
+
value: >
|
| 9 |
+
Thanks for taking the time to fill out this bug report!
|
| 10 |
+
This issue template is intended for bug reports where the model evaluation results
|
| 11 |
+
(i.e. the generated text) are incorrect or llama.cpp crashes during model evaluation.
|
| 12 |
+
If you encountered the issue while using an external UI (e.g. ollama),
|
| 13 |
+
please reproduce your issue using one of the examples/binaries in this repository.
|
| 14 |
+
The `llama-completion` binary can be used for simple and reproducible model inference.
|
| 15 |
+
|
| 16 |
+
Please fill out this template yourself, copypasting language model outputs is [strictly prohibited](https://github.com/ggml-org/llama.cpp/blob/master/CONTRIBUTING.md#ai-usage-policy).
|
| 17 |
+
- type: textarea
|
| 18 |
+
id: version
|
| 19 |
+
attributes:
|
| 20 |
+
label: Name and Version
|
| 21 |
+
description: Which version of our software are you running? (use `--version` to get a version string)
|
| 22 |
+
placeholder: |
|
| 23 |
+
$./llama-cli --version
|
| 24 |
+
version: 2999 (42b4109e)
|
| 25 |
+
built with cc (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0 for x86_64-linux-gnu
|
| 26 |
+
validations:
|
| 27 |
+
required: true
|
| 28 |
+
- type: dropdown
|
| 29 |
+
id: operating-system
|
| 30 |
+
attributes:
|
| 31 |
+
label: Operating systems
|
| 32 |
+
description: Which operating systems do you know to be affected?
|
| 33 |
+
multiple: true
|
| 34 |
+
options:
|
| 35 |
+
- Linux
|
| 36 |
+
- Mac
|
| 37 |
+
- Windows
|
| 38 |
+
- BSD
|
| 39 |
+
- Other? (Please let us know in description)
|
| 40 |
+
validations:
|
| 41 |
+
required: true
|
| 42 |
+
- type: dropdown
|
| 43 |
+
id: backends
|
| 44 |
+
attributes:
|
| 45 |
+
label: GGML backends
|
| 46 |
+
description: Which GGML backends do you know to be affected?
|
| 47 |
+
options: [AMX, BLAS, CANN, CPU, CUDA, Hexagon, HIP, Metal, Musa, OpenCL, OpenVINO, RPC, SYCL, VirtGPU, Vulkan, WebGPU, zDNN, ZenDNN]
|
| 48 |
+
multiple: true
|
| 49 |
+
validations:
|
| 50 |
+
required: true
|
| 51 |
+
- type: textarea
|
| 52 |
+
id: hardware
|
| 53 |
+
attributes:
|
| 54 |
+
label: Hardware
|
| 55 |
+
description: Which CPUs/GPUs are you using?
|
| 56 |
+
placeholder: >
|
| 57 |
+
e.g. Ryzen 5950X + 2x RTX 4090
|
| 58 |
+
validations:
|
| 59 |
+
required: true
|
| 60 |
+
- type: textarea
|
| 61 |
+
id: model
|
| 62 |
+
attributes:
|
| 63 |
+
label: Models
|
| 64 |
+
description: >
|
| 65 |
+
Which model(s) at which quantization were you using when encountering the bug?
|
| 66 |
+
If you downloaded a GGUF file off of Huggingface, please provide a link.
|
| 67 |
+
placeholder: >
|
| 68 |
+
e.g. Meta LLaMA 3.1 Instruct 8b q4_K_M
|
| 69 |
+
validations:
|
| 70 |
+
required: false
|
| 71 |
+
- type: textarea
|
| 72 |
+
id: info
|
| 73 |
+
attributes:
|
| 74 |
+
label: Problem description & steps to reproduce
|
| 75 |
+
description: >
|
| 76 |
+
Please give us a summary of the problem and tell us how to reproduce it.
|
| 77 |
+
If you can narrow down the bug to specific hardware, compile flags, or command line arguments,
|
| 78 |
+
that information would be very much appreciated by us.
|
| 79 |
+
|
| 80 |
+
If possible, please try to reproduce the issue using `llama-completion` with `-fit off`.
|
| 81 |
+
If you can only reproduce the issue with `-fit on`, please provide logs both with and without `--verbose`.
|
| 82 |
+
placeholder: >
|
| 83 |
+
e.g. when I run llama-completion with `-fa on` I get garbled outputs for very long prompts.
|
| 84 |
+
With short prompts or `-fa off` it works correctly.
|
| 85 |
+
Here are the exact commands that I used: ...
|
| 86 |
+
validations:
|
| 87 |
+
required: true
|
| 88 |
+
- type: textarea
|
| 89 |
+
id: first_bad_commit
|
| 90 |
+
attributes:
|
| 91 |
+
label: First Bad Commit
|
| 92 |
+
description: >
|
| 93 |
+
If the bug was not present on an earlier version: when did it start appearing?
|
| 94 |
+
If possible, please do a git bisect and identify the exact commit that introduced the bug.
|
| 95 |
+
validations:
|
| 96 |
+
required: false
|
| 97 |
+
- type: textarea
|
| 98 |
+
id: logs
|
| 99 |
+
attributes:
|
| 100 |
+
label: Relevant log output
|
| 101 |
+
description: >
|
| 102 |
+
Please copy and paste any relevant log output, including the command that you entered and any generated text.
|
| 103 |
+
For very long logs (thousands of lines), please upload them as files instead; the `--log-file` CLI argument can be used for this purpose.
|
| 104 |
+
On Linux you can alternatively redirect the console output of any command into a file by appending ` > llama.log 2>&1` to your command.
|
| 105 |
+
value: |
|
| 106 |
+
<details>
|
| 107 |
+
<summary>Logs</summary>
|
| 108 |
+
<!-- Copy-pasted short logs go into the "console" area here -->
|
| 109 |
+
|
| 110 |
+
```console
|
| 111 |
+
|
| 112 |
+
```
|
| 113 |
+
</details>
|
| 114 |
+
|
| 115 |
+
<!-- Long logs that you upload as files go here, outside the "console" area -->
|
| 116 |
+
validations:
|
| 117 |
+
required: true
|
backend/llama.cpp/.github/ISSUE_TEMPLATE/019-bug-misc.yml
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: Bug (misc.)
|
| 2 |
+
description: Something is not working the way it should (and it's not covered by any of the above cases).
|
| 3 |
+
title: "Misc. bug: "
|
| 4 |
+
labels: ["bug-unconfirmed"]
|
| 5 |
+
body:
|
| 6 |
+
- type: markdown
|
| 7 |
+
attributes:
|
| 8 |
+
value: >
|
| 9 |
+
Thanks for taking the time to fill out this bug report!
|
| 10 |
+
This issue template is intended for miscellaneous bugs that don't fit into any other category.
|
| 11 |
+
If you encountered the issue while using an external UI (e.g. ollama),
|
| 12 |
+
please reproduce your issue using one of the examples/binaries in this repository.
|
| 13 |
+
|
| 14 |
+
Please fill out this template yourself, copypasting language model outputs is [strictly prohibited](https://github.com/ggml-org/llama.cpp/blob/master/CONTRIBUTING.md#ai-usage-policy).
|
| 15 |
+
- type: textarea
|
| 16 |
+
id: version
|
| 17 |
+
attributes:
|
| 18 |
+
label: Name and Version
|
| 19 |
+
description: Which version of our software is affected? (You can use `--version` to get a version string.)
|
| 20 |
+
placeholder: |
|
| 21 |
+
$./llama-cli --version
|
| 22 |
+
version: 2999 (42b4109e)
|
| 23 |
+
built with cc (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0 for x86_64-linux-gnu
|
| 24 |
+
validations:
|
| 25 |
+
required: true
|
| 26 |
+
- type: dropdown
|
| 27 |
+
id: operating-system
|
| 28 |
+
attributes:
|
| 29 |
+
label: Operating systems
|
| 30 |
+
description: Which operating systems do you know to be affected?
|
| 31 |
+
multiple: true
|
| 32 |
+
options:
|
| 33 |
+
- Linux
|
| 34 |
+
- Mac
|
| 35 |
+
- Windows
|
| 36 |
+
- BSD
|
| 37 |
+
- Other? (Please let us know in description)
|
| 38 |
+
validations:
|
| 39 |
+
required: false
|
| 40 |
+
- type: dropdown
|
| 41 |
+
id: module
|
| 42 |
+
attributes:
|
| 43 |
+
label: Which llama.cpp modules do you know to be affected?
|
| 44 |
+
multiple: true
|
| 45 |
+
options:
|
| 46 |
+
- Documentation/Github
|
| 47 |
+
- libllama (core library)
|
| 48 |
+
- llama-cli
|
| 49 |
+
- llama-server
|
| 50 |
+
- llama-bench
|
| 51 |
+
- llama-quantize
|
| 52 |
+
- Python/Bash scripts
|
| 53 |
+
- Test code
|
| 54 |
+
- Other (Please specify in the next section)
|
| 55 |
+
validations:
|
| 56 |
+
required: false
|
| 57 |
+
- type: textarea
|
| 58 |
+
id: command
|
| 59 |
+
attributes:
|
| 60 |
+
label: Command line
|
| 61 |
+
description: >
|
| 62 |
+
Please provide the exact commands you entered, if applicable. For example: `llama-server -m ... -c ...`, `llama-cli -m ...`, etc.
|
| 63 |
+
This will be automatically formatted into code, so no need for backticks.
|
| 64 |
+
render: shell
|
| 65 |
+
validations:
|
| 66 |
+
required: false
|
| 67 |
+
- type: textarea
|
| 68 |
+
id: info
|
| 69 |
+
attributes:
|
| 70 |
+
label: Problem description & steps to reproduce
|
| 71 |
+
description: >
|
| 72 |
+
Please give us a summary of the problem and tell us how to reproduce it (if applicable).
|
| 73 |
+
validations:
|
| 74 |
+
required: true
|
| 75 |
+
- type: textarea
|
| 76 |
+
id: first_bad_commit
|
| 77 |
+
attributes:
|
| 78 |
+
label: First Bad Commit
|
| 79 |
+
description: >
|
| 80 |
+
If the bug was not present on an earlier version and it's not trivial to track down: when did it start appearing?
|
| 81 |
+
If possible, please do a git bisect and identify the exact commit that introduced the bug.
|
| 82 |
+
validations:
|
| 83 |
+
required: false
|
| 84 |
+
- type: textarea
|
| 85 |
+
id: logs
|
| 86 |
+
attributes:
|
| 87 |
+
label: Relevant log output
|
| 88 |
+
description: >
|
| 89 |
+
If applicable, please copy and paste any relevant log output, including any generated text.
|
| 90 |
+
If you are encountering problems specifically with the `llama_params_fit` module, always upload `--verbose` logs as well.
|
| 91 |
+
For very long logs (thousands of lines), please upload them as files instead; the `--log-file` CLI argument can be used for this purpose.
|
| 92 |
+
On Linux you can alternatively redirect the console output of any command into a file by appending ` > llama.log 2>&1` to your command.
|
| 93 |
+
value: |
|
| 94 |
+
<details>
|
| 95 |
+
<summary>Logs</summary>
|
| 96 |
+
<!-- Copy-pasted short logs go into the "console" area here -->
|
| 97 |
+
|
| 98 |
+
```console
|
| 99 |
+
|
| 100 |
+
```
|
| 101 |
+
</details>
|
| 102 |
+
|
| 103 |
+
<!-- Long logs that you upload as files go here, outside the "console" area -->
|
| 104 |
+
validations:
|
| 105 |
+
required: false
|
backend/llama.cpp/.github/ISSUE_TEMPLATE/020-enhancement.yml
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: Enhancement
|
| 2 |
+
description: Used to request enhancements for llama.cpp.
|
| 3 |
+
title: "Feature Request: "
|
| 4 |
+
labels: ["enhancement"]
|
| 5 |
+
body:
|
| 6 |
+
- type: markdown
|
| 7 |
+
attributes:
|
| 8 |
+
value: |
|
| 9 |
+
[Please post your idea first in Discussion if there is not yet a consensus for this enhancement request. This will help to keep this issue tracker focused on enhancements that the community has agreed needs to be implemented.](https://github.com/ggml-org/llama.cpp/discussions/categories/ideas)
|
| 10 |
+
|
| 11 |
+
Please fill out this template yourself, copypasting language model outputs is [strictly prohibited](https://github.com/ggml-org/llama.cpp/blob/master/CONTRIBUTING.md#ai-usage-policy).
|
| 12 |
+
|
| 13 |
+
- type: checkboxes
|
| 14 |
+
id: prerequisites
|
| 15 |
+
attributes:
|
| 16 |
+
label: Prerequisites
|
| 17 |
+
description: Please confirm the following before submitting your enhancement request.
|
| 18 |
+
options:
|
| 19 |
+
- label: I am running the latest code. Mention the version if possible as well.
|
| 20 |
+
required: true
|
| 21 |
+
- label: I carefully followed the [README.md](https://github.com/ggml-org/llama.cpp/blob/master/README.md).
|
| 22 |
+
required: true
|
| 23 |
+
- label: I searched using keywords relevant to my issue to make sure that I am creating a new issue that is not already open (or closed).
|
| 24 |
+
required: true
|
| 25 |
+
- label: I reviewed the [Discussions](https://github.com/ggml-org/llama.cpp/discussions), and have a new and useful enhancement to share.
|
| 26 |
+
required: true
|
| 27 |
+
|
| 28 |
+
- type: textarea
|
| 29 |
+
id: feature-description
|
| 30 |
+
attributes:
|
| 31 |
+
label: Feature Description
|
| 32 |
+
description: Please provide a detailed written description of what you were trying to do, and what you expected `llama.cpp` to do as an enhancement.
|
| 33 |
+
placeholder: Detailed description of the enhancement
|
| 34 |
+
validations:
|
| 35 |
+
required: true
|
| 36 |
+
|
| 37 |
+
- type: textarea
|
| 38 |
+
id: motivation
|
| 39 |
+
attributes:
|
| 40 |
+
label: Motivation
|
| 41 |
+
description: Please provide a detailed written description of reasons why this feature is necessary and how it is useful to `llama.cpp` users.
|
| 42 |
+
placeholder: Explanation of why this feature is needed and its benefits
|
| 43 |
+
validations:
|
| 44 |
+
required: true
|
| 45 |
+
|
| 46 |
+
- type: textarea
|
| 47 |
+
id: possible-implementation
|
| 48 |
+
attributes:
|
| 49 |
+
label: Possible Implementation
|
| 50 |
+
description: If you have an idea as to how it can be implemented, please write a detailed description. Feel free to give links to external sources or share visuals that might be helpful to understand the details better.
|
| 51 |
+
placeholder: Detailed description of potential implementation
|
| 52 |
+
validations:
|
| 53 |
+
required: false
|
backend/llama.cpp/.github/ISSUE_TEMPLATE/030-research.yml
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: Research
|
| 2 |
+
description: Track new technical research area.
|
| 3 |
+
title: "Research: "
|
| 4 |
+
labels: ["research 🔬"]
|
| 5 |
+
body:
|
| 6 |
+
- type: markdown
|
| 7 |
+
attributes:
|
| 8 |
+
value: |
|
| 9 |
+
Don't forget to check for any [duplicate research issue tickets](https://github.com/ggml-org/llama.cpp/issues?q=is%3Aopen+is%3Aissue+label%3A%22research+%F0%9F%94%AC%22)
|
| 10 |
+
|
| 11 |
+
Please fill out this template yourself, copypasting language model outputs is [strictly prohibited](https://github.com/ggml-org/llama.cpp/blob/master/CONTRIBUTING.md#ai-usage-policy).
|
| 12 |
+
|
| 13 |
+
- type: checkboxes
|
| 14 |
+
id: research-stage
|
| 15 |
+
attributes:
|
| 16 |
+
label: Research Stage
|
| 17 |
+
description: Track general state of this research ticket
|
| 18 |
+
options:
|
| 19 |
+
- label: Background Research (Let's try to avoid reinventing the wheel)
|
| 20 |
+
- label: Hypothesis Formed (How do you think this will work and it's effect?)
|
| 21 |
+
- label: Strategy / Implementation Forming
|
| 22 |
+
- label: Analysis of results
|
| 23 |
+
- label: Debrief / Documentation (So people in the future can learn from us)
|
| 24 |
+
|
| 25 |
+
- type: textarea
|
| 26 |
+
id: background
|
| 27 |
+
attributes:
|
| 28 |
+
label: Previous existing literature and research
|
| 29 |
+
description: Whats the current state of the art and whats the motivation for this research?
|
| 30 |
+
|
| 31 |
+
- type: textarea
|
| 32 |
+
id: hypothesis
|
| 33 |
+
attributes:
|
| 34 |
+
label: Hypothesis
|
| 35 |
+
description: How do you think this will work and it's effect?
|
| 36 |
+
|
| 37 |
+
- type: textarea
|
| 38 |
+
id: implementation
|
| 39 |
+
attributes:
|
| 40 |
+
label: Implementation
|
| 41 |
+
description: Got an approach? e.g. a PR ready to go?
|
| 42 |
+
|
| 43 |
+
- type: textarea
|
| 44 |
+
id: analysis
|
| 45 |
+
attributes:
|
| 46 |
+
label: Analysis
|
| 47 |
+
description: How does the proposed implementation behave?
|
| 48 |
+
|
| 49 |
+
- type: textarea
|
| 50 |
+
id: logs
|
| 51 |
+
attributes:
|
| 52 |
+
label: Relevant log output
|
| 53 |
+
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
| 54 |
+
render: shell
|
backend/llama.cpp/.github/ISSUE_TEMPLATE/040-refactor.yml
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: Refactor (Maintainers)
|
| 2 |
+
description: Used to track refactoring opportunities.
|
| 3 |
+
title: "Refactor: "
|
| 4 |
+
labels: ["refactor"]
|
| 5 |
+
body:
|
| 6 |
+
- type: markdown
|
| 7 |
+
attributes:
|
| 8 |
+
value: |
|
| 9 |
+
Don't forget to [check for existing refactor issue tickets](https://github.com/ggml-org/llama.cpp/issues?q=is%3Aopen+is%3Aissue+label%3Arefactoring) in case it's already covered.
|
| 10 |
+
Also you may want to check [Pull request refactor label as well](https://github.com/ggml-org/llama.cpp/pulls?q=is%3Aopen+is%3Apr+label%3Arefactoring) for duplicates too.
|
| 11 |
+
|
| 12 |
+
Please fill out this template yourself, copypasting language model outputs is [strictly prohibited](https://github.com/ggml-org/llama.cpp/blob/master/CONTRIBUTING.md#ai-usage-policy).
|
| 13 |
+
|
| 14 |
+
- type: textarea
|
| 15 |
+
id: background-description
|
| 16 |
+
attributes:
|
| 17 |
+
label: Background Description
|
| 18 |
+
description: Please provide a detailed written description of the pain points you are trying to solve.
|
| 19 |
+
placeholder: Detailed description behind your motivation to request refactor
|
| 20 |
+
validations:
|
| 21 |
+
required: true
|
| 22 |
+
|
| 23 |
+
- type: textarea
|
| 24 |
+
id: possible-approaches
|
| 25 |
+
attributes:
|
| 26 |
+
label: Possible Refactor Approaches
|
| 27 |
+
description: If you have some idea of possible approaches to solve this problem. You may want to make it a todo list.
|
| 28 |
+
placeholder: Your idea of possible refactoring opportunity/approaches
|
| 29 |
+
validations:
|
| 30 |
+
required: false
|
backend/llama.cpp/.github/ISSUE_TEMPLATE/config.yml
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
blank_issues_enabled: true
|
| 2 |
+
contact_links:
|
| 3 |
+
- name: Got an idea?
|
| 4 |
+
url: https://github.com/ggml-org/llama.cpp/discussions/categories/ideas
|
| 5 |
+
about: Pop it there. It may then become an enhancement ticket.
|
| 6 |
+
- name: Got a question?
|
| 7 |
+
url: https://github.com/ggml-org/llama.cpp/discussions/categories/q-a
|
| 8 |
+
about: Ask a question there!
|
| 9 |
+
- name: Want to contribute?
|
| 10 |
+
url: https://github.com/ggml-org/llama.cpp/wiki/contribute
|
| 11 |
+
about: Head to the contribution guide page of the wiki for areas you can help with
|
backend/llama.cpp/.github/actions/ccache-clear/action.yml
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: "ccache-clear"
|
| 2 |
+
description: "Delete all GitHub Actions caches matching a key prefix"
|
| 3 |
+
inputs:
|
| 4 |
+
key:
|
| 5 |
+
description: "Cache key prefix to match and delete"
|
| 6 |
+
required: true
|
| 7 |
+
|
| 8 |
+
runs:
|
| 9 |
+
using: "composite"
|
| 10 |
+
steps:
|
| 11 |
+
- name: Clear caches
|
| 12 |
+
shell: bash
|
| 13 |
+
run: |
|
| 14 |
+
CACHES=$(gh cache list --key "ccache-${{ inputs.key }}" --json id,key --jq '.[] | "\(.id) \(.key)"' 2>/dev/null)
|
| 15 |
+
if [ -z "$CACHES" ]; then
|
| 16 |
+
echo "No caches found with key prefix: ${{ inputs.key }}"
|
| 17 |
+
exit 0
|
| 18 |
+
fi
|
| 19 |
+
while read -r id key; do
|
| 20 |
+
echo "Deleting cache: $id ($key)"
|
| 21 |
+
gh cache delete "$id"
|
| 22 |
+
done <<< "$CACHES"
|
backend/llama.cpp/.github/actions/get-tag-name/action.yml
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: "Determine tag name"
|
| 2 |
+
description: "Determine the tag name to use for a release"
|
| 3 |
+
outputs:
|
| 4 |
+
name:
|
| 5 |
+
description: "The name of the tag"
|
| 6 |
+
value: ${{ steps.tag.outputs.name }}
|
| 7 |
+
|
| 8 |
+
runs:
|
| 9 |
+
using: "composite"
|
| 10 |
+
steps:
|
| 11 |
+
- name: Determine tag name
|
| 12 |
+
id: tag
|
| 13 |
+
shell: bash
|
| 14 |
+
run: |
|
| 15 |
+
BUILD_NUMBER="$(git rev-list --count HEAD)"
|
| 16 |
+
SHORT_HASH="$(git rev-parse --short=7 HEAD)"
|
| 17 |
+
if [[ "${{ env.BRANCH_NAME }}" == "master" ]]; then
|
| 18 |
+
echo "name=b${BUILD_NUMBER}" >> $GITHUB_OUTPUT
|
| 19 |
+
else
|
| 20 |
+
SAFE_NAME=$(echo "${{ env.BRANCH_NAME }}" | tr '/' '-')
|
| 21 |
+
echo "name=${SAFE_NAME}-b${BUILD_NUMBER}-${SHORT_HASH}" >> $GITHUB_OUTPUT
|
| 22 |
+
fi
|
backend/llama.cpp/.github/actions/install-exe/action.yml
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: "Install exe"
|
| 2 |
+
description: "Download and install exe"
|
| 3 |
+
inputs:
|
| 4 |
+
url:
|
| 5 |
+
description: "URL of the exe installer"
|
| 6 |
+
required: true
|
| 7 |
+
args:
|
| 8 |
+
description: "Installer arguments"
|
| 9 |
+
required: true
|
| 10 |
+
timeout:
|
| 11 |
+
description: "Timeout (in ms)"
|
| 12 |
+
required: false
|
| 13 |
+
default: "600000"
|
| 14 |
+
|
| 15 |
+
runs:
|
| 16 |
+
using: "composite"
|
| 17 |
+
steps:
|
| 18 |
+
- name: Install EXE
|
| 19 |
+
shell: pwsh
|
| 20 |
+
run: |
|
| 21 |
+
$ErrorActionPreference = "Stop"
|
| 22 |
+
write-host "Downloading Installer EXE"
|
| 23 |
+
Invoke-WebRequest -Uri "${{ inputs.url }}" -OutFile "${env:RUNNER_TEMP}\temp-install.exe"
|
| 24 |
+
write-host "Installing"
|
| 25 |
+
$proc = Start-Process "${env:RUNNER_TEMP}\temp-install.exe" -ArgumentList '${{ inputs.args }}' -NoNewWindow -PassThru
|
| 26 |
+
$completed = $proc.WaitForExit(${{ inputs.timeout }})
|
| 27 |
+
if (-not $completed) {
|
| 28 |
+
Write-Error "Installer timed out. Killing the process"
|
| 29 |
+
$proc.Kill()
|
| 30 |
+
exit 1
|
| 31 |
+
}
|
| 32 |
+
if ($proc.ExitCode -ne 0) {
|
| 33 |
+
Write-Error "Installer failed with exit code $($proc.ExitCode)"
|
| 34 |
+
exit 1
|
| 35 |
+
}
|
| 36 |
+
write-host "Completed installation"
|
backend/llama.cpp/.github/actions/linux-setup-openvino/action.yml
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: "Linux - Setup OpenVINO Toolkit"
|
| 2 |
+
description: "Setup OpenVINO Toolkit for Linux"
|
| 3 |
+
inputs:
|
| 4 |
+
path:
|
| 5 |
+
description: "Installation path"
|
| 6 |
+
required: true
|
| 7 |
+
version_major:
|
| 8 |
+
description: "OpenVINO major version (e.g., 2025.3)"
|
| 9 |
+
required: true
|
| 10 |
+
version_full:
|
| 11 |
+
description: "OpenVINO full version (e.g., 2025.3.0.19807.44526285f24)"
|
| 12 |
+
required: true
|
| 13 |
+
|
| 14 |
+
runs:
|
| 15 |
+
using: "composite"
|
| 16 |
+
steps:
|
| 17 |
+
- name: Setup OpenVINO Toolkit
|
| 18 |
+
id: setup
|
| 19 |
+
uses: ./.github/actions/unarchive-tar
|
| 20 |
+
with:
|
| 21 |
+
url: https://storage.openvinotoolkit.org/repositories/openvino/packages/${{ inputs.version_major }}/linux/openvino_toolkit_ubuntu24_${{ inputs.version_full }}_x86_64.tgz
|
| 22 |
+
path: ${{ inputs.path }}
|
| 23 |
+
type: z
|
| 24 |
+
strip: 1
|
| 25 |
+
|
backend/llama.cpp/.github/actions/linux-setup-spacemit/action.yml
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: "Linux - Setup SpacemiT Toolchain"
|
| 2 |
+
description: "Setup SpacemiT Toolchain for Linux"
|
| 3 |
+
inputs:
|
| 4 |
+
path:
|
| 5 |
+
description: "Installation path"
|
| 6 |
+
required: true
|
| 7 |
+
version:
|
| 8 |
+
description: "SpacemiT toolchain version"
|
| 9 |
+
required: true
|
| 10 |
+
|
| 11 |
+
runs:
|
| 12 |
+
using: "composite"
|
| 13 |
+
steps:
|
| 14 |
+
- name: Setup SpacemiT Toolchain
|
| 15 |
+
id: setup
|
| 16 |
+
uses: ./.github/actions/unarchive-tar
|
| 17 |
+
with:
|
| 18 |
+
url: https://github.com/spacemit-com/toolchain/releases/download/v${{ inputs.version }}/spacemit-toolchain-linux-glibc-x86_64-v${{ inputs.version }}.tar.xz
|
| 19 |
+
path: ${{ inputs.path }}
|
| 20 |
+
strip: 1
|
backend/llama.cpp/.github/actions/linux-setup-vulkan/action.yml
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: "Linux - Setup Vulkan SDK"
|
| 2 |
+
description: "Setup Vulkan SDK for Linux"
|
| 3 |
+
inputs:
|
| 4 |
+
path:
|
| 5 |
+
description: "Installation path"
|
| 6 |
+
required: true
|
| 7 |
+
version:
|
| 8 |
+
description: "Vulkan SDK version"
|
| 9 |
+
required: true
|
| 10 |
+
|
| 11 |
+
runs:
|
| 12 |
+
using: "composite"
|
| 13 |
+
steps:
|
| 14 |
+
- name: Setup Vulkan SDK
|
| 15 |
+
id: setup
|
| 16 |
+
uses: ./.github/actions/unarchive-tar
|
| 17 |
+
with:
|
| 18 |
+
url: https://sdk.lunarg.com/sdk/download/${{ inputs.version }}/linux/vulkan_sdk.tar.xz
|
| 19 |
+
path: ${{ inputs.path }}
|
| 20 |
+
strip: 1
|