-
Notifications
You must be signed in to change notification settings - Fork 3
Expand file tree
/
Copy pathconform.example.toml
More file actions
42 lines (38 loc) · 1.32 KB
/
Copy pathconform.example.toml
File metadata and controls
42 lines (38 loc) · 1.32 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
# Copy this to conform.toml and edit. conform.toml is gitignored, so your
# local ports and model names stay yours.
#
# kind = "reference" loads the model in-process with transformers on CPU.
# Slow, no server needed, and treated as ground truth.
# kind = "openai" talks to any OpenAI-compatible /v1/completions server.
#
# capabilities is optional. Omit it to accept the adapter's defaults; set it
# explicitly to tell the suite what a server really supports, so unsupported
# features are skipped instead of reported as failures.
[[engines]]
name = "reference"
kind = "reference"
model = "HuggingFaceTB/SmolLM2-135M-Instruct"
[[engines]]
name = "ollama"
kind = "openai"
base_url = "http://localhost:11434"
model = "qwen2.5:0.5b"
capabilities = ["stop_sequences"]
# vLLM's CPU backend. Start it with:
# vllm serve Qwen/Qwen2.5-0.5B-Instruct --device cpu --port 8000
[[engines]]
name = "vllm-cpu"
kind = "openai"
base_url = "http://localhost:8000"
model = "Qwen/Qwen2.5-0.5B-Instruct"
capabilities = ["stop_sequences", "logprobs", "top_k", "tokenize"]
enabled = false
# llama.cpp's built-in server:
# llama-server -m model.gguf --port 8080
[[engines]]
name = "llama-cpp"
kind = "openai"
base_url = "http://localhost:8080"
model = "local"
capabilities = ["stop_sequences", "logprobs", "top_k"]
enabled = false