-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathfabula.config.example.json
More file actions
102 lines (102 loc) · 3.55 KB
/
Copy pathfabula.config.example.json
File metadata and controls
102 lines (102 loc) · 3.55 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
{
"share": "disabled",
"model": "lmstudio/your-local-model-id",
// The «Enhance prompt» button uses the model SELECTED in the composer by default, with a 45s timeout (so a
// slow reasoning model still completes instead of failing). This optional section tunes it PER MODEL: keys
// are model refs ('provider/model'), '_default' applies to all. Each entry may set request params merged
// into the /chat/completions body (max_tokens, temperature, reasoning flags), a 'timeout_ms', and — if you
// want a slow reasoning model to enhance FAST — a 'model' override pointing enhance at a faster model for
// that selected model. Remove this section to just use the selected model as-is.
"enhance": {
"_default": {
"max_tokens": 1024,
"timeout_ms": 45000
},
"provider/some-slow-reasoning-model": {
"model": "provider/some-fast-instruct-model"
}
},
// The background checkpoint writer is OFF here (`thresholds: []` — the engine logs "checkpoint schedule
// empty" and takes none). With ONE local model there is one inference slot, and the writer competes for
// it with the agent it is summarising: measured, main-agent steps took 317/180/727/319 s while a writer
// held the slot versus 8-23 s when it was free, writers spent 1.32x the agent's own tokens, and 121 of 121
// compactions went through the overflow path anyway. If you serve more than one slot, delete the
// `thresholds` line to get the default schedule back; `fork: true` then lets the writer reuse the agent's
// prefill cache instead of evicting it.
"checkpoint": {
"fork": true,
"thresholds": []
},
// Loaded into EVERY request of EVERY project, so every byte here is a per-turn cost that competes with the
// task for the model's attention — keep this list small and general. Project-specific rules do NOT belong
// here: the engine already auto-discovers the nearest project-level agent instructions file for the
// repository being worked on. Situational, deep material belongs in skills, loaded on demand.
"instructions": [
"system-prompt.example.md",
"prompts/runtime-tools-addendum.md",
"prompts/agent-rules.md"
],
"provider": {
"lmstudio": {
"name": "LM Studio (FABULA-LLM local)",
"npm": "@ai-sdk/openai-compatible",
"options": {
"baseURL": "http://localhost:1235/v1"
},
"models": {
"your-local-model-id": {
"name": "Local model",
"family": "MLX · 4bit",
"tools": true,
"attachment": true,
"limit": {
"context": 131072,
"output": 32768
},
"modalities": {
"input": [
"text",
"image"
],
"output": [
"text"
]
}
}
}
},
"nvidia": {
"name": "NVIDIA (FABULA-LLM cloud, optional)",
"npm": "@ai-sdk/openai-compatible",
"options": {
"baseURL": "https://integrate.api.nvidia.com/v1",
"apiKey": "{env:NVIDIA_API_KEY}"
},
"models": {
"z-ai/glm-5.1": {
"name": "GLM-5.1 (agentic/coding)",
"tools": true
},
"deepseek-ai/deepseek-v4-flash": {
"name": "DeepSeek V4 Flash (fast)",
"tools": true
}
}
}
},
"mcp": {
"web-search-internet": {
"type": "local",
"command": [
"npx",
"-y",
"mcp-searxng"
],
"environment": {
"SEARXNG_URL": "http://localhost:8888"
},
"enabled": false
}
},
"lsp": true
}