@@ -446,7 +446,7 @@ if (model.reasoning) {
|
||||
const response = await completeSimple(model, {
|
||||
messages: [{ role: 'user', content: 'Solve: 2x + 5 = 13' }]
|
||||
}, {
|
||||
reasoning: 'medium' // 'minimal' | 'low' | 'medium' | 'high' | 'xhigh' (xhigh maps to high on non-OpenAI providers)
|
||||
reasoning: 'medium' // 'minimal' | 'low' | 'medium' | 'high' | 'xhigh'
|
||||
});
|
||||
|
||||
// Access thinking and text blocks
|
||||
@@ -821,6 +821,8 @@ const response = await stream(ollamaModel, context, {
|
||||
|
||||
Some OpenAI-compatible servers do not understand the `developer` role used for reasoning-capable models. For those providers, set `compat.supportsDeveloperRole` to `false` so the system prompt is sent as a `system` message instead. If the server also does not support `reasoning_effort`, set `compat.supportsReasoningEffort` to `false` too.
|
||||
|
||||
Use model-level `thinkingLevelMap` to describe model-specific thinking controls. Keys are pi thinking levels (`off`, `minimal`, `low`, `medium`, `high`, `xhigh`). Missing keys use provider defaults, string values are sent to the provider, and `null` marks a level unsupported.
|
||||
|
||||
This commonly applies to Ollama, vLLM, SGLang, and similar OpenAI-compatible servers. You can set `compat` at the provider level or per model.
|
||||
|
||||
```typescript
|
||||
@@ -835,6 +837,13 @@ const ollamaReasoningModel: Model<'openai-completions'> = {
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 131072,
|
||||
maxTokens: 32000,
|
||||
thinkingLevelMap: {
|
||||
minimal: null,
|
||||
low: null,
|
||||
medium: null,
|
||||
high: 'high',
|
||||
xhigh: null,
|
||||
},
|
||||
compat: {
|
||||
supportsDeveloperRole: false,
|
||||
supportsReasoningEffort: false,
|
||||
|
||||
Reference in New Issue
Block a user