feat(llm): add disableThinking option to suppress reasoning in extraction models (#228)

Support multiple inference engines and model providers via a strategy
enum so each receives its own thinking-disabling field in
chat-completion request bodies.

Strategies:
- "vllm"      → chat_template_kwargs: { enable_thinking: false }  (vLLM / SGLang)
- "deepseek"  → enable_thinking: false  (DeepSeek official API)
- "dashscope" → enable_thinking: false  (Alibaba DashScope / Qwen)
- "openai"    → reasoning_effort: "low"  (OpenAI o-series, cannot fully disable)
- "anthropic" → thinking: { type: "disabled" }  (Anthropic Claude)
- "kimi"      → thinking: { type: "disabled" }  (Kimi / Moonshot)
- "gemini"    → thinking_config: { thinking_budget: 0 }  (Google Gemini)

Defaults to false (no wrapper). Also accepts boolean true as a shorthand
for "vllm" for convenience.

Covered paths:
- StandaloneLLMRunner (OpenClaw plugin + gateway): llm.disableThinking,
  wired through parseConfig, tdai-core, seed-runtime, and gateway config
  (TDAI_LLM_DISABLE_THINKING env also accepts strategy names).
- Offload local mode (L1/L1.5/L2): separate offload.disableThinking.

The fetch wrapper lives in src/utils/no-think-fetch.ts with a
STRATEGY_TRANSFORMERS map for clean dispatch. StandaloneLLMRunner builds
it once in the constructor; LocalLlmClient caches it at construction
time and passes it through to callLlm().

Add vitest unit tests for all 7 strategies, normalization, validation,
embedding skip, and non-JSON tolerance (20 tests total).

Signed-off-by: jackson.jia <jiazhenghua0@gmail.com>
This commit is contained in:
jackson-jia-914
2026-06-24 18:02:25 +08:00
committed by GitHub
parent 38673b5d2b
commit e9c1af03f6
13 changed files with 514 additions and 2 deletions
+269
View File
@@ -0,0 +1,269 @@
import { describe, it, expect, vi, afterEach } from "vitest";
import {
createNoThinkFetch,
isValidDisableThinkingStrategy,
normalizeDisableThinking,
type DisableThinkingStrategy,
} from "./no-think-fetch";
/**
* Capture the (input, init) passed through to the real global fetch so we can
* assert on the (possibly rewritten) request body. The mock never blindly
* JSON.parses the body — it just records and returns a stub Response.
*/
function captureFetch() {
const calls: Array<{ input: unknown; init: RequestInit | undefined }> = [];
vi.spyOn(globalThis, "fetch").mockImplementation((async (input: unknown, init?: RequestInit) => {
calls.push({ input, init });
return new Response("{}", { status: 200 });
}) as typeof globalThis.fetch);
return calls;
}
afterEach(() => {
vi.restoreAllMocks();
});
describe("createNoThinkFetch", () => {
// ─── vllm strategy (original behavior) ────────────────────────────────────
describe("vllm strategy", () => {
it("injects chat_template_kwargs.enable_thinking=false into chat bodies", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("vllm");
await f("https://example/v1/chat/completions", {
method: "POST",
body: JSON.stringify({ model: "qwen3", messages: [{ role: "user", content: "hi" }] }),
});
const sent = JSON.parse(calls[0].init!.body as string);
expect(sent.chat_template_kwargs).toEqual({ enable_thinking: false });
expect(sent.messages).toHaveLength(1);
});
it("merges into an existing chat_template_kwargs instead of clobbering it", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("vllm");
await f("https://example", {
body: JSON.stringify({ messages: [], chat_template_kwargs: { foo: "bar", enable_thinking: true } }),
});
const sent = JSON.parse(calls[0].init!.body as string);
expect(sent.chat_template_kwargs).toEqual({ foo: "bar", enable_thinking: false });
});
it("leaves embedding requests (input, no messages) untouched", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("vllm");
const body = JSON.stringify({ model: "bge-m3", input: ["hello"] });
await f("https://example/v1/embeddings", { body });
expect(calls[0].init!.body).toBe(body);
expect(JSON.parse(calls[0].init!.body as string).chat_template_kwargs).toBeUndefined();
});
});
// ─── deepseek strategy ───────────────────────────────────────────────────
describe("deepseek strategy", () => {
it("injects top-level enable_thinking: false", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("deepseek");
await f("https://api.deepseek.com/v1/chat/completions", {
method: "POST",
body: JSON.stringify({ model: "deepseek-reasoner", messages: [{ role: "user", content: "hi" }] }),
});
const sent = JSON.parse(calls[0].init!.body as string);
expect(sent.enable_thinking).toBe(false);
expect(sent.chat_template_kwargs).toBeUndefined();
expect(sent.messages).toHaveLength(1);
});
});
// ─── dashscope strategy ─────────────────────────────────────────────────
describe("dashscope strategy", () => {
it("injects top-level enable_thinking: false", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("dashscope");
await f("https://dashscope.aliyuncs.com/compatible-mode/v1/chat/completions", {
method: "POST",
body: JSON.stringify({ model: "qwen-plus", messages: [{ role: "user", content: "hi" }] }),
});
const sent = JSON.parse(calls[0].init!.body as string);
expect(sent.enable_thinking).toBe(false);
expect(sent.chat_template_kwargs).toBeUndefined();
});
});
// ─── openai strategy ────────────────────────────────────────────────────
describe("openai strategy", () => {
it("injects reasoning_effort: low", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("openai");
await f("https://api.openai.com/v1/chat/completions", {
method: "POST",
body: JSON.stringify({ model: "o3-mini", messages: [{ role: "user", content: "hi" }] }),
});
const sent = JSON.parse(calls[0].init!.body as string);
expect(sent.reasoning_effort).toBe("low");
expect(sent.chat_template_kwargs).toBeUndefined();
expect(sent.enable_thinking).toBeUndefined();
});
});
// ─── anthropic strategy ─────────────────────────────────────────────────
describe("anthropic strategy", () => {
it("injects thinking: { type: disabled }", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("anthropic");
await f("https://api.anthropic.com/v1/messages", {
method: "POST",
body: JSON.stringify({ model: "claude-sonnet-4-20250514", messages: [{ role: "user", content: "hi" }] }),
});
const sent = JSON.parse(calls[0].init!.body as string);
expect(sent.thinking).toEqual({ type: "disabled" });
expect(sent.chat_template_kwargs).toBeUndefined();
});
});
// ─── kimi strategy ─────────────────────────────────────────────────────
describe("kimi strategy", () => {
it("injects thinking: { type: disabled }", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("kimi");
await f("https://api.moonshot.cn/v1/chat/completions", {
method: "POST",
body: JSON.stringify({ model: "kimi-k2.6", messages: [{ role: "user", content: "hi" }] }),
});
const sent = JSON.parse(calls[0].init!.body as string);
expect(sent.thinking).toEqual({ type: "disabled" });
expect(sent.chat_template_kwargs).toBeUndefined();
expect(sent.enable_thinking).toBeUndefined();
});
});
// ─── gemini strategy ────────────────────────────────────────────────────
describe("gemini strategy", () => {
it("injects thinking_config: { thinking_budget: 0 }", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("gemini");
await f("https://generativelanguage.googleapis.com/v1beta/models/gemini-2.5-flash:generateContent", {
method: "POST",
body: JSON.stringify({ messages: [{ role: "user", content: "hi" }] }),
});
const sent = JSON.parse(calls[0].init!.body as string);
expect(sent.thinking_config).toEqual({ thinking_budget: 0 });
expect(sent.chat_template_kwargs).toBeUndefined();
});
});
// ─── strategy === false (passthrough) ───────────────────────────────────
describe("false strategy", () => {
it("returns globalThis.fetch directly", () => {
const f = createNoThinkFetch(false);
expect(f).toBe(globalThis.fetch);
});
it("default parameter returns globalThis.fetch", () => {
const f = createNoThinkFetch();
expect(f).toBe(globalThis.fetch);
});
});
// ─── Common behavior across all strategies ──────────────────────────────
describe("common behavior", () => {
it("forwards a non-JSON string body unchanged", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("vllm");
await f("https://example", { body: "not-json" });
expect(calls[0].init!.body).toBe("not-json");
});
it("forwards requests with no init and with a non-string body unchanged", async () => {
const calls = captureFetch();
const f = createNoThinkFetch("deepseek");
await f("https://example");
expect(calls[0].init).toBeUndefined();
const blob = new Uint8Array([1, 2, 3]);
await f("https://example", { body: blob });
expect(calls[1].init!.body).toBe(blob);
});
});
});
// ─── Validation helpers ─────────────────────────────────────────────────────
describe("isValidDisableThinkingStrategy", () => {
it("returns true for all valid strategies", () => {
const valid: DisableThinkingStrategy[] = [false, "vllm", "deepseek", "dashscope", "openai", "anthropic", "kimi", "gemini"];
for (const v of valid) {
expect(isValidDisableThinkingStrategy(v)).toBe(true);
}
});
it("returns false for invalid values", () => {
const invalid = [true, "invalid", "sglang", "VLLM", 0, null, undefined, "", "true"];
for (const v of invalid) {
expect(isValidDisableThinkingStrategy(v)).toBe(false);
}
});
});
describe("normalizeDisableThinking", () => {
it("maps true to vllm (shorthand)", () => {
expect(normalizeDisableThinking(true)).toBe("vllm");
});
it("maps false to false", () => {
expect(normalizeDisableThinking(false)).toBe(false);
});
it("maps undefined to false", () => {
expect(normalizeDisableThinking(undefined)).toBe(false);
});
it("passes through valid strategy strings", () => {
expect(normalizeDisableThinking("vllm")).toBe("vllm");
expect(normalizeDisableThinking("deepseek")).toBe("deepseek");
expect(normalizeDisableThinking("dashscope")).toBe("dashscope");
expect(normalizeDisableThinking("openai")).toBe("openai");
expect(normalizeDisableThinking("anthropic")).toBe("anthropic");
expect(normalizeDisableThinking("kimi")).toBe("kimi");
expect(normalizeDisableThinking("gemini")).toBe("gemini");
});
it("warns and returns false for unknown strategies", () => {
const warnSpy = vi.spyOn(console, "warn").mockImplementation(() => {});
expect(normalizeDisableThinking("unknown_strategy")).toBe(false);
expect(warnSpy).toHaveBeenCalledWith(
expect.stringContaining('Unknown disableThinking strategy "unknown_strategy"'),
);
warnSpy.mockRestore();
});
});
+135
View File
@@ -0,0 +1,135 @@
/**
* Multi-strategy fetch wrapper for disabling thinking/reasoning across
* different inference engines and model providers.
*
* Each strategy injects provider-specific fields into chat-completion
* request bodies. Non-chat requests (embeddings, etc.) pass through
* unchanged.
*
* Strategies:
* - `"vllm"`: vLLM / SGLang — `chat_template_kwargs.enable_thinking = false`
* - `"deepseek"`: DeepSeek official API — top-level `enable_thinking: false`
* - `"dashscope"`: Alibaba DashScope (Qwen) — top-level `enable_thinking: false`
* - `"openai"`: OpenAI o-series — `reasoning_effort: "low"` (cannot fully disable)
* - `"anthropic"`: Anthropic Claude — `thinking: { type: "disabled" }`
* - `"kimi"`: Kimi (Moonshot) — `thinking: { type: "disabled" }`
* - `"gemini"`: Google Gemini — `thinking_config: { thinking_budget: 0 }`
*/
// ─── Type & validation ────────────────────────────────────────────────────────
export type DisableThinkingStrategy =
| false
| "vllm"
| "deepseek"
| "dashscope"
| "openai"
| "anthropic"
| "kimi"
| "gemini";
export const VALID_DISABLE_THINKING_STRATEGIES: readonly DisableThinkingStrategy[] = [
false, "vllm", "deepseek", "dashscope", "openai", "anthropic", "kimi", "gemini",
] as const;
/** Check if a value is a valid DisableThinkingStrategy. */
export function isValidDisableThinkingStrategy(value: unknown): value is DisableThinkingStrategy {
return (VALID_DISABLE_THINKING_STRATEGIES as readonly unknown[]).includes(value);
}
/**
* Normalize a raw boolean-or-string config value into a DisableThinkingStrategy.
*
* true → "vllm" (shorthand for the most common self-hosted scenario)
* false / undefined → false
*
* Unknown string values fall back to false with a console warning.
*/
export function normalizeDisableThinking(raw: boolean | string | undefined): DisableThinkingStrategy {
if (raw === undefined || raw === false) return false;
if (raw === true) return "vllm";
// raw is a string
if (isValidDisableThinkingStrategy(raw)) return raw;
console.warn(
`[memory-tdai] Unknown disableThinking strategy "${raw}", ` +
`valid values: false, true, "vllm", "deepseek", "dashscope", "openai", "anthropic", "kimi", "gemini". ` +
`Thinking will NOT be disabled.`,
);
return false;
}
// ─── Per-provider body transformers ───────────────────────────────────────────
function applyVllm(body: Record<string, unknown>): void {
const existing = body.chat_template_kwargs;
const base = (existing && typeof existing === "object" && !Array.isArray(existing))
? existing as Record<string, unknown>
: {};
body.chat_template_kwargs = { ...base, enable_thinking: false };
}
function applyDeepSeek(body: Record<string, unknown>): void {
body.enable_thinking = false;
}
function applyDashScope(body: Record<string, unknown>): void {
body.enable_thinking = false;
}
function applyOpenAI(body: Record<string, unknown>): void {
body.reasoning_effort = "low";
}
function applyAnthropic(body: Record<string, unknown>): void {
body.thinking = { type: "disabled" };
}
function applyGemini(body: Record<string, unknown>): void {
body.thinking_config = { thinking_budget: 0 };
}
const STRATEGY_TRANSFORMERS: Record<
Exclude<DisableThinkingStrategy, false>,
(body: Record<string, unknown>) => void
> = {
vllm: applyVllm,
deepseek: applyDeepSeek,
dashscope: applyDashScope,
openai: applyOpenAI,
anthropic: applyAnthropic,
kimi: applyAnthropic,
gemini: applyGemini,
};
// ─── Factory ──────────────────────────────────────────────────────────────────
/**
* Create a fetch wrapper that injects provider-specific thinking-disabling
* fields into chat-completion request bodies.
*
* When `strategy` is `false`, returns `globalThis.fetch` directly (no wrapper).
*
* Only requests with a `messages` array in the body are modified — embedding
* and other non-chat requests pass through unchanged.
*/
export function createNoThinkFetch(strategy: DisableThinkingStrategy = false): typeof globalThis.fetch {
if (strategy === false) return globalThis.fetch;
const transform = STRATEGY_TRANSFORMERS[strategy];
if (!transform) return globalThis.fetch; // defensive: unknown strategy → passthrough
return (async (input, init) => {
if (init && typeof init.body === "string") {
try {
const body = JSON.parse(init.body);
if (body && Array.isArray(body.messages)) {
transform(body);
init = { ...init, body: JSON.stringify(body) };
}
} catch {
// non-JSON body — forward unchanged
}
}
return globalThis.fetch(input, init);
}) as typeof globalThis.fetch;
}