diff --git a/app.py b/app.py
index e7ccea4..2e9e319 100644
--- a/app.py
+++ b/app.py
@@ -1899,7 +1899,14 @@ def handle_message(data):
gevent.spawn(generate_dalle_image, room_name, message, username)
else:
# All other models (Groq, Together, Mistral, etc.) use OpenAI client
- gevent.spawn(chat_gpt, username, room_name, model_name=model)
+ enable_thinking = data.get("enable_thinking", True)
+ gevent.spawn(
+ chat_gpt,
+ username,
+ room_name,
+ model_name=model,
+ enable_thinking=enable_thinking,
+ )
@socketio.on("delete_message")
@@ -2120,7 +2127,7 @@ def chat_claude(
socketio.emit("delete_processing_message", msg_id, room=room_name)
-def chat_gpt(username, room_name, model_name="gpt-4o-mini"):
+def chat_gpt(username, room_name, model_name="gpt-4o-mini", enable_thinking=True):
openai_client, model_name = get_openai_client_and_model(model_name)
temperature = 0
@@ -2194,23 +2201,29 @@ def chat_gpt(username, room_name, model_name="gpt-4o-mini"):
first_chunk = True
+ create_kwargs = {
+ "model": model_name,
+ "messages": chat_history,
+ "n": 1,
+ "stream": True,
+ }
+ if "o3" not in model_name:
+ # o3 does not support temperature at all!
+ create_kwargs["temperature"] = temperature
+ if not enable_thinking:
+ # Qwen3 / vLLM-style switch to suppress chain-of-thought and save
+ # tokens. OpenAI's hosted API rejects unknown body params, so only
+ # send to self-hosted OpenAI-compatible endpoints; o1/o3 think
+ # unconditionally and would 400 on this anyway. Models whose chat
+ # template ignores the kwarg (e.g. Hermes) simply drop it.
+ base_url = str(getattr(openai_client, "base_url", "") or "")
+ if "api.openai.com" not in base_url:
+ create_kwargs["extra_body"] = {
+ "chat_template_kwargs": {"enable_thinking": False}
+ }
+
try:
- if "o3" in model_name:
- # o3 does not support temperature at all!
- chunks = openai_client.chat.completions.create(
- model=model_name,
- messages=chat_history,
- n=1,
- stream=True,
- )
- else:
- chunks = openai_client.chat.completions.create(
- model=model_name,
- messages=chat_history,
- n=1,
- temperature=temperature,
- stream=True,
- )
+ chunks = openai_client.chat.completions.create(**create_kwargs)
except Exception as e:
with app.app_context():
message_content = f"{model_name} Error: {e}"
diff --git a/templates/base.html b/templates/base.html
index e9146cf..74fba9e 100644
--- a/templates/base.html
+++ b/templates/base.html
@@ -103,6 +103,11 @@
Auto-Play TTS: OFF
+
+
+
Activities
diff --git a/templates/chat.html b/templates/chat.html
index 91ddf75..a3a046b 100644
--- a/templates/chat.html
+++ b/templates/chat.html
@@ -59,6 +59,11 @@
Auto-Play TTS: OFF
+
+
+
Activities
@@ -148,6 +153,13 @@ let autoPlayTTS = localStorage.getItem('autoPlayTTS') === 'true' || false;
let ttsQueue = [];
let isPlayingTTS = false;
+// Thinking-mode state. Default OFF. When OFF we send enable_thinking=false
+// with each chat_message so the server asks the model (Qwen3-style) to skip
+// chain-of-thought entirely — saving tokens, not just hiding output. As a
+// fallback, any reasoning_content that still arrives is dropped client-side
+// (see message_chunk handler). Toggle persists to localStorage.
+let showThinking = localStorage.getItem('showThinking') === 'true';
+
// Vision model state for auto alt-text
let visionAvailable = false;
let visionModel = null;
@@ -375,6 +387,9 @@ document.addEventListener('DOMContentLoaded', (event) => {
// Initialize auto-play TTS button state from localStorage
updateAutoPlayTTSDisplay();
+ // Initialize show-thinking button state from localStorage
+ updateShowThinkingDisplay();
+
// Check for vision model availability (enables image hover descriptions)
initVisionCapability();
@@ -660,7 +675,8 @@ function sendMessage() {
"username": username,
"message": messageToSend,
"model": model, // Pass model as a separate attribute
- "room_name": room_name
+ "room_name": room_name,
+ "enable_thinking": showThinking // false => ask model to skip chain-of-thought
});
messageTextarea.value = "";
// Reset textarea height after sending
@@ -1112,6 +1128,29 @@ function processNextTTS() {
});
}
+// Update show-thinking button labels/colors on both desktop + mobile.
+function updateShowThinkingDisplay() {
+ const btn = document.getElementById("show-thinking-btn");
+ const btnMobile = document.getElementById("show-thinking-btn-mobile");
+ const label = showThinking ? "Thinking: ON" : "Thinking: OFF";
+ const color = showThinking ? "#4CAF50" : "#f44336";
+ if (btn) {
+ btn.textContent = label;
+ btn.style.backgroundColor = color;
+ }
+ if (btnMobile) {
+ btnMobile.textContent = label;
+ btnMobile.style.backgroundColor = color;
+ }
+}
+
+// Toggle reasoning_content display. Off by default; persisted in localStorage.
+function toggleShowThinking() {
+ showThinking = !showThinking;
+ localStorage.setItem('showThinking', showThinking.toString());
+ updateShowThinkingDisplay();
+}
+
// Function to update auto-play TTS button display
function updateAutoPlayTTSDisplay() {
const autoPlayBtn = document.getElementById("auto-play-tts-btn");
@@ -1448,7 +1487,10 @@ socket.on("message_chunk", (data) => {
// the message-content div. Only appears if the server actually streams
// delta.reasoning_content (i.e., thinking is on AND the model is using it).
// Auto-collapses below on the first content delta.
+ // Fallback guard: with Thinking OFF the server already suppresses reasoning
+ // at the model, but if a model ignores the switch and streams anyway, drop it.
if (data.reasoning_content) {
+ if (!showThinking) return;
let thinkingDetails = messageWrapper.querySelector(".message-thinking");
if (!thinkingDetails) {
// Ensure a message-body wrapper exists to anchor against