Просмотр исходного кода

fix: enable multi-turn tool calling with proper tool_calls propagation

- Add tool_calls field to ChatMessage in provider interface
- Add ToolCall struct and tool_calls field to session Message
- Include tool_calls in assistant messages when building OpenAI API requests
- Serialize/deserialize tool_calls in database session store
- Copy tool_calls from session to provider messages in BuildChatRequest
- Add tools and tool_calls handling in SubmitToolResults
- Fix SubmitToolResultsStream to send tool_calls as separate chunks
- Add query_records tool to webserver tool service
- Add debug logging for API requests and responses

The model can now make multiple sequential tool calls (e.g., list_views
followed by query_records) to complete complex queries like "list my todos".
fszontagh 7 месяцев назад
Родитель
Сommit
a614511e06

+ 5 - 0
llm/include/smartbotic/llm/provider/provider_interface.hpp

@@ -61,10 +61,14 @@ struct ContentPart {
     bool is_error = false;
 };
 
+/// Tool call requested by the model (forward declaration for ChatMessage)
+struct ToolCall;
+
 /// A chat message
 struct ChatMessage {
     MessageRole role = MessageRole::kUser;
     std::vector<ContentPart> content;
+    std::vector<ToolCall> tool_calls;  // For assistant messages with tool calls
 };
 
 /// Tool definition for function calling
@@ -130,6 +134,7 @@ struct ChatResponse {
 /// Stream chunk during streaming
 struct StreamChunk {
     std::string content_delta;
+    std::string thinking_delta;           // For reasoning/thinking content (e.g., <think> tags)
     std::optional<ToolCall> tool_call;
     bool is_done = false;
     std::optional<UsageInfo> usage;       // Present in final chunk

+ 10 - 1
llm/include/smartbotic/llm/session/session.hpp

@@ -29,7 +29,8 @@ enum class ContentPartType {
     kText,
     kImage,
     kToolUse,
-    kToolResult
+    kToolResult,
+    kThinking       // For reasoning/thinking content
 };
 
 /// A content part in a message
@@ -48,11 +49,19 @@ struct ContentPart {
     bool is_error = false;
 };
 
+/// Tool call requested by the model
+struct ToolCall {
+    std::string id;
+    std::string name;
+    std::string arguments;  // JSON
+};
+
 /// A chat message
 struct Message {
     std::string id;
     MessageRole role = MessageRole::kUser;
     std::vector<ContentPart> content;
+    std::vector<ToolCall> tool_calls;  // For assistant messages with tool calls
     std::chrono::system_clock::time_point created_at;
     std::optional<PageContext> page_context;
 };

+ 94 - 1
llm/src/grpc/llm_service.cpp

@@ -169,11 +169,24 @@ auto LLMServiceImpl::BuildChatRequest(const ::smartbotic::llm::ChatRequest& requ
                     content_part.text = part.text;
                     content_part.is_error = part.is_error;
                     break;
+                case session::ContentPartType::kThinking:
+                    // Skip thinking content when building chat context for the provider
+                    // We store it for display purposes but don't include it in the LLM context
+                    continue;
             }
 
             chat_msg.content.push_back(std::move(content_part));
         }
 
+        // Copy tool_calls for assistant messages
+        for (const auto& tc : msg.tool_calls) {
+            provider::ToolCall provider_tc;
+            provider_tc.id = tc.id;
+            provider_tc.name = tc.name;
+            provider_tc.arguments = tc.arguments;
+            chat_msg.tool_calls.push_back(std::move(provider_tc));
+        }
+
         chat_request.messages.push_back(std::move(chat_msg));
     }
 
@@ -392,6 +405,7 @@ auto LLMServiceImpl::BuildChatRequest(const ::smartbotic::llm::ChatRequest& requ
     provider::FinishReason final_reason = provider::FinishReason::kStop;
 
     // Stream callback
+    std::string accumulated_thinking;
     auto callback = [&](const provider::StreamChunk& chunk) -> bool {
         if (context->IsCancelled()) {
             return false;
@@ -405,6 +419,11 @@ auto LLMServiceImpl::BuildChatRequest(const ::smartbotic::llm::ChatRequest& requ
             accumulated_content += chunk.content_delta;
         }
 
+        if (!chunk.thinking_delta.empty()) {
+            proto_chunk.set_thinking_delta(chunk.thinking_delta);
+            accumulated_thinking += chunk.thinking_delta;
+        }
+
         if (chunk.tool_call.has_value()) {
             *proto_chunk.mutable_tool_call() = ToolCallToProto(*chunk.tool_call);
             accumulated_tool_calls.push_back(*chunk.tool_call);
@@ -435,6 +454,14 @@ auto LLMServiceImpl::BuildChatRequest(const ::smartbotic::llm::ChatRequest& requ
     assistant_message.role = session::MessageRole::kAssistant;
     assistant_message.created_at = session::Now();
 
+    // Add thinking content first (if any) - it comes before the main response
+    if (!accumulated_thinking.empty()) {
+        session::ContentPart thinking_part;
+        thinking_part.type = session::ContentPartType::kThinking;
+        thinking_part.text = accumulated_thinking;
+        assistant_message.content.push_back(std::move(thinking_part));
+    }
+
     if (!accumulated_content.empty()) {
         session::ContentPart text_part;
         text_part.type = session::ContentPartType::kText;
@@ -449,6 +476,13 @@ auto LLMServiceImpl::BuildChatRequest(const ::smartbotic::llm::ChatRequest& requ
         tool_part.tool_name = tc.name;
         tool_part.tool_arguments = tc.arguments;
         assistant_message.content.push_back(std::move(tool_part));
+
+        // Also add to tool_calls for proper OpenAI API format
+        session::ToolCall session_tc;
+        session_tc.id = tc.id;
+        session_tc.name = tc.name;
+        session_tc.arguments = tc.arguments;
+        assistant_message.tool_calls.push_back(std::move(session_tc));
     }
 
     session_store_->AddMessage(session.id, session.user_id, assistant_message);
@@ -537,6 +571,10 @@ auto LLMServiceImpl::BuildChatRequest(const ::smartbotic::llm::ChatRequest& requ
                       : provider::MessageRole::kSystem;
 
         for (const auto& part : msg.content) {
+            // Skip kThinking content - it's for display only
+            if (part.type == session::ContentPartType::kThinking) {
+                continue;
+            }
             provider::ContentPart cp;
             cp.type = part.type == session::ContentPartType::kText ? provider::ContentPart::Type::kText
                     : part.type == session::ContentPartType::kImage ? provider::ContentPart::Type::kImage
@@ -552,9 +590,37 @@ auto LLMServiceImpl::BuildChatRequest(const ::smartbotic::llm::ChatRequest& requ
             chat_msg.content.push_back(std::move(cp));
         }
 
+        // Copy tool_calls for assistant messages (required for OpenAI API format)
+        for (const auto& tc : msg.tool_calls) {
+            provider::ToolCall provider_tc;
+            provider_tc.id = tc.id;
+            provider_tc.name = tc.name;
+            provider_tc.arguments = tc.arguments;
+            chat_msg.tool_calls.push_back(std::move(provider_tc));
+        }
+
         chat_request.messages.push_back(std::move(chat_msg));
     }
 
+    // Add tools from request
+    for (const auto& tool : request->tools()) {
+        provider::ToolDefinition def;
+        def.name = tool.name();
+        def.description = tool.description();
+        def.input_schema = tool.input_schema();
+        chat_request.tools.push_back(std::move(def));
+    }
+
+    // Add tools from registry
+    auto workspace_tools = tool_registry_->GetToolsForWorkspace(session.workspace_id);
+    for (const auto& tool : workspace_tools) {
+        provider::ToolDefinition def;
+        def.name = tool.name;
+        def.description = tool.description;
+        def.input_schema = tool.input_schema;
+        chat_request.tools.push_back(std::move(def));
+    }
+
     auto chat_result = provider->Chat(chat_request);
     if (!chat_result.success) {
         return ::grpc::Status(::grpc::StatusCode::INTERNAL, chat_result.error);
@@ -580,6 +646,13 @@ auto LLMServiceImpl::BuildChatRequest(const ::smartbotic::llm::ChatRequest& requ
         tool_part.tool_name = tc.name;
         tool_part.tool_arguments = tc.arguments;
         assistant_message.content.push_back(std::move(tool_part));
+
+        // Also add to tool_calls for proper OpenAI API format
+        session::ToolCall session_tc;
+        session_tc.id = tc.id;
+        session_tc.name = tc.name;
+        session_tc.arguments = tc.arguments;
+        assistant_message.tool_calls.push_back(std::move(session_tc));
     }
 
     session_store_->AddMessage(session.id, session.user_id, assistant_message);
@@ -610,7 +683,27 @@ auto LLMServiceImpl::BuildChatRequest(const ::smartbotic::llm::ChatRequest& requ
         return status;
     }
 
-    // Send as single chunk with metadata
+    // Send content delta if there's text content
+    if (!response.assistant_message().content().empty()) {
+        for (const auto& part : response.assistant_message().content()) {
+            if (part.has_text() && !part.text().empty()) {
+                ::smartbotic::llm::ChatStreamChunk content_chunk;
+                content_chunk.set_session_id(response.session_id());
+                content_chunk.set_content_delta(part.text());
+                writer->Write(content_chunk);
+            }
+        }
+    }
+
+    // Send each tool_call as a separate chunk (so frontend can process them)
+    for (const auto& tc : response.tool_calls()) {
+        ::smartbotic::llm::ChatStreamChunk tc_chunk;
+        tc_chunk.set_session_id(response.session_id());
+        *tc_chunk.mutable_tool_call() = tc;
+        writer->Write(tc_chunk);
+    }
+
+    // Send final chunk with metadata
     ::smartbotic::llm::ChatStreamChunk chunk;
     chunk.set_session_id(response.session_id());
     auto* metadata = chunk.mutable_metadata();

+ 5 - 0
llm/src/grpc/proto_convert.cpp

@@ -130,6 +130,11 @@ auto ContentPartToProto(const session::ContentPart& part) -> ::smartbotic::llm::
             tool_result->set_is_error(part.is_error);
             break;
         }
+        case session::ContentPartType::kThinking: {
+            auto* thinking = proto.mutable_thinking();
+            thinking->set_content(part.text);
+            break;
+        }
     }
 
     return proto;

+ 174 - 6
llm/src/provider/openai_provider.cpp

@@ -247,6 +247,26 @@ auto OpenAIProvider::BuildChatRequestBody(const ChatRequest& request, bool strea
             }
         }
 
+        // Add tool_calls for assistant messages that have them
+        if (msg.role == MessageRole::kAssistant && !msg.tool_calls.empty()) {
+            nlohmann::json tool_calls_json = nlohmann::json::array();
+            for (const auto& tc : msg.tool_calls) {
+                tool_calls_json.push_back({
+                    {"id", tc.id},
+                    {"type", "function"},
+                    {"function", {
+                        {"name", tc.name},
+                        {"arguments", tc.arguments}
+                    }}
+                });
+            }
+            message["tool_calls"] = tool_calls_json;
+            // OpenAI requires content to be null or empty string for tool call messages
+            if (!message.contains("content") || message["content"].empty()) {
+                message["content"] = nullptr;
+            }
+        }
+
         // Add tool_call_id for tool messages
         if (msg.role == MessageRole::kTool && !msg.content.empty()) {
             for (const auto& part : msg.content) {
@@ -263,6 +283,33 @@ auto OpenAIProvider::BuildChatRequestBody(const ChatRequest& request, bool strea
 
     body["messages"] = messages;
 
+    // Debug: Log the full request being sent
+    spdlog::info("OpenAI API Request - {} messages, {} tools", messages.size(), request.tools.size());
+    for (size_t i = 0; i < messages.size(); ++i) {
+        const auto& m = messages[i];
+        std::string role = m.value("role", "unknown");
+        std::string content_preview;
+        if (m.contains("content")) {
+            if (m["content"].is_string()) {
+                content_preview = m["content"].get<std::string>().substr(0, 100);
+            } else if (m["content"].is_null()) {
+                content_preview = "(null)";
+            } else {
+                content_preview = "(array)";
+            }
+        }
+        bool has_tool_calls = m.contains("tool_calls");
+        bool has_tool_call_id = m.contains("tool_call_id");
+        spdlog::info("  Message[{}]: role={}, content={}, has_tool_calls={}, has_tool_call_id={}",
+                     i, role, content_preview, has_tool_calls, has_tool_call_id);
+        if (has_tool_calls) {
+            spdlog::info("    tool_calls: {}", m["tool_calls"].dump());
+        }
+        if (has_tool_call_id) {
+            spdlog::info("    tool_call_id: {}", m["tool_call_id"].get<std::string>());
+        }
+    }
+
     // Add tools if present
     if (!request.tools.empty()) {
         nlohmann::json tools = nlohmann::json::array();
@@ -382,6 +429,7 @@ auto OpenAIProvider::ParseStreamChunk(const std::string& data) -> std::optional<
 
                 // Parse tool calls delta
                 if (delta.contains("tool_calls") && delta["tool_calls"].is_array()) {
+                    spdlog::info("Found tool_calls in delta: {}", delta["tool_calls"].dump());
                     for (const auto& tc : delta["tool_calls"]) {
                         ToolCall call;
                         call.id = tc.value("id", "");
@@ -398,6 +446,7 @@ auto OpenAIProvider::ParseStreamChunk(const std::string& data) -> std::optional<
             // Parse finish reason
             if (choice.contains("finish_reason") && !choice["finish_reason"].is_null()) {
                 std::string finish_reason = choice["finish_reason"].get<std::string>();
+                spdlog::info("Stream finish_reason: {}", finish_reason);
                 if (finish_reason == "stop") {
                     chunk.finish_reason = FinishReason::kStop;
                 } else if (finish_reason == "length") {
@@ -456,7 +505,17 @@ auto OpenAIProvider::Chat(const ChatRequest& request) -> Result<ChatResponse> {
                 "Chat failed: HTTP " + std::to_string(res->status));
         }
 
-        return ParseChatResponse(res->body);
+        auto result = ParseChatResponse(res->body);
+        if (result.success) {
+            spdlog::info("Chat response: content={}, tool_calls={}, finish_reason={}",
+                         result.value.content.substr(0, 100),
+                         result.value.tool_calls.size(),
+                         static_cast<int>(result.value.finish_reason));
+            for (const auto& tc : result.value.tool_calls) {
+                spdlog::info("  Tool call: id={}, name={}, args={}", tc.id, tc.name, tc.arguments.substr(0, 100));
+            }
+        }
+        return result;
     } catch (const std::exception& e) {
         return Result<ChatResponse>::Error(std::string("Exception: ") + e.what());
     }
@@ -473,11 +532,95 @@ auto OpenAIProvider::ChatStream(const ChatRequest& request, StreamCallback callb
         // For accumulating the full response
         ChatResponse final_response;
         std::string accumulated_content;
+        std::string accumulated_thinking;
         std::unordered_map<int, ToolCall> tool_calls;  // index -> ToolCall
         int last_tool_index = -1;
         std::string raw_response;  // Capture raw response for error handling
         bool is_sse_response = false;
 
+        // State for parsing <think> tags in streaming content
+        bool inside_thinking = false;
+        std::string pending_buffer;  // Buffer for incomplete tag detection
+
+        // Helper to process content and separate thinking from regular content
+        auto process_content_delta = [&](const std::string& delta, StreamChunk& output_chunk) {
+            std::string buffer = pending_buffer + delta;
+            pending_buffer.clear();
+
+            size_t pos = 0;
+            while (pos < buffer.size()) {
+                if (inside_thinking) {
+                    // Look for </think>
+                    size_t end_tag = buffer.find("</think>", pos);
+                    if (end_tag != std::string::npos) {
+                        // Add content up to end tag as thinking
+                        std::string thinking_part = buffer.substr(pos, end_tag - pos);
+                        output_chunk.thinking_delta += thinking_part;
+                        accumulated_thinking += thinking_part;
+                        pos = end_tag + 8;  // Skip "</think>"
+                        inside_thinking = false;
+                    } else {
+                        // Check if we might have a partial </think> tag
+                        size_t potential_tag = buffer.find_last_of('<', buffer.size() - 1);
+                        if (potential_tag != std::string::npos && potential_tag >= pos &&
+                            buffer.size() - potential_tag < 8) {
+                            // Keep potential partial tag in buffer
+                            std::string thinking_part = buffer.substr(pos, potential_tag - pos);
+                            output_chunk.thinking_delta += thinking_part;
+                            accumulated_thinking += thinking_part;
+                            pending_buffer = buffer.substr(potential_tag);
+                            return;
+                        }
+                        // All remaining is thinking content
+                        std::string thinking_part = buffer.substr(pos);
+                        output_chunk.thinking_delta += thinking_part;
+                        accumulated_thinking += thinking_part;
+                        return;
+                    }
+                } else {
+                    // Look for <think> or orphan </think> tags
+                    size_t start_tag = buffer.find("<think>", pos);
+                    size_t orphan_end_tag = buffer.find("</think>", pos);
+
+                    // Determine which comes first
+                    if (start_tag != std::string::npos &&
+                        (orphan_end_tag == std::string::npos || start_tag < orphan_end_tag)) {
+                        // <think> comes first - add content up to it as regular content
+                        std::string content_part = buffer.substr(pos, start_tag - pos);
+                        output_chunk.content_delta += content_part;
+                        accumulated_content += content_part;
+                        pos = start_tag + 7;  // Skip "<think>"
+                        inside_thinking = true;
+                    } else if (orphan_end_tag != std::string::npos) {
+                        // Orphan </think> found (no matching <think> before it)
+                        // Add content up to the orphan tag as regular content, then skip the tag
+                        std::string content_part = buffer.substr(pos, orphan_end_tag - pos);
+                        output_chunk.content_delta += content_part;
+                        accumulated_content += content_part;
+                        pos = orphan_end_tag + 8;  // Skip "</think>"
+                        // Continue to look for more tags
+                    } else {
+                        // Check if we might have a partial <think> or </think> tag
+                        size_t potential_tag = buffer.find_last_of('<', buffer.size() - 1);
+                        if (potential_tag != std::string::npos && potential_tag >= pos &&
+                            buffer.size() - potential_tag < 8) {  // 8 = max tag length "</think>"
+                            // Keep potential partial tag in buffer
+                            std::string content_part = buffer.substr(pos, potential_tag - pos);
+                            output_chunk.content_delta += content_part;
+                            accumulated_content += content_part;
+                            pending_buffer = buffer.substr(potential_tag);
+                            return;
+                        }
+                        // All remaining is regular content
+                        std::string content_part = buffer.substr(pos);
+                        output_chunk.content_delta += content_part;
+                        accumulated_content += content_part;
+                        return;
+                    }
+                }
+            }
+        };
+
         auto content_receiver = [&](const char* data, size_t data_length) -> bool {
             std::string chunk_data(data, data_length);
             raw_response += chunk_data;  // Capture for error handling
@@ -503,6 +646,19 @@ auto OpenAIProvider::ChatStream(const ChatRequest& request, StreamCallback callb
                     auto chunk = ParseStreamChunk(json_data);
                     if (chunk) {
                         if (chunk->is_done) {
+                            // Flush any pending buffer as content
+                            if (!pending_buffer.empty()) {
+                                StreamChunk flush_chunk;
+                                if (inside_thinking) {
+                                    flush_chunk.thinking_delta = pending_buffer;
+                                    accumulated_thinking += pending_buffer;
+                                } else {
+                                    flush_chunk.content_delta = pending_buffer;
+                                    accumulated_content += pending_buffer;
+                                }
+                                pending_buffer.clear();
+                                callback(flush_chunk);
+                            }
                             StreamChunk done_chunk;
                             done_chunk.is_done = true;
                             done_chunk.usage = final_response.usage;
@@ -511,11 +667,15 @@ auto OpenAIProvider::ChatStream(const ChatRequest& request, StreamCallback callb
                             return true;
                         }
 
-                        // Accumulate content
+                        // Process content delta to separate thinking from regular content
+                        StreamChunk processed_chunk;
                         if (!chunk->content_delta.empty()) {
-                            accumulated_content += chunk->content_delta;
+                            process_content_delta(chunk->content_delta, processed_chunk);
                         }
 
+                        // Copy tool call if present
+                        processed_chunk.tool_call = chunk->tool_call;
+
                         // Accumulate tool calls
                         if (chunk->tool_call) {
                             // Tool calls come with an index in the delta
@@ -533,14 +693,22 @@ auto OpenAIProvider::ChatStream(const ChatRequest& request, StreamCallback callb
                         // Update usage and finish reason if present
                         if (chunk->usage) {
                             final_response.usage = *chunk->usage;
+                            processed_chunk.usage = chunk->usage;
                         }
                         if (chunk->finish_reason) {
                             final_response.finish_reason = *chunk->finish_reason;
+                            processed_chunk.finish_reason = chunk->finish_reason;
                         }
 
-                        // Call the callback
-                        if (!callback(*chunk)) {
-                            return false;  // Client wants to stop
+                        // Call the callback with processed chunk (only if there's something to send)
+                        if (!processed_chunk.content_delta.empty() ||
+                            !processed_chunk.thinking_delta.empty() ||
+                            processed_chunk.tool_call ||
+                            processed_chunk.usage ||
+                            processed_chunk.finish_reason) {
+                            if (!callback(processed_chunk)) {
+                                return false;  // Client wants to stop
+                            }
                         }
                     }
                 }

+ 43 - 0
llm/src/session/database_session_store.cpp

@@ -205,6 +205,10 @@ auto DatabaseSessionStore::MessageToValue(const Message& message)
                 SetStringValue(&(*part_map)["text"], part.text);
                 SetBoolValue(&(*part_map)["is_error"], part.is_error);
                 break;
+            case ContentPartType::kThinking:
+                SetStringValue(&(*part_map)["type"], "thinking");
+                SetStringValue(&(*part_map)["text"], part.text);
+                break;
         }
     }
 
@@ -218,6 +222,18 @@ auto DatabaseSessionStore::MessageToValue(const Message& message)
         SetStringValue(&(*ctx_map)["view_id"], message.page_context->view_id);
     }
 
+    // Tool calls (for assistant messages)
+    if (!message.tool_calls.empty()) {
+        auto* tc_array = (*msg_map)["tool_calls"].mutable_array_value();
+        for (const auto& tc : message.tool_calls) {
+            auto* tc_value = tc_array->add_values();
+            auto* tc_map = tc_value->mutable_map_value()->mutable_fields();
+            SetStringValue(&(*tc_map)["id"], tc.id);
+            SetStringValue(&(*tc_map)["name"], tc.name);
+            SetStringValue(&(*tc_map)["arguments"], tc.arguments);
+        }
+    }
+
     return value;
 }
 
@@ -294,6 +310,11 @@ auto DatabaseSessionStore::ValueToMessage(const ::smartbotic::database::Value& v
                 if (part_fields.contains("is_error")) {
                     part.is_error = GetBoolValue(part_fields.at("is_error"));
                 }
+            } else if (type_str == "thinking") {
+                part.type = ContentPartType::kThinking;
+                if (part_fields.contains("text")) {
+                    part.text = GetStringValue(part_fields.at("text"));
+                }
             }
 
             message.content.push_back(std::move(part));
@@ -324,6 +345,28 @@ auto DatabaseSessionStore::ValueToMessage(const ::smartbotic::database::Value& v
         message.page_context = ctx;
     }
 
+    // Parse tool_calls (for assistant messages)
+    if (fields.contains("tool_calls") && fields.at("tool_calls").has_array_value()) {
+        for (const auto& tc_value : fields.at("tool_calls").array_value().values()) {
+            if (!tc_value.has_map_value()) continue;
+
+            ToolCall tc;
+            const auto& tc_fields = tc_value.map_value().fields();
+
+            if (tc_fields.contains("id")) {
+                tc.id = GetStringValue(tc_fields.at("id"));
+            }
+            if (tc_fields.contains("name")) {
+                tc.name = GetStringValue(tc_fields.at("name"));
+            }
+            if (tc_fields.contains("arguments")) {
+                tc.arguments = GetStringValue(tc_fields.at("arguments"));
+            }
+
+            message.tool_calls.push_back(std::move(tc));
+        }
+    }
+
     return message;
 }
 

+ 2 - 0
webserver/include/smartbotic/webserver/tool_service.hpp

@@ -231,6 +231,8 @@ private:
                                             const ToolExecutionContext& ctx) -> ToolExecutionResult;
     [[nodiscard]] auto HandleDeleteDocument(const nlohmann::json& args,
                                              const ToolExecutionContext& ctx) -> ToolExecutionResult;
+    [[nodiscard]] auto HandleQueryRecords(const nlohmann::json& args,
+                                           const ToolExecutionContext& ctx) -> ToolExecutionResult;
 
     AuthorizationService& authorization_service_;
     WorkspaceService& workspace_service_;

+ 24 - 4
webserver/src/http_server.cpp

@@ -5922,12 +5922,15 @@ void HttpServer::HandleLlmGetSession(const httplib::Request& req, httplib::Respo
             m["id"] = msg.id();
             m["role"] = MessageRoleToString(msg.role());
 
-            // Extract text content and tool_calls from message parts
+            // Extract text content, thinking, and tool_calls from message parts
             std::string msg_text;
+            std::string msg_thinking;
             nlohmann::json tool_calls_array = nlohmann::json::array();
             for (const auto& part : msg.content()) {
                 if (part.has_text()) {
                     msg_text += part.text();
+                } else if (part.has_thinking()) {
+                    msg_thinking += part.thinking().content();
                 } else if (part.has_tool_use()) {
                     tool_calls_array.push_back({
                         {"id", part.tool_use().id()},
@@ -5937,6 +5940,9 @@ void HttpServer::HandleLlmGetSession(const httplib::Request& req, httplib::Respo
                 }
             }
             m["content"] = msg_text;
+            if (!msg_thinking.empty()) {
+                m["thinking"] = msg_thinking;
+            }
             if (!tool_calls_array.empty()) {
                 m["tool_calls"] = tool_calls_array;
             }
@@ -6103,14 +6109,20 @@ void HttpServer::HandleLlmSendMessage(const httplib::Request& req, httplib::Resp
         response["session_id"] = chat_response.session_id();
         response["message_id"] = chat_response.assistant_message().id();
 
-        // Extract text content from assistant message
+        // Extract text content and thinking from assistant message
         std::string content_text;
+        std::string thinking_text;
         for (const auto& part : chat_response.assistant_message().content()) {
             if (part.has_text()) {
                 content_text += part.text();
+            } else if (part.has_thinking()) {
+                thinking_text += part.thinking().content();
             }
         }
         response["content"] = content_text;
+        if (!thinking_text.empty()) {
+            response["thinking"] = thinking_text;
+        }
         response["finish_reason"] = static_cast<int>(chat_response.finish_reason());
 
         if (chat_response.tool_calls_size() > 0) {
@@ -6229,13 +6241,15 @@ void HttpServer::HandleLlmStreamMessage(const httplib::Request& req, httplib::Re
         // Add available tools for this user
         if (toolService_) {
             auto tools = toolService_->GetAvailableTools(*auth_user, session.workspace_id());
+            std::string tool_names;
             for (const auto& tool : tools) {
                 auto* tool_def = request.add_tools();
                 tool_def->set_name(tool.name);
                 tool_def->set_description(tool.description);
                 tool_def->set_input_schema(tool.input_schema);
+                tool_names += tool.name + ", ";
             }
-            spdlog::debug("Added {} tools to chat request", tools.size());
+            spdlog::info("Added {} tools to chat request: {}", tools.size(), tool_names);
         }
 
         // Set up SSE response
@@ -6429,14 +6443,20 @@ void HttpServer::HandleLlmCompressSession(const httplib::Request& req, httplib::
             msg_json["id"] = msg.id();
             msg_json["role"] = MessageRoleToString(msg.role());
 
-            // Extract text content
+            // Extract text content and thinking
             std::string content_text;
+            std::string thinking_text;
             for (const auto& part : msg.content()) {
                 if (part.has_text()) {
                     content_text += part.text();
+                } else if (part.has_thinking()) {
+                    thinking_text += part.thinking().content();
                 }
             }
             msg_json["content"] = content_text;
+            if (!thinking_text.empty()) {
+                msg_json["thinking"] = thinking_text;
+            }
             msg_json["created_at"] = msg.created_at().seconds();
             messages_array.push_back(msg_json);
         }

+ 96 - 1
webserver/src/tool_service.cpp

@@ -697,6 +697,34 @@ void ToolService::RegisterDocumentTools() {
             });
     }
 
+    // query_records - alias for list_documents that also supports view_id parameter
+    // Some models expect this tool name format
+    {
+        nlohmann::json props;
+        props["view_id"] = {{"type", "string"}, {"description", "View ID to query records from"}};
+        props["collection"] = {{"type", "string"}, {"description", "Collection name (alternative to view_id)"}};
+        props["filter"] = {{"type", "object"}, {"description", "Filter conditions (field: value pairs)"}};
+        props["sort_field"] = {{"type", "string"}, {"description", "Field to sort by"}};
+        props["sort_ascending"] = {{"type", "boolean"}, {"description", "Sort ascending (default: true)"}};
+        props["limit"] = {{"type", "integer"}, {"description", "Maximum records to return (default: 20)"}};
+
+        RegisterTool(
+            {
+                .name = "query_records",
+                .description = "Query and list records/documents from a view or collection. Use view_id from list_views results.",
+                .input_schema = MakeObjectSchema(props, {}),  // No required fields - view_id OR collection
+                .category = "document",
+                .permissions = {
+                    .any_of = {std::string(permissions::kCollectionReadAll), std::string(permissions::kCollectionReadOwn)},
+                    .all_of = {},
+                    .requires_workspace_context = true
+                }
+            },
+            [this](const nlohmann::json& args, const ToolExecutionContext& ctx) {
+                return HandleQueryRecords(args, ctx);
+            });
+    }
+
     // delete_document
     {
         nlohmann::json props;
@@ -720,7 +748,7 @@ void ToolService::RegisterDocumentTools() {
             });
     }
 
-    spdlog::debug("Registered {} document tools", 5);
+    spdlog::debug("Registered {} document tools", 6);
 }
 
 // =============================================================================
@@ -1556,6 +1584,73 @@ auto ToolService::HandleListDocuments(const nlohmann::json& args,
     return {true, response.dump(), false, std::nullopt};
 }
 
+auto ToolService::HandleQueryRecords(const nlohmann::json& args,
+                                      const ToolExecutionContext& ctx) -> ToolExecutionResult {
+    std::string collection;
+
+    // Support both view_id and collection parameters
+    if (args.contains("view_id") && args["view_id"].is_string()) {
+        std::string view_id = args["view_id"].get<std::string>();
+        // Look up view to get collection name
+        auto view_result = view_service_.GetView(ctx.workspace_id, view_id);
+        if (!view_result.view) {
+            return {false, "View not found: " + view_id, true, std::nullopt};
+        }
+        collection = view_result.view->collection_name;
+    } else if (args.contains("collection") && args["collection"].is_string()) {
+        collection = args["collection"].get<std::string>();
+    } else {
+        return {false, "Either view_id or collection is required", true, std::nullopt};
+    }
+
+    // Build query with the resolved collection
+    DocumentQuery query;
+    query.workspace_id = ctx.workspace_id;
+    query.collection = collection;
+
+    // If user only has read_own, filter to their documents
+    if (!authorization_service_.CanReadAllDocuments(ctx.user, ctx.workspace_id, collection)) {
+        if (!authorization_service_.CanReadCollection(ctx.user, ctx.workspace_id, collection)) {
+            return {false, "Permission denied to read collection: " + collection, true, std::nullopt};
+        }
+        query.owner_filter = ctx.user_id;
+    }
+
+    if (args.contains("filter") && args["filter"].is_object()) {
+        query.filter = args["filter"];
+    }
+    if (args.contains("sort_field") && args["sort_field"].is_string()) {
+        query.sort_field = args["sort_field"].get<std::string>();
+    }
+    if (args.contains("sort_ascending") && args["sort_ascending"].is_boolean()) {
+        query.sort_ascending = args["sort_ascending"].get<bool>();
+    }
+    if (args.contains("limit") && args["limit"].is_number_integer()) {
+        query.limit = args["limit"].get<int32_t>();
+        if (query.limit > 100) query.limit = 100;  // Cap at 100
+    } else {
+        query.limit = 20;  // Default limit
+    }
+
+    auto result = document_service_.ListDocuments(query);
+    if (!result.success) {
+        return {false, result.error, true, std::nullopt};
+    }
+
+    nlohmann::json response = nlohmann::json::array();
+    for (auto& doc : result.documents) {
+        // Filter fields for each document
+        authorization_service_.FilterDocumentFields(ctx.user, ctx.workspace_id, collection, doc.data);
+
+        response.push_back({
+            {"id", doc.id},
+            {"data", doc.data}
+        });
+    }
+
+    return {true, response.dump(), false, std::nullopt};
+}
+
 auto ToolService::HandleDeleteDocument(const nlohmann::json& args,
                                         const ToolExecutionContext& ctx) -> ToolExecutionResult {
     if (!args.contains("collection") || !args["collection"].is_string()) {

+ 14 - 0
webui/src/pages/WorkspaceEdit.tsx

@@ -43,6 +43,20 @@ async function generateSystemPrompt(context) {
   return \`You are a helpful AI assistant for \${workspace.name}.
 You are helping \${user.name} (\${user.email}).
 
+## Tool Usage Rules (CRITICAL)
+You have access to tools for querying and managing data. Follow these rules:
+
+1. NEVER make up or hallucinate data. If you need information about records, documents, or any data stored in the system, you MUST use the appropriate tool to fetch it first.
+
+2. When the user asks about their data:
+   - First use the available tools to find what collections/views exist
+   - Then use the appropriate query tool with the view_id to get actual records
+   - Only present data that was returned by the tools
+
+3. If a tool returns empty results, tell the user honestly. Do not invent placeholder data.
+
+4. Always complete the full tool chain: discover views -> query records -> present results.
+
 Be concise and helpful. Focus on the user's needs.
 \`;
 }