|
|
@@ -247,6 +247,26 @@ auto OpenAIProvider::BuildChatRequestBody(const ChatRequest& request, bool strea
|
|
|
}
|
|
|
}
|
|
|
|
|
|
+ // Add tool_calls for assistant messages that have them
|
|
|
+ if (msg.role == MessageRole::kAssistant && !msg.tool_calls.empty()) {
|
|
|
+ nlohmann::json tool_calls_json = nlohmann::json::array();
|
|
|
+ for (const auto& tc : msg.tool_calls) {
|
|
|
+ tool_calls_json.push_back({
|
|
|
+ {"id", tc.id},
|
|
|
+ {"type", "function"},
|
|
|
+ {"function", {
|
|
|
+ {"name", tc.name},
|
|
|
+ {"arguments", tc.arguments}
|
|
|
+ }}
|
|
|
+ });
|
|
|
+ }
|
|
|
+ message["tool_calls"] = tool_calls_json;
|
|
|
+ // OpenAI requires content to be null or empty string for tool call messages
|
|
|
+ if (!message.contains("content") || message["content"].empty()) {
|
|
|
+ message["content"] = nullptr;
|
|
|
+ }
|
|
|
+ }
|
|
|
+
|
|
|
// Add tool_call_id for tool messages
|
|
|
if (msg.role == MessageRole::kTool && !msg.content.empty()) {
|
|
|
for (const auto& part : msg.content) {
|
|
|
@@ -263,6 +283,33 @@ auto OpenAIProvider::BuildChatRequestBody(const ChatRequest& request, bool strea
|
|
|
|
|
|
body["messages"] = messages;
|
|
|
|
|
|
+ // Debug: Log the full request being sent
|
|
|
+ spdlog::info("OpenAI API Request - {} messages, {} tools", messages.size(), request.tools.size());
|
|
|
+ for (size_t i = 0; i < messages.size(); ++i) {
|
|
|
+ const auto& m = messages[i];
|
|
|
+ std::string role = m.value("role", "unknown");
|
|
|
+ std::string content_preview;
|
|
|
+ if (m.contains("content")) {
|
|
|
+ if (m["content"].is_string()) {
|
|
|
+ content_preview = m["content"].get<std::string>().substr(0, 100);
|
|
|
+ } else if (m["content"].is_null()) {
|
|
|
+ content_preview = "(null)";
|
|
|
+ } else {
|
|
|
+ content_preview = "(array)";
|
|
|
+ }
|
|
|
+ }
|
|
|
+ bool has_tool_calls = m.contains("tool_calls");
|
|
|
+ bool has_tool_call_id = m.contains("tool_call_id");
|
|
|
+ spdlog::info(" Message[{}]: role={}, content={}, has_tool_calls={}, has_tool_call_id={}",
|
|
|
+ i, role, content_preview, has_tool_calls, has_tool_call_id);
|
|
|
+ if (has_tool_calls) {
|
|
|
+ spdlog::info(" tool_calls: {}", m["tool_calls"].dump());
|
|
|
+ }
|
|
|
+ if (has_tool_call_id) {
|
|
|
+ spdlog::info(" tool_call_id: {}", m["tool_call_id"].get<std::string>());
|
|
|
+ }
|
|
|
+ }
|
|
|
+
|
|
|
// Add tools if present
|
|
|
if (!request.tools.empty()) {
|
|
|
nlohmann::json tools = nlohmann::json::array();
|
|
|
@@ -382,6 +429,7 @@ auto OpenAIProvider::ParseStreamChunk(const std::string& data) -> std::optional<
|
|
|
|
|
|
// Parse tool calls delta
|
|
|
if (delta.contains("tool_calls") && delta["tool_calls"].is_array()) {
|
|
|
+ spdlog::info("Found tool_calls in delta: {}", delta["tool_calls"].dump());
|
|
|
for (const auto& tc : delta["tool_calls"]) {
|
|
|
ToolCall call;
|
|
|
call.id = tc.value("id", "");
|
|
|
@@ -398,6 +446,7 @@ auto OpenAIProvider::ParseStreamChunk(const std::string& data) -> std::optional<
|
|
|
// Parse finish reason
|
|
|
if (choice.contains("finish_reason") && !choice["finish_reason"].is_null()) {
|
|
|
std::string finish_reason = choice["finish_reason"].get<std::string>();
|
|
|
+ spdlog::info("Stream finish_reason: {}", finish_reason);
|
|
|
if (finish_reason == "stop") {
|
|
|
chunk.finish_reason = FinishReason::kStop;
|
|
|
} else if (finish_reason == "length") {
|
|
|
@@ -456,7 +505,17 @@ auto OpenAIProvider::Chat(const ChatRequest& request) -> Result<ChatResponse> {
|
|
|
"Chat failed: HTTP " + std::to_string(res->status));
|
|
|
}
|
|
|
|
|
|
- return ParseChatResponse(res->body);
|
|
|
+ auto result = ParseChatResponse(res->body);
|
|
|
+ if (result.success) {
|
|
|
+ spdlog::info("Chat response: content={}, tool_calls={}, finish_reason={}",
|
|
|
+ result.value.content.substr(0, 100),
|
|
|
+ result.value.tool_calls.size(),
|
|
|
+ static_cast<int>(result.value.finish_reason));
|
|
|
+ for (const auto& tc : result.value.tool_calls) {
|
|
|
+ spdlog::info(" Tool call: id={}, name={}, args={}", tc.id, tc.name, tc.arguments.substr(0, 100));
|
|
|
+ }
|
|
|
+ }
|
|
|
+ return result;
|
|
|
} catch (const std::exception& e) {
|
|
|
return Result<ChatResponse>::Error(std::string("Exception: ") + e.what());
|
|
|
}
|
|
|
@@ -473,11 +532,95 @@ auto OpenAIProvider::ChatStream(const ChatRequest& request, StreamCallback callb
|
|
|
// For accumulating the full response
|
|
|
ChatResponse final_response;
|
|
|
std::string accumulated_content;
|
|
|
+ std::string accumulated_thinking;
|
|
|
std::unordered_map<int, ToolCall> tool_calls; // index -> ToolCall
|
|
|
int last_tool_index = -1;
|
|
|
std::string raw_response; // Capture raw response for error handling
|
|
|
bool is_sse_response = false;
|
|
|
|
|
|
+ // State for parsing <think> tags in streaming content
|
|
|
+ bool inside_thinking = false;
|
|
|
+ std::string pending_buffer; // Buffer for incomplete tag detection
|
|
|
+
|
|
|
+ // Helper to process content and separate thinking from regular content
|
|
|
+ auto process_content_delta = [&](const std::string& delta, StreamChunk& output_chunk) {
|
|
|
+ std::string buffer = pending_buffer + delta;
|
|
|
+ pending_buffer.clear();
|
|
|
+
|
|
|
+ size_t pos = 0;
|
|
|
+ while (pos < buffer.size()) {
|
|
|
+ if (inside_thinking) {
|
|
|
+ // Look for </think>
|
|
|
+ size_t end_tag = buffer.find("</think>", pos);
|
|
|
+ if (end_tag != std::string::npos) {
|
|
|
+ // Add content up to end tag as thinking
|
|
|
+ std::string thinking_part = buffer.substr(pos, end_tag - pos);
|
|
|
+ output_chunk.thinking_delta += thinking_part;
|
|
|
+ accumulated_thinking += thinking_part;
|
|
|
+ pos = end_tag + 8; // Skip "</think>"
|
|
|
+ inside_thinking = false;
|
|
|
+ } else {
|
|
|
+ // Check if we might have a partial </think> tag
|
|
|
+ size_t potential_tag = buffer.find_last_of('<', buffer.size() - 1);
|
|
|
+ if (potential_tag != std::string::npos && potential_tag >= pos &&
|
|
|
+ buffer.size() - potential_tag < 8) {
|
|
|
+ // Keep potential partial tag in buffer
|
|
|
+ std::string thinking_part = buffer.substr(pos, potential_tag - pos);
|
|
|
+ output_chunk.thinking_delta += thinking_part;
|
|
|
+ accumulated_thinking += thinking_part;
|
|
|
+ pending_buffer = buffer.substr(potential_tag);
|
|
|
+ return;
|
|
|
+ }
|
|
|
+ // All remaining is thinking content
|
|
|
+ std::string thinking_part = buffer.substr(pos);
|
|
|
+ output_chunk.thinking_delta += thinking_part;
|
|
|
+ accumulated_thinking += thinking_part;
|
|
|
+ return;
|
|
|
+ }
|
|
|
+ } else {
|
|
|
+ // Look for <think> or orphan </think> tags
|
|
|
+ size_t start_tag = buffer.find("<think>", pos);
|
|
|
+ size_t orphan_end_tag = buffer.find("</think>", pos);
|
|
|
+
|
|
|
+ // Determine which comes first
|
|
|
+ if (start_tag != std::string::npos &&
|
|
|
+ (orphan_end_tag == std::string::npos || start_tag < orphan_end_tag)) {
|
|
|
+ // <think> comes first - add content up to it as regular content
|
|
|
+ std::string content_part = buffer.substr(pos, start_tag - pos);
|
|
|
+ output_chunk.content_delta += content_part;
|
|
|
+ accumulated_content += content_part;
|
|
|
+ pos = start_tag + 7; // Skip "<think>"
|
|
|
+ inside_thinking = true;
|
|
|
+ } else if (orphan_end_tag != std::string::npos) {
|
|
|
+ // Orphan </think> found (no matching <think> before it)
|
|
|
+ // Add content up to the orphan tag as regular content, then skip the tag
|
|
|
+ std::string content_part = buffer.substr(pos, orphan_end_tag - pos);
|
|
|
+ output_chunk.content_delta += content_part;
|
|
|
+ accumulated_content += content_part;
|
|
|
+ pos = orphan_end_tag + 8; // Skip "</think>"
|
|
|
+ // Continue to look for more tags
|
|
|
+ } else {
|
|
|
+ // Check if we might have a partial <think> or </think> tag
|
|
|
+ size_t potential_tag = buffer.find_last_of('<', buffer.size() - 1);
|
|
|
+ if (potential_tag != std::string::npos && potential_tag >= pos &&
|
|
|
+ buffer.size() - potential_tag < 8) { // 8 = max tag length "</think>"
|
|
|
+ // Keep potential partial tag in buffer
|
|
|
+ std::string content_part = buffer.substr(pos, potential_tag - pos);
|
|
|
+ output_chunk.content_delta += content_part;
|
|
|
+ accumulated_content += content_part;
|
|
|
+ pending_buffer = buffer.substr(potential_tag);
|
|
|
+ return;
|
|
|
+ }
|
|
|
+ // All remaining is regular content
|
|
|
+ std::string content_part = buffer.substr(pos);
|
|
|
+ output_chunk.content_delta += content_part;
|
|
|
+ accumulated_content += content_part;
|
|
|
+ return;
|
|
|
+ }
|
|
|
+ }
|
|
|
+ }
|
|
|
+ };
|
|
|
+
|
|
|
auto content_receiver = [&](const char* data, size_t data_length) -> bool {
|
|
|
std::string chunk_data(data, data_length);
|
|
|
raw_response += chunk_data; // Capture for error handling
|
|
|
@@ -503,6 +646,19 @@ auto OpenAIProvider::ChatStream(const ChatRequest& request, StreamCallback callb
|
|
|
auto chunk = ParseStreamChunk(json_data);
|
|
|
if (chunk) {
|
|
|
if (chunk->is_done) {
|
|
|
+ // Flush any pending buffer as content
|
|
|
+ if (!pending_buffer.empty()) {
|
|
|
+ StreamChunk flush_chunk;
|
|
|
+ if (inside_thinking) {
|
|
|
+ flush_chunk.thinking_delta = pending_buffer;
|
|
|
+ accumulated_thinking += pending_buffer;
|
|
|
+ } else {
|
|
|
+ flush_chunk.content_delta = pending_buffer;
|
|
|
+ accumulated_content += pending_buffer;
|
|
|
+ }
|
|
|
+ pending_buffer.clear();
|
|
|
+ callback(flush_chunk);
|
|
|
+ }
|
|
|
StreamChunk done_chunk;
|
|
|
done_chunk.is_done = true;
|
|
|
done_chunk.usage = final_response.usage;
|
|
|
@@ -511,11 +667,15 @@ auto OpenAIProvider::ChatStream(const ChatRequest& request, StreamCallback callb
|
|
|
return true;
|
|
|
}
|
|
|
|
|
|
- // Accumulate content
|
|
|
+ // Process content delta to separate thinking from regular content
|
|
|
+ StreamChunk processed_chunk;
|
|
|
if (!chunk->content_delta.empty()) {
|
|
|
- accumulated_content += chunk->content_delta;
|
|
|
+ process_content_delta(chunk->content_delta, processed_chunk);
|
|
|
}
|
|
|
|
|
|
+ // Copy tool call if present
|
|
|
+ processed_chunk.tool_call = chunk->tool_call;
|
|
|
+
|
|
|
// Accumulate tool calls
|
|
|
if (chunk->tool_call) {
|
|
|
// Tool calls come with an index in the delta
|
|
|
@@ -533,14 +693,22 @@ auto OpenAIProvider::ChatStream(const ChatRequest& request, StreamCallback callb
|
|
|
// Update usage and finish reason if present
|
|
|
if (chunk->usage) {
|
|
|
final_response.usage = *chunk->usage;
|
|
|
+ processed_chunk.usage = chunk->usage;
|
|
|
}
|
|
|
if (chunk->finish_reason) {
|
|
|
final_response.finish_reason = *chunk->finish_reason;
|
|
|
+ processed_chunk.finish_reason = chunk->finish_reason;
|
|
|
}
|
|
|
|
|
|
- // Call the callback
|
|
|
- if (!callback(*chunk)) {
|
|
|
- return false; // Client wants to stop
|
|
|
+ // Call the callback with processed chunk (only if there's something to send)
|
|
|
+ if (!processed_chunk.content_delta.empty() ||
|
|
|
+ !processed_chunk.thinking_delta.empty() ||
|
|
|
+ processed_chunk.tool_call ||
|
|
|
+ processed_chunk.usage ||
|
|
|
+ processed_chunk.finish_reason) {
|
|
|
+ if (!callback(processed_chunk)) {
|
|
|
+ return false; // Client wants to stop
|
|
|
+ }
|
|
|
}
|
|
|
}
|
|
|
}
|