|
| 1 | +#include "model.h" |
| 2 | +#include "test_utils.h" |
| 3 | +#include <fstream> |
| 4 | +#include <iostream> |
| 5 | +#include <sstream> |
| 6 | +#include <stdexcept> |
| 7 | +#include <string> |
| 8 | + |
| 9 | +#ifndef LLAMA_TEMPLATES_DIR |
| 10 | +#error "LLAMA_TEMPLATES_DIR must be defined" |
| 11 | +#endif |
| 12 | + |
| 13 | +namespace { |
| 14 | + |
| 15 | +// Progress markers on stderr, unbuffered: if a test hangs (as this one did on |
| 16 | +// Windows) the ctest timeout output shows exactly which phase it stalled in. |
| 17 | +void |
| 18 | +trace(const std::string& phase) |
| 19 | +{ |
| 20 | + std::cerr << "[trace] " << phase << std::endl; |
| 21 | +} |
| 22 | + |
| 23 | +std::string |
| 24 | +read_file(const std::string& path) |
| 25 | +{ |
| 26 | + std::ifstream file(path); |
| 27 | + if (!file) { |
| 28 | + throw std::runtime_error("failed to open " + path); |
| 29 | + } |
| 30 | + std::ostringstream buffer; |
| 31 | + buffer << file.rdbuf(); |
| 32 | + return buffer.str(); |
| 33 | +} |
| 34 | + |
| 35 | +// Mirrors how Model::generate() renders a turn, so the parser under test is |
| 36 | +// fed the same common_chat_params the real code path produces. |
| 37 | +common_chat_params |
| 38 | +apply_template(const std::string& template_name, bool with_tools) |
| 39 | +{ |
| 40 | + const std::string template_path = |
| 41 | + std::string(LLAMA_TEMPLATES_DIR) + "/" + template_name; |
| 42 | + |
| 43 | + trace("reading " + template_path); |
| 44 | + const std::string template_source = read_file(template_path); |
| 45 | + |
| 46 | + trace("initializing templates"); |
| 47 | + auto tmpls = common_chat_templates_init(nullptr, template_source); |
| 48 | + |
| 49 | + common_chat_msg user_msg; |
| 50 | + user_msg.role = "user"; |
| 51 | + user_msg.content = "Calculate 3 + 4"; |
| 52 | + |
| 53 | + common_chat_templates_inputs inputs; |
| 54 | + inputs.messages = { user_msg }; |
| 55 | + if (with_tools) { |
| 56 | + common_chat_tool calculator; |
| 57 | + calculator.name = "calculator"; |
| 58 | + calculator.description = "Performs arithmetic"; |
| 59 | + calculator.parameters = |
| 60 | + R"({"type":"object","properties":{"a":{"type":"number"},)" |
| 61 | + R"("b":{"type":"number"},"operation":{"type":"string"}},)" |
| 62 | + R"("required":["a","b","operation"]})"; |
| 63 | + inputs.tools = { calculator }; |
| 64 | + } |
| 65 | + inputs.tool_choice = COMMON_CHAT_TOOL_CHOICE_AUTO; |
| 66 | + inputs.add_generation_prompt = true; |
| 67 | + inputs.enable_thinking = false; |
| 68 | + |
| 69 | + trace("applying template"); |
| 70 | + auto params = common_chat_templates_apply(tmpls.get(), inputs); |
| 71 | + trace("template applied"); |
| 72 | + |
| 73 | + return params; |
| 74 | +} |
| 75 | + |
| 76 | +} |
| 77 | + |
| 78 | +// Regression test for tool calls being returned as raw text instead of being |
| 79 | +// parsed. llama.cpp ggml-org/llama.cpp#18675 moved parsing to a PEG parser |
| 80 | +// derived from the chat template; parse_response() must forward both the |
| 81 | +// derived parser and the generation prompt it was built against, otherwise |
| 82 | +// common_chat_parse() falls back to a pure-content parser. |
| 83 | +TEST(test_tool_call_is_parsed_from_response) |
| 84 | +{ |
| 85 | + auto params = apply_template("ibm-granite-granite-4.0.jinja", true); |
| 86 | + ASSERT_TRUE(!params.parser.empty()); |
| 87 | + |
| 88 | + const std::string response = |
| 89 | + "<tool_call>\n" |
| 90 | + R"({"name": "calculator", "arguments": {"a": 3, "b": 4, "operation": "add"}})" |
| 91 | + "\n</tool_call>"; |
| 92 | + |
| 93 | + auto parsed = agent_cpp::parse_response(params, response); |
| 94 | + |
| 95 | + ASSERT_EQ(parsed.role, std::string("assistant")); |
| 96 | + ASSERT_EQ(parsed.tool_calls.size(), static_cast<size_t>(1)); |
| 97 | + ASSERT_EQ(parsed.tool_calls[0].name, std::string("calculator")); |
| 98 | + ASSERT_TRUE(parsed.tool_calls[0].arguments.find("\"operation\"") != |
| 99 | + std::string::npos); |
| 100 | + ASSERT_TRUE(parsed.content.empty()); |
| 101 | +} |
| 102 | + |
| 103 | +// A plain answer must still come back as content, with no spurious tool calls. |
| 104 | +TEST(test_plain_response_is_parsed_as_content) |
| 105 | +{ |
| 106 | + auto params = apply_template("ibm-granite-granite-4.0.jinja", true); |
| 107 | + |
| 108 | + auto parsed = |
| 109 | + agent_cpp::parse_response(params, "The result of 3 + 4 is 7."); |
| 110 | + |
| 111 | + ASSERT_TRUE(parsed.tool_calls.empty()); |
| 112 | + ASSERT_EQ(parsed.content, std::string("The result of 3 + 4 is 7.")); |
| 113 | +} |
| 114 | + |
| 115 | +// Templates rendered without tools still parse ordinary content. |
| 116 | +TEST(test_response_without_tools_is_parsed_as_content) |
| 117 | +{ |
| 118 | + auto params = apply_template("ibm-granite-granite-4.0.jinja", false); |
| 119 | + |
| 120 | + auto parsed = agent_cpp::parse_response(params, "Hello!"); |
| 121 | + |
| 122 | + ASSERT_TRUE(parsed.tool_calls.empty()); |
| 123 | + ASSERT_EQ(parsed.content, std::string("Hello!")); |
| 124 | +} |
| 125 | + |
| 126 | +int |
| 127 | +main() |
| 128 | +{ |
| 129 | + std::cout << "\n=== Running Chat Parser Unit Tests ===\n" << std::endl; |
| 130 | + |
| 131 | + try { |
| 132 | + RUN_TEST(test_tool_call_is_parsed_from_response); |
| 133 | + RUN_TEST(test_plain_response_is_parsed_as_content); |
| 134 | + RUN_TEST(test_response_without_tools_is_parsed_as_content); |
| 135 | + |
| 136 | + std::cout << "\n=== All tests passed! ✓ ===\n" << std::endl; |
| 137 | + return 0; |
| 138 | + } catch (const std::exception& e) { |
| 139 | + std::cerr << "\n✗ TEST FAILED: " << e.what() << std::endl; |
| 140 | + return 1; |
| 141 | + } |
| 142 | +} |
0 commit comments