How to use Gemma 4 function calling with the Tool Use API of ailia LLM 1.5. This guide covers the JSON history, returning results with tool_call_id, image and audio input, and retrieving the response with GetResponseJson.
With tool use, ailiaLLMSetPromptJson is used from the very first user message.
While tools are set with ailiaLLMSetTools, the conventional ailiaLLMSetPrompt and ailiaLLMSetMultimodalPrompt return
AILIA_LLM_STATUS_INVALID_STATE. They become available again once the tools are cleared.
ailiaLLMSetTools.ailiaLLMSetPromptJson.ailiaLLMGenerate until it completes. Deltas can be previewed with ailiaLLMGetDeltaText, but retrieving and concatenating them is not required.ailiaLLMGetResponseJsonSize / ailiaLLMGetResponseJson.The SDK buffers the generated response, so the application does not have to carry the raw output around. GetResponseJson is a read and does not clear the buffer. It is reset by a new prompt.
[
{"role":"user","content":"What is the weather in Tokyo?"},
{"role":"assistant","content":"","tool_calls":[
{"id":"call_0","type":"function","function":{
"name":"get_weather","arguments":"{\"city\":\"Tokyo\"}"}}
]},
{"role":"tool","tool_call_id":"call_0","content":"Snow, -3 C"}
]
The content itself is never re-parsed as tool syntax. Assistant thinking is given in reasoning_content.
arguments is a string that holds a JSON object. The id and name of a call cannot be empty, and IDs are unique within an assistant message.
IDs may be reused in a different turn. Consecutive tool results are matched to the calls of the immediately preceding assistant message by ID.
The order of the results does not matter. An unknown ID, a duplicated result or a conflicting tool name is INVALID_ARGUMENT.
The content of a tool result must be a string; convert objects into a JSON string.
After clearing the tools with setTools(null), past tool_calls and tool results can still be passed to setPromptJson to generate a summary or a final answer. ID matching is still validated. Clearing the tools disables the output parser, so retrieve the response JSON before clearing them.
After loading the corresponding projector with ailiaLLMOpenMultimodalProjectorFileA, specify the media in order inside the content array of a user message.
Media can be given in JSON whether or not tools are set.
[{"role":"user","content":[
{"type":"text","text":"Describe this image and audio."},
{"type":"image","file_path":"/path/to/image.jpg"},
{"type":"audio","file_path":"/path/to/audio.wav"}
]}]
An image or audio entry must specify exactly one of file_path or data. data is the file content encoded in
standard Base64 (with padding). JPEG/PNG, WAV/MP3/FLAC and similar formats can be used. Fetching from a URL, video, and raw RGB/PCM are not supported.
A history that contains images or audio is re-evaluated every time. A model or projector without media support, or a projector that has not been loaded, is INVALID_STATE.
Image and audio support can be checked with ailiaLLMGetMultimodalCapabilities.
The full flow: opening the model, setting the tools with ailiaLLMSetTools, generating, returning tool results, and clearing the tools. The history is edited with nlohmann/json.
#include "ailia_llm.h"
#include <nlohmann/json.hpp>
#include <cstdio>
#include <string>
#include <vector>
using json = nlohmann::json;
// Runs one turn: SetPromptJson → Generate → GetResponseJson.
static int generate_turn(AILIALLM* llm, const json& messages, json& response) {
std::string input = messages.dump();
int status = ailiaLLMSetPromptJson(llm, input.c_str());
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
unsigned int done = 0;
while (!done) {
status = ailiaLLMGenerate(llm, &done);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
// Preview with GetDeltaText only if needed. No concatenation for the history.
}
unsigned int size = 0;
status = ailiaLLMGetResponseJsonSize(llm, &size);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
std::vector<char> output(size);
status = ailiaLLMGetResponseJson(llm, output.data(), size);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
response = json::parse(output.data());
return AILIA_LLM_STATUS_SUCCESS;
}
static int run_tool_use(AILIALLM* llm) {
const char* tools = R"([{"type":"function","function":{
"name":"get_weather","description":"Get the current weather.",
"parameters":{"type":"object","properties":{"city":{"type":"string"}},
"required":["city"]}}}])";
int status = ailiaLLMSetTools(llm, tools);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
json messages = json::array({{{"role", "user"}, {"content", "What is the weather in Tokyo?"}}});
for (int turn = 0; turn < 8; turn++) {
json response;
status = generate_turn(llm, messages, response);
if (status == AILIA_LLM_STATUS_PARSE_ERROR) {
// Keep the existing history; never append the failed assistant or a tool result.
messages.push_back({{"role", "user"},
{"content", "Please retry with a complete response or tool call."}});
continue;
}
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
messages.push_back(response);
if (!response.contains("tool_calls") || response["tool_calls"].empty()) {
if (response["content"].is_string()) {
printf("%s\n", response["content"].get<std::string>().c_str());
}
break;
}
for (const auto& call : response["tool_calls"]) {
const auto& function = call["function"];
if (function["name"] != "get_weather") return AILIA_LLM_STATUS_INVALID_ARGUMENT;
json args = json::parse(function["arguments"].get<std::string>());
// Validate the arguments and execute the tool. Mock result here.
json result = {{"city", args["city"]}, {"weather", "sunny"}};
messages.push_back({{"role", "tool"}, {"tool_call_id", call["id"]},
{"content", result.dump()}}); // The result is a JSON string
}
}
return AILIA_LLM_STATUS_SUCCESS;
}
int main() {
AILIALLM* llm = nullptr;
if (ailiaLLMCreate(&llm) != AILIA_LLM_STATUS_SUCCESS) return 1;
int status = ailiaLLMOpenModelFileA(llm, "gemma-4-E2B-it-Q4_K_M.gguf", 4096);
if (status == AILIA_LLM_STATUS_SUCCESS) {
status = run_tool_use(llm);
ailiaLLMSetTools(llm, nullptr); // Clear the tools even on error.
}
if (status != AILIA_LLM_STATUS_SUCCESS) {
fprintf(stderr, "error %d: %s\n", status, ailiaLLMGetErrorDetail(llm));
}
ailiaLLMDestroy(llm);
return status == AILIA_LLM_STATUS_SUCCESS ? 0 : 1;
}
| Language | Set the JSON history and generate | Get the response buffered in the SDK |
|---|---|---|
| C / C++ | ailiaLLMSetPromptJson → ailiaLLMGenerate |
ailiaLLMGetResponseJsonSize / ailiaLLMGetResponseJson |
| Python | generate_json(messages) |
get_response_json() (dict) |
| Unity | SetPromptJson(messagesJson) → Generate |
GetResponseJson() (JSON string, empty string on failure) |
| Flutter | setPromptJson(messages) → generate |
getResponseJson() (Map) |
| Kotlin | setPromptJson(messagesJson) → generate |
getResponseJson() (JSON string) |
Python, Flutter and Kotlin raise exceptions on error. SetPromptJson in Unity returns a bool. For tool use, use the JSON history from the very first turn.
Pass an AiliaLLM instance with the model already loaded. The sample uses org.json from Android. On JVM environments other than Android, add an equivalent JSON library.
import axip.ailia_llm.AiliaLLM
import org.json.JSONArray
import org.json.JSONObject
// llm has the model loaded. JSON is handled with org.json from Android.
fun runToolUse(llm: AiliaLLM) {
try {
llm.setTools("""[{"type":"function","function":{
"name":"get_weather","description":"Get the current weather.",
"parameters":{"type":"object","properties":{"city":{"type":"string"}},
"required":["city"]}
}}]""")
val messages = JSONArray().put(
JSONObject().put("role", "user").put("content", "What is the weather in Tokyo?")
)
for (turn in 0 until 8) {
llm.setPromptJson(messages.toString())
while (!llm.generate()) {
print(llm.getDeltaText()) // Optional preview. No concatenation needed.
}
// Throws on incomplete output. Decide on a retry without executing any tool.
val response = JSONObject(llm.getResponseJson())
messages.put(response)
val calls = response.optJSONArray("tool_calls")
if (calls == null || calls.length() == 0) {
println(response.optString("content"))
break
}
for (i in 0 until calls.length()) {
val call = calls.getJSONObject(i)
val function = call.getJSONObject("function")
require(function.getString("name") == "get_weather")
val args = JSONObject(function.getString("arguments"))
val city = args.getString("city")
// Validate the arguments and execute the tool. Mock result here.
val result = JSONObject().put("city", city).put("weather", "sunny")
messages.put(JSONObject()
.put("role", "tool")
.put("tool_call_id", call.getString("id"))
.put("content", result.toString())) // The result is a JSON string
}
}
} finally {
llm.setTools(null)
}
}
Pass an AiliaLLMModel instance with the model already loaded. The content of a tool result must be converted into a JSON string instead of being left as a Map.
import 'dart:convert';
import 'dart:io';
import 'package:ailia_llm/ailia_llm_model.dart';
// llm has the model loaded.
void runToolUse(AiliaLLMModel llm) {
try {
llm.setTools([{
'type': 'function',
'function': {
'name': 'get_weather',
'description': 'Get the current weather.',
'parameters': {
'type': 'object',
'properties': {'city': {'type': 'string'}},
'required': ['city'],
},
},
}]);
final messages = <Map<String, dynamic>>[
{'role': 'user', 'content': 'What is the weather in Tokyo?'},
];
for (var turn = 0; turn < 8; turn++) {
llm.setPromptJson(messages);
String? delta;
while ((delta = llm.generate()) != null) {
stdout.write(delta); // Optional preview. No concatenation needed.
}
// Throws on incomplete output. Decide on a retry without executing any tool.
final response = llm.getResponseJson();
messages.add(response);
final calls = response['tool_calls'] as List? ?? [];
if (calls.isEmpty) {
print(response['content']);
break;
}
for (final call in calls) {
final function = call['function'];
if (function['name'] != 'get_weather') {
throw StateError('Unknown tool: ${function['name']}');
}
final args = jsonDecode(function['arguments'] as String);
final city = args['city'] as String;
// Validate the arguments and execute the tool. Mock result here.
final result = {'city': city, 'weather': 'sunny'};
messages.add({
'role': 'tool',
'tool_call_id': call['id'],
'content': jsonEncode(result), // The result is a JSON string
});
}
}
} finally {
llm.setTools(null);
}
}
Only PARSE_ERROR is retried. The failed assistant message is not appended to the history; a user message asking for a retry is appended to the existing history instead. Every sample clears the tools in a finally block, including on exceptions.
# Tool Use with SDK-buffered responses. Deltas are optional previews.
import json
import ailia_llm
model = ailia_llm.AiliaLLM()
model.open("gemma-4-E2B-it-Q4_K_M.gguf", n_ctx=4096)
try:
model.set_tools([{"type":"function","function":{
"name":"get_weather", "description":"Get the current weather.",
"parameters":{"type":"object","properties":{"city":{"type":"string"}},"required":["city"]}
}}])
messages = [{"role":"user","content":"What is the weather in Tokyo?"}]
for turn in range(8):
for delta in model.generate_json(messages):
print(delta, end="", flush=True) # optional preview; no accumulation
try:
response = model.get_response_json()
except ailia_llm.AiliaLLMError as error:
if error.code != ailia_llm.AILIA_LLM_STATUS_PARSE_ERROR:
raise
# Keep existing history; never append the failed assistant or a tool result.
messages.append({"role": "user", "content": "Please retry with a complete response or tool call."})
continue
messages.append(response)
if not response.get("tool_calls"):
print(response["content"])
break
for call in response["tool_calls"]:
args = json.loads(call["function"]["arguments"])
# Validate and execute the tool in the application. Mock result here.
result = {"city": args["city"], "weather": "sunny"}
messages.append({"role":"tool", "tool_call_id":call["id"],
"content":json.dumps(result)})
finally:
model.set_tools(None)