How to load a multimodal projector in ailia LLM and input images and audio. There are two methods: ailiaLLMSetMultimodalPrompt, which takes structs, and ailiaLLMSetPromptJson, which takes a JSON history.
After opening the model, load the corresponding projector with ailiaLLMOpenMultimodalProjectorFileA.
Image and audio support can be checked with vision_support / audio_support of ailiaLLMGetMultimodalCapabilities. Specifying images or audio with a model or projector without media support, or before the projector is loaded, is INVALID_STATE.
ailiaLLMSetMultimodalPrompt | ailiaLLMSetPromptJson | |
|---|---|---|
| Messages | Array of AILIALLMMultimodalChatMessage | JSON messages array (UTF-8) |
| Media position | <__media__> placeholder in content | Order in the content array |
| Input formats | File path, encoded buffer, raw RGB | File path, Base64 (encoded file) |
| Response | Concatenate ailiaLLMGetDeltaText | Retrieve with ailiaLLMGetResponseJson |
| Tool Use | INVALID_STATE while tools are set | Available whether or not tools are set |
Either method can be used to input images and audio. Use ailiaLLMSetMultimodalPrompt to pass raw RGB, and ailiaLLMSetPromptJson to combine media with Tool Use.
See the Tool Use Guide for details on Tool Use.
Include <__media__> placeholders in the content of a message, set the corresponding media in order in media_data, and set their number in media_count.
The content of the messages is copied internally, so it can be freed after the call.
Set media_type of AILIALLMMediaData to "image" or "audio". Both images and audio can be used. Media is given in one of the following ways.
file_path (UTF-8). Ignored when data is set.data and data_size, with width / height set to 0. Images can be JPG, PNG, TGA, BMP, PSD, GIF, HDR or PIC, and audio can be WAV, MP3, FLAC and similar formats.width * height * 3 bytes of RGB data in data, and set width / height. Images only.Specify images and audio in order inside the content array of a user message, alongside text entries.
[{"role":"user","content":[
{"type":"text","text":"Describe this image and audio."},
{"type":"image","file_path":"/path/to/image.jpg"},
{"type":"audio","file_path":"/path/to/audio.wav"}
]}]
An image or audio entry must specify exactly one of file_path or data. data is the file content encoded in
standard Base64 (with padding). JPEG/PNG, WAV/MP3/FLAC and similar formats can be used. Fetching from a URL, video, and raw RGB/PCM are not supported.
A history that contains images or audio is re-evaluated every time.
Opens the model and the projector, passes an image with structs, and generates a response. The response is built by concatenating GetDeltaText.
#include "ailia_llm.h"
#include <cstdio>
#include <string>
#include <vector>
static int open_projector(AILIALLM* llm) {
int status = ailiaLLMOpenMultimodalProjectorFileA(llm, "/path/to/mmproj.gguf");
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
unsigned int vision = 0, audio = 0;
status = ailiaLLMGetMultimodalCapabilities(llm, &vision, &audio);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
// Check image and audio support with vision_support / audio_support.
return vision ? AILIA_LLM_STATUS_SUCCESS : AILIA_LLM_STATUS_INVALID_STATE;
}
static int describe_image(AILIALLM* llm) {
AILIALLMMediaData media = {};
media.media_type = "image";
media.file_path = "/path/to/image.jpg"; // a buffer can also be given with data and data_size
AILIALLMMultimodalChatMessage message = {};
message.role = "user";
message.content = "Describe this image. <__media__>";
message.media_data = &media;
message.media_count = 1;
int status = ailiaLLMSetMultimodalPrompt(llm, &message, 1);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
std::string text;
unsigned int done = 0;
while (!done) {
status = ailiaLLMGenerate(llm, &done);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
if (done) break;
unsigned int size = 0;
status = ailiaLLMGetDeltaTextSize(llm, &size);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
std::vector<char> delta(size);
status = ailiaLLMGetDeltaText(llm, delta.data(), size);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
text += delta.data();
}
printf("%s\n", text.c_str());
return AILIA_LLM_STATUS_SUCCESS;
}
int main() {
AILIALLM* llm = nullptr;
if (ailiaLLMCreate(&llm) != AILIA_LLM_STATUS_SUCCESS) return 1;
int status = ailiaLLMOpenModelFileA(llm, "gemma-4-E2B-it-Q4_K_M.gguf", 4096);
if (status == AILIA_LLM_STATUS_SUCCESS) status = open_projector(llm);
if (status == AILIA_LLM_STATUS_SUCCESS) status = describe_image(llm);
if (status != AILIA_LLM_STATUS_SUCCESS) {
fprintf(stderr, "error %d: %s\n", status, ailiaLLMGetErrorDetail(llm));
}
ailiaLLMDestroy(llm);
return status == AILIA_LLM_STATUS_SUCCESS ? 0 : 1;
}
The same process with a JSON history. The history is built with nlohmann/json.
#include "ailia_llm.h"
#include <nlohmann/json.hpp>
#include <cstdio>
#include <string>
#include <vector>
using json = nlohmann::json;
static int open_projector(AILIALLM* llm) {
int status = ailiaLLMOpenMultimodalProjectorFileA(llm, "/path/to/mmproj.gguf");
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
unsigned int vision = 0, audio = 0;
status = ailiaLLMGetMultimodalCapabilities(llm, &vision, &audio);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
// Check image and audio support with vision_support / audio_support.
return vision ? AILIA_LLM_STATUS_SUCCESS : AILIA_LLM_STATUS_INVALID_STATE;
}
static int describe_image_json(AILIALLM* llm) {
json messages = json::array({{{"role", "user"}, {"content", json::array({
{{"type", "text"}, {"text", "Describe this image."}},
{{"type", "image"}, {"file_path", "/path/to/image.jpg"}}
})}}});
std::string input = messages.dump();
int status = ailiaLLMSetPromptJson(llm, input.c_str());
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
unsigned int done = 0;
while (!done) {
status = ailiaLLMGenerate(llm, &done);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
}
unsigned int size = 0;
status = ailiaLLMGetResponseJsonSize(llm, &size);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
std::vector<char> output(size);
status = ailiaLLMGetResponseJson(llm, output.data(), size);
if (status != AILIA_LLM_STATUS_SUCCESS) return status;
try {
json response = json::parse(output.data());
printf("%s\n", response.at("content").get<std::string>().c_str());
} catch (const json::exception&) {
return AILIA_LLM_STATUS_INVALID_ARGUMENT;
}
return AILIA_LLM_STATUS_SUCCESS;
}
int main() {
AILIALLM* llm = nullptr;
if (ailiaLLMCreate(&llm) != AILIA_LLM_STATUS_SUCCESS) return 1;
int status = ailiaLLMOpenModelFileA(llm, "gemma-4-E2B-it-Q4_K_M.gguf", 4096);
if (status == AILIA_LLM_STATUS_SUCCESS) status = open_projector(llm);
if (status == AILIA_LLM_STATUS_SUCCESS) status = describe_image_json(llm);
if (status != AILIA_LLM_STATUS_SUCCESS) {
fprintf(stderr, "error %d: %s\n", status, ailiaLLMGetErrorDetail(llm));
}
ailiaLLMDestroy(llm);
return status == AILIA_LLM_STATUS_SUCCESS ? 0 : 1;
}