Specifications
| Property | Value |
|---|---|
| Parameters | 587M (381M shared trunk + 94M vision encoder + 112M audio encoder) |
| Context Length | 16K tokens |
| Architecture | LFM2.5-Encoder (Bidirectional) — 350M backbone + SigLIP2 vision + FastConformer audio |
Multimodal
Text + images or text + audio (up to 30 s)
Zero Output Tokens
Answers read from the distribution, no generation
Edge-Sized
587M parameters for on-device deployment
Quick Start
- Transformers
- llama.cpp
Install:Text input:Image input:Audio input (16 kHz mono, 0.5–30 s):
pip install "transformers>=5.15" torch torchvision pillow soundfile
from transformers import AutoModel
model_id = "LiquidAI/d1-omni-600M"
model = AutoModel.from_pretrained(model_id, trust_remote_code=True, dtype="float16", device_map="auto")
state = "I was charged twice this month, please refund one of them."
questions = {
"refund": {"type": "noul", "instructions": "Is the customer asking for a refund?"},
"team": {"type": "choice", "instructions": "Which team should handle this?",
"criteria": {"billing": "Charges, refunds, invoices", "technical": "App or site faults",
"fraud": "Suspected unauthorised use"}},
"urgency": {"type": "score", "instructions": "How urgent is this?",
"criteria": ["Can wait", "Today", "Blocking the customer now"]},
}
output = model.system_one(state, questions)
print(output)
import io
import urllib.request
from PIL import Image
from transformers import AutoModel
model_id = "LiquidAI/d1-omni-600M"
model = AutoModel.from_pretrained(model_id, trust_remote_code=True, dtype="float16", device_map="auto")
url = "http://images.cocodataset.org/val2017/000000039769.jpg"
photo = Image.open(io.BytesIO(urllib.request.urlopen(url).read()))
questions = {"cats": {"type": "choice", "instructions": "How many cats are there?",
"criteria": {"one": "One", "two": "Two", "more": "Three or more"}}}
output = model.system_one(None, questions, images=[photo])
print(output)
import soundfile as sf
from transformers import AutoModel
model_id = "LiquidAI/d1-omni-600M"
model = AutoModel.from_pretrained(model_id, trust_remote_code=True, dtype="float16", device_map="auto")
wave, rate = sf.read("command.wav", dtype="int16")
assert rate == 16000
questions = {"kind": {"type": "choice", "instructions": "What kind of utterance is this?",
"criteria": {"request": "A request to do something",
"question": "A question asking for information",
"conversation": "Small talk or a greeting"}}}
output = model.system_one(None, questions, audio=wave)
print(output)
Start the server:Text input:Image input:Audio input:
llama-server -hf LiquidAI/d1-omni-600M-GGUF:Q8_0 -b 4096 -ub 4096
curl http://127.0.0.1:8080/v1/systemone -H "Content-Type: application/json" -d '{
"state": "I was charged twice this month, please refund one of them.",
"questions": {
"refund": {"type": "noul", "instructions": "Is the customer asking for a refund?"},
"team": {"type": "choice", "instructions": "Which team should handle this?",
"criteria": {"billing": "Charges, refunds, invoices", "technical": "App or site faults",
"fraud": "Suspected unauthorised use"}},
"urgency": {"type": "score", "instructions": "How urgent is this?",
"criteria": ["Can wait", "Today", "Blocking the customer now"]}
}
}'
curl -sL -o cats.jpg http://images.cocodataset.org/val2017/000000039769.jpg
curl http://127.0.0.1:8080/v1/systemone -H "Content-Type: application/json" -d @- <<JSON
{
"images": ["data:image/jpeg;base64,$(base64 < cats.jpg | tr -d '\n')"],
"questions": {
"cats": {"type": "choice", "instructions": "How many cats are there?",
"criteria": {"one": "One", "two": "Two", "more": "Three or more"}}
}
}
JSON
curl http://127.0.0.1:8080/v1/systemone -H "Content-Type: application/json" -d @- <<JSON
{
"audio": "data:audio/wav;base64,$(base64 < command.wav | tr -d '\n')",
"questions": {
"kind": {"type": "choice", "instructions": "What kind of utterance is this?",
"criteria": {"request": "A request to do something",
"question": "A question asking for information",
"conversation": "Small talk or a greeting"}}
}
}
JSON