Convert images to PNG before sending to llama-server
llama-server uses stb_image internally which does not support WebP, TIFF, AVIF, and other formats that browsers accept for upload. Uploading a WebP image to a vision GGUF model caused a 400 error: "Failed to load image or audio file" / "failed to decode image bytes". Convert all uploaded images to PNG via PIL before base64-encoding and forwarding to llama-server. This handles WebP, TIFF, BMP, GIF, AVIF, and any other format PIL supports. RGBA images are converted to RGB first since PNG with alpha can cause issues in some vision pipelines.
This commit is contained in:
parent
c842e019d8
commit
d417407087
1 changed files with 17 additions and 0 deletions
|
|
@ -860,6 +860,23 @@ async def openai_chat_completions(
|
|||
detail = "Image provided but current GGUF model does not support vision.",
|
||||
)
|
||||
|
||||
# Convert image to PNG for llama-server (stb_image has limited format support)
|
||||
if image_b64:
|
||||
try:
|
||||
import base64 as _b64
|
||||
from io import BytesIO as _BytesIO
|
||||
from PIL import Image as _Image
|
||||
|
||||
raw = _b64.b64decode(image_b64)
|
||||
img = _Image.open(_BytesIO(raw))
|
||||
if img.mode == "RGBA":
|
||||
img = img.convert("RGB")
|
||||
buf = _BytesIO()
|
||||
img.save(buf, format = "PNG")
|
||||
image_b64 = _b64.b64encode(buf.getvalue()).decode("ascii")
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code = 400, detail = f"Failed to process image: {e}")
|
||||
|
||||
# Build message list with system prompt prepended
|
||||
gguf_messages = []
|
||||
if system_prompt:
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue