From c017ce802fdc44815d6dbd6d435d8fed36981381 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Mon, 23 Feb 2026 17:37:40 +0400 Subject: [PATCH 01/49] Remove firebase-debug.log and setup_leo.sh from tracking and add to .gitignore --- .gitignore | 4 + firebase-debug.log | 296 --------------------------------------------- setup_leo.sh | 241 ------------------------------------ 3 files changed, 4 insertions(+), 537 deletions(-) delete mode 100644 firebase-debug.log delete mode 100644 setup_leo.sh diff --git a/.gitignore b/.gitignore index f9be87df48..3901c1c699 100755 --- a/.gitignore +++ b/.gitignore @@ -38,6 +38,9 @@ unsloth_training_checkpoints/ .DS_Store Thumbs.db +# Firebase +firebase-debug.log + # Other resources/ tmp/ @@ -45,3 +48,4 @@ auth.db studio/frontend/package-lock.json log_rtx.txt log.txt +setup_leo.sh diff --git a/firebase-debug.log b/firebase-debug.log deleted file mode 100644 index a6bea586c8..0000000000 --- a/firebase-debug.log +++ /dev/null @@ -1,296 +0,0 @@ -[debug] [2026-02-18T11:41:02.773Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T11:41:02.774Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T11:41:02.774Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T11:41:02.774Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T11:41:02.782Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T11:41:02.782Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T11:41:03.470Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T11:41:03.470Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T11:41:03.471Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T11:41:03.471Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T11:41:03.483Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T11:41:03.483Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T11:41:03.484Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T11:41:03.484Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T12:08:15.283Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T12:08:15.285Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T12:08:15.285Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T12:08:15.285Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T12:08:15.292Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T12:08:15.292Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T12:08:15.724Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T12:08:15.725Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T12:08:15.726Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T12:08:15.726Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T12:08:15.727Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T12:08:15.727Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T12:08:15.727Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T12:08:15.727Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T13:46:32.892Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T13:46:32.893Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T13:46:32.894Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T13:46:32.894Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T13:46:32.900Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T13:46:32.900Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T13:46:33.376Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T13:46:33.376Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T13:46:33.377Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T13:46:33.378Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T13:46:33.482Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T13:46:33.483Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T13:46:33.483Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T13:46:33.484Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:08:25.135Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:08:25.136Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:08:25.137Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:08:25.137Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:08:25.145Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:08:25.145Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:08:25.574Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:08:25.574Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:08:25.575Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:08:25.575Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:08:25.579Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:08:25.579Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:08:25.580Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:08:25.580Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:14:07.782Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:14:07.783Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:14:07.783Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:14:07.783Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:14:07.790Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:14:07.791Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:14:08.126Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:14:08.127Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:14:08.128Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:14:08.128Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:14:08.152Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:14:08.153Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T14:14:08.153Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T14:14:08.153Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T16:19:05.291Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T16:19:05.292Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T16:19:05.292Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T16:19:05.292Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T16:19:05.299Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T16:19:05.299Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T16:19:05.733Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T16:19:05.733Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T16:19:05.734Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T16:19:05.734Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T16:19:05.747Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T16:19:05.747Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-18T16:19:05.748Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-18T16:19:05.748Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T05:51:29.275Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T05:51:29.277Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T05:51:29.277Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T05:51:29.277Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T05:51:29.284Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T05:51:29.284Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T05:51:29.631Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T05:51:29.631Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T05:51:29.632Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T05:51:29.632Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T05:51:29.633Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T05:51:29.634Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T05:51:29.634Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T05:51:29.634Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:13:16.833Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:13:16.834Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:13:16.834Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:13:16.834Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:13:16.842Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:13:16.842Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:13:17.305Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:13:17.306Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:13:17.307Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:13:17.307Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:13:17.310Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:13:17.310Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:13:17.311Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:13:17.311Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:14:26.095Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:14:26.096Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:14:26.096Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:14:26.096Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:14:26.103Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:14:26.103Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:14:26.454Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:14:26.454Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:14:26.455Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:14:26.456Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:14:26.458Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:14:26.458Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T06:14:26.459Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T06:14:26.459Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:04:34.310Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:04:34.312Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:04:34.312Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:04:34.312Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:04:34.319Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:04:34.319Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:04:34.689Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:04:34.690Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:04:34.691Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:04:34.691Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:04:34.694Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:04:34.694Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:04:34.695Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:04:34.695Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:37:42.030Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:37:42.031Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:37:42.031Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:37:42.031Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:37:42.038Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:37:42.039Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:37:42.404Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:37:42.405Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:37:42.406Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:37:42.406Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:37:42.408Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:37:42.408Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T07:37:42.409Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T07:37:42.409Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T08:56:00.592Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T08:56:00.593Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T08:56:00.593Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T08:56:00.593Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T08:56:00.601Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T08:56:00.601Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T08:56:01.042Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T08:56:01.042Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T08:56:01.043Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T08:56:01.044Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T08:56:01.045Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T08:56:01.045Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T08:56:01.046Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T08:56:01.046Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T09:08:43.860Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T09:08:43.861Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T09:08:43.861Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T09:08:43.861Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T09:08:43.868Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T09:08:43.868Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T09:08:44.206Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T09:08:44.207Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T09:08:44.208Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T09:08:44.209Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T09:08:44.212Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T09:08:44.212Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T09:08:44.213Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T09:08:44.213Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T11:08:31.233Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T11:08:31.234Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T11:08:31.234Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T11:08:31.234Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T11:08:31.242Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T11:08:31.242Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T11:08:31.688Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T11:08:31.688Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T11:08:31.689Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T11:08:31.690Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T11:08:31.694Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T11:08:31.694Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T11:08:31.695Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T11:08:31.695Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T14:05:39.367Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T14:05:39.368Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T14:05:39.368Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T14:05:39.368Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T14:05:39.376Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T14:05:39.376Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T14:05:39.808Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T14:05:39.808Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T14:05:39.810Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T14:05:39.810Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T14:05:39.813Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T14:05:39.813Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T14:05:39.814Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T14:05:39.814Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T18:50:53.945Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T18:50:53.946Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T18:50:53.947Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T18:50:53.947Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T18:50:53.954Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T18:50:53.954Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T18:50:54.408Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T18:50:54.409Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T18:50:54.410Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T18:50:54.410Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T18:50:54.412Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T18:50:54.412Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-19T18:50:54.413Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-19T18:50:54.413Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T05:58:38.962Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T05:58:38.964Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T05:58:38.964Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T05:58:38.964Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T05:58:38.972Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T05:58:38.972Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T05:58:39.404Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T05:58:39.404Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T05:58:39.405Z] >>> [apiv2][query] POST https://developerknowledge.googleapis.com/mcp [none] -[debug] [2026-02-20T05:58:39.405Z] >>> [apiv2][body] POST https://developerknowledge.googleapis.com/mcp {"method":"tools/list","jsonrpc":"2.0","id":1} -[debug] [2026-02-20T05:58:39.407Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T05:58:39.407Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T05:58:39.408Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T05:58:39.408Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T05:58:40.459Z] <<< [apiv2][status] POST https://developerknowledge.googleapis.com/mcp 200 -[debug] [2026-02-20T05:58:40.459Z] <<< [apiv2][body] POST https://developerknowledge.googleapis.com/mcp {"id":1,"jsonrpc":"2.0","result":{"tools":[{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to find documentation about Google developer products.\nThe documents contain official APIs, code snippets, release notes, best\npractices, guides, debugging info, and more. It covers the following\nproducts and domains:\n\n* Android: developer.android.com\n* Apigee: docs.apigee.com\n* Chrome: developer.chrome.com\n* Firebase: firebase.google.com\n* Fuchsia: fuchsia.dev\n* Google AI: ai.google.dev\n* Google Cloud: docs.cloud.google.com\n* Google Developers, Ads, Search, Google Maps, Youtube: developers.google.com\n* Google Home: developers.home.google.com\n* TensorFlow: www.tensorflow.org\n* Web: web.dev\n\nThis tool returns chunks of text, names, and URLs for matching documents.\nIf the returned chunks are not detailed enough to answer the\nuser's question, use `get_document` or `batch_get_documents` with the\n`parent` from this tool's output to retrieve the full document\ncontent.","inputSchema":{"description":"Request schema for search_documents. Use the query field to search for related Google developer documentation.","properties":{"query":{"description":"Required. The raw query string provided by the user, such as \"How to create a Cloud Storage bucket?\".","type":"string"}},"required":["query"],"type":"object"},"name":"search_documents","outputSchema":{"$defs":{"DocumentChunk":{"description":"A DocumentChunk represents a piece of content from a Document in the DeveloperKnowledge corpus. To fetch the entire document content, pass the `parent` to get_document or batch_get_documents.","properties":{"content":{"description":"Output only. The content of the document chunk.","readOnly":true,"type":"string"},"id":{"description":"Output only. The ID of this chunk within the document. The chunk ID is unique within a document, but not globally unique across documents. The chunk ID is not stable and may change over time.","readOnly":true,"type":"string"},"parent":{"description":"Output only. The resource name of the document this chunk is from. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","readOnly":true,"type":"string"}},"type":"object"}},"description":"Response schema for search_documents.","properties":{"results":{"description":"The search results for the given query. Each Document in this list contains a snippet of content relevant to the search query. Use the Document.name field of each result with get_document or batch_get_documents to retrieve the full document content.","items":{"$ref":"#/$defs/DocumentChunk"},"type":"array"}},"type":"object"}},{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to retrieve the full content of a single document. The\ndocument name should be obtained from the `parent` field of results from a\ncall to the `search_documents` tool. If you need to retrieve multiple\ndocuments, use `batch_get_documents` instead.","inputSchema":{"description":"Request schema for get_document.","properties":{"name":{"description":"Required. The name of the document to retrieve. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string"}},"required":["name"],"type":"object"},"name":"get_document","outputSchema":{"description":"A Document represents a piece of content from the Developer Knowledge corpus.","properties":{"content":{"description":"Output only. The content of the document in Markdown format. If this document is returned by search_documents, this field contains a snippet of text relevant to the search query. If this document is returned by get_document or batch_get_documents, this field contains the full document content.","readOnly":true,"type":"string"},"description":{"description":"Output only. A description of the document.","readOnly":true,"type":"string"},"name":{"description":"Identifier. The resource name of the document. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string","x-google-identifier":true},"uri":{"description":"Output only. The URI of the content, such as `https://cloud.google.com/storage/docs/creating-buckets`.","readOnly":true,"type":"string"}},"type":"object"}},{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to retrieve the full content of up to 20 documents in a\nsingle call. The document names should be obtained from the `parent` field\nof results from a call to the `search_documents` tool. Use this tool\ninstead of calling `get_document` multiple times to fetch multiple\ndocuments.","inputSchema":{"description":"Request schema for batch_get_documents.","properties":{"names":{"description":"Required. The names of the documents to retrieve, as returned by search_documents. A maximum of 20 documents can be retrieved in a batch. The documents are returned in the same order as the `names` in the request. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","items":{"type":"string"},"type":"array"}},"required":["names"],"type":"object"},"name":"batch_get_documents","outputSchema":{"$defs":{"Document":{"description":"A Document represents a piece of content from the Developer Knowledge corpus.","properties":{"content":{"description":"Output only. The content of the document in Markdown format. If this document is returned by search_documents, this field contains a snippet of text relevant to the search query. If this document is returned by get_document or batch_get_documents, this field contains the full document content.","readOnly":true,"type":"string"},"description":{"description":"Output only. A description of the document.","readOnly":true,"type":"string"},"name":{"description":"Identifier. The resource name of the document. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string","x-google-identifier":true},"uri":{"description":"Output only. The URI of the content, such as `https://cloud.google.com/storage/docs/creating-buckets`.","readOnly":true,"type":"string"}},"type":"object"}},"description":"Response schema for batch_get_documents.","properties":{"documents":{"description":"Documents requested.","items":{"$ref":"#/$defs/Document"},"type":"array"}},"type":"object"}}]}} -[debug] [2026-02-20T05:58:40.462Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T05:58:40.462Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T11:50:13.855Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T11:50:13.856Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T11:50:13.857Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T11:50:13.857Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T11:50:13.864Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T11:50:13.864Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T11:50:14.206Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T11:50:14.206Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T11:50:14.207Z] >>> [apiv2][query] POST https://developerknowledge.googleapis.com/mcp [none] -[debug] [2026-02-20T11:50:14.207Z] >>> [apiv2][body] POST https://developerknowledge.googleapis.com/mcp {"method":"tools/list","jsonrpc":"2.0","id":1} -[debug] [2026-02-20T11:50:14.212Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T11:50:14.212Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T11:50:14.212Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T11:50:14.213Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T11:50:15.300Z] <<< [apiv2][status] POST https://developerknowledge.googleapis.com/mcp 200 -[debug] [2026-02-20T11:50:15.300Z] <<< [apiv2][body] POST https://developerknowledge.googleapis.com/mcp {"id":1,"jsonrpc":"2.0","result":{"tools":[{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to find documentation about Google developer products.\nThe documents contain official APIs, code snippets, release notes, best\npractices, guides, debugging info, and more. It covers the following\nproducts and domains:\n\n* Android: developer.android.com\n* Apigee: docs.apigee.com\n* Chrome: developer.chrome.com\n* Firebase: firebase.google.com\n* Fuchsia: fuchsia.dev\n* Google AI: ai.google.dev\n* Google Cloud: docs.cloud.google.com\n* Google Developers, Ads, Search, Google Maps, Youtube: developers.google.com\n* Google Home: developers.home.google.com\n* TensorFlow: www.tensorflow.org\n* Web: web.dev\n\nThis tool returns chunks of text, names, and URLs for matching documents.\nIf the returned chunks are not detailed enough to answer the\nuser's question, use `get_document` or `batch_get_documents` with the\n`parent` from this tool's output to retrieve the full document\ncontent.","inputSchema":{"description":"Request schema for search_documents. Use the query field to search for related Google developer documentation.","properties":{"query":{"description":"Required. The raw query string provided by the user, such as \"How to create a Cloud Storage bucket?\".","type":"string"}},"required":["query"],"type":"object"},"name":"search_documents","outputSchema":{"$defs":{"DocumentChunk":{"description":"A DocumentChunk represents a piece of content from a Document in the DeveloperKnowledge corpus. To fetch the entire document content, pass the `parent` to get_document or batch_get_documents.","properties":{"content":{"description":"Output only. The content of the document chunk.","readOnly":true,"type":"string"},"id":{"description":"Output only. The ID of this chunk within the document. The chunk ID is unique within a document, but not globally unique across documents. The chunk ID is not stable and may change over time.","readOnly":true,"type":"string"},"parent":{"description":"Output only. The resource name of the document this chunk is from. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","readOnly":true,"type":"string"}},"type":"object"}},"description":"Response schema for search_documents.","properties":{"results":{"description":"The search results for the given query. Each Document in this list contains a snippet of content relevant to the search query. Use the Document.name field of each result with get_document or batch_get_documents to retrieve the full document content.","items":{"$ref":"#/$defs/DocumentChunk"},"type":"array"}},"type":"object"}},{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to retrieve the full content of a single document. The\ndocument name should be obtained from the `parent` field of results from a\ncall to the `search_documents` tool. If you need to retrieve multiple\ndocuments, use `batch_get_documents` instead.","inputSchema":{"description":"Request schema for get_document.","properties":{"name":{"description":"Required. The name of the document to retrieve. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string"}},"required":["name"],"type":"object"},"name":"get_document","outputSchema":{"description":"A Document represents a piece of content from the Developer Knowledge corpus.","properties":{"content":{"description":"Output only. The content of the document in Markdown format. If this document is returned by search_documents, this field contains a snippet of text relevant to the search query. If this document is returned by get_document or batch_get_documents, this field contains the full document content.","readOnly":true,"type":"string"},"description":{"description":"Output only. A description of the document.","readOnly":true,"type":"string"},"name":{"description":"Identifier. The resource name of the document. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string","x-google-identifier":true},"uri":{"description":"Output only. The URI of the content, such as `https://cloud.google.com/storage/docs/creating-buckets`.","readOnly":true,"type":"string"}},"type":"object"}},{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to retrieve the full content of up to 20 documents in a\nsingle call. The document names should be obtained from the `parent` field\nof results from a call to the `search_documents` tool. Use this tool\ninstead of calling `get_document` multiple times to fetch multiple\ndocuments.","inputSchema":{"description":"Request schema for batch_get_documents.","properties":{"names":{"description":"Required. The names of the documents to retrieve, as returned by search_documents. A maximum of 20 documents can be retrieved in a batch. The documents are returned in the same order as the `names` in the request. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","items":{"type":"string"},"type":"array"}},"required":["names"],"type":"object"},"name":"batch_get_documents","outputSchema":{"$defs":{"Document":{"description":"A Document represents a piece of content from the Developer Knowledge corpus.","properties":{"content":{"description":"Output only. The content of the document in Markdown format. If this document is returned by search_documents, this field contains a snippet of text relevant to the search query. If this document is returned by get_document or batch_get_documents, this field contains the full document content.","readOnly":true,"type":"string"},"description":{"description":"Output only. A description of the document.","readOnly":true,"type":"string"},"name":{"description":"Identifier. The resource name of the document. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string","x-google-identifier":true},"uri":{"description":"Output only. The URI of the content, such as `https://cloud.google.com/storage/docs/creating-buckets`.","readOnly":true,"type":"string"}},"type":"object"}},"description":"Response schema for batch_get_documents.","properties":{"documents":{"description":"Documents requested.","items":{"$ref":"#/$defs/Document"},"type":"array"}},"type":"object"}}]}} -[debug] [2026-02-20T11:50:15.302Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T11:50:15.302Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T19:09:28.241Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T19:09:28.242Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T19:09:28.242Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T19:09:28.242Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T19:09:28.249Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T19:09:28.249Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T19:09:28.579Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T19:09:28.579Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T19:09:28.580Z] >>> [apiv2][query] POST https://developerknowledge.googleapis.com/mcp [none] -[debug] [2026-02-20T19:09:28.581Z] >>> [apiv2][body] POST https://developerknowledge.googleapis.com/mcp {"method":"tools/list","jsonrpc":"2.0","id":1} -[debug] [2026-02-20T19:09:28.586Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T19:09:28.586Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T19:09:28.586Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T19:09:28.586Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-20T19:09:29.538Z] <<< [apiv2][status] POST https://developerknowledge.googleapis.com/mcp 200 -[debug] [2026-02-20T19:09:29.539Z] <<< [apiv2][body] POST https://developerknowledge.googleapis.com/mcp {"id":1,"jsonrpc":"2.0","result":{"tools":[{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to find documentation about Google developer products.\nThe documents contain official APIs, code snippets, release notes, best\npractices, guides, debugging info, and more. It covers the following\nproducts and domains:\n\n* Android: developer.android.com\n* Apigee: docs.apigee.com\n* Chrome: developer.chrome.com\n* Firebase: firebase.google.com\n* Fuchsia: fuchsia.dev\n* Google AI: ai.google.dev\n* Google Cloud: docs.cloud.google.com\n* Google Developers, Ads, Search, Google Maps, Youtube: developers.google.com\n* Google Home: developers.home.google.com\n* TensorFlow: www.tensorflow.org\n* Web: web.dev\n\nThis tool returns chunks of text, names, and URLs for matching documents.\nIf the returned chunks are not detailed enough to answer the\nuser's question, use `get_document` or `batch_get_documents` with the\n`parent` from this tool's output to retrieve the full document\ncontent.","inputSchema":{"description":"Request schema for search_documents. Use the query field to search for related Google developer documentation.","properties":{"query":{"description":"Required. The raw query string provided by the user, such as \"How to create a Cloud Storage bucket?\".","type":"string"}},"required":["query"],"type":"object"},"name":"search_documents","outputSchema":{"$defs":{"DocumentChunk":{"description":"A DocumentChunk represents a piece of content from a Document in the DeveloperKnowledge corpus. To fetch the entire document content, pass the `parent` to get_document or batch_get_documents.","properties":{"content":{"description":"Output only. The content of the document chunk.","readOnly":true,"type":"string"},"id":{"description":"Output only. The ID of this chunk within the document. The chunk ID is unique within a document, but not globally unique across documents. The chunk ID is not stable and may change over time.","readOnly":true,"type":"string"},"parent":{"description":"Output only. The resource name of the document this chunk is from. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","readOnly":true,"type":"string"}},"type":"object"}},"description":"Response schema for search_documents.","properties":{"results":{"description":"The search results for the given query. Each Document in this list contains a snippet of content relevant to the search query. Use the Document.name field of each result with get_document or batch_get_documents to retrieve the full document content.","items":{"$ref":"#/$defs/DocumentChunk"},"type":"array"}},"type":"object"}},{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to retrieve the full content of a single document. The\ndocument name should be obtained from the `parent` field of results from a\ncall to the `search_documents` tool. If you need to retrieve multiple\ndocuments, use `batch_get_documents` instead.","inputSchema":{"description":"Request schema for get_document.","properties":{"name":{"description":"Required. The name of the document to retrieve. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string"}},"required":["name"],"type":"object"},"name":"get_document","outputSchema":{"description":"A Document represents a piece of content from the Developer Knowledge corpus.","properties":{"content":{"description":"Output only. The content of the document in Markdown format. If this document is returned by search_documents, this field contains a snippet of text relevant to the search query. If this document is returned by get_document or batch_get_documents, this field contains the full document content.","readOnly":true,"type":"string"},"description":{"description":"Output only. A description of the document.","readOnly":true,"type":"string"},"name":{"description":"Identifier. The resource name of the document. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string","x-google-identifier":true},"uri":{"description":"Output only. The URI of the content, such as `https://cloud.google.com/storage/docs/creating-buckets`.","readOnly":true,"type":"string"}},"type":"object"}},{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to retrieve the full content of up to 20 documents in a\nsingle call. The document names should be obtained from the `parent` field\nof results from a call to the `search_documents` tool. Use this tool\ninstead of calling `get_document` multiple times to fetch multiple\ndocuments.","inputSchema":{"description":"Request schema for batch_get_documents.","properties":{"names":{"description":"Required. The names of the documents to retrieve, as returned by search_documents. A maximum of 20 documents can be retrieved in a batch. The documents are returned in the same order as the `names` in the request. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","items":{"type":"string"},"type":"array"}},"required":["names"],"type":"object"},"name":"batch_get_documents","outputSchema":{"$defs":{"Document":{"description":"A Document represents a piece of content from the Developer Knowledge corpus.","properties":{"content":{"description":"Output only. The content of the document in Markdown format. If this document is returned by search_documents, this field contains a snippet of text relevant to the search query. If this document is returned by get_document or batch_get_documents, this field contains the full document content.","readOnly":true,"type":"string"},"description":{"description":"Output only. A description of the document.","readOnly":true,"type":"string"},"name":{"description":"Identifier. The resource name of the document. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string","x-google-identifier":true},"uri":{"description":"Output only. The URI of the content, such as `https://cloud.google.com/storage/docs/creating-buckets`.","readOnly":true,"type":"string"}},"type":"object"}},"description":"Response schema for batch_get_documents.","properties":{"documents":{"description":"Documents requested.","items":{"$ref":"#/$defs/Document"},"type":"array"}},"type":"object"}}]}} -[debug] [2026-02-20T19:09:29.541Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-20T19:09:29.541Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-23T10:47:00.254Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-23T10:47:00.256Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-23T10:47:00.256Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-23T10:47:00.256Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-23T10:47:00.263Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-23T10:47:00.263Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-23T10:47:00.736Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-23T10:47:00.737Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-23T10:47:00.738Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-23T10:47:00.738Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-23T10:47:00.740Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-23T10:47:00.741Z] > authorizing via signed-in user (support@gonovel.co) -[debug] [2026-02-23T10:47:00.742Z] >>> [apiv2][query] POST https://developerknowledge.googleapis.com/mcp [none] -[debug] [2026-02-23T10:47:00.742Z] >>> [apiv2][body] POST https://developerknowledge.googleapis.com/mcp {"method":"tools/list","jsonrpc":"2.0","id":1} -[debug] [2026-02-23T10:47:01.893Z] <<< [apiv2][status] POST https://developerknowledge.googleapis.com/mcp 200 -[debug] [2026-02-23T10:47:01.893Z] <<< [apiv2][body] POST https://developerknowledge.googleapis.com/mcp {"id":1,"jsonrpc":"2.0","result":{"tools":[{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to find documentation about Google developer products.\nThe documents contain official APIs, code snippets, release notes, best\npractices, guides, debugging info, and more. It covers the following\nproducts and domains:\n\n* Android: developer.android.com\n* Apigee: docs.apigee.com\n* Chrome: developer.chrome.com\n* Firebase: firebase.google.com\n* Fuchsia: fuchsia.dev\n* Google AI: ai.google.dev\n* Google Cloud: docs.cloud.google.com\n* Google Developers, Ads, Search, Google Maps, Youtube: developers.google.com\n* Google Home: developers.home.google.com\n* TensorFlow: www.tensorflow.org\n* Web: web.dev\n\nThis tool returns chunks of text, names, and URLs for matching documents.\nIf the returned chunks are not detailed enough to answer the\nuser's question, use `get_document` or `batch_get_documents` with the\n`parent` from this tool's output to retrieve the full document\ncontent.","inputSchema":{"description":"Request schema for search_documents. Use the query field to search for related Google developer documentation.","properties":{"query":{"description":"Required. The raw query string provided by the user, such as \"How to create a Cloud Storage bucket?\".","type":"string"}},"required":["query"],"type":"object"},"name":"search_documents","outputSchema":{"$defs":{"DocumentChunk":{"description":"A DocumentChunk represents a piece of content from a Document in the DeveloperKnowledge corpus. To fetch the entire document content, pass the `parent` to get_document or batch_get_documents.","properties":{"content":{"description":"Output only. The content of the document chunk.","readOnly":true,"type":"string"},"id":{"description":"Output only. The ID of this chunk within the document. The chunk ID is unique within a document, but not globally unique across documents. The chunk ID is not stable and may change over time.","readOnly":true,"type":"string"},"parent":{"description":"Output only. The resource name of the document this chunk is from. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","readOnly":true,"type":"string"}},"type":"object"}},"description":"Response schema for search_documents.","properties":{"results":{"description":"The search results for the given query. Each Document in this list contains a snippet of content relevant to the search query. Use the Document.name field of each result with get_document or batch_get_documents to retrieve the full document content.","items":{"$ref":"#/$defs/DocumentChunk"},"type":"array"}},"type":"object"}},{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to retrieve the full content of a single document. The\ndocument name should be obtained from the `parent` field of results from a\ncall to the `search_documents` tool. If you need to retrieve multiple\ndocuments, use `batch_get_documents` instead.","inputSchema":{"description":"Request schema for get_document.","properties":{"name":{"description":"Required. The name of the document to retrieve. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string"}},"required":["name"],"type":"object"},"name":"get_document","outputSchema":{"description":"A Document represents a piece of content from the Developer Knowledge corpus.","properties":{"content":{"description":"Output only. The content of the document in Markdown format. If this document is returned by search_documents, this field contains a snippet of text relevant to the search query. If this document is returned by get_document or batch_get_documents, this field contains the full document content.","readOnly":true,"type":"string"},"description":{"description":"Output only. A description of the document.","readOnly":true,"type":"string"},"name":{"description":"Identifier. The resource name of the document. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string","x-google-identifier":true},"uri":{"description":"Output only. The URI of the content, such as `https://cloud.google.com/storage/docs/creating-buckets`.","readOnly":true,"type":"string"}},"type":"object"}},{"annotations":{"destructiveHint":false,"idempotentHint":true,"openWorldHint":false,"readOnlyHint":true},"description":"Use this tool to retrieve the full content of up to 20 documents in a\nsingle call. The document names should be obtained from the `parent` field\nof results from a call to the `search_documents` tool. Use this tool\ninstead of calling `get_document` multiple times to fetch multiple\ndocuments.","inputSchema":{"description":"Request schema for batch_get_documents.","properties":{"names":{"description":"Required. The names of the documents to retrieve, as returned by search_documents. A maximum of 20 documents can be retrieved in a batch. The documents are returned in the same order as the `names` in the request. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","items":{"type":"string"},"type":"array"}},"required":["names"],"type":"object"},"name":"batch_get_documents","outputSchema":{"$defs":{"Document":{"description":"A Document represents a piece of content from the Developer Knowledge corpus.","properties":{"content":{"description":"Output only. The content of the document in Markdown format. If this document is returned by search_documents, this field contains a snippet of text relevant to the search query. If this document is returned by get_document or batch_get_documents, this field contains the full document content.","readOnly":true,"type":"string"},"description":{"description":"Output only. A description of the document.","readOnly":true,"type":"string"},"name":{"description":"Identifier. The resource name of the document. Format: `documents/{uri_without_scheme}` Example: `documents/docs.cloud.google.com/storage/docs/creating-buckets`","type":"string","x-google-identifier":true},"uri":{"description":"Output only. The URI of the content, such as `https://cloud.google.com/storage/docs/creating-buckets`.","readOnly":true,"type":"string"}},"type":"object"}},"description":"Response schema for batch_get_documents.","properties":{"documents":{"description":"Documents requested.","items":{"$ref":"#/$defs/Document"},"type":"array"}},"type":"object"}}]}} -[debug] [2026-02-23T10:47:01.896Z] > command requires scopes: ["email","openid","https://www.googleapis.com/auth/cloudplatformprojects.readonly","https://www.googleapis.com/auth/firebase","https://www.googleapis.com/auth/cloud-platform"] -[debug] [2026-02-23T10:47:01.896Z] > authorizing via signed-in user (support@gonovel.co) diff --git a/setup_leo.sh b/setup_leo.sh deleted file mode 100644 index 98d878421f..0000000000 --- a/setup_leo.sh +++ /dev/null @@ -1,241 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" - -# ── Helper: run command quietly, show output only on failure ── -run_quiet() { - local label="$1" - shift - local tmplog - tmplog=$(mktemp) - if "$@" > "$tmplog" 2>&1; then - rm -f "$tmplog" - else - local exit_code=$? - echo "❌ $label failed (exit code $exit_code):" - cat "$tmplog" - rm -f "$tmplog" - exit $exit_code - fi -} - -echo "╔══════════════════════════════════════╗" -echo "║ Unsloth Studio Setup Script ║" -echo "╚══════════════════════════════════════╝" - -# ── Detect Colab (like unsloth does) ── -IS_COLAB=false -keynames=$'\n'$(printenv | cut -d= -f1) -if [[ "$keynames" == *$'\nCOLAB_'* ]]; then - IS_COLAB=true -fi - -# ── 1. Check existing Node/npm versions ── -NEED_NODE=true -if command -v node &>/dev/null && command -v npm &>/dev/null; then - NODE_MAJOR=$(node -v | sed 's/v//' | cut -d. -f1) - NPM_MAJOR=$(npm -v | cut -d. -f1) - if [ "$NODE_MAJOR" -ge 20 ] && [ "$NPM_MAJOR" -ge 11 ]; then - echo "✅ Node $(node -v) and npm $(npm -v) already meet requirements. Skipping nvm install." - NEED_NODE=false - else - if [ "$IS_COLAB" = true ]; then - echo "✅ Node $(node -v) and npm $(npm -v) detected in Colab." - # In Colab, just upgrade npm directly - nvm doesn't work well - if [ "$NPM_MAJOR" -lt 11 ]; then - echo " Upgrading npm to latest..." - npm install -g npm@latest > /dev/null 2>&1 - fi - NEED_NODE=false - else - echo "⚠️ Node $(node -v) / npm $(npm -v) too old. Installing via nvm..." - fi - fi -else - echo "⚠️ Node/npm not found. Installing via nvm..." -fi - -if [ "$NEED_NODE" = true ]; then - # ── 2. Install nvm ── - echo "Installing nvm..." - curl -so- https://raw.githubusercontent.com/nvm-sh/nvm/v0.40.1/install.sh | bash > /dev/null 2>&1 - - # Load nvm (source ~/.bashrc won't work inside a script) - export NVM_DIR="$HOME/.nvm" - [ -s "$NVM_DIR/nvm.sh" ] && \. "$NVM_DIR/nvm.sh" - - # ── 3. Install Node LTS ── - echo "Installing Node LTS..." - run_quiet "nvm install" nvm install --lts - nvm use --lts > /dev/null 2>&1 - - # ── 4. Verify versions ── - NODE_MAJOR=$(node -v | sed 's/v//' | cut -d. -f1) - NPM_MAJOR=$(npm -v | cut -d. -f1) - - if [ "$NODE_MAJOR" -lt 20 ]; then - echo "❌ ERROR: Node version must be >= 20 (got $(node -v))" - exit 1 - fi - if [ "$NPM_MAJOR" -lt 11 ]; then - echo "⚠️ npm version is $(npm -v), expected >= 11. Updating..." - run_quiet "npm update" npm install -g npm@latest - fi -fi - -echo "✅ Node $(node -v) | npm $(npm -v)" - -# ── 5. Build frontend ── -echo "" -echo "Building frontend..." -cd "$SCRIPT_DIR/studio/frontend" -run_quiet "npm install" npm install -run_quiet "npm run build" npm run build -cd "$SCRIPT_DIR" -echo "✅ Frontend built to studio/frontend/dist" - -# ── 6. Python venv + deps ── -echo "" -echo "Setting up Python environment..." - -# ── 6a. Discover best Python <= 3.12.x ── -BEST_PY="" -BEST_MAJOR=0 -BEST_MINOR=0 - -# Collect candidate python3 binaries (python3, python3.9, python3.10, …) -for candidate in $(compgen -c python3 2>/dev/null | grep -E '^python3(\.[0-9]+)?$' | sort -u); do - if ! command -v "$candidate" &>/dev/null; then - continue - fi - # Get version string, e.g. "Python 3.11.5" - ver_str=$("$candidate" --version 2>&1 | awk '{print $2}') - py_major=$(echo "$ver_str" | cut -d. -f1) - py_minor=$(echo "$ver_str" | cut -d. -f2) - - # Skip anything that isn't Python 3 - if [ "$py_major" -ne 3 ] 2>/dev/null; then - continue - fi - - # Skip versions above 3.12 - if [ "$py_minor" -gt 12 ] 2>/dev/null; then - continue - fi - - # Keep the highest qualifying version - if [ "$py_minor" -gt "$BEST_MINOR" ]; then - BEST_PY="$candidate" - BEST_MAJOR="$py_major" - BEST_MINOR="$py_minor" - fi -done - -if [ -z "$BEST_PY" ]; then - echo "❌ ERROR: No Python version <= 3.12.x found on this system." - echo " Detected Python 3 installations:" - for candidate in $(compgen -c python3 2>/dev/null | grep -E '^python3(\.[0-9]+)?$' | sort -u); do - if command -v "$candidate" &>/dev/null; then - echo " - $candidate ($($candidate --version 2>&1))" - fi - done - echo "" - echo " Please install Python <= 3.12.x for maximum compatibility." - echo " For example: sudo apt install python3.12 python3.12-venv" - exit 1 -fi - -BEST_VER=$("$BEST_PY" --version 2>&1 | awk '{print $2}') -echo "✅ Using $BEST_PY ($BEST_VER) — compatible (≤ 3.12.x)" - -if [ "$IS_COLAB" = true ]; then - # Colab: install packages directly without venv - run_quiet "pip upgrade" pip install --upgrade pip - echo " Installing unsloth-zoo + unsloth..." - run_quiet "pip install unsloth" pip install unsloth-zoo unsloth - echo " Installing studio dependencies..." - run_quiet "pip install extras" pip install typer fastapi uvicorn pydantic matplotlib pandas nest_asyncio "datasets==4.3.0" pyjwt easydict addict - echo "✅ Python dependencies installed" -else - # Local: create venv - "$BEST_PY" -m venv .venv - source .venv/bin/activate - run_quiet "pip upgrade" pip install --upgrade pip - echo " Installing unsloth-zoo + unsloth..." - run_quiet "pip install unsloth" pip install unsloth-zoo unsloth - echo " Installing studio dependencies..." - run_quiet "pip install extras" pip install typer fastapi uvicorn pydantic matplotlib pandas nest_asyncio "datasets==4.3.0" pyjwt easydict addict - echo "✅ Python dependencies installed" -fi - -# ── 7. Add shell alias (skip in Colab) ── -# Note: venv activation does NOT persist across terminal sessions. -# This alias hardcodes the venv python path so users don't need to activate. -if [ "$IS_COLAB" = false ]; then -echo "" -REPO_DIR="$SCRIPT_DIR" - -# Detect the user's default shell and pick the right rc file -USER_SHELL="$(basename "${SHELL:-/bin/bash}")" -case "$USER_SHELL" in - zsh) - SHELL_RC="$HOME/.zshrc" - ALIAS_BLOCK="alias unsloth-ui='${REPO_DIR}/.venv/bin/python ${REPO_DIR}/cli.py ui -f ${REPO_DIR}/studio/frontend/dist'" - ;; - fish) - SHELL_RC="$HOME/.config/fish/config.fish" - # fish uses 'abbr' or 'function'; a simple alias works via 'alias' in config.fish - ALIAS_BLOCK="alias unsloth-ui '${REPO_DIR}/.venv/bin/python ${REPO_DIR}/cli.py ui -f ${REPO_DIR}/studio/frontend/dist'" - ;; - ksh) - SHELL_RC="$HOME/.kshrc" - ALIAS_BLOCK="alias unsloth-ui='${REPO_DIR}/.venv/bin/python ${REPO_DIR}/cli.py ui -f ${REPO_DIR}/studio/frontend/dist'" - ;; - *) - # Default to bash for bash and any other POSIX-compatible shell - SHELL_RC="$HOME/.bashrc" - ALIAS_BLOCK="alias unsloth-ui='${REPO_DIR}/.venv/bin/python ${REPO_DIR}/cli.py ui -f ${REPO_DIR}/studio/frontend/dist'" - ;; -esac - -echo " Detected shell: $USER_SHELL → $SHELL_RC" - -ALIAS_ADDED=false -if ! grep -qF "unsloth-ui" "$SHELL_RC" 2>/dev/null; then - mkdir -p "$(dirname "$SHELL_RC")" # needed for fish's nested config path - cat >> "$SHELL_RC" < Date: Mon, 23 Feb 2026 15:44:06 +0000 Subject: [PATCH 02/49] feat: sort and filter dataset search results by model type relevance --- .../components/steps/dataset-step.tsx | 3 + .../studio/sections/dataset-section.tsx | 3 + .../src/hooks/use-hf-dataset-search.ts | 194 +++++++++++++++++- 3 files changed, 195 insertions(+), 5 deletions(-) diff --git a/studio/frontend/src/features/onboarding/components/steps/dataset-step.tsx b/studio/frontend/src/features/onboarding/components/steps/dataset-step.tsx index 484b10ddb4..bdb2d8e69d 100644 --- a/studio/frontend/src/features/onboarding/components/steps/dataset-step.tsx +++ b/studio/frontend/src/features/onboarding/components/steps/dataset-step.tsx @@ -77,6 +77,7 @@ export function DatasetStep() { setDatasetSplit, uploadedFile, setUploadedFile, + modelType, } = useTrainingConfigStore( useShallow((s) => ({ hfToken: s.hfToken, @@ -93,6 +94,7 @@ export function DatasetStep() { setDatasetSplit: s.setDatasetSplit, uploadedFile: s.uploadedFile, setUploadedFile: s.setUploadedFile, + modelType: s.modelType, })), ); @@ -106,6 +108,7 @@ export function DatasetStep() { fetchMore, error: hfSearchError, } = useHfDatasetSearch(debouncedQuery, { + modelType, accessToken: hfToken || undefined, }); diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index d142ae5f54..320c248e22 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -67,6 +67,7 @@ export function DatasetSection() { datasetSplit, setDatasetSplit, hfToken, + modelType, } = useTrainingConfigStore( useShallow((s) => ({ dataset: s.dataset, @@ -78,6 +79,7 @@ export function DatasetSection() { datasetSplit: s.datasetSplit, setDatasetSplit: s.setDatasetSplit, hfToken: s.hfToken, + modelType: s.modelType, })), ); @@ -105,6 +107,7 @@ export function DatasetSection() { fetchMore, error: hfSearchError, } = useHfDatasetSearch(debouncedQuery, { + modelType, accessToken: hfToken || undefined, }); diff --git a/studio/frontend/src/hooks/use-hf-dataset-search.ts b/studio/frontend/src/hooks/use-hf-dataset-search.ts index ac94a427a6..9236a60486 100644 --- a/studio/frontend/src/hooks/use-hf-dataset-search.ts +++ b/studio/frontend/src/hooks/use-hf-dataset-search.ts @@ -1,5 +1,6 @@ import { listDatasets } from "@huggingface/hub"; -import { useCallback } from "react"; +import { useCallback, useMemo } from "react"; +import type { ModelType } from "@/types/training"; import { useHfPaginatedSearch } from "./use-hf-paginated-search"; interface DatasetInfoSplit { @@ -47,6 +48,7 @@ export interface HfDatasetResult { likes: number; totalExamples?: number; sizeCategory?: string; + taskCategories: string[]; } function mapDataset(raw: unknown): HfDatasetResult { @@ -54,32 +56,214 @@ function mapDataset(raw: unknown): HfDatasetResult { name: string; downloads: number; likes: number; + tags?: string[]; cardData?: unknown; }; const card = ds.cardData as CardDataWithInfo | undefined; + const taskCategories = (ds.tags ?? []) + .filter((t) => t.startsWith("task_categories:")) + .map((t) => t.slice("task_categories:".length)); return { id: ds.name, downloads: ds.downloads, likes: ds.likes, totalExamples: extractTotalExamples(card), sizeCategory: card?.size_categories?.[0], + taskCategories, }; } +function withTrendingSort( + input: Parameters[0], + init?: Parameters[1], +): ReturnType { + const rawUrl = + typeof input === "string" + ? input + : input instanceof URL + ? input.toString() + : input.url; + const url = new URL(rawUrl); + + if (!url.searchParams.has("sort")) { + url.searchParams.set("sort", "trendingScore"); + } + if (!url.searchParams.has("direction")) { + url.searchParams.set("direction", "-1"); + } + + return fetch(url, init); +} + +const RELEVANT_TASK_CATEGORIES: Record> = { + text: new Set([ + "text-generation", + "text2text-generation", + "question-answering", + "summarization", + "conversational", + ]), + vision: new Set([ + "image-text-to-text", + "visual-question-answering", + "image-to-text", + "image-captioning", + ]), + tts: new Set([ + "text-to-speech", + "text-to-audio", + "automatic-speech-recognition", + ]), + embeddings: new Set([ + "feature-extraction", + "sentence-similarity", + "text-retrieval", + ]), +}; + +const INCOMPATIBLE_TASK_CATEGORIES: Record> = { + text: new Set([ + "text-to-3d", + "image-to-3d", + "text-to-image", + "image-to-image", + "image-to-video", + "text-to-video", + "image-classification", + "image-feature-extraction", + "image-text-to-image", + "zero-shot-image-classification", + "keypoint-detection", + "object-detection", + "image-segmentation", + "depth-estimation", + "text-to-speech", + "text-to-audio", + "audio-classification", + "audio-to-audio", + "automatic-speech-recognition", + "video-classification", + "robotics", + "reinforcement-learning", + "tabular-classification", + "tabular-regression", + "time-series-forecasting", + "visual-document-retrieval", + ]), + vision: new Set([ + "text-to-3d", + "image-to-3d", + "text-to-speech", + "text-to-audio", + "audio-classification", + "audio-to-audio", + "automatic-speech-recognition", + "robotics", + "reinforcement-learning", + "tabular-classification", + "tabular-regression", + "time-series-forecasting", + ]), + tts: new Set([ + "text-to-3d", + "image-to-3d", + "text-to-image", + "image-to-image", + "image-to-video", + "text-to-video", + "image-classification", + "image-feature-extraction", + "image-text-to-image", + "zero-shot-image-classification", + "keypoint-detection", + "object-detection", + "image-segmentation", + "depth-estimation", + "video-classification", + "robotics", + "reinforcement-learning", + "tabular-classification", + "tabular-regression", + "time-series-forecasting", + "visual-document-retrieval", + ]), + embeddings: new Set([ + "text-to-3d", + "image-to-3d", + "text-to-image", + "image-to-image", + "image-to-video", + "text-to-video", + "image-classification", + "image-feature-extraction", + "image-text-to-image", + "zero-shot-image-classification", + "keypoint-detection", + "object-detection", + "image-segmentation", + "depth-estimation", + "text-to-speech", + "text-to-audio", + "audio-classification", + "audio-to-audio", + "automatic-speech-recognition", + "video-classification", + "robotics", + "reinforcement-learning", + "tabular-classification", + "tabular-regression", + "time-series-forecasting", + "visual-document-retrieval", + ]), +}; + +function classifyDataset( + dataset: HfDatasetResult, + modelType: ModelType, +): -1 | 0 | 1 { + const { taskCategories } = dataset; + if (taskCategories.length === 0) return 0; + + const relevant = RELEVANT_TASK_CATEGORIES[modelType]; + const incompatible = INCOMPATIBLE_TASK_CATEGORIES[modelType]; + + if (taskCategories.some((t) => relevant.has(t))) return 1; + if (taskCategories.every((t) => incompatible.has(t))) return -1; + return 0; +} + export function useHfDatasetSearch( query: string, - options?: { accessToken?: string }, + options?: { modelType?: ModelType | null; accessToken?: string }, ) { - const { accessToken } = options ?? {}; + const { modelType, accessToken } = options ?? {}; const createIter = useCallback( () => listDatasets({ search: query.trim() ? { query } : {}, - additionalFields: ["cardData"], + additionalFields: ["cardData", "tags"], + fetch: withTrendingSort, ...(accessToken ? { credentials: { accessToken } } : {}), }) as AsyncGenerator, [query, accessToken], ); - return useHfPaginatedSearch(createIter, mapDataset); + const search = useHfPaginatedSearch(createIter, mapDataset); + + const results = useMemo(() => { + if (!modelType) return search.results; + + const boosted: HfDatasetResult[] = []; + const neutral: HfDatasetResult[] = []; + + for (const ds of search.results) { + const rank = classifyDataset(ds, modelType); + if (rank === 1) boosted.push(ds); + else if (rank !== -1) neutral.push(ds); + } + + return [...boosted, ...neutral]; + }, [search.results, modelType]); + + return { ...search, results }; } From 32d5cd71981a90ce9aed4bc2181fa6dddae2c616 Mon Sep 17 00:00:00 2001 From: samit Date: Mon, 23 Feb 2026 17:37:23 -0800 Subject: [PATCH 03/49] resolved unbound variable error --- setup.sh | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/setup.sh b/setup.sh index 1f55e141fd..df5b0fddcd 100755 --- a/setup.sh +++ b/setup.sh @@ -69,13 +69,14 @@ if [ "$NEED_NODE" = true ]; then # Load nvm (source ~/.bashrc won't work inside a script) export NVM_DIR="$HOME/.nvm" + set +u [ -s "$NVM_DIR/nvm.sh" ] && \. "$NVM_DIR/nvm.sh" # ── 3. Install Node LTS ── echo "Installing Node LTS..." run_quiet "nvm install" nvm install --lts nvm use --lts > /dev/null 2>&1 - + set -u # ── 4. Verify versions ── NODE_MAJOR=$(node -v | sed 's/v//' | cut -d. -f1) NPM_MAJOR=$(npm -v | cut -d. -f1) From a40ebb1aabc7911cb5556a88351f72cffccaea53 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Tue, 24 Feb 2026 17:40:05 +0400 Subject: [PATCH 04/49] Add GGUF model inference via llama-server backend --- .gitignore | 3 + setup.sh | 62 ++- studio/backend/core/inference/__init__.py | 2 + studio/backend/core/inference/llama_cpp.py | 414 ++++++++++++++++++ studio/backend/models/inference.py | 2 + studio/backend/models/models.py | 1 + studio/backend/routes/inference.py | 268 ++++++++++-- studio/backend/routes/models.py | 18 + studio/backend/utils/models/model_config.py | 50 ++- .../chat/hooks/use-chat-model-runtime.ts | 6 +- .../frontend/src/features/chat/types/api.ts | 3 + .../src/features/chat/types/runtime.ts | 1 + .../frontend/src/hooks/use-hf-model-search.ts | 1 - 13 files changed, 791 insertions(+), 40 deletions(-) create mode 100644 studio/backend/core/inference/llama_cpp.py diff --git a/.gitignore b/.gitignore index 3901c1c699..e24c38c2b0 100755 --- a/.gitignore +++ b/.gitignore @@ -24,6 +24,9 @@ unsloth_training_checkpoints/ *.gguf *.safetensors +# Built binaries (llama-server etc.) +bin/ + # IDE / Editors .vscode/ .idea/ diff --git a/setup.sh b/setup.sh index 1f55e141fd..be17370df4 100755 --- a/setup.sh +++ b/setup.sh @@ -206,7 +206,67 @@ else fi fi -# ── 8. Add shell alias (skip in Colab) ── +# ── 8. Build llama-server for GGUF inference ── +# Builds in an isolated temp directory to avoid conflicts with unsloth-zoo's +# own llama.cpp management (used for GGUF export). Only the llama-server +# binary is extracted to $REPO/bin/. +LLAMA_SERVER_BIN="$SCRIPT_DIR/bin/llama-server" +if [ -f "$LLAMA_SERVER_BIN" ]; then + echo "" + echo "✅ llama-server already exists at $LLAMA_SERVER_BIN" +else + # Check prerequisites + if ! command -v cmake &>/dev/null; then + echo "" + echo "⚠️ cmake not found — skipping llama-server build (GGUF inference won't be available)" + echo " Install cmake and re-run setup.sh to enable GGUF inference." + elif ! command -v git &>/dev/null; then + echo "" + echo "⚠️ git not found — skipping llama-server build (GGUF inference won't be available)" + else + echo "" + echo "Building llama-server for GGUF inference..." + LLAMA_BUILD_TMP=$(mktemp -d) + + BUILD_OK=true + run_quiet "clone llama.cpp" git clone --depth 1 https://github.com/ggml-org/llama.cpp.git "$LLAMA_BUILD_TMP/llama.cpp" || BUILD_OK=false + + if [ "$BUILD_OK" = true ]; then + CMAKE_ARGS="" + if command -v nvcc &>/dev/null; then + echo " Building with CUDA support..." + CMAKE_ARGS="-DGGML_CUDA=ON" + else + echo " Building CPU-only (no CUDA detected)..." + fi + + NCPU=$(nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 4) + + run_quiet "cmake llama.cpp" cmake -S "$LLAMA_BUILD_TMP/llama.cpp" -B "$LLAMA_BUILD_TMP/llama.cpp/build" $CMAKE_ARGS || BUILD_OK=false + fi + + if [ "$BUILD_OK" = true ]; then + run_quiet "build llama-server" cmake --build "$LLAMA_BUILD_TMP/llama.cpp/build" --config Release --target llama-server -j"$NCPU" || BUILD_OK=false + fi + + if [ "$BUILD_OK" = true ]; then + mkdir -p "$SCRIPT_DIR/bin" + if [ -f "$LLAMA_BUILD_TMP/llama.cpp/build/bin/llama-server" ]; then + cp "$LLAMA_BUILD_TMP/llama.cpp/build/bin/llama-server" "$LLAMA_SERVER_BIN" + echo "✅ llama-server built and installed to $LLAMA_SERVER_BIN" + else + echo "⚠️ llama-server binary not found after build — GGUF inference won't be available" + fi + else + echo "⚠️ llama-server build failed — GGUF inference won't be available, but everything else works" + fi + + # Clean up temp build directory + rm -rf "$LLAMA_BUILD_TMP" + fi +fi + +# ── 9. Add shell alias (skip in Colab) ── # Note: venv activation does NOT persist across terminal sessions. # This alias hardcodes the venv python path so users don't need to activate. if [ "$IS_COLAB" = false ]; then diff --git a/studio/backend/core/inference/__init__.py b/studio/backend/core/inference/__init__.py index 494229a087..ff8b75d36a 100644 --- a/studio/backend/core/inference/__init__.py +++ b/studio/backend/core/inference/__init__.py @@ -2,8 +2,10 @@ Inference submodule - Inference backend for model loading and generation """ from .inference import InferenceBackend, get_inference_backend +from .llama_cpp import LlamaCppBackend __all__ = [ 'InferenceBackend', 'get_inference_backend', + 'LlamaCppBackend', ] diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py new file mode 100644 index 0000000000..dae8d20489 --- /dev/null +++ b/studio/backend/core/inference/llama_cpp.py @@ -0,0 +1,414 @@ +""" +llama-server inference backend for GGUF models. + +Manages a llama-server subprocess and proxies chat completions +through its /v1/completions endpoint. +""" +import atexit +import json +import logging +import shutil +import signal +import socket +import subprocess +import threading +import time +from pathlib import Path +from typing import Generator, Optional + +import httpx + +logger = logging.getLogger(__name__) + + +class LlamaCppBackend: + """ + Manages a llama-server subprocess for GGUF model inference. + + Lifecycle: + 1. load_model() — starts llama-server with the GGUF file + 2. generate_chat_completion() — formats prompt, proxies to /v1/completions, streams back + 3. unload_model() — terminates llama-server subprocess + """ + + def __init__(self): + self._process: Optional[subprocess.Popen] = None + self._port: Optional[int] = None + self._model_identifier: Optional[str] = None + self._gguf_path: Optional[str] = None + self._healthy = False + self._lock = threading.Lock() + self._chat_template: Optional[str] = None + + atexit.register(self._cleanup) + + # ── Properties ──────────────────────────────────────────────── + + @property + def is_loaded(self) -> bool: + return self._process is not None and self._healthy + + @property + def base_url(self) -> str: + return f"http://127.0.0.1:{self._port}" + + @property + def model_identifier(self) -> Optional[str]: + return self._model_identifier + + # ── Binary discovery ────────────────────────────────────────── + + @staticmethod + def _find_llama_server_binary() -> Optional[str]: + """ + Locate the llama-server binary. + + Search order: + 1. LLAMA_SERVER_PATH environment variable + 2. ./bin/llama-server (built by setup.sh) + 3. llama-server on PATH (system install) + 4. ./llama.cpp/llama-server (unsloth-zoo build output) + """ + import os + + # 1. Env var + env_path = os.environ.get("LLAMA_SERVER_PATH") + if env_path and Path(env_path).is_file(): + return env_path + + # 2. Project bin/ directory (setup.sh output) + project_root = Path(__file__).resolve().parents[3] # core/inference/ → backend/ → studio/ → root + bin_path = project_root / "bin" / "llama-server" + if bin_path.is_file(): + return str(bin_path) + + # 3. System PATH + system_path = shutil.which("llama-server") + if system_path: + return system_path + + # 4. unsloth-zoo build output (from GGUF export) + llama_cpp_path = project_root / "llama.cpp" / "llama-server" + if llama_cpp_path.is_file(): + return str(llama_cpp_path) + + return None + + # ── Port allocation ─────────────────────────────────────────── + + @staticmethod + def _find_free_port() -> int: + """Find an available TCP port.""" + with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s: + s.bind(("", 0)) + return s.getsockname()[1] + + # ── Lifecycle ───────────────────────────────────────────────── + + def load_model( + self, + gguf_path: str, + model_identifier: str, + n_ctx: int = 4096, + n_gpu_layers: int = -1, + n_threads: Optional[int] = None, + ) -> bool: + """ + Start llama-server with the given GGUF file. + + Args: + gguf_path: Path to the .gguf file + model_identifier: Display identifier for the model + n_ctx: Context window size + n_gpu_layers: Number of layers to offload to GPU (-1 = all) + n_threads: Number of CPU threads (None = auto) + + Returns: + True if server started and health check passed. + """ + with self._lock: + # Kill existing process if any + self._kill_process() + + binary = self._find_llama_server_binary() + if not binary: + raise RuntimeError( + "llama-server binary not found. " + "Run setup.sh to build it, install llama.cpp, " + "or set LLAMA_SERVER_PATH environment variable." + ) + + if not Path(gguf_path).is_file(): + raise FileNotFoundError(f"GGUF file not found: {gguf_path}") + + self._port = self._find_free_port() + cmd = [ + binary, + "-m", gguf_path, + "--port", str(self._port), + "-c", str(n_ctx), + "-ngl", str(n_gpu_layers), + ] + if n_threads is not None: + cmd.extend(["--threads", str(n_threads)]) + + logger.info(f"Starting llama-server: {' '.join(cmd)}") + + self._process = subprocess.Popen( + cmd, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + ) + + self._gguf_path = gguf_path + self._model_identifier = model_identifier + + # Wait for health + if not self._wait_for_health(timeout=120.0): + self._kill_process() + raise RuntimeError( + "llama-server failed to start. " + "Check that the GGUF file is valid and you have enough memory." + ) + + self._healthy = True + + # Try to read chat template from GGUF metadata + self._chat_template = self._read_gguf_chat_template(gguf_path) + + logger.info( + f"llama-server ready on port {self._port} " + f"for model '{model_identifier}'" + ) + return True + + def unload_model(self) -> bool: + """Terminate the llama-server subprocess and clean up state.""" + with self._lock: + self._kill_process() + logger.info(f"Unloaded GGUF model: {self._model_identifier}") + self._model_identifier = None + self._gguf_path = None + self._port = None + self._healthy = False + self._chat_template = None + return True + + def _kill_process(self): + """Terminate the subprocess if running.""" + if self._process is None: + return + try: + self._process.terminate() + self._process.wait(timeout=5) + except subprocess.TimeoutExpired: + logger.warning("llama-server did not exit on SIGTERM, sending SIGKILL") + self._process.kill() + self._process.wait(timeout=5) + except Exception as e: + logger.warning(f"Error killing llama-server process: {e}") + finally: + self._process = None + + def _cleanup(self): + """atexit handler to ensure llama-server is terminated.""" + self._kill_process() + + def _wait_for_health(self, timeout: float = 120.0, interval: float = 0.5) -> bool: + """ + Poll llama-server's /health endpoint until it responds 200. + + Also monitors subprocess for early exit/crash. + """ + deadline = time.monotonic() + timeout + url = f"http://127.0.0.1:{self._port}/health" + + while time.monotonic() < deadline: + # Check if process crashed + if self._process.poll() is not None: + # Read remaining output for error info + output = self._process.stdout.read() if self._process.stdout else "" + logger.error( + f"llama-server exited with code {self._process.returncode}. " + f"Output: {output[:2000]}" + ) + return False + + try: + resp = httpx.get(url, timeout=2.0) + if resp.status_code == 200: + return True + except (httpx.ConnectError, httpx.TimeoutException): + pass + + time.sleep(interval) + + logger.error(f"llama-server health check timed out after {timeout}s") + return False + + # ── Chat template ───────────────────────────────────────────── + + @staticmethod + def _read_gguf_chat_template(gguf_path: str) -> Optional[str]: + """ + Try to read the chat_template from GGUF file metadata. + + Uses the gguf Python library if available. + Returns the Jinja2 template string, or None. + """ + try: + from gguf import GGUFReader + + reader = GGUFReader(gguf_path) + for field_name in reader.fields: + if field_name == "tokenizer.chat_template": + field = reader.fields[field_name] + # Field data is an array of bytes + template_bytes = bytes(field.parts[field.data[0]]) + template = template_bytes.decode("utf-8") + logger.info(f"Read chat template from GGUF metadata ({len(template)} chars)") + return template + except ImportError: + logger.debug("gguf library not available, cannot read chat template from GGUF metadata") + except Exception as e: + logger.warning(f"Could not read chat template from GGUF: {e}") + + return None + + def format_prompt(self, messages: list[dict], system_prompt: str = "") -> str: + """ + Format chat messages into a raw prompt string for /v1/completions. + + Attempts to: + 1. Render the GGUF's embedded chat_template with Jinja2 + 2. Fallback to ChatML format + """ + # Build full message list with system prompt + full_messages = [] + if system_prompt: + full_messages.append({"role": "system", "content": system_prompt}) + full_messages.extend(messages) + + # Try Jinja2 rendering if we have a template + if self._chat_template: + try: + return self._render_jinja_template(full_messages) + except Exception as e: + logger.warning(f"Jinja2 template rendering failed, falling back to ChatML: {e}") + + # Fallback: ChatML format + return self._format_chatml(full_messages) + + def _render_jinja_template(self, messages: list[dict]) -> str: + """Render messages using the GGUF's Jinja2 chat template.""" + from jinja2 import BaseLoader, Environment + + env = Environment(loader=BaseLoader(), keep_trailing_newline=True) + # Add common template globals + env.globals["raise_exception"] = lambda msg: (_ for _ in ()).throw(ValueError(msg)) + + template = env.from_string(self._chat_template) + rendered = template.render( + messages=messages, + add_generation_prompt=True, + bos_token="", + eos_token="", + ) + return rendered + + @staticmethod + def _format_chatml(messages: list[dict]) -> str: + """Format messages using ChatML template (universal fallback).""" + parts = [] + for msg in messages: + role = msg.get("role", "user") + content = msg.get("content", "") + parts.append(f"<|im_start|>{role}\n{content}<|im_end|>") + parts.append("<|im_start|>assistant") + return "\n".join(parts) + "\n" + + # ── Generation (proxy to llama-server) ──────────────────────── + + def generate_chat_completion( + self, + prompt: str, + temperature: float = 0.7, + top_p: float = 0.9, + top_k: int = 40, + min_p: float = 0.0, + max_tokens: int = 512, + repetition_penalty: float = 1.1, + stop: Optional[list[str]] = None, + cancel_event: Optional[threading.Event] = None, + ) -> Generator[str, None, None]: + """ + Send a completion request to llama-server and stream tokens back. + + Uses /v1/completions (NOT /v1/chat/completions) so we control + the prompt format entirely. + + Yields cumulative text (matching InferenceBackend's convention). + """ + if not self.is_loaded: + raise RuntimeError("llama-server is not loaded") + + payload = { + "prompt": prompt, + "stream": True, + "temperature": temperature, + "top_p": top_p, + "top_k": top_k if top_k >= 0 else 0, + "min_p": min_p, + "n_predict": max_tokens, + "repeat_penalty": repetition_penalty, + } + if stop: + payload["stop"] = stop + + url = f"{self.base_url}/v1/completions" + cumulative = "" + + try: + with httpx.Client(timeout=None) as client: + with client.stream("POST", url, json=payload) as response: + if response.status_code != 200: + error_body = response.read().decode() + raise RuntimeError( + f"llama-server returned {response.status_code}: {error_body}" + ) + + buffer = "" + for raw_chunk in response.iter_text(): + if cancel_event is not None and cancel_event.is_set(): + break + + buffer += raw_chunk + while "\n" in buffer: + line, buffer = buffer.split("\n", 1) + line = line.strip() + + if not line: + continue + if line == "data: [DONE]": + return + if not line.startswith("data: "): + continue + + try: + data = json.loads(line[6:]) + choices = data.get("choices", []) + if choices: + token = choices[0].get("text", "") + if token: + cumulative += token + yield cumulative + except json.JSONDecodeError: + logger.debug(f"Skipping malformed SSE line: {line[:100]}") + + except httpx.ConnectError: + raise RuntimeError("Lost connection to llama-server") + except Exception as e: + if cancel_event is not None and cancel_event.is_set(): + return + raise diff --git a/studio/backend/models/inference.py b/studio/backend/models/inference.py index d2d98d7944..c4a062ae3a 100644 --- a/studio/backend/models/inference.py +++ b/studio/backend/models/inference.py @@ -43,6 +43,7 @@ class LoadResponse(BaseModel): display_name: str = Field(..., description="Display name of the model") is_vision: bool = Field(False, description="Whether model is a vision model") is_lora: bool = Field(False, description="Whether model is a LoRA adapter") + is_gguf: bool = Field(False, description="Whether model is a GGUF model (llama.cpp)") inference: dict = Field(..., description="Inference parameters (temperature, top_p, top_k, min_p)") @@ -56,6 +57,7 @@ class InferenceStatusResponse(BaseModel): """Current inference backend status""" active_model: Optional[str] = Field(None, description="Currently active model identifier") is_vision: bool = Field(False, description="Whether the active model is a vision model") + is_gguf: bool = Field(False, description="Whether the active model is a GGUF model (llama.cpp)") loading: List[str] = Field(default_factory=list, description="Models currently being loaded") loaded: List[str] = Field(default_factory=list, description="Models currently loaded") diff --git a/studio/backend/models/models.py b/studio/backend/models/models.py index 5542db76d5..b3f9b50ed2 100644 --- a/studio/backend/models/models.py +++ b/studio/backend/models/models.py @@ -53,6 +53,7 @@ class ModelDetails(BaseModel): config: Optional[Dict[str, Any]] = Field(None, description="Model configuration dictionary") is_vision: bool = Field(False, description="Whether model is a vision model") is_lora: bool = Field(False, description="Whether model is a LoRA adapter") + is_gguf: bool = Field(False, description="Whether model is a GGUF model (llama.cpp format)") base_model: Optional[str] = Field(None, description="Base model if this is a LoRA adapter") diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index c30a1638d4..156578b960 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -23,6 +23,7 @@ if str(backend_path) not in sys.path: # Import backend functions try: from core.inference import get_inference_backend + from core.inference.llama_cpp import LlamaCppBackend from utils.models import ModelConfig from utils.inference import load_inference_config except ImportError: @@ -30,6 +31,7 @@ except ImportError: if str(parent_backend) not in sys.path: sys.path.insert(0, str(parent_backend)) from core.inference import get_inference_backend + from core.inference.llama_cpp import LlamaCppBackend from utils.models import ModelConfig from utils.inference import load_inference_config @@ -61,60 +63,111 @@ if not logger.handlers: logger.addHandler(handler) logger.setLevel(logging.INFO) +# GGUF inference backend (llama-server) +_llama_cpp_backend = LlamaCppBackend() + +def get_llama_cpp_backend() -> LlamaCppBackend: + return _llama_cpp_backend + @router.post("/load", response_model=LoadResponse) async def load_model(request: LoadRequest): """ Load a model for inference. - + The model_path should be a clean identifier from GET /models/list. Returns inference configuration parameters (temperature, top_p, top_k, min_p) from the model's YAML config, falling back to default.yaml for missing values. + + GGUF models are loaded via llama-server (llama.cpp) instead of Unsloth. """ try: - backend = get_inference_backend() - # Create config using clean factory method # is_lora is auto-detected from adapter_config.json on disk/HF config = ModelConfig.from_identifier( model_id=request.model_path, hf_token=request.hf_token, ) - + if not config: raise HTTPException( status_code=400, detail=f"Invalid model identifier: {request.model_path}" ) - - # Load the model + + # ── GGUF path: load via llama-server ────────────────────── + if config.is_gguf: + llama_backend = get_llama_cpp_backend() + unsloth_backend = get_inference_backend() + + # Unload any active Unsloth model first to free VRAM + if unsloth_backend.active_model_name: + logger.info(f"Unloading Unsloth model '{unsloth_backend.active_model_name}' before loading GGUF") + unsloth_backend.unload_model(unsloth_backend.active_model_name) + + success = llama_backend.load_model( + gguf_path=config.gguf_file, + model_identifier=config.identifier, + n_ctx=request.max_seq_length, + ) + + if not success: + raise HTTPException( + status_code=500, + detail=f"Failed to load GGUF model: {config.display_name}" + ) + + logger.info(f"Loaded GGUF model via llama-server: {config.identifier}") + + inference_config = load_inference_config(config.identifier) + + return LoadResponse( + status="loaded", + model=config.identifier, + display_name=config.display_name, + is_vision=False, + is_lora=False, + is_gguf=True, + inference=inference_config, + ) + + # ── Standard path: load via Unsloth/transformers ────────── + backend = get_inference_backend() + + # Unload any active GGUF model first + llama_backend = get_llama_cpp_backend() + if llama_backend.is_loaded: + logger.info("Unloading GGUF model before loading Unsloth model") + llama_backend.unload_model() + success = backend.load_model( config=config, max_seq_length=request.max_seq_length, load_in_4bit=request.load_in_4bit, hf_token=request.hf_token, ) - + if not success: raise HTTPException( status_code=500, detail=f"Failed to load model: {config.display_name}" ) - + logger.info(f"Loaded model: {config.identifier}") - + # Load inference configuration parameters inference_config = load_inference_config(config.identifier) - + return LoadResponse( status="loaded", model=config.identifier, display_name=config.display_name, is_vision=config.is_vision, is_lora=config.is_lora, + is_gguf=False, inference=inference_config, ) - + except HTTPException: raise except Exception as e: @@ -129,13 +182,22 @@ async def load_model(request: LoadRequest): async def unload_model(request: UnloadRequest): """ Unload a model from memory. + Routes to the correct backend (llama-server for GGUF, Unsloth otherwise). """ try: + # Check if the GGUF backend has this model loaded + llama_backend = get_llama_cpp_backend() + if llama_backend.is_loaded and llama_backend.model_identifier == request.model_path: + llama_backend.unload_model() + logger.info(f"Unloaded GGUF model: {request.model_path}") + return UnloadResponse(status="unloaded", model=request.model_path) + + # Otherwise, unload from Unsloth backend backend = get_inference_backend() backend.unload_model(request.model_path) logger.info(f"Unloaded model: {request.model_path}") return UnloadResponse(status="unloaded", model=request.model_path) - + except Exception as e: logger.error(f"Error unloading model: {e}", exc_info=True) raise HTTPException( @@ -221,22 +283,37 @@ async def generate_stream(request: GenerateRequest): async def get_status(): """ Get current inference backend status. + Reports whichever backend (Unsloth or llama-server) is currently active. """ try: + llama_backend = get_llama_cpp_backend() + + # If a GGUF model is loaded via llama-server, report that + if llama_backend.is_loaded: + return InferenceStatusResponse( + active_model=llama_backend.model_identifier, + is_vision=False, + is_gguf=True, + loading=[], + loaded=[llama_backend.model_identifier], + ) + + # Otherwise, report Unsloth backend status backend = get_inference_backend() - + is_vision = False if backend.active_model_name: model_info = backend.models.get(backend.active_model_name, {}) is_vision = model_info.get("is_vision", False) - + return InferenceStatusResponse( active_model=backend.active_model_name, is_vision=is_vision, + is_gguf=False, loading=list(getattr(backend, 'loading_models', set())), loaded=list(backend.models.keys()), ) - + except Exception as e: logger.error(f"Error getting status: {e}", exc_info=True) raise HTTPException( @@ -315,29 +392,157 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque Streaming (default): returns SSE chunks matching OpenAI's format. Non-streaming: returns a single ChatCompletion JSON object. - """ - backend = get_inference_backend() - if not backend.active_model_name: - raise HTTPException( - status_code=400, - detail="No model loaded. Call POST /inference/load first.", - ) + Automatically routes to the correct backend: + - GGUF models → llama-server via LlamaCppBackend + - Other models → Unsloth/transformers via InferenceBackend + """ + llama_backend = get_llama_cpp_backend() + using_gguf = llama_backend.is_loaded + + # ── Determine which backend is active ───────────────────── + if using_gguf: + model_name = llama_backend.model_identifier or payload.model + else: + backend = get_inference_backend() + if not backend.active_model_name: + raise HTTPException( + status_code=400, + detail="No model loaded. Call POST /inference/load first.", + ) + model_name = backend.active_model_name or payload.model # ── Parse messages (handles multimodal content parts) ───── system_prompt, chat_messages, extracted_image_b64 = _extract_content_parts( payload.messages ) - # If no non-system messages were provided, error out if not chat_messages: raise HTTPException( status_code=400, detail="At least one non-system message is required.", ) - # ── Decode image (from content parts OR legacy field) ───── - # Content-part images take priority; fall back to legacy field + # ── GGUF path: format prompt → proxy to llama-server ────── + if using_gguf: + # GGUF models don't support vision + image_b64 = extracted_image_b64 or payload.image_base64 + if image_b64: + raise HTTPException( + status_code=400, + detail="Image provided but GGUF models do not support vision.", + ) + + prompt = llama_backend.format_prompt(chat_messages, system_prompt) + cancel_event = threading.Event() + + completion_id = f"chatcmpl-{uuid.uuid4().hex[:12]}" + created = int(time.time()) + + def gguf_generate(): + return llama_backend.generate_chat_completion( + prompt=prompt, + temperature=payload.temperature, + top_p=payload.top_p, + top_k=payload.top_k, + min_p=payload.min_p, + max_tokens=payload.max_tokens or 512, + repetition_penalty=payload.repetition_penalty, + cancel_event=cancel_event, + ) + + if payload.stream: + async def gguf_stream_chunks(): + try: + # First chunk: role + first_chunk = ChatCompletionChunk( + id=completion_id, + created=created, + model=model_name, + choices=[ChunkChoice( + delta=ChoiceDelta(role="assistant"), + finish_reason=None, + )], + ) + yield f"data: {first_chunk.model_dump_json(exclude_none=True)}\n\n" + + # Content chunks — llama backend yields cumulative text + prev_text = "" + for cumulative in gguf_generate(): + if await request.is_disconnected(): + cancel_event.set() + return + new_text = cumulative[len(prev_text):] + prev_text = cumulative + if not new_text: + continue + chunk = ChatCompletionChunk( + id=completion_id, + created=created, + model=model_name, + choices=[ChunkChoice( + delta=ChoiceDelta(content=new_text), + finish_reason=None, + )], + ) + yield f"data: {chunk.model_dump_json(exclude_none=True)}\n\n" + + # Final chunk + final_chunk = ChatCompletionChunk( + id=completion_id, + created=created, + model=model_name, + choices=[ChunkChoice( + delta=ChoiceDelta(), + finish_reason="stop", + )], + ) + yield f"data: {final_chunk.model_dump_json(exclude_none=True)}\n\n" + yield "data: [DONE]\n\n" + + except asyncio.CancelledError: + cancel_event.set() + raise + except Exception as e: + logger.error(f"Error during GGUF streaming: {e}", exc_info=True) + error_chunk = { + "error": {"message": str(e), "type": "server_error"}, + } + yield f"data: {json.dumps(error_chunk)}\n\n" + + return StreamingResponse( + gguf_stream_chunks(), + media_type="text/event-stream", + headers={ + "Cache-Control": "no-cache", + "Connection": "keep-alive", + "X-Accel-Buffering": "no", + }, + ) + else: + try: + full_text = "" + for token in gguf_generate(): + full_text = token + + response = ChatCompletion( + id=completion_id, + created=created, + model=model_name, + choices=[CompletionChoice( + message=CompletionMessage(content=full_text), + finish_reason="stop", + )], + ) + return JSONResponse(content=response.model_dump()) + + except Exception as e: + logger.error(f"Error during GGUF completion: {e}", exc_info=True) + raise HTTPException(status_code=500, detail=str(e)) + + # ── Standard Unsloth path ───────────────────────────────── + + # Decode image (from content parts OR legacy field) image_b64 = extracted_image_b64 or payload.image_base64 image = None @@ -363,7 +568,7 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque except Exception as e: raise HTTPException(status_code=400, detail=f"Failed to decode image: {e}") - # ── Shared generation kwargs ────────────────────────────── + # Shared generation kwargs gen_kwargs = dict( messages=chat_messages, system_prompt=system_prompt, @@ -376,11 +581,10 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque repetition_penalty=payload.repetition_penalty, ) - # ── Choose generation path (adapter-controlled or standard) ── + # Choose generation path (adapter-controlled or standard) cancel_event = threading.Event() if payload.use_adapter is not None: - # Compare mode: toggle adapter state atomically with generation def generate(): return backend.generate_with_adapter_control( use_adapter=payload.use_adapter, @@ -388,11 +592,9 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque **gen_kwargs, ) else: - # Standard path: no adapter toggling def generate(): return backend.generate_chat_response(cancel_event=cancel_event, **gen_kwargs) - model_name = backend.active_model_name or payload.model completion_id = f"chatcmpl-{uuid.uuid4().hex[:12]}" created = int(time.time()) @@ -400,7 +602,6 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque if payload.stream: async def stream_chunks(): try: - # First chunk: send the role first_chunk = ChatCompletionChunk( id=completion_id, created=created, @@ -412,8 +613,6 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque ) yield f"data: {first_chunk.model_dump_json(exclude_none=True)}\n\n" - # Content chunks — generate_chat_response yields cumulative - # text, so we diff to get incremental deltas. prev_text = "" for cumulative in generate(): if await request.is_disconnected(): @@ -435,7 +634,6 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque ) yield f"data: {chunk.model_dump_json(exclude_none=True)}\n\n" - # Final chunk: finish_reason = stop final_chunk = ChatCompletionChunk( id=completion_id, created=created, @@ -475,7 +673,7 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque try: full_text = "" for token in generate(): - full_text = token # generate_stream yields cumulative text + full_text = token response = ChatCompletion( id=completion_id, diff --git a/studio/backend/routes/models.py b/studio/backend/routes/models.py index 761c04d3e7..881f399c3d 100644 --- a/studio/backend/routes/models.py +++ b/studio/backend/routes/models.py @@ -88,6 +88,7 @@ def _scan_models_dir(models_dir: Path) -> List[LocalModelInfo]: or (child / "adapter_config.json").exists() or any(child.glob("*.safetensors")) or any(child.glob("*.bin")) + or any(child.glob("*.gguf")) ) if not has_model_files: continue @@ -104,6 +105,23 @@ def _scan_models_dir(models_dir: Path) -> List[LocalModelInfo]: updated_at=updated_at, ), ) + # Also scan for standalone .gguf files directly in the models directory + for gguf_file in models_dir.glob("*.gguf"): + if gguf_file.is_file(): + try: + updated_at = gguf_file.stat().st_mtime + except OSError: + updated_at = None + found.append( + LocalModelInfo( + id=str(gguf_file), + display_name=gguf_file.stem, + path=str(gguf_file), + source="models_dir", + updated_at=updated_at, + ), + ) + return found diff --git a/studio/backend/utils/models/model_config.py b/studio/backend/utils/models/model_config.py index fdf89fce39..27fbe3ee44 100644 --- a/studio/backend/utils/models/model_config.py +++ b/studio/backend/utils/models/model_config.py @@ -422,6 +422,32 @@ def is_vision_model(model_name: str, hf_token: Optional[str] = None) -> bool: pass +def detect_gguf_model(path: str) -> Optional[str]: + """ + Check if the given path is or contains a GGUF model file. + + Handles three cases: + 1. path is a direct .gguf file path + 2. path is a directory containing .gguf files + 3. path is an HuggingFace repo with GGUF files (not yet — future enhancement) + + Returns the full path to the .gguf file if found, None otherwise. + """ + p = Path(path) + + # Case 1: direct .gguf file + if p.suffix == ".gguf" and p.is_file(): + return str(p.resolve()) + + # Case 2: directory containing .gguf files + if p.is_dir(): + gguf_files = sorted(p.glob("*.gguf"), key=lambda f: f.stat().st_size, reverse=True) + if gguf_files: + return str(gguf_files[0].resolve()) + + return None + + def scan_trained_loras(outputs_dir: str = "./outputs") -> List[Tuple[str, str]]: """ Scan outputs folder for trained LoRA adapters. @@ -595,6 +621,8 @@ class ModelConfig: is_cached: bool # Is this already in HF cache? is_vision: bool # Is this a vision model? is_lora: bool # Is this a lora adapter? + is_gguf: bool = False # Is this a GGUF model? + gguf_file: Optional[str] = None # Full path to the .gguf file base_model: Optional[str] = None # Base model (for LoRAs) @classmethod @@ -675,12 +703,30 @@ class ModelConfig: identifier = model_id.strip() is_local = is_local_path(identifier) path = normalize_path(identifier) if is_local else identifier - + # Add unsloth/ prefix for shorthand HF models if not is_local and "/" not in identifier: identifier = f"unsloth/{identifier}" path = identifier - + + # Auto-detect GGUF models (check before LoRA/vision detection) + if is_local: + gguf_file = detect_gguf_model(path) + if gguf_file: + display_name = Path(gguf_file).stem + logger.info(f"Detected GGUF model: {gguf_file}") + return cls( + identifier=identifier, + display_name=display_name, + path=path, + is_local=True, + is_cached=True, + is_vision=False, + is_lora=False, + is_gguf=True, + gguf_file=gguf_file, + ) + # Auto-detect LoRA for local paths (check adapter_config.json on disk) if not is_lora and is_local: detected_base = get_base_model_from_lora(path) diff --git a/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts b/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts index 3dca94d7f5..030fe55bff 100644 --- a/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts +++ b/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts @@ -41,11 +41,13 @@ function stripTrailingEpoch(input: string): string { function describeModel(model: { is_lora?: boolean; is_vision?: boolean; + is_gguf?: boolean; }): string | undefined { const tags: string[] = []; + if (model.is_gguf) tags.push("GGUF"); if (model.is_lora) tags.push("LoRA"); if (model.is_vision) tags.push("Vision"); - if (!model.is_lora && !model.is_vision) tags.push("Base"); + if (!model.is_lora && !model.is_vision && !model.is_gguf) tags.push("Base"); return tags.join(" · "); } @@ -54,6 +56,7 @@ function toChatModelSummary(model: { name?: string | null; is_lora?: boolean; is_vision?: boolean; + is_gguf?: boolean; }): ChatModelSummary { return { id: model.id, @@ -61,6 +64,7 @@ function toChatModelSummary(model: { description: describeModel(model), isLora: Boolean(model.is_lora), isVision: Boolean(model.is_vision), + isGguf: Boolean(model.is_gguf), }; } diff --git a/studio/frontend/src/features/chat/types/api.ts b/studio/frontend/src/features/chat/types/api.ts index cadcad152d..bc9268e2c2 100644 --- a/studio/frontend/src/features/chat/types/api.ts +++ b/studio/frontend/src/features/chat/types/api.ts @@ -3,6 +3,7 @@ export interface BackendModelDetails { name?: string | null; is_vision?: boolean; is_lora?: boolean; + is_gguf?: boolean; } export interface ListModelsResponse { @@ -35,6 +36,7 @@ export interface LoadModelResponse { display_name: string; is_vision: boolean; is_lora: boolean; + is_gguf?: boolean; inference?: { temperature?: number; top_p?: number; @@ -50,6 +52,7 @@ export interface UnloadModelRequest { export interface InferenceStatusResponse { active_model: string | null; is_vision: boolean; + is_gguf?: boolean; loading: string[]; loaded: string[]; } diff --git a/studio/frontend/src/features/chat/types/runtime.ts b/studio/frontend/src/features/chat/types/runtime.ts index 47478e5aae..9f392fe08d 100644 --- a/studio/frontend/src/features/chat/types/runtime.ts +++ b/studio/frontend/src/features/chat/types/runtime.ts @@ -26,6 +26,7 @@ export interface ChatModelSummary { description?: string; isVision: boolean; isLora: boolean; + isGguf?: boolean; } export interface ChatLoraSummary { diff --git a/studio/frontend/src/hooks/use-hf-model-search.ts b/studio/frontend/src/hooks/use-hf-model-search.ts index 6ba70a4d5c..0029f13317 100644 --- a/studio/frontend/src/hooks/use-hf-model-search.ts +++ b/studio/frontend/src/hooks/use-hf-model-search.ts @@ -11,7 +11,6 @@ export interface HfModelResult { } const EXCLUDED_TAGS = new Set([ - "gguf", "gptq", "awq", "exl2", From 70c912d788859c448747cefe5532c69ea119a424 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Tue, 24 Feb 2026 17:45:17 +0400 Subject: [PATCH 05/49] Fix CUDA detection for llama-server build on multi-GPU machines --- setup.sh | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/setup.sh b/setup.sh index be17370df4..68c2d010da 100755 --- a/setup.sh +++ b/setup.sh @@ -233,9 +233,25 @@ else if [ "$BUILD_OK" = true ]; then CMAKE_ARGS="" + # Detect CUDA: check nvcc on PATH, then common install locations + NVCC_PATH="" if command -v nvcc &>/dev/null; then - echo " Building with CUDA support..." + NVCC_PATH="$(command -v nvcc)" + elif [ -x /usr/local/cuda/bin/nvcc ]; then + NVCC_PATH="/usr/local/cuda/bin/nvcc" + export PATH="/usr/local/cuda/bin:$PATH" + elif ls /usr/local/cuda-*/bin/nvcc &>/dev/null 2>&1; then + # Pick the newest cuda-XX.X directory + NVCC_PATH="$(ls -d /usr/local/cuda-*/bin/nvcc 2>/dev/null | sort -V | tail -1)" + export PATH="$(dirname "$NVCC_PATH"):$PATH" + fi + + if [ -n "$NVCC_PATH" ]; then + echo " Building with CUDA support (nvcc: $NVCC_PATH)..." CMAKE_ARGS="-DGGML_CUDA=ON" + elif [ -d /usr/local/cuda ] || nvidia-smi &>/dev/null; then + echo " CUDA driver detected but nvcc not found — building CPU-only" + echo " To enable GPU: install cuda-toolkit or add nvcc to PATH" else echo " Building CPU-only (no CUDA detected)..." fi From a900eb9ad70479a373d59ce2f28825b35a6c5fa6 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Tue, 24 Feb 2026 17:49:09 +0400 Subject: [PATCH 06/49] Fix GGUF detection for HuggingFace repo IDs (not just local paths) --- studio/backend/utils/models/model_config.py | 101 +++++++++++++++++++- 1 file changed, 97 insertions(+), 4 deletions(-) diff --git a/studio/backend/utils/models/model_config.py b/studio/backend/utils/models/model_config.py index 27fbe3ee44..53ac28be65 100644 --- a/studio/backend/utils/models/model_config.py +++ b/studio/backend/utils/models/model_config.py @@ -424,14 +424,14 @@ pass def detect_gguf_model(path: str) -> Optional[str]: """ - Check if the given path is or contains a GGUF model file. + Check if the given local path is or contains a GGUF model file. - Handles three cases: + Handles two cases: 1. path is a direct .gguf file path 2. path is a directory containing .gguf files - 3. path is an HuggingFace repo with GGUF files (not yet — future enhancement) Returns the full path to the .gguf file if found, None otherwise. + For HuggingFace repo detection, use detect_gguf_model_remote() instead. """ p = Path(path) @@ -448,6 +448,76 @@ def detect_gguf_model(path: str) -> Optional[str]: return None +# Preferred GGUF quantization levels, in descending priority. +# Q4_K_M is a good default: small, fast, acceptable quality. +_GGUF_QUANT_PREFERENCE = [ + "Q4_K_M", "Q4_K_S", "Q5_K_M", "Q5_K_S", + "Q6_K", "Q8_0", "Q3_K_M", "Q3_K_L", "Q2_K", + "F16", "BF16", "F32", +] + + +def _pick_best_gguf(filenames: list[str]) -> Optional[str]: + """ + Pick the best GGUF file from a list of filenames. + + Prefers quantization levels in _GGUF_QUANT_PREFERENCE order. + Falls back to the first .gguf file found. + """ + gguf_files = [f for f in filenames if f.endswith(".gguf")] + if not gguf_files: + return None + + # Try preferred quantization levels + for quant in _GGUF_QUANT_PREFERENCE: + for f in gguf_files: + if quant in f: + return f + + # Fallback: first GGUF file + return gguf_files[0] + + +def detect_gguf_model_remote( + repo_id: str, + hf_token: Optional[str] = None, +) -> Optional[str]: + """ + Check if a HuggingFace repo contains GGUF files. + + Returns the filename of the best GGUF file in the repo, or None. + """ + try: + from huggingface_hub import model_info as hf_model_info + + info = hf_model_info(repo_id, token=hf_token) + repo_files = [s.rfilename for s in info.siblings] + return _pick_best_gguf(repo_files) + except Exception as e: + logger.debug(f"Could not check GGUF files for '{repo_id}': {e}") + return None + + +def download_gguf_file( + repo_id: str, + filename: str, + hf_token: Optional[str] = None, +) -> str: + """ + Download a specific GGUF file from a HuggingFace repo. + + Returns the local path to the downloaded file. + """ + from huggingface_hub import hf_hub_download + + local_path = hf_hub_download( + repo_id=repo_id, + filename=filename, + token=hf_token, + ) + return local_path + + def scan_trained_loras(outputs_dir: str = "./outputs") -> List[Tuple[str, str]]: """ Scan outputs folder for trained LoRA adapters. @@ -714,7 +784,7 @@ class ModelConfig: gguf_file = detect_gguf_model(path) if gguf_file: display_name = Path(gguf_file).stem - logger.info(f"Detected GGUF model: {gguf_file}") + logger.info(f"Detected local GGUF model: {gguf_file}") return cls( identifier=identifier, display_name=display_name, @@ -726,6 +796,29 @@ class ModelConfig: is_gguf=True, gguf_file=gguf_file, ) + else: + # Check if the HF repo contains GGUF files + gguf_filename = detect_gguf_model_remote(identifier, hf_token=hf_token) + if gguf_filename: + logger.info(f"Detected remote GGUF repo '{identifier}', file: {gguf_filename}") + logger.info(f"Downloading GGUF file '{gguf_filename}' from '{identifier}'...") + local_gguf_path = download_gguf_file( + repo_id=identifier, + filename=gguf_filename, + hf_token=hf_token, + ) + display_name = Path(gguf_filename).stem + return cls( + identifier=identifier, + display_name=display_name, + path=local_gguf_path, + is_local=False, + is_cached=True, + is_vision=False, + is_lora=False, + is_gguf=True, + gguf_file=local_gguf_path, + ) # Auto-detect LoRA for local paths (check adapter_config.json on disk) if not is_lora and is_local: From 4e88092452e1122c014b86735da5e6cd74d08c91 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Tue, 24 Feb 2026 18:02:43 +0400 Subject: [PATCH 07/49] Preflight llama-server check before downloading remote GGUF files --- studio/backend/utils/models/model_config.py | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/studio/backend/utils/models/model_config.py b/studio/backend/utils/models/model_config.py index 53ac28be65..601ea75770 100644 --- a/studio/backend/utils/models/model_config.py +++ b/studio/backend/utils/models/model_config.py @@ -800,6 +800,15 @@ class ModelConfig: # Check if the HF repo contains GGUF files gguf_filename = detect_gguf_model_remote(identifier, hf_token=hf_token) if gguf_filename: + # Preflight: verify llama-server binary exists before downloading + # a potentially multi-GB GGUF file + from core.inference.llama_cpp import LlamaCppBackend + if not LlamaCppBackend._find_llama_server_binary(): + raise RuntimeError( + "llama-server binary not found — cannot load GGUF models. " + "Run setup.sh to build it, or set LLAMA_SERVER_PATH." + ) + logger.info(f"Detected remote GGUF repo '{identifier}', file: {gguf_filename}") logger.info(f"Downloading GGUF file '{gguf_filename}' from '{identifier}'...") local_gguf_path = download_gguf_file( From 08aeeaee4b134a3138b470b50ef08df7e045bb63 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Tue, 24 Feb 2026 18:19:29 +0400 Subject: [PATCH 08/49] Fix llama-server: build in-tree, fix path resolution, add LD_LIBRARY_PATH --- .gitignore | 3 ++ setup.sh | 32 +++++++++++---------- studio/backend/core/inference/llama_cpp.py | 33 ++++++++++++++-------- 3 files changed, 42 insertions(+), 26 deletions(-) diff --git a/.gitignore b/.gitignore index e24c38c2b0..2ede66ec5b 100755 --- a/.gitignore +++ b/.gitignore @@ -24,6 +24,9 @@ unsloth_training_checkpoints/ *.gguf *.safetensors +# llama.cpp build (built by setup.sh, shared with unsloth-zoo export) +llama.cpp/ + # Built binaries (llama-server etc.) bin/ diff --git a/setup.sh b/setup.sh index 68c2d010da..36165dac65 100755 --- a/setup.sh +++ b/setup.sh @@ -207,10 +207,10 @@ else fi # ── 8. Build llama-server for GGUF inference ── -# Builds in an isolated temp directory to avoid conflicts with unsloth-zoo's -# own llama.cpp management (used for GGUF export). Only the llama-server -# binary is extracted to $REPO/bin/. -LLAMA_SERVER_BIN="$SCRIPT_DIR/bin/llama-server" +# Builds in-tree at $REPO/llama.cpp/. This directory is shared with +# unsloth-zoo's GGUF export pipeline — if converter/quantize are missing, +# unsloth-zoo will rebuild them on first export. We only build llama-server here. +LLAMA_SERVER_BIN="$SCRIPT_DIR/llama.cpp/build/bin/llama-server" if [ -f "$LLAMA_SERVER_BIN" ]; then echo "" echo "✅ llama-server already exists at $LLAMA_SERVER_BIN" @@ -226,10 +226,17 @@ else else echo "" echo "Building llama-server for GGUF inference..." - LLAMA_BUILD_TMP=$(mktemp -d) + LLAMA_CPP_DIR="$SCRIPT_DIR/llama.cpp" BUILD_OK=true - run_quiet "clone llama.cpp" git clone --depth 1 https://github.com/ggml-org/llama.cpp.git "$LLAMA_BUILD_TMP/llama.cpp" || BUILD_OK=false + if [ -d "$LLAMA_CPP_DIR/.git" ]; then + echo " llama.cpp repo already cloned, pulling latest..." + run_quiet "pull llama.cpp" git -C "$LLAMA_CPP_DIR" pull || true + else + # Remove any non-git llama.cpp directory (stale build artifacts) + rm -rf "$LLAMA_CPP_DIR" + run_quiet "clone llama.cpp" git clone --depth 1 https://github.com/ggml-org/llama.cpp.git "$LLAMA_CPP_DIR" || BUILD_OK=false + fi if [ "$BUILD_OK" = true ]; then CMAKE_ARGS="" @@ -258,27 +265,22 @@ else NCPU=$(nproc 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || echo 4) - run_quiet "cmake llama.cpp" cmake -S "$LLAMA_BUILD_TMP/llama.cpp" -B "$LLAMA_BUILD_TMP/llama.cpp/build" $CMAKE_ARGS || BUILD_OK=false + run_quiet "cmake llama.cpp" cmake -S "$LLAMA_CPP_DIR" -B "$LLAMA_CPP_DIR/build" $CMAKE_ARGS || BUILD_OK=false fi if [ "$BUILD_OK" = true ]; then - run_quiet "build llama-server" cmake --build "$LLAMA_BUILD_TMP/llama.cpp/build" --config Release --target llama-server -j"$NCPU" || BUILD_OK=false + run_quiet "build llama-server" cmake --build "$LLAMA_CPP_DIR/build" --config Release --target llama-server -j"$NCPU" || BUILD_OK=false fi if [ "$BUILD_OK" = true ]; then - mkdir -p "$SCRIPT_DIR/bin" - if [ -f "$LLAMA_BUILD_TMP/llama.cpp/build/bin/llama-server" ]; then - cp "$LLAMA_BUILD_TMP/llama.cpp/build/bin/llama-server" "$LLAMA_SERVER_BIN" - echo "✅ llama-server built and installed to $LLAMA_SERVER_BIN" + if [ -f "$LLAMA_SERVER_BIN" ]; then + echo "✅ llama-server built at $LLAMA_SERVER_BIN" else echo "⚠️ llama-server binary not found after build — GGUF inference won't be available" fi else echo "⚠️ llama-server build failed — GGUF inference won't be available, but everything else works" fi - - # Clean up temp build directory - rm -rf "$LLAMA_BUILD_TMP" fi fi diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index dae8d20489..1489365686 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -65,9 +65,9 @@ class LlamaCppBackend: Search order: 1. LLAMA_SERVER_PATH environment variable - 2. ./bin/llama-server (built by setup.sh) + 2. ./llama.cpp/build/bin/llama-server (built by setup.sh in-tree) 3. llama-server on PATH (system install) - 4. ./llama.cpp/llama-server (unsloth-zoo build output) + 4. ./bin/llama-server (legacy: extracted binary) """ import os @@ -76,21 +76,23 @@ class LlamaCppBackend: if env_path and Path(env_path).is_file(): return env_path - # 2. Project bin/ directory (setup.sh output) - project_root = Path(__file__).resolve().parents[3] # core/inference/ → backend/ → studio/ → root - bin_path = project_root / "bin" / "llama-server" - if bin_path.is_file(): - return str(bin_path) + # Project root: llama_cpp.py → inference/ → core/ → backend/ → studio/ → root + project_root = Path(__file__).resolve().parents[4] + + # 2. In-tree llama.cpp build (setup.sh builds here) + build_path = project_root / "llama.cpp" / "build" / "bin" / "llama-server" + if build_path.is_file(): + return str(build_path) # 3. System PATH system_path = shutil.which("llama-server") if system_path: return system_path - # 4. unsloth-zoo build output (from GGUF export) - llama_cpp_path = project_root / "llama.cpp" / "llama-server" - if llama_cpp_path.is_file(): - return str(llama_cpp_path) + # 4. Legacy: extracted to bin/ + bin_path = project_root / "bin" / "llama-server" + if bin_path.is_file(): + return str(bin_path) return None @@ -154,11 +156,20 @@ class LlamaCppBackend: logger.info(f"Starting llama-server: {' '.join(cmd)}") + # Set LD_LIBRARY_PATH so llama-server can find its shared libs + # (libmtmd.so, libllama.so, etc.) which live next to the binary + import os + env = os.environ.copy() + binary_dir = str(Path(binary).parent) + existing_ld = env.get("LD_LIBRARY_PATH", "") + env["LD_LIBRARY_PATH"] = f"{binary_dir}:{existing_ld}" if existing_ld else binary_dir + self._process = subprocess.Popen( cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, + env=env, ) self._gguf_path = gguf_path From ef1cd3ac983b2396d526b031e3c2a3d97dba0331 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Tue, 24 Feb 2026 19:03:06 +0400 Subject: [PATCH 09/49] Use llama-server -hf mode, add GGUF variant selector, fix vision detection Replace Python-side GGUF download with llama-server's native -hf flag for HuggingFace repos. Add frontend variant picker so users can choose quantization (Q4_K_M, Q8_0, BF16, etc.) with file sizes. Fix vision detection via mmproj files instead of hardcoding is_vision=False. --- studio/backend/core/inference/llama_cpp.py | 90 +++++--- studio/backend/models/inference.py | 1 + studio/backend/models/models.py | 15 ++ studio/backend/routes/inference.py | 35 ++- studio/backend/routes/models.py | 48 ++++ studio/backend/utils/models/__init__.py | 4 + studio/backend/utils/models/model_config.py | 132 +++++++++-- .../assistant-ui/model-selector/pickers.tsx | 208 +++++++++++++++--- .../assistant-ui/model-selector/types.ts | 1 + .../src/features/chat/api/chat-api.ts | 11 + .../frontend/src/features/chat/chat-page.tsx | 8 +- .../chat/hooks/use-chat-model-runtime.ts | 4 + .../frontend/src/features/chat/types/api.ts | 14 ++ 13 files changed, 489 insertions(+), 82 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index 1489365686..b5c7d3c50d 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -36,6 +36,9 @@ class LlamaCppBackend: self._port: Optional[int] = None self._model_identifier: Optional[str] = None self._gguf_path: Optional[str] = None + self._hf_repo: Optional[str] = None + self._hf_variant: Optional[str] = None + self._is_vision: bool = False self._healthy = False self._lock = threading.Lock() self._chat_template: Optional[str] = None @@ -56,6 +59,10 @@ class LlamaCppBackend: def model_identifier(self) -> Optional[str]: return self._model_identifier + @property + def is_vision(self) -> bool: + return self._is_vision + # ── Binary discovery ────────────────────────────────────────── @staticmethod @@ -109,27 +116,33 @@ class LlamaCppBackend: def load_model( self, - gguf_path: str, + *, + # Local mode: pass a path to a .gguf file + gguf_path: Optional[str] = None, + # HF mode: let llama-server download via -hf "repo:quant" + hf_repo: Optional[str] = None, + hf_variant: Optional[str] = None, + hf_token: Optional[str] = None, + # Common model_identifier: str, + is_vision: bool = False, n_ctx: int = 4096, n_gpu_layers: int = -1, n_threads: Optional[int] = None, ) -> bool: """ - Start llama-server with the given GGUF file. + Start llama-server with a GGUF model. - Args: - gguf_path: Path to the .gguf file - model_identifier: Display identifier for the model - n_ctx: Context window size - n_gpu_layers: Number of layers to offload to GPU (-1 = all) - n_threads: Number of CPU threads (None = auto) + Two modes: + - Local: ``gguf_path="/path/to/model.gguf"`` → uses ``-m`` + - HF: ``hf_repo="unsloth/gemma-3-4b-it-GGUF", hf_variant="Q4_K_M"`` → uses ``-hf`` - Returns: - True if server started and health check passed. + In HF mode, llama-server handles downloading, caching, and + auto-loading mmproj files for vision models. + + Returns True if server started and health check passed. """ with self._lock: - # Kill existing process if any self._kill_process() binary = self._find_llama_server_binary() @@ -140,17 +153,33 @@ class LlamaCppBackend: "or set LLAMA_SERVER_PATH environment variable." ) - if not Path(gguf_path).is_file(): - raise FileNotFoundError(f"GGUF file not found: {gguf_path}") - self._port = self._find_free_port() - cmd = [ - binary, - "-m", gguf_path, - "--port", str(self._port), - "-c", str(n_ctx), - "-ngl", str(n_gpu_layers), - ] + + # Build command based on mode + if hf_repo: + hf_spec = f"{hf_repo}:{hf_variant}" if hf_variant else hf_repo + cmd = [ + binary, + "-hf", hf_spec, + "--port", str(self._port), + "-c", str(n_ctx), + "-ngl", str(n_gpu_layers), + ] + if hf_token: + cmd.extend(["--hf-token", hf_token]) + elif gguf_path: + if not Path(gguf_path).is_file(): + raise FileNotFoundError(f"GGUF file not found: {gguf_path}") + cmd = [ + binary, + "-m", gguf_path, + "--port", str(self._port), + "-c", str(n_ctx), + "-ngl", str(n_gpu_layers), + ] + else: + raise ValueError("Either gguf_path or hf_repo must be provided") + if n_threads is not None: cmd.extend(["--threads", str(n_threads)]) @@ -173,10 +202,14 @@ class LlamaCppBackend: ) self._gguf_path = gguf_path + self._hf_repo = hf_repo + self._hf_variant = hf_variant + self._is_vision = is_vision self._model_identifier = model_identifier - # Wait for health - if not self._wait_for_health(timeout=120.0): + # HF mode: llama-server downloads before becoming healthy — need longer timeout + timeout = 600.0 if hf_repo else 120.0 + if not self._wait_for_health(timeout=timeout): self._kill_process() raise RuntimeError( "llama-server failed to start. " @@ -185,8 +218,12 @@ class LlamaCppBackend: self._healthy = True - # Try to read chat template from GGUF metadata - self._chat_template = self._read_gguf_chat_template(gguf_path) + # Read chat template from local GGUF metadata (skip in HF mode — + # llama-server handles template application internally) + if gguf_path: + self._chat_template = self._read_gguf_chat_template(gguf_path) + else: + self._chat_template = None logger.info( f"llama-server ready on port {self._port} " @@ -201,6 +238,9 @@ class LlamaCppBackend: logger.info(f"Unloaded GGUF model: {self._model_identifier}") self._model_identifier = None self._gguf_path = None + self._hf_repo = None + self._hf_variant = None + self._is_vision = False self._port = None self._healthy = False self._chat_template = None diff --git a/studio/backend/models/inference.py b/studio/backend/models/inference.py index c4a062ae3a..fc3f788ac6 100644 --- a/studio/backend/models/inference.py +++ b/studio/backend/models/inference.py @@ -17,6 +17,7 @@ class LoadRequest(BaseModel): max_seq_length: int = Field(2048, ge=128, le=32768, description="Maximum sequence length") load_in_4bit: bool = Field(True, description="Load model in 4-bit quantization") is_lora: bool = Field(False, description="Whether this is a LoRA adapter") + gguf_variant: Optional[str] = Field(None, description="GGUF quantization variant (e.g. 'Q4_K_M')") class UnloadRequest(BaseModel): diff --git a/studio/backend/models/models.py b/studio/backend/models/models.py index b3f9b50ed2..bd035fcbee 100644 --- a/studio/backend/models/models.py +++ b/studio/backend/models/models.py @@ -76,6 +76,21 @@ class ModelListResponse(BaseModel): default_models: List[str] = Field(default_factory=list, description="List of default model IDs") +class GgufVariantDetail(BaseModel): + """A single GGUF quantization variant in a HuggingFace repo.""" + filename: str = Field(..., description="GGUF filename (e.g., 'gemma-3-4b-it-Q4_K_M.gguf')") + quant: str = Field(..., description="Quantization label (e.g., 'Q4_K_M')") + size_bytes: int = Field(0, description="File size in bytes") + + +class GgufVariantsResponse(BaseModel): + """Response for listing GGUF quantization variants in a HuggingFace repo.""" + repo_id: str = Field(..., description="HuggingFace repo ID") + variants: List[GgufVariantDetail] = Field(default_factory=list, description="Available GGUF variants") + has_vision: bool = Field(False, description="Whether the model has vision support (mmproj files)") + default_variant: Optional[str] = Field(None, description="Recommended default quantization variant") + + class LocalModelInfo(BaseModel): """Discovered local model candidate.""" id: str = Field(..., description="Identifier to use for loading/training") diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index 156578b960..b7a7a33b43 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -87,6 +87,7 @@ async def load_model(request: LoadRequest): config = ModelConfig.from_identifier( model_id=request.model_path, hf_token=request.hf_token, + gguf_variant=request.gguf_variant, ) if not config: @@ -105,11 +106,25 @@ async def load_model(request: LoadRequest): logger.info(f"Unloading Unsloth model '{unsloth_backend.active_model_name}' before loading GGUF") unsloth_backend.unload_model(unsloth_backend.active_model_name) - success = llama_backend.load_model( - gguf_path=config.gguf_file, - model_identifier=config.identifier, - n_ctx=request.max_seq_length, - ) + # Route to HF mode or local mode based on config + if config.gguf_hf_repo: + # HF mode: llama-server downloads via -hf "repo:quant" + success = llama_backend.load_model( + hf_repo=config.gguf_hf_repo, + hf_variant=config.gguf_variant, + hf_token=request.hf_token, + model_identifier=config.identifier, + is_vision=config.is_vision, + n_ctx=request.max_seq_length, + ) + else: + # Local mode: llama-server loads via -m + success = llama_backend.load_model( + gguf_path=config.gguf_file, + model_identifier=config.identifier, + is_vision=config.is_vision, + n_ctx=request.max_seq_length, + ) if not success: raise HTTPException( @@ -125,7 +140,7 @@ async def load_model(request: LoadRequest): status="loaded", model=config.identifier, display_name=config.display_name, - is_vision=False, + is_vision=config.is_vision, is_lora=False, is_gguf=True, inference=inference_config, @@ -292,7 +307,7 @@ async def get_status(): if llama_backend.is_loaded: return InferenceStatusResponse( active_model=llama_backend.model_identifier, - is_vision=False, + is_vision=llama_backend.is_vision, is_gguf=True, loading=[], loaded=[llama_backend.model_identifier], @@ -425,12 +440,12 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque # ── GGUF path: format prompt → proxy to llama-server ────── if using_gguf: - # GGUF models don't support vision + # Reject images if this GGUF model doesn't support vision image_b64 = extracted_image_b64 or payload.image_base64 - if image_b64: + if image_b64 and not llama_backend.is_vision: raise HTTPException( status_code=400, - detail="Image provided but GGUF models do not support vision.", + detail="Image provided but current GGUF model does not support vision.", ) prompt = llama_backend.format_prompt(chat_messages, system_prompt) diff --git a/studio/backend/routes/models.py b/studio/backend/routes/models.py index 881f399c3d..31cc36cc45 100644 --- a/studio/backend/routes/models.py +++ b/studio/backend/routes/models.py @@ -22,8 +22,10 @@ try: get_base_model_from_lora, is_vision_model, scan_checkpoints, + list_gguf_variants, ModelConfig, ) + from utils.models.model_config import _pick_best_gguf, _extract_quant_label from core.inference import get_inference_backend except ImportError: # Fallback: try to import from parent directory @@ -36,8 +38,10 @@ except ImportError: get_base_model_from_lora, is_vision_model, scan_checkpoints, + list_gguf_variants, ModelConfig, ) + from utils.models.model_config import _pick_best_gguf, _extract_quant_label from core.inference import get_inference_backend from models import ( @@ -51,6 +55,7 @@ from models import ( LoRAInfo, ModelListResponse, ) +from models.models import GgufVariantDetail, GgufVariantsResponse from models.responses import LoRABaseModelResponse, VisionCheckResponse router = APIRouter() @@ -405,6 +410,49 @@ async def check_vision_model( detail=f"Failed to check vision model: {str(e)}" ) +@router.get("/gguf-variants", response_model=GgufVariantsResponse) +async def get_gguf_variants( + repo_id: str = Query(..., description="HuggingFace repo ID (e.g. 'unsloth/gemma-3-4b-it-GGUF')"), + hf_token: Optional[str] = Query(None, description="HuggingFace token for private repos"), + current_subject: str = Depends(get_current_subject), +): + """ + List available GGUF quantization variants for a HuggingFace repo. + + Returns all available quantization variants (Q4_K_M, Q8_0, BF16, etc.) + with file sizes, whether the model supports vision, and the recommended + default variant. + """ + try: + variants, has_vision = list_gguf_variants(repo_id, hf_token=hf_token) + + # Determine default variant + filenames = [v.filename for v in variants] + best = _pick_best_gguf(filenames) + default_variant = _extract_quant_label(best) if best else None + + return GgufVariantsResponse( + repo_id=repo_id, + variants=[ + GgufVariantDetail( + filename=v.filename, + quant=v.quant, + size_bytes=v.size_bytes, + ) + for v in variants + ], + has_vision=has_vision, + default_variant=default_variant, + ) + + except Exception as e: + logger.error(f"Error listing GGUF variants for '{repo_id}': {e}", exc_info=True) + raise HTTPException( + status_code=500, + detail=f"Failed to list GGUF variants: {str(e)}", + ) + + @router.get("/checkpoints", response_model=CheckpointListResponse) async def list_checkpoints( outputs_dir: str = Query( diff --git a/studio/backend/utils/models/__init__.py b/studio/backend/utils/models/__init__.py index 505fd35edd..4006e63908 100644 --- a/studio/backend/utils/models/__init__.py +++ b/studio/backend/utils/models/__init__.py @@ -3,11 +3,13 @@ Model and LoRA configuration handling """ from .model_config import ( ModelConfig, + GgufVariantInfo, is_vision_model, scan_trained_loras, load_model_defaults, get_base_model_from_lora, load_model_config, + list_gguf_variants, MODEL_NAME_MAPPING, UI_STATUS_INDICATORS, ) @@ -15,11 +17,13 @@ from .checkpoints import scan_checkpoints __all__ = [ 'ModelConfig', + 'GgufVariantInfo', 'is_vision_model', 'scan_trained_loras', 'load_model_defaults', 'get_base_model_from_lora', 'load_model_config', + 'list_gguf_variants', 'MODEL_NAME_MAPPING', 'UI_STATUS_INDICATORS', 'scan_checkpoints', diff --git a/studio/backend/utils/models/model_config.py b/studio/backend/utils/models/model_config.py index 601ea75770..0c29d1c869 100644 --- a/studio/backend/utils/models/model_config.py +++ b/studio/backend/utils/models/model_config.py @@ -478,6 +478,83 @@ def _pick_best_gguf(filenames: list[str]) -> Optional[str]: return gguf_files[0] +@dataclass +class GgufVariantInfo: + """A single GGUF quantization variant from a HuggingFace repo.""" + filename: str # e.g., "gemma-3-4b-it-Q4_K_M.gguf" + quant: str # e.g., "Q4_K_M" (extracted from filename) + size_bytes: int # file size + + +def _extract_quant_label(filename: str) -> str: + """ + Extract quantization label like Q4_K_M, IQ4_XS, BF16 from a GGUF filename. + + Examples: + "gemma-3-4b-it-Q4_K_M.gguf" → "Q4_K_M" + "model-IQ4_NL.gguf" → "IQ4_NL" + "model-BF16.gguf" → "BF16" + "model-UD-IQ1_S.gguf" → "UD-IQ1_S" + """ + import re + stem = filename.rsplit(".", 1)[0] # Remove .gguf + # Match known quantization patterns (UD- prefix, IQ, Q, BF/F variants) + match = re.search( + r'(UD-)?' # Optional UD- prefix (Ultra Discrete) + r'(IQ[0-9]+_[A-Z]+(?:_[A-Z0-9]+)?' # IQ variants: IQ4_XS, IQ4_NL, IQ1_S + r'|Q[0-9]+_K_[A-Z]+' # K-quant: Q4_K_M, Q3_K_S + r'|Q[0-9]+_[0-9]+' # Standard: Q8_0, Q5_1 + r'|Q[0-9]+_K' # Short K-quant: Q6_K + r'|BF16|F16|F32)', # Full precision + stem, re.IGNORECASE, + ) + if match: + prefix = match.group(1) or "" + return f"{prefix}{match.group(2)}" + # Fallback: last segment after hyphen + return stem.split("-")[-1] + + +def list_gguf_variants( + repo_id: str, + hf_token: Optional[str] = None, +) -> tuple[list[GgufVariantInfo], bool]: + """ + List all GGUF quantization variants in a HuggingFace repo. + + Separates main model files from mmproj (vision projection) files. + The presence of mmproj files indicates a vision-capable model. + + Returns: + (variants, has_vision): list of non-mmproj GGUF variants + vision flag. + """ + from huggingface_hub import model_info as hf_model_info + + info = hf_model_info(repo_id, token=hf_token) + variants: list[GgufVariantInfo] = [] + has_vision = False + + for sibling in info.siblings: + fname = sibling.rfilename + if not fname.endswith(".gguf"): + continue + size = sibling.size or 0 + + # mmproj files are vision projection models, not main model files + if "mmproj" in fname.lower(): + has_vision = True + continue + + quant = _extract_quant_label(fname) + variants.append(GgufVariantInfo( + filename=fname, + quant=quant, + size_bytes=size, + )) + + return variants, has_vision + + def detect_gguf_model_remote( repo_id: str, hf_token: Optional[str] = None, @@ -692,7 +769,9 @@ class ModelConfig: is_vision: bool # Is this a vision model? is_lora: bool # Is this a lora adapter? is_gguf: bool = False # Is this a GGUF model? - gguf_file: Optional[str] = None # Full path to the .gguf file + gguf_file: Optional[str] = None # Full path to the .gguf file (local mode) + gguf_hf_repo: Optional[str] = None # HF repo ID for -hf mode (e.g. "unsloth/gemma-3-4b-it-GGUF") + gguf_variant: Optional[str] = None # Quantization variant (e.g. "Q4_K_M") base_model: Optional[str] = None # Base model (for LoRAs) @classmethod @@ -748,28 +827,32 @@ class ModelConfig: cls, model_id: str, hf_token: Optional[str] = None, - is_lora: bool = False + is_lora: bool = False, + gguf_variant: Optional[str] = None, ) -> Optional['ModelConfig']: """ Create ModelConfig from a clean model identifier. - + For FastAPI routes where the frontend sends sanitized model paths. No Gradio dropdown parsing - expects clean identifiers like: - "unsloth/Meta-Llama-3.1-8B-Instruct-bnb-4bit" - "./outputs/my_lora_adapter" - "/absolute/path/to/model" - + Args: model_id: Clean model identifier (HF repo name or local path) hf_token: Optional HF token for vision detection on gated models is_lora: Whether this is a LoRA adapter - + gguf_variant: Optional GGUF quantization variant (e.g. "Q4_K_M"). + For remote GGUF repos, specifies which quant to load via -hf. + If None, auto-selects using _pick_best_gguf(). + Returns: ModelConfig or None if configuration cannot be created """ if not model_id or not model_id.strip(): return None - + identifier = model_id.strip() is_local = is_local_path(identifier) path = normalize_path(identifier) if is_local else identifier @@ -800,8 +883,8 @@ class ModelConfig: # Check if the HF repo contains GGUF files gguf_filename = detect_gguf_model_remote(identifier, hf_token=hf_token) if gguf_filename: - # Preflight: verify llama-server binary exists before downloading - # a potentially multi-GB GGUF file + # Preflight: verify llama-server binary exists BEFORE user waits + # for a multi-GB download that llama-server handles natively from core.inference.llama_cpp import LlamaCppBackend if not LlamaCppBackend._find_llama_server_binary(): raise RuntimeError( @@ -809,24 +892,35 @@ class ModelConfig: "Run setup.sh to build it, or set LLAMA_SERVER_PATH." ) - logger.info(f"Detected remote GGUF repo '{identifier}', file: {gguf_filename}") - logger.info(f"Downloading GGUF file '{gguf_filename}' from '{identifier}'...") - local_gguf_path = download_gguf_file( - repo_id=identifier, - filename=gguf_filename, - hf_token=hf_token, + # Use list_gguf_variants() to detect vision & resolve variant + variants, has_vision = list_gguf_variants(identifier, hf_token=hf_token) + variant = gguf_variant + if not variant: + # Auto-select best quantization + variant_filenames = [v.filename for v in variants] + best = _pick_best_gguf(variant_filenames) + if best: + variant = _extract_quant_label(best) + else: + variant = "Q4_K_M" # Fallback — llama-server's own default + + display_name = f"{identifier.split('/')[-1]} ({variant})" + logger.info( + f"Detected remote GGUF repo '{identifier}', " + f"variant={variant}, vision={has_vision}" ) - display_name = Path(gguf_filename).stem return cls( identifier=identifier, display_name=display_name, - path=local_gguf_path, + path=identifier, is_local=False, - is_cached=True, - is_vision=False, + is_cached=False, + is_vision=has_vision, is_lora=False, is_gguf=True, - gguf_file=local_gguf_path, + gguf_file=None, + gguf_hf_repo=identifier, + gguf_variant=variant, ) # Auto-detect LoRA for local paths (check adapter_config.json on disk) diff --git a/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx b/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx index 80e9dd0b6b..c61f33785b 100644 --- a/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx +++ b/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx @@ -5,6 +5,8 @@ import { TooltipContent, TooltipTrigger, } from "@/components/ui/tooltip"; +import { listGgufVariants } from "@/features/chat/api/chat-api"; +import type { GgufVariantDetail } from "@/features/chat/types/api"; import { useDebouncedValue, useGpuInfo, @@ -17,7 +19,7 @@ import type { VramFitStatus } from "@/lib/vram"; import { checkVramFit, estimateLoadingVram } from "@/lib/vram"; import { Search01Icon } from "@hugeicons/core-free-icons"; import { HugeiconsIcon } from "@hugeicons/react"; -import { useMemo, useState, type ReactNode } from "react"; +import { useCallback, useEffect, useMemo, useState, type ReactNode } from "react"; import type { LoraModelOption, ModelOption, @@ -36,6 +38,15 @@ function ListLabel({ children }: { children: ReactNode }) { ); } +/** Format bytes to a human-readable size string. */ +function formatBytes(bytes: number): string { + if (bytes === 0) return "0 B"; + const units = ["B", "KB", "MB", "GB", "TB"]; + const i = Math.floor(Math.log(bytes) / Math.log(1024)); + const value = bytes / 1024 ** i; + return `${value.toFixed(value < 10 ? 1 : 0)} ${units[i]}`; +} + function ModelRow({ label, meta, @@ -114,6 +125,124 @@ function ModelRow({ return content; } +// ── GGUF Variant Expander ──────────────────────────────────── + +function GgufVariantExpander({ + repoId, + onSelect, +}: { + repoId: string; + onSelect: (id: string, meta: ModelSelectorChangeMeta) => void; +}) { + const [variants, setVariants] = useState(null); + const [defaultVariant, setDefaultVariant] = useState(null); + const [hasVision, setHasVision] = useState(false); + const [loading, setLoading] = useState(true); + const [error, setError] = useState(null); + + useEffect(() => { + let canceled = false; + setLoading(true); + setError(null); + + listGgufVariants(repoId) + .then((res) => { + if (canceled) return; + setVariants(res.variants); + setDefaultVariant(res.default_variant); + setHasVision(res.has_vision); + }) + .catch((err) => { + if (canceled) return; + setError(err instanceof Error ? err.message : "Failed to load variants"); + }) + .finally(() => { + if (!canceled) setLoading(false); + }); + + return () => { + canceled = true; + }; + }, [repoId]); + + const handleVariantClick = useCallback( + (quant: string) => { + onSelect(repoId, { + source: "hub", + isLora: false, + ggufVariant: quant, + }); + }, + [repoId, onSelect], + ); + + if (loading) { + return ( +
+ + Loading variants… +
+ ); + } + + if (error) { + return ( +
{error}
+ ); + } + + if (!variants || variants.length === 0) { + return ( +
+ No GGUF variants found. +
+ ); + } + + return ( +
+
+ + Quantizations + + {hasVision && ( + Vision + )} +
+ {variants.map((v) => ( + + ))} +
+ ); +} + +// ── Detect GGUF repos by naming convention ──────────────────── + +function isGgufRepo(id: string): boolean { + return id.toUpperCase().includes("-GGUF"); +} + +// ── Hub Model Picker ────────────────────────────────────────── + export function HubModelPicker({ models, value, @@ -130,6 +259,9 @@ export function HubModelPicker({ debouncedQuery, ); + // Track which GGUF repo is expanded for variant selection + const [expandedGguf, setExpandedGguf] = useState(null); + const recommendedIds = useMemo( () => dedupe([...models.map((model) => model.id), value ?? ""]), [models, value], @@ -199,6 +331,19 @@ export function HubModelPicker({ const { scrollRef, sentinelRef } = useInfiniteScroll(fetchMore, results.length); + /** Handle clicking a model row — GGUF repos expand, others load directly. */ + const handleModelClick = useCallback( + (id: string) => { + if (isGgufRepo(id)) { + // Toggle GGUF variant expander + setExpandedGguf((prev) => (prev === id ? null : id)); + } else { + onSelect(id, { source: "hub", isLora: false }); + } + }, + [onSelect], + ); + return (
@@ -230,18 +375,24 @@ export function HubModelPicker({ recommendedIds.map((id) => { const vram = recommendedVramMap.get(id); return ( - - onSelect(id, { source: "hub", isLora: false }) - } - vramStatus={vram?.status ?? null} - vramEst={vram?.est} - gpuGb={gpu.available ? gpu.memoryTotalGb : undefined} - /> +
+ handleModelClick(id)} + vramStatus={isGgufRepo(id) ? null : vram?.status ?? null} + vramEst={isGgufRepo(id) ? undefined : vram?.est} + gpuGb={gpu.available ? gpu.memoryTotalGb : undefined} + /> + {expandedGguf === id && ( + + )} +
); }) )} @@ -259,18 +410,24 @@ export function HubModelPicker({ hfIds.map((id) => { const vram = vramMap.get(id); return ( - - onSelect(id, { source: "hub", isLora: false }) - } - vramStatus={vram?.status ?? null} - vramEst={vram?.est} - gpuGb={gpu.available ? gpu.memoryTotalGb : undefined} - /> +
+ handleModelClick(id)} + vramStatus={isGgufRepo(id) ? null : vram?.status ?? null} + vramEst={isGgufRepo(id) ? undefined : vram?.est} + gpuGb={gpu.available ? gpu.memoryTotalGb : undefined} + /> + {expandedGguf === id && ( + + )} +
); }) )} @@ -382,4 +539,3 @@ export function LoraModelPicker({
); } - diff --git a/studio/frontend/src/components/assistant-ui/model-selector/types.ts b/studio/frontend/src/components/assistant-ui/model-selector/types.ts index dcf110bfb7..a94d3dd931 100644 --- a/studio/frontend/src/components/assistant-ui/model-selector/types.ts +++ b/studio/frontend/src/components/assistant-ui/model-selector/types.ts @@ -15,5 +15,6 @@ export interface LoraModelOption extends ModelOption { export interface ModelSelectorChangeMeta { source: "hub" | "lora"; isLora: boolean; + ggufVariant?: string; } diff --git a/studio/frontend/src/features/chat/api/chat-api.ts b/studio/frontend/src/features/chat/api/chat-api.ts index 72baf9a6f6..5d5a9551ef 100644 --- a/studio/frontend/src/features/chat/api/chat-api.ts +++ b/studio/frontend/src/features/chat/api/chat-api.ts @@ -1,5 +1,6 @@ import { authFetch } from "@/features/auth"; import type { + GgufVariantsResponse, InferenceStatusResponse, ListLorasResponse, ListModelsResponse, @@ -74,6 +75,16 @@ export async function unloadModel(payload: UnloadModelRequest): Promise { await parseJsonOrThrow(response); } +export async function listGgufVariants( + repoId: string, + hfToken?: string, +): Promise { + const params = new URLSearchParams({ repo_id: repoId }); + if (hfToken) params.set("hf_token", hfToken); + const response = await authFetch(`/api/models/gguf-variants?${params}`); + return parseJsonOrThrow(response); +} + function parseSseEvent(rawEvent: string): string[] { const dataLines: string[] = []; for (const line of rawEvent.split(/\r?\n/)) { diff --git a/studio/frontend/src/features/chat/chat-page.tsx b/studio/frontend/src/features/chat/chat-page.tsx index c363704d61..7c57bdf0a4 100644 --- a/studio/frontend/src/features/chat/chat-page.tsx +++ b/studio/frontend/src/features/chat/chat-page.tsx @@ -300,7 +300,7 @@ export function ChatPage(): ReactElement { }, [inferenceParams.checkpoint, lorasFromStore]); const handleCheckpointChange = useCallback( - (value: string, meta?: { isLora: boolean }) => { + (value: string, meta?: { isLora: boolean; ggufVariant?: string }) => { const currentCheckpoint = useChatRuntimeStore.getState().params.checkpoint; if (!value || value === currentCheckpoint) return; @@ -309,7 +309,11 @@ export function ChatPage(): ReactElement { if (currentCheckpoint) { await ejectModel(); } - await selectModel({ id: value, isLora: meta?.isLora }); + await selectModel({ + id: value, + isLora: meta?.isLora, + ggufVariant: meta?.ggufVariant, + }); })(); }, [selectModel, ejectModel], diff --git a/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts b/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts index 030fe55bff..24a6b3c01f 100644 --- a/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts +++ b/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts @@ -20,6 +20,7 @@ const DEFAULT_MODEL_MAX_SEQ_LENGTH = 2048; type SelectedModelInput = { id: string; isLora?: boolean; + ggufVariant?: string; }; const LORA_SUFFIX_RE = /_(\d{9,})$/; @@ -159,6 +160,8 @@ export function useChatModelRuntime() { const explicitIsLora = typeof selection === "string" ? undefined : selection.isLora; + const ggufVariant = + typeof selection === "string" ? undefined : selection.ggufVariant; const model = models.find((entry) => entry.id === modelId); const lora = loras.find((entry) => entry.id === modelId); const isLora = @@ -181,6 +184,7 @@ export function useChatModelRuntime() { max_seq_length: DEFAULT_MODEL_MAX_SEQ_LENGTH, load_in_4bit: true, is_lora: isLora, + gguf_variant: ggufVariant ?? null, }); const currentParams = useChatRuntimeStore.getState().params; diff --git a/studio/frontend/src/features/chat/types/api.ts b/studio/frontend/src/features/chat/types/api.ts index bc9268e2c2..f0a10a6cae 100644 --- a/studio/frontend/src/features/chat/types/api.ts +++ b/studio/frontend/src/features/chat/types/api.ts @@ -28,6 +28,20 @@ export interface LoadModelRequest { max_seq_length: number; load_in_4bit: boolean; is_lora: boolean; + gguf_variant?: string | null; +} + +export interface GgufVariantDetail { + filename: string; + quant: string; + size_bytes: number; +} + +export interface GgufVariantsResponse { + repo_id: string; + variants: GgufVariantDetail[]; + has_vision: boolean; + default_variant: string | null; } export interface LoadModelResponse { From 0e7c8a2e5eeef97b01b208e6d8daf9ca08979c70 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Tue, 24 Feb 2026 19:21:01 +0400 Subject: [PATCH 10/49] Switch GGUF backend from /v1/completions to /v1/chat/completions Fixes two bugs: 1. Chat template tags (<|im_start|>, <|im_end|>) leaking into output because /v1/completions treated them as literal text 2. Image hallucination because image_b64 was never passed to llama-server Now llama-server handles chat templates natively and receives images as OpenAI-format multimodal content parts for vision models. --- studio/backend/core/inference/llama_cpp.py | 133 +++++++-------------- studio/backend/routes/inference.py | 12 +- 2 files changed, 51 insertions(+), 94 deletions(-) diff --git a/studio/backend/core/inference/llama_cpp.py b/studio/backend/core/inference/llama_cpp.py index b5c7d3c50d..30fe660404 100644 --- a/studio/backend/core/inference/llama_cpp.py +++ b/studio/backend/core/inference/llama_cpp.py @@ -2,7 +2,7 @@ llama-server inference backend for GGUF models. Manages a llama-server subprocess and proxies chat completions -through its /v1/completions endpoint. +through its OpenAI-compatible /v1/chat/completions endpoint. """ import atexit import json @@ -27,7 +27,7 @@ class LlamaCppBackend: Lifecycle: 1. load_model() — starts llama-server with the GGUF file - 2. generate_chat_completion() — formats prompt, proxies to /v1/completions, streams back + 2. generate_chat_completion() — proxies to /v1/chat/completions, streams back 3. unload_model() — terminates llama-server subprocess """ @@ -41,7 +41,6 @@ class LlamaCppBackend: self._is_vision: bool = False self._healthy = False self._lock = threading.Lock() - self._chat_template: Optional[str] = None atexit.register(self._cleanup) @@ -218,13 +217,6 @@ class LlamaCppBackend: self._healthy = True - # Read chat template from local GGUF metadata (skip in HF mode — - # llama-server handles template application internally) - if gguf_path: - self._chat_template = self._read_gguf_chat_template(gguf_path) - else: - self._chat_template = None - logger.info( f"llama-server ready on port {self._port} " f"for model '{model_identifier}'" @@ -243,7 +235,6 @@ class LlamaCppBackend: self._is_vision = False self._port = None self._healthy = False - self._chat_template = None return True def _kill_process(self): @@ -298,92 +289,49 @@ class LlamaCppBackend: logger.error(f"llama-server health check timed out after {timeout}s") return False - # ── Chat template ───────────────────────────────────────────── + # ── Message building (OpenAI format) ────────────────────────── @staticmethod - def _read_gguf_chat_template(gguf_path: str) -> Optional[str]: + def _build_openai_messages( + messages: list[dict], + image_b64: Optional[str] = None, + ) -> list[dict]: """ - Try to read the chat_template from GGUF file metadata. + Build OpenAI-format messages, optionally injecting an image_url + content part into the last user message for vision models. - Uses the gguf Python library if available. - Returns the Jinja2 template string, or None. + If no image is provided, returns messages as-is. """ - try: - from gguf import GGUFReader + if not image_b64: + return messages - reader = GGUFReader(gguf_path) - for field_name in reader.fields: - if field_name == "tokenizer.chat_template": - field = reader.fields[field_name] - # Field data is an array of bytes - template_bytes = bytes(field.parts[field.data[0]]) - template = template_bytes.decode("utf-8") - logger.info(f"Read chat template from GGUF metadata ({len(template)} chars)") - return template - except ImportError: - logger.debug("gguf library not available, cannot read chat template from GGUF metadata") - except Exception as e: - logger.warning(f"Could not read chat template from GGUF: {e}") + # Find the last user message and convert to multimodal content parts + result = [msg.copy() for msg in messages] + last_user_idx = None + for i, msg in enumerate(result): + if msg["role"] == "user": + last_user_idx = i - return None + if last_user_idx is not None: + text_content = result[last_user_idx].get("content", "") + result[last_user_idx]["content"] = [ + {"type": "text", "text": text_content}, + { + "type": "image_url", + "image_url": { + "url": f"data:image/png;base64,{image_b64}", + }, + }, + ] - def format_prompt(self, messages: list[dict], system_prompt: str = "") -> str: - """ - Format chat messages into a raw prompt string for /v1/completions. - - Attempts to: - 1. Render the GGUF's embedded chat_template with Jinja2 - 2. Fallback to ChatML format - """ - # Build full message list with system prompt - full_messages = [] - if system_prompt: - full_messages.append({"role": "system", "content": system_prompt}) - full_messages.extend(messages) - - # Try Jinja2 rendering if we have a template - if self._chat_template: - try: - return self._render_jinja_template(full_messages) - except Exception as e: - logger.warning(f"Jinja2 template rendering failed, falling back to ChatML: {e}") - - # Fallback: ChatML format - return self._format_chatml(full_messages) - - def _render_jinja_template(self, messages: list[dict]) -> str: - """Render messages using the GGUF's Jinja2 chat template.""" - from jinja2 import BaseLoader, Environment - - env = Environment(loader=BaseLoader(), keep_trailing_newline=True) - # Add common template globals - env.globals["raise_exception"] = lambda msg: (_ for _ in ()).throw(ValueError(msg)) - - template = env.from_string(self._chat_template) - rendered = template.render( - messages=messages, - add_generation_prompt=True, - bos_token="", - eos_token="", - ) - return rendered - - @staticmethod - def _format_chatml(messages: list[dict]) -> str: - """Format messages using ChatML template (universal fallback).""" - parts = [] - for msg in messages: - role = msg.get("role", "user") - content = msg.get("content", "") - parts.append(f"<|im_start|>{role}\n{content}<|im_end|>") - parts.append("<|im_start|>assistant") - return "\n".join(parts) + "\n" + return result # ── Generation (proxy to llama-server) ──────────────────────── def generate_chat_completion( self, - prompt: str, + messages: list[dict], + image_b64: Optional[str] = None, temperature: float = 0.7, top_p: float = 0.9, top_k: int = 40, @@ -394,30 +342,32 @@ class LlamaCppBackend: cancel_event: Optional[threading.Event] = None, ) -> Generator[str, None, None]: """ - Send a completion request to llama-server and stream tokens back. + Send a chat completion request to llama-server and stream tokens back. - Uses /v1/completions (NOT /v1/chat/completions) so we control - the prompt format entirely. + Uses /v1/chat/completions — llama-server handles chat template + application and vision (multimodal image_url parts) natively. Yields cumulative text (matching InferenceBackend's convention). """ if not self.is_loaded: raise RuntimeError("llama-server is not loaded") + openai_messages = self._build_openai_messages(messages, image_b64) + payload = { - "prompt": prompt, + "messages": openai_messages, "stream": True, "temperature": temperature, "top_p": top_p, "top_k": top_k if top_k >= 0 else 0, "min_p": min_p, - "n_predict": max_tokens, + "max_tokens": max_tokens, "repeat_penalty": repetition_penalty, } if stop: payload["stop"] = stop - url = f"{self.base_url}/v1/completions" + url = f"{self.base_url}/v1/chat/completions" cumulative = "" try: @@ -450,7 +400,8 @@ class LlamaCppBackend: data = json.loads(line[6:]) choices = data.get("choices", []) if choices: - token = choices[0].get("text", "") + delta = choices[0].get("delta", {}) + token = delta.get("content", "") if token: cumulative += token yield cumulative diff --git a/studio/backend/routes/inference.py b/studio/backend/routes/inference.py index b7a7a33b43..bc450add0d 100644 --- a/studio/backend/routes/inference.py +++ b/studio/backend/routes/inference.py @@ -438,7 +438,7 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque detail="At least one non-system message is required.", ) - # ── GGUF path: format prompt → proxy to llama-server ────── + # ── GGUF path: proxy to llama-server /v1/chat/completions ── if using_gguf: # Reject images if this GGUF model doesn't support vision image_b64 = extracted_image_b64 or payload.image_base64 @@ -448,7 +448,12 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque detail="Image provided but current GGUF model does not support vision.", ) - prompt = llama_backend.format_prompt(chat_messages, system_prompt) + # Build message list with system prompt prepended + gguf_messages = [] + if system_prompt: + gguf_messages.append({"role": "system", "content": system_prompt}) + gguf_messages.extend(chat_messages) + cancel_event = threading.Event() completion_id = f"chatcmpl-{uuid.uuid4().hex[:12]}" @@ -456,7 +461,8 @@ async def openai_chat_completions(payload: ChatCompletionRequest, request: Reque def gguf_generate(): return llama_backend.generate_chat_completion( - prompt=prompt, + messages=gguf_messages, + image_b64=image_b64, temperature=payload.temperature, top_p=payload.top_p, top_k=payload.top_k, From a3daae1c40f2cb59e3866ea21ac5caaad4100ac2 Mon Sep 17 00:00:00 2001 From: Leo Borcherding Date: Tue, 24 Feb 2026 14:37:00 -0600 Subject: [PATCH 11/49] fix: replace datetime.UTC with timezone.utc for Python 3.9+ compatibility - Replace datetime.UTC with datetime.timezone.utc in authentication.py and storage.py - Fixes ImportError on Python versions < 3.11 - timezone.utc works on Python 3.9+ Resolves #237 --- studio/backend/auth/authentication.py | 6 +++--- studio/backend/auth/storage.py | 6 +++--- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/studio/backend/auth/authentication.py b/studio/backend/auth/authentication.py index 6ea668db3e..e725834630 100644 --- a/studio/backend/auth/authentication.py +++ b/studio/backend/auth/authentication.py @@ -1,5 +1,5 @@ import secrets -from datetime import UTC, datetime, timedelta +from datetime import datetime, timedelta, timezone from typing import Optional from fastapi import Depends, HTTPException, status @@ -34,7 +34,7 @@ def create_access_token( Tokens are valid across restarts because SECRET_KEY is stored in SQLite. """ to_encode = {"sub": subject} - expire = datetime.now(UTC) + ( + expire = datetime.now(timezone.utc) + ( expires_delta or timedelta(minutes=ACCESS_TOKEN_EXPIRE_MINUTES) ) to_encode.update({"exp": expire}) @@ -48,7 +48,7 @@ def create_refresh_token(subject: str) -> str: Refresh tokens are opaque (not JWTs) and expire after REFRESH_TOKEN_EXPIRE_DAYS. """ token = secrets.token_urlsafe(48) - expires_at = datetime.now(UTC) + timedelta(days=REFRESH_TOKEN_EXPIRE_DAYS) + expires_at = datetime.now(timezone.utc) + timedelta(days=REFRESH_TOKEN_EXPIRE_DAYS) save_refresh_token(token, subject, expires_at.isoformat()) return token diff --git a/studio/backend/auth/storage.py b/studio/backend/auth/storage.py index faea6266e3..e5a486bca2 100644 --- a/studio/backend/auth/storage.py +++ b/studio/backend/auth/storage.py @@ -3,7 +3,7 @@ SQLite storage for authentication data (user credentials + JWT secret). """ import hashlib import sqlite3 -from datetime import UTC, datetime +from datetime import datetime, timezone from pathlib import Path from typing import Optional, Tuple @@ -218,7 +218,7 @@ def verify_refresh_token(token: str) -> Optional[str]: # Clean up any expired tokens while we're here conn.execute( "DELETE FROM refresh_tokens WHERE expires_at < ?", - (datetime.now(UTC).isoformat(),), + (datetime.now(timezone.utc).isoformat(),), ) conn.commit() @@ -235,7 +235,7 @@ def verify_refresh_token(token: str) -> Optional[str]: # Check expiry expires_at = datetime.fromisoformat(row["expires_at"]) - if datetime.now(UTC) > expires_at: + if datetime.now(timezone.utc) > expires_at: conn.execute("DELETE FROM refresh_tokens WHERE id = ?", (row["id"],)) conn.commit() return None From 7adb69581e410a77a5fcd3ff2c421187495415ea Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Wed, 25 Feb 2026 03:30:54 +0400 Subject: [PATCH 12/49] Fix GGUF export cwd confusion: remove os.chdir, use absolute paths Remove os.chdir(save_directory) from export.py which was causing all of unsloth-zoo's relative-path internals (check_llama_cpp, use_local_gguf, _download_convert_hf_to_gguf) to resolve against the export directory instead of the repo root. This caused llama.cpp to be cloned inside each export dir and destroyed the repo root's llama-server build on cleanup. Now passes absolute paths to save_pretrained_gguf so unsloth resolves llama.cpp from the repo root where setup.sh already built it. Also builds llama-quantize in setup.sh (needed by unsloth-zoo's export pipeline) and symlinks it to llama.cpp root for check_llama_cpp(). --- setup.sh | 20 +++++++-- studio/backend/core/export/export.py | 62 ++++++++++------------------ 2 files changed, 38 insertions(+), 44 deletions(-) diff --git a/setup.sh b/setup.sh index 36165dac65..0d8821d50a 100755 --- a/setup.sh +++ b/setup.sh @@ -206,10 +206,11 @@ else fi fi -# ── 8. Build llama-server for GGUF inference ── +# ── 8. Build llama.cpp binaries for GGUF inference + export ── # Builds in-tree at $REPO/llama.cpp/. This directory is shared with -# unsloth-zoo's GGUF export pipeline — if converter/quantize are missing, -# unsloth-zoo will rebuild them on first export. We only build llama-server here. +# unsloth-zoo's GGUF export pipeline. We build: +# - llama-server: for GGUF model inference +# - llama-quantize: for GGUF export quantization (symlinked to root for check_llama_cpp()) LLAMA_SERVER_BIN="$SCRIPT_DIR/llama.cpp/build/bin/llama-server" if [ -f "$LLAMA_SERVER_BIN" ]; then echo "" @@ -272,12 +273,25 @@ else run_quiet "build llama-server" cmake --build "$LLAMA_CPP_DIR/build" --config Release --target llama-server -j"$NCPU" || BUILD_OK=false fi + # Also build llama-quantize (needed by unsloth-zoo's GGUF export pipeline) + if [ "$BUILD_OK" = true ]; then + run_quiet "build llama-quantize" cmake --build "$LLAMA_CPP_DIR/build" --config Release --target llama-quantize -j"$NCPU" || true + # Symlink to llama.cpp root — check_llama_cpp() looks for the binary there + QUANTIZE_BIN="$LLAMA_CPP_DIR/build/bin/llama-quantize" + if [ -f "$QUANTIZE_BIN" ]; then + ln -sf build/bin/llama-quantize "$LLAMA_CPP_DIR/llama-quantize" + fi + fi + if [ "$BUILD_OK" = true ]; then if [ -f "$LLAMA_SERVER_BIN" ]; then echo "✅ llama-server built at $LLAMA_SERVER_BIN" else echo "⚠️ llama-server binary not found after build — GGUF inference won't be available" fi + if [ -f "$LLAMA_CPP_DIR/llama-quantize" ]; then + echo "✅ llama-quantize available for GGUF export" + fi else echo "⚠️ llama-server build failed — GGUF inference won't be available, but everything else works" fi diff --git a/studio/backend/core/export/export.py b/studio/backend/core/export/export.py index da5b11c60d..865900cb65 100644 --- a/studio/backend/core/export/export.py +++ b/studio/backend/core/export/export.py @@ -378,53 +378,33 @@ class ExportBackend: # Save locally if requested if save_directory: - logger.info(f"Saving GGUF model locally to: {save_directory}") + # Resolve to absolute path so unsloth's relative-path internals + # (check_llama_cpp, use_local_gguf, _download_convert_hf_to_gguf) + # all resolve against the repo root cwd, NOT the export directory. + abs_save_dir = os.path.abspath(save_directory) + logger.info(f"Saving GGUF model locally to: {abs_save_dir}") # Create the directory if it doesn't exist - os.makedirs(save_directory, exist_ok=True) + os.makedirs(abs_save_dir, exist_ok=True) - # Get the base filename for the GGUF file - import shutil - original_dir = os.getcwd() + # On WSL, patch out sudo check before llama.cpp build + _apply_wsl_sudo_patch() - try: - # Change to target directory - os.chdir(save_directory) - logger.info(f"Changed directory to: {save_directory}") + # Enable verbose logging so subprocess errors are printed + os.environ["UNSLOTH_ENABLE_LOGGING"] = "1" - # On WSL, patch out sudo check before llama.cpp build - _apply_wsl_sudo_patch() + # Pass absolute path — no os.chdir needed. + # unsloth saves model files into this directory, while + # check_llama_cpp("llama.cpp") resolves against cwd (repo root) + # where setup.sh already built llama.cpp with quantizer. + model_save_path = os.path.join(abs_save_dir, "model") + self.current_model.save_pretrained_gguf( + model_save_path, + self.current_tokenizer, + quantization_method=quant_method + ) - # Now save (will save in current directory) - self.current_model.save_pretrained_gguf( - "model", # Base filename - self.current_tokenizer, - quantization_method=quant_method - ) - - logger.info(f"GGUF model saved successfully in {save_directory}") - - # Check if llama.cpp directory was created here - llama_cpp_in_target = os.path.join(save_directory, "llama.cpp") - llama_cpp_in_original = os.path.join(original_dir, "llama.cpp") - - if os.path.exists(llama_cpp_in_target): - logger.info(f"Found llama.cpp directory in {save_directory}") - - # Remove llama.cpp from original directory if it exists - if os.path.exists(llama_cpp_in_original): - logger.info(f"Removing existing llama.cpp in {original_dir}") - shutil.rmtree(llama_cpp_in_original) - - # Move llama.cpp back to original directory - logger.info(f"Moving llama.cpp to {original_dir}") - shutil.move(llama_cpp_in_target, llama_cpp_in_original) - logger.info(f"Successfully moved llama.cpp back to original directory") - - finally: - # Always change back to original directory - os.chdir(original_dir) - logger.info(f"Changed back to original directory: {original_dir}") + logger.info(f"GGUF model saved successfully in {abs_save_dir}") # Push to hub if requested if push_to_hub: From dbf5acf486ec5d77d3e37958cbdf60411ad75660 Mon Sep 17 00:00:00 2001 From: imagineer99 Date: Wed, 25 Feb 2026 03:00:40 +0000 Subject: [PATCH 13/49] feat: filter pretraining datasets from search results --- .../src/hooks/use-hf-dataset-search.ts | 100 ++++++++++-------- 1 file changed, 58 insertions(+), 42 deletions(-) diff --git a/studio/frontend/src/hooks/use-hf-dataset-search.ts b/studio/frontend/src/hooks/use-hf-dataset-search.ts index 9236a60486..b00dcee1ed 100644 --- a/studio/frontend/src/hooks/use-hf-dataset-search.ts +++ b/studio/frontend/src/hooks/use-hf-dataset-search.ts @@ -49,6 +49,7 @@ export interface HfDatasetResult { totalExamples?: number; sizeCategory?: string; taskCategories: string[]; + plainTags: string[]; } function mapDataset(raw: unknown): HfDatasetResult { @@ -60,9 +61,11 @@ function mapDataset(raw: unknown): HfDatasetResult { cardData?: unknown; }; const card = ds.cardData as CardDataWithInfo | undefined; - const taskCategories = (ds.tags ?? []) + const tags = ds.tags ?? []; + const taskCategories = tags .filter((t) => t.startsWith("task_categories:")) .map((t) => t.slice("task_categories:".length)); + const plainTags = tags.filter((t) => !t.includes(":")); return { id: ds.name, downloads: ds.downloads, @@ -70,6 +73,7 @@ function mapDataset(raw: unknown): HfDatasetResult { totalExamples: extractTotalExamples(card), sizeCategory: card?.size_categories?.[0], taskCategories, + plainTags, }; } @@ -95,7 +99,9 @@ function withTrendingSort( return fetch(url, init); } -const RELEVANT_TASK_CATEGORIES: Record> = { +type DatasetRelevance = "incompatible" | "neutral" | "boosted"; + +const BOOSTED_TASK_CATEGORIES: Record> = { text: new Set([ "text-generation", "text2text-generation", @@ -121,10 +127,28 @@ const RELEVANT_TASK_CATEGORIES: Record> = { ]), }; -const INCOMPATIBLE_TASK_CATEGORIES: Record> = { +const INCOMPATIBLE_TASKS_ALL_MODELS = new Set([ + "text-to-3d", + "image-to-3d", + "robotics", + "reinforcement-learning", + "tabular-classification", + "tabular-regression", + "time-series-forecasting", +]); + +const PRETRAINING_PLAIN_TAGS = new Set(["pretraining", "pre-training"]); + +const PRETRAINING_SIZE_CATEGORIES = new Set([ + "100M1T", +]); + +const INCOMPATIBLE_TASKS_BY_MODEL: Record> = { text: new Set([ - "text-to-3d", - "image-to-3d", "text-to-image", "image-to-image", "image-to-video", @@ -143,30 +167,16 @@ const INCOMPATIBLE_TASK_CATEGORIES: Record> = { "audio-to-audio", "automatic-speech-recognition", "video-classification", - "robotics", - "reinforcement-learning", - "tabular-classification", - "tabular-regression", - "time-series-forecasting", "visual-document-retrieval", ]), vision: new Set([ - "text-to-3d", - "image-to-3d", "text-to-speech", "text-to-audio", "audio-classification", "audio-to-audio", "automatic-speech-recognition", - "robotics", - "reinforcement-learning", - "tabular-classification", - "tabular-regression", - "time-series-forecasting", ]), tts: new Set([ - "text-to-3d", - "image-to-3d", "text-to-image", "image-to-image", "image-to-video", @@ -180,16 +190,9 @@ const INCOMPATIBLE_TASK_CATEGORIES: Record> = { "image-segmentation", "depth-estimation", "video-classification", - "robotics", - "reinforcement-learning", - "tabular-classification", - "tabular-regression", - "time-series-forecasting", "visual-document-retrieval", ]), embeddings: new Set([ - "text-to-3d", - "image-to-3d", "text-to-image", "image-to-image", "image-to-video", @@ -208,28 +211,41 @@ const INCOMPATIBLE_TASK_CATEGORIES: Record> = { "audio-to-audio", "automatic-speech-recognition", "video-classification", - "robotics", - "reinforcement-learning", - "tabular-classification", - "tabular-regression", - "time-series-forecasting", "visual-document-retrieval", ]), }; -function classifyDataset( +function isPretrainingDataset(dataset: HfDatasetResult): boolean { + if (dataset.plainTags.some((t) => PRETRAINING_PLAIN_TAGS.has(t.toLowerCase()))) + return true; + if ( + dataset.sizeCategory && + PRETRAINING_SIZE_CATEGORIES.has(dataset.sizeCategory) + ) + return true; + return false; +} + +function rankDatasetRelevance( dataset: HfDatasetResult, modelType: ModelType, -): -1 | 0 | 1 { +): DatasetRelevance { + if (isPretrainingDataset(dataset)) return "incompatible"; + const { taskCategories } = dataset; - if (taskCategories.length === 0) return 0; + if (taskCategories.length === 0) return "neutral"; - const relevant = RELEVANT_TASK_CATEGORIES[modelType]; - const incompatible = INCOMPATIBLE_TASK_CATEGORIES[modelType]; + const boosted = BOOSTED_TASK_CATEGORIES[modelType]; + const modelIncompat = INCOMPATIBLE_TASKS_BY_MODEL[modelType]; - if (taskCategories.some((t) => relevant.has(t))) return 1; - if (taskCategories.every((t) => incompatible.has(t))) return -1; - return 0; + if (taskCategories.some((t) => boosted.has(t))) return "boosted"; + if ( + taskCategories.every( + (t) => INCOMPATIBLE_TASKS_ALL_MODELS.has(t) || modelIncompat.has(t), + ) + ) + return "incompatible"; + return "neutral"; } export function useHfDatasetSearch( @@ -257,9 +273,9 @@ export function useHfDatasetSearch( const neutral: HfDatasetResult[] = []; for (const ds of search.results) { - const rank = classifyDataset(ds, modelType); - if (rank === 1) boosted.push(ds); - else if (rank !== -1) neutral.push(ds); + const relevance = rankDatasetRelevance(ds, modelType); + if (relevance === "boosted") boosted.push(ds); + else if (relevance !== "incompatible") neutral.push(ds); } return [...boosted, ...neutral]; From 6f0b7bc38aef0cb3f39e0525e38eea0af9cc9266 Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 25 Feb 2026 10:29:05 +0000 Subject: [PATCH 14/49] fix: use raw github URL for vision.py patch + add VLM processor diagnostic logging --- setup.sh | 4 ++++ studio/backend/core/training/trainer.py | 9 +++++++++ 2 files changed, 13 insertions(+) diff --git a/setup.sh b/setup.sh index 0d8821d50a..a911d9ed7a 100755 --- a/setup.sh +++ b/setup.sh @@ -169,6 +169,10 @@ if [ "$IS_COLAB" = true ]; then LLAMA_CPP_DST="$(pip show unsloth-zoo | grep -i '^Location:' | awk '{print $2}')/unsloth_zoo/llama_cpp.py" curl -sSL "https://raw.githubusercontent.com/unslothai/unsloth-zoo/refs/heads/main/unsloth_zoo/llama_cpp.py" \ -o "$LLAMA_CPP_DST" + # Patch: override vision.py with fix from unsloth PR: https://github.com/unslothai/unsloth/pull/4091 until next pypi release + VISION_DST="$(pip show unsloth | grep -i '^Location:' | awk '{print $2}')/unsloth/vision.py" + curl -sSL "https://raw.githubusercontent.com/unslothai/unsloth/80e0108a684c882965a02a8ed851e3473c1145ab/unsloth/models/vision.py" \ + -o "$VISION_DST" echo " Installing studio dependencies..." run_quiet "pip install studio" pip install -r "$SCRIPT_DIR/studio/backend/requirements/studio.txt" echo "✅ Python dependencies installed" diff --git a/studio/backend/core/training/trainer.py b/studio/backend/core/training/trainer.py index 804baa33c0..c56b4a8a89 100644 --- a/studio/backend/core/training/trainer.py +++ b/studio/backend/core/training/trainer.py @@ -173,6 +173,15 @@ class UnslothTrainer: token=hf_token, ) logger.info("Loaded vision model") + + # Diagnostic: check if FastVisionModel returned a real Processor or a raw tokenizer + from transformers import ProcessorMixin + tok = self.tokenizer + has_image_proc = isinstance(tok, ProcessorMixin) or hasattr(tok, "image_processor") + print(f"\n[VLM Diagnostic] FastVisionModel returned: {type(tok).__name__}") + print(f"[VLM Diagnostic] Is ProcessorMixin: {isinstance(tok, ProcessorMixin)}") + print(f"[VLM Diagnostic] Has image_processor: {hasattr(tok, 'image_processor')}") + print(f"[VLM Diagnostic] Usable as vision processor: {has_image_proc}\n") else: # Load text model - returns (model, tokenizer) self.model, self.tokenizer = FastLanguageModel.from_pretrained( From 299ce774674f6ca6a6a0f30acea5a1d93300d2af Mon Sep 17 00:00:00 2001 From: Roland Tannous Date: Wed, 25 Feb 2026 11:39:17 +0000 Subject: [PATCH 15/49] added vision.py patch for vision processor from PR#260 --- setup.sh | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/setup.sh b/setup.sh index a911d9ed7a..a6d227de2b 100755 --- a/setup.sh +++ b/setup.sh @@ -170,7 +170,7 @@ if [ "$IS_COLAB" = true ]; then curl -sSL "https://raw.githubusercontent.com/unslothai/unsloth-zoo/refs/heads/main/unsloth_zoo/llama_cpp.py" \ -o "$LLAMA_CPP_DST" # Patch: override vision.py with fix from unsloth PR: https://github.com/unslothai/unsloth/pull/4091 until next pypi release - VISION_DST="$(pip show unsloth | grep -i '^Location:' | awk '{print $2}')/unsloth/vision.py" + VISION_DST="$(pip show unsloth | grep -i '^Location:' | awk '{print $2}')/unsloth/models/vision.py" curl -sSL "https://raw.githubusercontent.com/unslothai/unsloth/80e0108a684c882965a02a8ed851e3473c1145ab/unsloth/models/vision.py" \ -o "$VISION_DST" echo " Installing studio dependencies..." @@ -193,6 +193,10 @@ else LLAMA_CPP_DST="$(pip show unsloth-zoo | grep -i '^Location:' | awk '{print $2}')/unsloth_zoo/llama_cpp.py" curl -sSL "https://raw.githubusercontent.com/unslothai/unsloth-zoo/refs/heads/main/unsloth_zoo/llama_cpp.py" \ -o "$LLAMA_CPP_DST" + # Patch: override vision.py with fix from unsloth PR: https://github.com/unslothai/unsloth/pull/4091 until next pypi release + VISION_DST="$(pip show unsloth | grep -i '^Location:' | awk '{print $2}')/unsloth/models/vision.py" + curl -sSL "https://raw.githubusercontent.com/unslothai/unsloth/80e0108a684c882965a02a8ed851e3473c1145ab/unsloth/models/vision.py" \ + -o "$VISION_DST" echo " Installing studio dependencies..." run_quiet "pip install studio" pip install -r "$SCRIPT_DIR/studio/backend/requirements/studio.txt" echo "✅ Python dependencies installed" From a7fe8a388cc90f0453329984e69a1b6e86b96180 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Wed, 25 Feb 2026 15:47:45 +0400 Subject: [PATCH 16/49] Filter GGUF models from training page model selectors GGUF models can't be fine-tuned, so hide them from the training/studio page while keeping them available for inference on the chat page. - Add "gguf" to EXCLUDED_TAGS in HF model search hook - Filter local models with .gguf extension or -GGUF in ID --- .../studio/sections/model-section.tsx | 21 ++++++++++++++----- .../frontend/src/hooks/use-hf-model-search.ts | 1 + 2 files changed, 17 insertions(+), 5 deletions(-) diff --git a/studio/frontend/src/features/studio/sections/model-section.tsx b/studio/frontend/src/features/studio/sections/model-section.tsx index 0c4a71a8c3..5040e5bd77 100644 --- a/studio/frontend/src/features/studio/sections/model-section.tsx +++ b/studio/frontend/src/features/studio/sections/model-section.tsx @@ -170,14 +170,25 @@ export function ModelSection() { return ids; }, [hfResults, selectedModel]); + // Filter out GGUF models — they can't be used for training + const trainableLocalModels = useMemo( + () => + localModels.filter((m) => { + if (m.path.endsWith(".gguf")) return false; + if (m.id.toLowerCase().includes("-gguf")) return false; + return true; + }), + [localModels], + ); + const localMetaById = useMemo(() => { const map = new Map(); - for (const model of localModels) map.set(model.id, model); + for (const model of trainableLocalModels) map.set(model.id, model); return map; - }, [localModels]); + }, [trainableLocalModels]); const localResultIds = useMemo(() => { - const ids = localModels.map((model) => model.id); + const ids = trainableLocalModels.map((model) => model.id); const manual = localModelInput.trim(); if (manual && !ids.includes(manual)) { ids.unshift(manual); @@ -341,8 +352,8 @@ export function ModelSection() {

{localModelsError}

) : (

- {localModels.length > 0 - ? `${localModels.length} local/cached models found` + {trainableLocalModels.length > 0 + ? `${trainableLocalModels.length} local/cached models found` : "No local models found. Enter path manually."}

)} diff --git a/studio/frontend/src/hooks/use-hf-model-search.ts b/studio/frontend/src/hooks/use-hf-model-search.ts index 0029f13317..6ba70a4d5c 100644 --- a/studio/frontend/src/hooks/use-hf-model-search.ts +++ b/studio/frontend/src/hooks/use-hf-model-search.ts @@ -11,6 +11,7 @@ export interface HfModelResult { } const EXCLUDED_TAGS = new Set([ + "gguf", "gptq", "awq", "exl2", From a1e064b1c44924e8789a0190d5036c78b7c9179b Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Wed, 25 Feb 2026 16:00:24 +0400 Subject: [PATCH 17/49] Remove UNSLOTH_ENABLE_LOGGING from export pipeline --- studio/backend/core/export/export.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/studio/backend/core/export/export.py b/studio/backend/core/export/export.py index 865900cb65..dbe11ece52 100644 --- a/studio/backend/core/export/export.py +++ b/studio/backend/core/export/export.py @@ -390,9 +390,6 @@ class ExportBackend: # On WSL, patch out sudo check before llama.cpp build _apply_wsl_sudo_patch() - # Enable verbose logging so subprocess errors are printed - os.environ["UNSLOTH_ENABLE_LOGGING"] = "1" - # Pass absolute path — no os.chdir needed. # unsloth saves model files into this directory, while # check_llama_cpp("llama.cpp") resolves against cwd (repo root) From a8b5b7ed588b309f83ca401c07f2b66355533619 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Wed, 25 Feb 2026 16:21:08 +0400 Subject: [PATCH 18/49] Fix GGUF models missing from chat page model search GGUF was in the global EXCLUDED_TAGS set which filtered it from all consumers of useHfModelSearch, including the chat page. Move GGUF exclusion to an opt-in excludeGguf option so only training and onboarding pages filter out GGUF models. --- .../components/steps/model-selection-step.tsx | 1 + .../studio/sections/model-section.tsx | 1 + .../frontend/src/hooks/use-hf-model-search.ts | 43 +++++++++++-------- 3 files changed, 26 insertions(+), 19 deletions(-) diff --git a/studio/frontend/src/features/onboarding/components/steps/model-selection-step.tsx b/studio/frontend/src/features/onboarding/components/steps/model-selection-step.tsx index 747311c9a5..5cf4ebc0b5 100644 --- a/studio/frontend/src/features/onboarding/components/steps/model-selection-step.tsx +++ b/studio/frontend/src/features/onboarding/components/steps/model-selection-step.tsx @@ -85,6 +85,7 @@ export function ModelSelectionStep() { } = useHfModelSearch(debouncedQuery, { task, accessToken: hfToken || undefined, + excludeGguf: true, }); const { error: tokenValidationError, isChecking: isCheckingToken } = diff --git a/studio/frontend/src/features/studio/sections/model-section.tsx b/studio/frontend/src/features/studio/sections/model-section.tsx index 5b88664fd1..3d3ede7c1f 100644 --- a/studio/frontend/src/features/studio/sections/model-section.tsx +++ b/studio/frontend/src/features/studio/sections/model-section.tsx @@ -159,6 +159,7 @@ export function ModelSection() { } = useHfModelSearch(debouncedQuery, { task, accessToken: hfToken || undefined, + excludeGguf: true, }); const { error: tokenValidationError, isChecking: isCheckingToken } = diff --git a/studio/frontend/src/hooks/use-hf-model-search.ts b/studio/frontend/src/hooks/use-hf-model-search.ts index 6ba70a4d5c..8fc0b32cf8 100644 --- a/studio/frontend/src/hooks/use-hf-model-search.ts +++ b/studio/frontend/src/hooks/use-hf-model-search.ts @@ -11,7 +11,6 @@ export interface HfModelResult { } const EXCLUDED_TAGS = new Set([ - "gguf", "gptq", "awq", "exl2", @@ -45,22 +44,27 @@ function withPopularitySort( return fetch(url, init); } -function mapModel(raw: unknown): HfModelResult | null { - const m = raw as { - name: string; - downloads: number; - likes: number; - safetensors?: { total: number }; - tags?: string[]; - }; - if (m.tags?.some((t) => EXCLUDED_TAGS.has(t))) { - return null; - } - return { - id: m.name, - downloads: m.downloads, - likes: m.likes, - totalParams: m.safetensors?.total, +function makeMapModel(excludeGguf: boolean) { + return (raw: unknown): HfModelResult | null => { + const m = raw as { + name: string; + downloads: number; + likes: number; + safetensors?: { total: number }; + tags?: string[]; + }; + if (m.tags?.some((t) => EXCLUDED_TAGS.has(t))) { + return null; + } + if (excludeGguf && m.tags?.includes("gguf")) { + return null; + } + return { + id: m.name, + downloads: m.downloads, + likes: m.likes, + totalParams: m.safetensors?.total, + }; }; } @@ -113,9 +117,9 @@ async function* mergedModelIterator( export function useHfModelSearch( query: string, - options?: { task?: PipelineType; accessToken?: string }, + options?: { task?: PipelineType; accessToken?: string; excludeGguf?: boolean }, ) { - const { task, accessToken } = options ?? {}; + const { task, accessToken, excludeGguf = false } = options ?? {}; const createIter = useCallback( () => { @@ -135,6 +139,7 @@ export function useHfModelSearch( [query, task, accessToken], ); + const mapModel = useMemo(() => makeMapModel(excludeGguf), [excludeGguf]); const search = useHfPaginatedSearch(createIter, mapModel); // Secondary sort guarantee: unsloth models always float to the top From bfb140303277bafe5ba1fade10586be8e55dd1ec Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Wed, 25 Feb 2026 18:54:39 +0400 Subject: [PATCH 19/49] Relocate GGUF exports into exports/ directory --- studio/backend/core/export/export.py | 33 +++++++++++++++++-- .../src/features/export/export-page.tsx | 7 +++- 2 files changed, 37 insertions(+), 3 deletions(-) diff --git a/studio/backend/core/export/export.py b/studio/backend/core/export/export.py index cb2cb01c22..079ea0277d 100644 --- a/studio/backend/core/export/export.py +++ b/studio/backend/core/export/export.py @@ -2,9 +2,11 @@ """ Export backend - handles model exporting in various formats """ +import glob import json import logging import os +import shutil from pathlib import Path from typing import Optional, Tuple, List from peft import PeftModel, PeftModelForCausalLM @@ -409,9 +411,15 @@ class ExportBackend: # On WSL, patch out sudo check before llama.cpp build _apply_wsl_sudo_patch() + # Snapshot existing .gguf files in cwd before conversion. + # unsloth's convert_to_gguf writes output files relative to + # cwd (repo root), so we diff afterwards and relocate them. + cwd = os.getcwd() + pre_existing_ggufs = set(glob.glob(os.path.join(cwd, "*.gguf"))) + # Pass absolute path — no os.chdir needed. - # unsloth saves model files into this directory, while - # check_llama_cpp("llama.cpp") resolves against cwd (repo root) + # unsloth saves intermediate HF model files into model_save_path, + # while check_llama_cpp("llama.cpp") resolves against cwd (repo root) # where setup.sh already built llama.cpp with quantizer. model_save_path = os.path.join(abs_save_dir, "model") self.current_model.save_pretrained_gguf( @@ -420,6 +428,27 @@ class ExportBackend: quantization_method=quant_method ) + # Relocate GGUF artifacts into the export directory. + # convert_to_gguf writes .gguf files to cwd (repo root) + # because --outfile is a relative path like "model.Q4_K_M.gguf". + new_ggufs = set(glob.glob(os.path.join(cwd, "*.gguf"))) - pre_existing_ggufs + for src in sorted(new_ggufs): + dest = os.path.join(abs_save_dir, os.path.basename(src)) + shutil.move(src, dest) + logger.info(f"Relocated GGUF: {os.path.basename(src)} → {abs_save_dir}/") + + # Also check model_save_path for any .gguf files + if os.path.isdir(model_save_path): + for src in glob.glob(os.path.join(model_save_path, "*.gguf")): + dest = os.path.join(abs_save_dir, os.path.basename(src)) + shutil.move(src, dest) + logger.info(f"Relocated GGUF: {os.path.basename(src)} → {abs_save_dir}/") + + # Clean up intermediate HF model files (safetensors, config, etc.) + # since we only need the final .gguf output + shutil.rmtree(model_save_path, ignore_errors=True) + logger.info("Cleaned up intermediate HF model files") + logger.info(f"GGUF model saved successfully in {abs_save_dir}") # Push to hub if requested diff --git a/studio/frontend/src/features/export/export-page.tsx b/studio/frontend/src/features/export/export-page.tsx index 9d3e5053de..fa43c5eb98 100644 --- a/studio/frontend/src/features/export/export-page.tsx +++ b/studio/frontend/src/features/export/export-page.tsx @@ -155,7 +155,12 @@ export function ExportPage() { setExportError(null); setExportSuccess(false); - const saveDir = `./exports/${selectedModelIdx ?? "model"}/${checkpoint}`; + // For GGUF, use a flat folder like "exports/gemma-3-4b-it-finetune-gguf" + // For other formats, nest under training-run/checkpoint + const saveDir = + exportMethod === "gguf" + ? `./exports/${(baseModelName.split("/").pop() ?? selectedModelIdx ?? "model")}-finetune-gguf` + : `./exports/${selectedModelIdx ?? "model"}/${checkpoint}`; const pushToHub = destination === "hub"; const repoId = pushToHub && hfUsername && modelName ? `${hfUsername}/${modelName}` From c21cf2ffcf5e2c15990be8c8860e41145c969703 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Wed, 25 Feb 2026 19:01:47 +0400 Subject: [PATCH 20/49] Add GGUF tag for exported models in chat page selector --- studio/backend/models/models.py | 2 +- studio/backend/utils/models/model_config.py | 45 ++++++++++++++++--- .../assistant-ui/model-selector/pickers.tsx | 11 +++-- .../assistant-ui/model-selector/types.ts | 2 +- .../frontend/src/features/chat/types/api.ts | 2 +- .../src/features/chat/types/runtime.ts | 2 +- 6 files changed, 50 insertions(+), 14 deletions(-) diff --git a/studio/backend/models/models.py b/studio/backend/models/models.py index 39034f8ce2..8c7d0c037d 100644 --- a/studio/backend/models/models.py +++ b/studio/backend/models/models.py @@ -63,7 +63,7 @@ class LoRAInfo(BaseModel): adapter_path: str = Field(..., description="Path to the LoRA adapter or exported model") base_model: Optional[str] = Field(None, description="Base model identifier") source: Optional[str] = Field(None, description="'training' or 'exported'") - export_type: Optional[str] = Field(None, description="'lora' or 'merged' (for exports)") + export_type: Optional[str] = Field(None, description="'lora', 'merged', or 'gguf' (for exports)") class LoRAScanResponse(BaseModel): diff --git a/studio/backend/utils/models/model_config.py b/studio/backend/utils/models/model_config.py index 98e84c7d6c..a37deffff2 100644 --- a/studio/backend/utils/models/model_config.py +++ b/studio/backend/utils/models/model_config.py @@ -640,14 +640,15 @@ def scan_trained_loras(outputs_dir: str = "./outputs") -> List[Tuple[str, str]]: def scan_exported_models(exports_dir: str = "./exports") -> List[Tuple[str, str, str, Optional[str]]]: """ - Scan exports folder for exported models (merged, LoRA, base). - Skips GGUF-only exports (not loadable by Unsloth inference backend). + Scan exports folder for exported models (merged, LoRA, GGUF). - The exports directory is two levels deep: {run}/{checkpoint}/ + Supports two directory layouts: + - Two-level: {run}/{checkpoint}/ (merged & LoRA exports) + - Flat: {name}-finetune-gguf/ (GGUF exports) Returns: List of tuples: [(display_name, model_path, export_type, base_model), ...] - export_type: "lora" | "merged" + export_type: "lora" | "merged" | "gguf" """ results = [] exports_path = Path(exports_dir) @@ -659,6 +660,26 @@ def scan_exported_models(exports_dir: str = "./exports") -> List[Tuple[str, str, for run_dir in exports_path.iterdir(): if not run_dir.is_dir(): continue + + # Check for flat GGUF export (e.g. exports/gemma-3-4b-it-finetune-gguf/) + gguf_files = list(run_dir.glob("*.gguf")) + if gguf_files: + base_model = None + export_meta = run_dir / "export_metadata.json" + try: + if export_meta.exists(): + meta = json.loads(export_meta.read_text()) + base_model = meta.get("base_model") + except Exception: + pass + + display_name = run_dir.name + model_path = str(gguf_files[0]) # path to the .gguf file + results.append((display_name, model_path, "gguf", base_model)) + logger.debug(f"Found GGUF export: {display_name}") + continue + + # Two-level: {run}/{checkpoint}/ for checkpoint_dir in run_dir.iterdir(): if not checkpoint_dir.is_dir(): continue @@ -683,7 +704,6 @@ def scan_exported_models(exports_dir: str = "./exports") -> List[Tuple[str, str, pass elif config_file.exists() and has_weights: export_type = "merged" - # Read base model from export_metadata.json (written at export time) export_meta = checkpoint_dir / "export_metadata.json" try: if export_meta.exists(): @@ -692,7 +712,20 @@ def scan_exported_models(exports_dir: str = "./exports") -> List[Tuple[str, str, except Exception: pass elif has_gguf: - # GGUF-only — not loadable by current inference backend + export_type = "gguf" + gguf_list = list(checkpoint_dir.glob("*.gguf")) + export_meta = checkpoint_dir / "export_metadata.json" + try: + if export_meta.exists(): + meta = json.loads(export_meta.read_text()) + base_model = meta.get("base_model") + except Exception: + pass + + display_name = f"{run_dir.name} / {checkpoint_dir.name}" + model_path = str(gguf_list[0]) if gguf_list else str(checkpoint_dir) + results.append((display_name, model_path, export_type, base_model)) + logger.debug(f"Found GGUF export: {display_name}") continue else: continue diff --git a/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx b/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx index 02205248ea..26f0e6e028 100644 --- a/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx +++ b/studio/frontend/src/components/assistant-ui/model-selector/pickers.tsx @@ -525,9 +525,12 @@ export function LoraModelPicker({ {adapters.map((adapter) => { const isExported = adapter.source === "exported"; const isMerged = adapter.exportType === "merged"; - const tag = isExported - ? isMerged ? "Merged" : "LoRA" - : "LoRA"; + const isGguf = adapter.exportType === "gguf"; + const tag = isGguf + ? "GGUF" + : isExported + ? isMerged ? "Merged" : "LoRA" + : "LoRA"; const meta = isExported ? `${tag} · Exported` : tag; return ( onSelect(adapter.id, { source: isExported ? "exported" : "lora", - isLora: !isMerged, + isLora: !isMerged && !isGguf, })} /> ); diff --git a/studio/frontend/src/components/assistant-ui/model-selector/types.ts b/studio/frontend/src/components/assistant-ui/model-selector/types.ts index 43f5e935b3..0e8cf5fb4d 100644 --- a/studio/frontend/src/components/assistant-ui/model-selector/types.ts +++ b/studio/frontend/src/components/assistant-ui/model-selector/types.ts @@ -11,7 +11,7 @@ export interface LoraModelOption extends ModelOption { baseModel?: string; updatedAt?: number; source?: "training" | "exported"; - exportType?: "lora" | "merged"; + exportType?: "lora" | "merged" | "gguf"; } export interface ModelSelectorChangeMeta { diff --git a/studio/frontend/src/features/chat/types/api.ts b/studio/frontend/src/features/chat/types/api.ts index 5dd0cd7a6b..d0b37f5cce 100644 --- a/studio/frontend/src/features/chat/types/api.ts +++ b/studio/frontend/src/features/chat/types/api.ts @@ -16,7 +16,7 @@ export interface BackendLoraInfo { adapter_path: string; base_model?: string | null; source?: "training" | "exported" | null; - export_type?: "lora" | "merged" | null; + export_type?: "lora" | "merged" | "gguf" | null; } export interface ListLorasResponse { diff --git a/studio/frontend/src/features/chat/types/runtime.ts b/studio/frontend/src/features/chat/types/runtime.ts index 953f4ebbaa..710898713b 100644 --- a/studio/frontend/src/features/chat/types/runtime.ts +++ b/studio/frontend/src/features/chat/types/runtime.ts @@ -35,5 +35,5 @@ export interface ChatLoraSummary { baseModel: string; updatedAt?: number; source?: "training" | "exported"; - exportType?: "lora" | "merged"; + exportType?: "lora" | "merged" | "gguf"; } From 62c6fd9f46757f96daa085f2b920b2ac967659ef Mon Sep 17 00:00:00 2001 From: Leo Borcherding Date: Wed, 25 Feb 2026 17:27:10 -0600 Subject: [PATCH 21/49] fix(attachment): replace never exhaustive check to fix Colab TS2322 build error `attachment.type` resolves to `string & {}` via @assistant-ui/store@0.1.6's generic type chain when installed through npm (package-lock.json), breaking the `const _exhaustiveCheck: never = type` exhaustive check pattern. Replace with a direct throw that compiles cleanly across library versions while preserving identical runtime behaviour. Fixes #263 Co-Authored-By: Claude Sonnet 4.6 --- studio/frontend/src/components/assistant-ui/attachment.tsx | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/studio/frontend/src/components/assistant-ui/attachment.tsx b/studio/frontend/src/components/assistant-ui/attachment.tsx index c53c134ea2..94e30fb4ac 100644 --- a/studio/frontend/src/components/assistant-ui/attachment.tsx +++ b/studio/frontend/src/components/assistant-ui/attachment.tsx @@ -149,10 +149,8 @@ const AttachmentUI: FC = () => { return "Document"; case "file": return "File"; - default: { - const _exhaustiveCheck: never = type; - throw new Error(`Unknown attachment type: ${_exhaustiveCheck}`); - } + default: + throw new Error(`Unknown attachment type: ${type as string}`); } }); From 7bc752d5f914db6d68bd123b1972030df67bb85d Mon Sep 17 00:00:00 2001 From: samit Date: Wed, 25 Feb 2026 18:08:02 -0800 Subject: [PATCH 22/49] passed checkpoint as a parameter to presets --- studio/frontend/src/features/chat/chat-settings-sheet.tsx | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/studio/frontend/src/features/chat/chat-settings-sheet.tsx b/studio/frontend/src/features/chat/chat-settings-sheet.tsx index 2d7ae781fa..1bc61cd25d 100644 --- a/studio/frontend/src/features/chat/chat-settings-sheet.tsx +++ b/studio/frontend/src/features/chat/chat-settings-sheet.tsx @@ -165,7 +165,11 @@ export function ChatSettingsPanel({ function applyPreset(name: string) { const p = presets.find((pr) => pr.name === name); if (p) { - onParamsChange({ ...p.params, systemPrompt: params.systemPrompt }); + onParamsChange({ + ...p.params, + systemPrompt: params.systemPrompt, + checkpoint: params.checkpoint, + }); setActivePreset(name); } } From 1c55e2fbaa98b2497c4d9d36bb1fbd32ec194238 Mon Sep 17 00:00:00 2001 From: imagineer99 Date: Thu, 26 Feb 2026 03:57:55 +0000 Subject: [PATCH 23/49] fix: remove dataset metadata badges from HF dataset dropdowns --- .../components/steps/dataset-step.tsx | 21 ++----------------- .../studio/sections/dataset-section.tsx | 21 +------------------ 2 files changed, 3 insertions(+), 39 deletions(-) diff --git a/studio/frontend/src/features/onboarding/components/steps/dataset-step.tsx b/studio/frontend/src/features/onboarding/components/steps/dataset-step.tsx index 1ad4fad8f0..99434f1285 100644 --- a/studio/frontend/src/features/onboarding/components/steps/dataset-step.tsx +++ b/studio/frontend/src/features/onboarding/components/steps/dataset-step.tsx @@ -38,7 +38,7 @@ import { useHfTokenValidation, useInfiniteScroll, } from "@/hooks"; -import { cn, formatCompact } from "@/lib/utils"; +import { cn } from "@/lib/utils"; import { HfDatasetSubsetSplitSelectors, useTrainingConfigStore, @@ -251,16 +251,8 @@ export function DatasetStep() { > {(id: string) => { - const r = hfResults.find((r) => r.id === id); - const detail = r?.totalExamples - ? `${formatCompact(r.totalExamples)} rows` - : (r?.sizeCategory ?? null); return ( - + @@ -274,15 +266,6 @@ export function DatasetStep() { {id} - {detail ? ( - - {detail} - - ) : r?.downloads != null ? ( - - ↓{formatCompact(r.downloads)} - - ) : null} ); }} diff --git a/studio/frontend/src/features/studio/sections/dataset-section.tsx b/studio/frontend/src/features/studio/sections/dataset-section.tsx index 6f6a6a401e..cf0fb7d90f 100644 --- a/studio/frontend/src/features/studio/sections/dataset-section.tsx +++ b/studio/frontend/src/features/studio/sections/dataset-section.tsx @@ -33,7 +33,6 @@ import { useHfTokenValidation, useInfiniteScroll, } from "@/hooks"; -import { formatCompact } from "@/lib/utils"; import { HfDatasetSubsetSplitSelectors, useDatasetPreviewDialogStore, @@ -221,21 +220,8 @@ export function DatasetSection() { > {(id: string) => { - const r = hfResults.find((ds) => ds.id === id); - let detail: string | null = null; - if (r?.totalExamples) { - detail = `${formatCompact(r.totalExamples)} rows`; - } else if (r?.sizeCategory) { - detail = r.sizeCategory; - } else if (r?.downloads != null) { - detail = `↓${formatCompact(r.downloads)}`; - } return ( - + @@ -249,11 +235,6 @@ export function DatasetSection() { {id} - {detail && ( - - {detail} - - )} ); }} From 6e535ed0ebf9c7c55aa9dde5c8f32f29560c3553 Mon Sep 17 00:00:00 2001 From: imagineer99 Date: Thu, 26 Feb 2026 06:27:52 +0000 Subject: [PATCH 24/49] fix: filter OCR datasets from non-vision hub results --- .../src/hooks/use-hf-dataset-search.ts | 34 +++++++++++++++++-- 1 file changed, 32 insertions(+), 2 deletions(-) diff --git a/studio/frontend/src/hooks/use-hf-dataset-search.ts b/studio/frontend/src/hooks/use-hf-dataset-search.ts index b00dcee1ed..32879fd015 100644 --- a/studio/frontend/src/hooks/use-hf-dataset-search.ts +++ b/studio/frontend/src/hooks/use-hf-dataset-search.ts @@ -138,6 +138,7 @@ const INCOMPATIBLE_TASKS_ALL_MODELS = new Set([ ]); const PRETRAINING_PLAIN_TAGS = new Set(["pretraining", "pre-training"]); +const OCR_PLAIN_TAGS = new Set(["ocr", "document-ocr"]); const PRETRAINING_SIZE_CATEGORIES = new Set([ "100M1T", ]); +const OCR_OR_VISION_TEXT_TASKS = new Set([ + "image-to-text", + "image-captioning", + "visual-question-answering", + "document-question-answering", +]); + const INCOMPATIBLE_TASKS_BY_MODEL: Record> = { text: new Set([ "text-to-image", @@ -232,6 +240,16 @@ function rankDatasetRelevance( ): DatasetRelevance { if (isPretrainingDataset(dataset)) return "incompatible"; + // Keep OCR / vision-text corpora out of non-vision defaults. + if (modelType !== "vision") { + if ( + dataset.plainTags.some((t) => OCR_PLAIN_TAGS.has(t.toLowerCase())) || + dataset.taskCategories.some((t) => OCR_OR_VISION_TEXT_TASKS.has(t)) + ) { + return "incompatible"; + } + } + const { taskCategories } = dataset; if (taskCategories.length === 0) return "neutral"; @@ -248,6 +266,13 @@ function rankDatasetRelevance( return "neutral"; } +function isOcrOrVisionTextDataset(dataset: HfDatasetResult): boolean { + return ( + dataset.plainTags.some((t) => OCR_PLAIN_TAGS.has(t.toLowerCase())) || + dataset.taskCategories.some((t) => OCR_OR_VISION_TEXT_TASKS.has(t)) + ); +} + export function useHfDatasetSearch( query: string, options?: { modelType?: ModelType | null; accessToken?: string }, @@ -267,12 +292,17 @@ export function useHfDatasetSearch( const search = useHfPaginatedSearch(createIter, mapDataset); const results = useMemo(() => { - if (!modelType) return search.results; + const hideOcr = modelType !== "vision"; + const baseResults = hideOcr + ? search.results.filter((ds) => !isOcrOrVisionTextDataset(ds)) + : search.results; + + if (!modelType) return baseResults; const boosted: HfDatasetResult[] = []; const neutral: HfDatasetResult[] = []; - for (const ds of search.results) { + for (const ds of baseResults) { const relevance = rankDatasetRelevance(ds, modelType); if (relevance === "boosted") boosted.push(ds); else if (relevance !== "incompatible") neutral.push(ds); From 852dff564ed28684b30d611cdd2f3c8830d392ab Mon Sep 17 00:00:00 2001 From: imagineer99 Date: Thu, 26 Feb 2026 06:32:45 +0000 Subject: [PATCH 25/49] feat: added datasets of size 5M and 10M to pretraining size category --- studio/frontend/src/hooks/use-hf-dataset-search.ts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/studio/frontend/src/hooks/use-hf-dataset-search.ts b/studio/frontend/src/hooks/use-hf-dataset-search.ts index 32879fd015..ec64ffd7e6 100644 --- a/studio/frontend/src/hooks/use-hf-dataset-search.ts +++ b/studio/frontend/src/hooks/use-hf-dataset-search.ts @@ -141,6 +141,8 @@ const PRETRAINING_PLAIN_TAGS = new Set(["pretraining", "pre-training"]); const OCR_PLAIN_TAGS = new Set(["ocr", "document-ocr"]); const PRETRAINING_SIZE_CATEGORIES = new Set([ + "5M Date: Thu, 26 Feb 2026 10:54:49 +0400 Subject: [PATCH 26/49] Add gguf to frontend export_type unions --- .../src/components/assistant-ui/model-selector/types.ts | 2 +- studio/frontend/src/features/chat/types/api.ts | 2 +- studio/frontend/src/features/chat/types/runtime.ts | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/studio/frontend/src/components/assistant-ui/model-selector/types.ts b/studio/frontend/src/components/assistant-ui/model-selector/types.ts index 43f5e935b3..0e8cf5fb4d 100644 --- a/studio/frontend/src/components/assistant-ui/model-selector/types.ts +++ b/studio/frontend/src/components/assistant-ui/model-selector/types.ts @@ -11,7 +11,7 @@ export interface LoraModelOption extends ModelOption { baseModel?: string; updatedAt?: number; source?: "training" | "exported"; - exportType?: "lora" | "merged"; + exportType?: "lora" | "merged" | "gguf"; } export interface ModelSelectorChangeMeta { diff --git a/studio/frontend/src/features/chat/types/api.ts b/studio/frontend/src/features/chat/types/api.ts index 5dd0cd7a6b..d0b37f5cce 100644 --- a/studio/frontend/src/features/chat/types/api.ts +++ b/studio/frontend/src/features/chat/types/api.ts @@ -16,7 +16,7 @@ export interface BackendLoraInfo { adapter_path: string; base_model?: string | null; source?: "training" | "exported" | null; - export_type?: "lora" | "merged" | null; + export_type?: "lora" | "merged" | "gguf" | null; } export interface ListLorasResponse { diff --git a/studio/frontend/src/features/chat/types/runtime.ts b/studio/frontend/src/features/chat/types/runtime.ts index 953f4ebbaa..710898713b 100644 --- a/studio/frontend/src/features/chat/types/runtime.ts +++ b/studio/frontend/src/features/chat/types/runtime.ts @@ -35,5 +35,5 @@ export interface ChatLoraSummary { baseModel: string; updatedAt?: number; source?: "training" | "exported"; - exportType?: "lora" | "merged"; + exportType?: "lora" | "merged" | "gguf"; } From 1b822a943c342f70d1d4005c2ee14baaadf63c5f Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Thu, 26 Feb 2026 10:56:26 +0400 Subject: [PATCH 27/49] Revert "Add gguf to frontend export_type unions" This reverts commit 782af3994911f426fbe6c128437dc11e2724b347. --- .../src/components/assistant-ui/model-selector/types.ts | 2 +- studio/frontend/src/features/chat/types/api.ts | 2 +- studio/frontend/src/features/chat/types/runtime.ts | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/studio/frontend/src/components/assistant-ui/model-selector/types.ts b/studio/frontend/src/components/assistant-ui/model-selector/types.ts index 0e8cf5fb4d..43f5e935b3 100644 --- a/studio/frontend/src/components/assistant-ui/model-selector/types.ts +++ b/studio/frontend/src/components/assistant-ui/model-selector/types.ts @@ -11,7 +11,7 @@ export interface LoraModelOption extends ModelOption { baseModel?: string; updatedAt?: number; source?: "training" | "exported"; - exportType?: "lora" | "merged" | "gguf"; + exportType?: "lora" | "merged"; } export interface ModelSelectorChangeMeta { diff --git a/studio/frontend/src/features/chat/types/api.ts b/studio/frontend/src/features/chat/types/api.ts index d0b37f5cce..5dd0cd7a6b 100644 --- a/studio/frontend/src/features/chat/types/api.ts +++ b/studio/frontend/src/features/chat/types/api.ts @@ -16,7 +16,7 @@ export interface BackendLoraInfo { adapter_path: string; base_model?: string | null; source?: "training" | "exported" | null; - export_type?: "lora" | "merged" | "gguf" | null; + export_type?: "lora" | "merged" | null; } export interface ListLorasResponse { diff --git a/studio/frontend/src/features/chat/types/runtime.ts b/studio/frontend/src/features/chat/types/runtime.ts index 710898713b..953f4ebbaa 100644 --- a/studio/frontend/src/features/chat/types/runtime.ts +++ b/studio/frontend/src/features/chat/types/runtime.ts @@ -35,5 +35,5 @@ export interface ChatLoraSummary { baseModel: string; updatedAt?: number; source?: "training" | "exported"; - exportType?: "lora" | "merged" | "gguf"; + exportType?: "lora" | "merged"; } From 2ce63f09c4a60fd7af6777de3f558b9c08c7bcd5 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Thu, 26 Feb 2026 10:59:39 +0400 Subject: [PATCH 28/49] Add gguf to toLoraSummary inline type --- .../frontend/src/features/chat/hooks/use-chat-model-runtime.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts b/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts index c3464f9071..a0ea17e304 100644 --- a/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts +++ b/studio/frontend/src/features/chat/hooks/use-chat-model-runtime.ts @@ -75,7 +75,7 @@ function toLoraSummary(lora: { adapter_path: string; base_model?: string | null; source?: "training" | "exported" | null; - export_type?: "lora" | "merged" | null; + export_type?: "lora" | "merged" | "gguf" | null; }): ChatLoraSummary { const idTail = lora.adapter_path.split("/").filter(Boolean).at(-1) ?? ""; const updatedAt = From 90f012a444279314d5ace74b376e0f1959ad5b0a Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Thu, 26 Feb 2026 11:24:32 +0400 Subject: [PATCH 29/49] Write export metadata for GGUF exports to fix Unknown base model --- studio/backend/core/export/export.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/studio/backend/core/export/export.py b/studio/backend/core/export/export.py index 079ea0277d..39e956eb6f 100644 --- a/studio/backend/core/export/export.py +++ b/studio/backend/core/export/export.py @@ -449,6 +449,9 @@ class ExportBackend: shutil.rmtree(model_save_path, ignore_errors=True) logger.info("Cleaned up intermediate HF model files") + # Write export metadata so the Chat page can identify the base model + self._write_export_metadata(abs_save_dir) + logger.info(f"GGUF model saved successfully in {abs_save_dir}") # Push to hub if requested From ed18f9b9dd51d4a4943f8b0cefbe858f0cce7a91 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Thu, 26 Feb 2026 11:35:04 +0400 Subject: [PATCH 30/49] Flatten GGUF subdirs in export and fix metadata lookup in scanner --- studio/backend/core/export/export.py | 24 +++++++++++---------- studio/backend/utils/models/model_config.py | 19 ++++++++++------ 2 files changed, 25 insertions(+), 18 deletions(-) diff --git a/studio/backend/core/export/export.py b/studio/backend/core/export/export.py index 39e956eb6f..bc4e267f75 100644 --- a/studio/backend/core/export/export.py +++ b/studio/backend/core/export/export.py @@ -437,17 +437,19 @@ class ExportBackend: shutil.move(src, dest) logger.info(f"Relocated GGUF: {os.path.basename(src)} → {abs_save_dir}/") - # Also check model_save_path for any .gguf files - if os.path.isdir(model_save_path): - for src in glob.glob(os.path.join(model_save_path, "*.gguf")): - dest = os.path.join(abs_save_dir, os.path.basename(src)) - shutil.move(src, dest) - logger.info(f"Relocated GGUF: {os.path.basename(src)} → {abs_save_dir}/") - - # Clean up intermediate HF model files (safetensors, config, etc.) - # since we only need the final .gguf output - shutil.rmtree(model_save_path, ignore_errors=True) - logger.info("Cleaned up intermediate HF model files") + # Flatten any .gguf files from subdirectories into abs_save_dir. + # save_pretrained_gguf may create subdirs (e.g. model_gguf/) + # with a name different from model_save_path. + for sub in list(Path(abs_save_dir).iterdir()): + if not sub.is_dir(): + continue + for src in sub.glob("*.gguf"): + dest = os.path.join(abs_save_dir, src.name) + shutil.move(str(src), dest) + logger.info(f"Relocated GGUF: {src.name} → {abs_save_dir}/") + # Clean up the subdirectory (intermediate HF files, etc.) + shutil.rmtree(str(sub), ignore_errors=True) + logger.info(f"Cleaned up subdirectory: {sub.name}") # Write export metadata so the Chat page can identify the base model self._write_export_metadata(abs_save_dir) diff --git a/studio/backend/utils/models/model_config.py b/studio/backend/utils/models/model_config.py index a37deffff2..8bb9e60ed2 100644 --- a/studio/backend/utils/models/model_config.py +++ b/studio/backend/utils/models/model_config.py @@ -714,13 +714,18 @@ def scan_exported_models(exports_dir: str = "./exports") -> List[Tuple[str, str, elif has_gguf: export_type = "gguf" gguf_list = list(checkpoint_dir.glob("*.gguf")) - export_meta = checkpoint_dir / "export_metadata.json" - try: - if export_meta.exists(): - meta = json.loads(export_meta.read_text()) - base_model = meta.get("base_model") - except Exception: - pass + # Check checkpoint_dir first, then fall back to parent run_dir + # (export.py writes metadata to the top-level export directory) + for meta_dir in (checkpoint_dir, run_dir): + export_meta = meta_dir / "export_metadata.json" + try: + if export_meta.exists(): + meta = json.loads(export_meta.read_text()) + base_model = meta.get("base_model") + if base_model: + break + except Exception: + pass display_name = f"{run_dir.name} / {checkpoint_dir.name}" model_path = str(gguf_list[0]) if gguf_list else str(checkpoint_dir) From e81516320d4669c326e39c76f0c154b1e9dbf4e8 Mon Sep 17 00:00:00 2001 From: Roland Tannous <115670425+rolandtannous@users.noreply.github.com> Date: Thu, 26 Feb 2026 11:54:16 +0400 Subject: [PATCH 31/49] fix(setup): restrict Python to >=3.11 and <3.14 Adds lower bound (>= 3.11) and tightens upper bound (< 3.14) for Python version discovery in setup.sh. Extracts bounds into MIN_PY_MINOR / MAX_PY_MINOR variables for easy future updates. --- setup.sh | 21 ++++++++++++++------- 1 file changed, 14 insertions(+), 7 deletions(-) diff --git a/setup.sh b/setup.sh index 55fa57c83f..8da67e5667 100755 --- a/setup.sh +++ b/setup.sh @@ -105,7 +105,9 @@ echo "✅ Frontend built to studio/frontend/dist" echo "" echo "Setting up Python environment..." -# ── 6a. Discover best Python <= 3.12.x ── +# ── 6a. Discover best Python >= 3.11 and < 3.14 (i.e. 3.11.x, 3.12.x, or 3.13.x) ── +MIN_PY_MINOR=11 # minimum minor version (>= 3.11) +MAX_PY_MINOR=13 # maximum minor version (< 3.14) BEST_PY="" BEST_MAJOR=0 BEST_MINOR=0 @@ -115,7 +117,7 @@ for candidate in $(compgen -c python3 2>/dev/null | grep -E '^python3(\.[0-9]+)? if ! command -v "$candidate" &>/dev/null; then continue fi - # Get version string, e.g. "Python 3.11.5" + # Get version string, e.g. "Python 3.12.5" ver_str=$("$candidate" --version 2>&1 | awk '{print $2}') py_major=$(echo "$ver_str" | cut -d. -f1) py_minor=$(echo "$ver_str" | cut -d. -f2) @@ -125,8 +127,13 @@ for candidate in $(compgen -c python3 2>/dev/null | grep -E '^python3(\.[0-9]+)? continue fi - # Skip versions above 3.12 - if [ "$py_minor" -gt 12 ] 2>/dev/null; then + # Skip versions below 3.12 (require > 3.11) + if [ "$py_minor" -lt "$MIN_PY_MINOR" ] 2>/dev/null; then + continue + fi + + # Skip versions above 3.13 (require < 3.14) + if [ "$py_minor" -gt "$MAX_PY_MINOR" ] 2>/dev/null; then continue fi @@ -139,7 +146,7 @@ for candidate in $(compgen -c python3 2>/dev/null | grep -E '^python3(\.[0-9]+)? done if [ -z "$BEST_PY" ]; then - echo "❌ ERROR: No Python version <= 3.12.x found on this system." + echo "❌ ERROR: No Python version between 3.${MIN_PY_MINOR} and 3.${MAX_PY_MINOR} found on this system." echo " Detected Python 3 installations:" for candidate in $(compgen -c python3 2>/dev/null | grep -E '^python3(\.[0-9]+)?$' | sort -u); do if command -v "$candidate" &>/dev/null; then @@ -147,13 +154,13 @@ if [ -z "$BEST_PY" ]; then fi done echo "" - echo " Please install Python <= 3.12.x for maximum compatibility." + echo " Please install Python 3.${MIN_PY_MINOR} or 3.${MAX_PY_MINOR}." echo " For example: sudo apt install python3.12 python3.12-venv" exit 1 fi BEST_VER=$("$BEST_PY" --version 2>&1 | awk '{print $2}') -echo "✅ Using $BEST_PY ($BEST_VER) — compatible (≤ 3.12.x)" +echo "✅ Using $BEST_PY ($BEST_VER) — compatible (3.${MIN_PY_MINOR}.x – 3.${MAX_PY_MINOR}.x)" REQ_ROOT="$SCRIPT_DIR/studio/backend/requirements" SINGLE_ENV_CONSTRAINTS="$REQ_ROOT/single-env/constraints.txt" From 1de4b7324405c067f4f9e4f2cd64760813d1ece8 Mon Sep 17 00:00:00 2001 From: Shine1i Date: Thu, 26 Feb 2026 10:03:47 +0100 Subject: [PATCH 32/49] fix: recipe studio dialog combobox click-select + simplify model provider form --- .../frontend/src/components/ui/combobox.tsx | 28 ++-- .../components/inline/inline-model.tsx | 26 ++-- .../recipe-studio/dialogs/llm/general-tab.tsx | 4 +- .../dialogs/models/model-provider-dialog.tsx | 135 ++++++++++-------- .../utils/payload/builders-model.ts | 2 +- 5 files changed, 103 insertions(+), 92 deletions(-) diff --git a/studio/frontend/src/components/ui/combobox.tsx b/studio/frontend/src/components/ui/combobox.tsx index 9c1e970c57..8ccc40c95f 100644 --- a/studio/frontend/src/components/ui/combobox.tsx +++ b/studio/frontend/src/components/ui/combobox.tsx @@ -162,20 +162,20 @@ function ComboboxContent({ - + align={align} + alignOffset={alignOffset} + anchor={anchor} + className="isolate z-[120] pointer-events-auto" + > + ); diff --git a/studio/frontend/src/features/recipe-studio/components/inline/inline-model.tsx b/studio/frontend/src/features/recipe-studio/components/inline/inline-model.tsx index 5316b0f297..2c2c2ffd41 100644 --- a/studio/frontend/src/features/recipe-studio/components/inline/inline-model.tsx +++ b/studio/frontend/src/features/recipe-studio/components/inline/inline-model.tsx @@ -14,19 +14,6 @@ export function InlineModel(props: InlineModelProps): ReactElement { if (props.config.kind === "model_provider") { return (
- - - props.onUpdate({ - // biome-ignore lint/style/useNamingConvention: api schema - provider_type: event.target.value, - }) - } - /> - props.onUpdate({ endpoint: event.target.value })} /> + + + props.onUpdate({ + // biome-ignore lint/style/useNamingConvention: api schema + api_key: event.target.value, + }) + } + /> +
); } diff --git a/studio/frontend/src/features/recipe-studio/dialogs/llm/general-tab.tsx b/studio/frontend/src/features/recipe-studio/dialogs/llm/general-tab.tsx index 41153626b0..a0ab8e23fc 100644 --- a/studio/frontend/src/features/recipe-studio/dialogs/llm/general-tab.tsx +++ b/studio/frontend/src/features/recipe-studio/dialogs/llm/general-tab.tsx @@ -147,7 +147,7 @@ export function LlmGeneralTab({ />