diff --git a/.github/workflows/mlx-ci.yml b/.github/workflows/mlx-ci.yml index 4e70ec487e..dbe35e721e 100644 --- a/.github/workflows/mlx-ci.yml +++ b/.github/workflows/mlx-ci.yml @@ -249,28 +249,30 @@ jobs: # at upstream ggml-org/llama.cpp; that repo doesn't ship the # llama-prebuilt-manifest.json asset Studio's default policy # expects, so the simple platform-specific policy maps - # Darwin+arm64 -> bin-macos-arm64 directly. setup.sh passes - # this flag too. + # Darwin+arm64 -> bin-macos-arm64 directly. studio/setup.sh + # passes both --published-repo ggml-org/llama.cpp AND + # --simple-policy automatically on macOS, so this CI step + # exercises the same code path users hit when they run + # `curl -fsSL https://unsloth.ai/install.sh | sh`. python studio/install_llama_prebuilt.py \ --install-dir "$INSTALL_DIR" \ --published-repo ggml-org/llama.cpp \ --published-release-tag b9049 \ --simple-policy - LLAMA_CLI="" - for c in \ - "$INSTALL_DIR/build/bin/llama-cli" \ - "$INSTALL_DIR/llama-cli" \ - "$INSTALL_DIR/bin/llama-cli"; do - if [ -x "$c" ]; then LLAMA_CLI="$c"; break; fi - done - if [ -z "$LLAMA_CLI" ]; then - echo "::error::llama-cli not found under $INSTALL_DIR" - find "$INSTALL_DIR" -maxdepth 4 -type f -name 'llama-*' || true - exit 1 - fi - echo "found llama-cli at: $LLAMA_CLI" - "$LLAMA_CLI" --version || true + # Studio bundles only llama-server + llama-quantize from the + # prebuilt (not llama-cli) -- inference goes through + # llama-server's HTTP /completion endpoint. Validate both: + # llama-quantize --help proves the dynamic libs link, then + # spin up llama-server and POST a /completion request on a + # tiny published GGUF. + LLAMA_SERVER="$INSTALL_DIR/build/bin/llama-server" + LLAMA_QUANT="$INSTALL_DIR/build/bin/llama-quantize" + [ -x "$LLAMA_SERVER" ] || { echo "::error::llama-server missing at $LLAMA_SERVER"; find "$INSTALL_DIR/build" -type f | head -40; exit 1; } + [ -x "$LLAMA_QUANT" ] || { echo "::error::llama-quantize missing at $LLAMA_QUANT"; exit 1; } + echo "llama-server : $LLAMA_SERVER" + echo "llama-quantize: $LLAMA_QUANT" + "$LLAMA_QUANT" --help >/dev/null && echo " llama-quantize loads OK" mkdir -p /tmp/ggufs python -c " @@ -283,26 +285,48 @@ jobs: print('downloaded:', p) " - PROMPT="Hello, my name is" - echo "=== llama-cli inference ===" - OUT=$("$LLAMA_CLI" \ + PORT=18080 + echo "=== starting llama-server on 127.0.0.1:$PORT ===" + "$LLAMA_SERVER" \ -m /tmp/ggufs/gemma-3-270m-it-Q4_K_M.gguf \ - -p "$PROMPT" \ + --host 127.0.0.1 \ + --port "$PORT" \ + -c 256 \ -n 16 \ - --temp 0 \ - --seed 3407 \ - -no-cnv \ - --no-warmup 2>&1) || { - echo "::error::llama-cli exited non-zero" - echo "$OUT" | head -80 - exit 1 - } - echo "$OUT" | tail -40 - if ! echo "$OUT" | grep -q "Hello"; then - echo "::error::llama-cli output did not contain the prompt echo 'Hello'" + --no-warmup \ + > /tmp/llama-server.log 2>&1 & + SERVER_PID=$! + trap 'kill "$SERVER_PID" 2>/dev/null || true' EXIT + + # Wait for /health to come up + for i in $(seq 1 30); do + if curl -sf "http://127.0.0.1:$PORT/health" >/dev/null 2>&1; then + echo " server up after ${i}s" + break + fi + sleep 1 + done + if ! curl -sf "http://127.0.0.1:$PORT/health" >/dev/null 2>&1; then + echo "::error::llama-server never became healthy" + tail -40 /tmp/llama-server.log exit 1 fi - echo "OK: Studio prebuilt llama.cpp on Mac M1 + GGUF inference works" + + PROMPT="Hello, my name is" + echo "=== POST /completion ===" + RESP=$(curl -sf -X POST "http://127.0.0.1:$PORT/completion" \ + -H 'Content-Type: application/json' \ + -d "{\"prompt\":\"$PROMPT\",\"n_predict\":16,\"temperature\":0,\"seed\":3407}") + echo "raw response (head): $(echo "$RESP" | head -c 600)" + CONTENT=$(echo "$RESP" | python -c "import json,sys; print(json.loads(sys.stdin.read()).get('content',''))") + echo "completion content: $CONTENT" + + if [ -z "$CONTENT" ]; then + echo "::error::llama-server /completion returned empty content" + tail -40 /tmp/llama-server.log + exit 1 + fi + echo "OK: Studio prebuilt llama.cpp on Mac M1 + GGUF /completion works" # Real MLX training + inference smoke test. Trains # unsloth/gemma-3-270m-it for 7 deterministic LoRA steps