Collapse null pyworker client to a single mode parameterized by --count

Now that the session model means no HTTP connection is held during the reservation, the dichotomy between "single reserve" and "trapezoid demo" collapses — both are "open N sessions, each held for H seconds, started I seconds apart, close." Replace --reserve/--demo/--duration/--plateau with --count/--hold/--interval. --session-cost becomes --cost. Client is now 64 lines (down from 120). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
Simplify null pyworker code and docs
2026-05-12 12:18:33 +01:00 · 2026-05-12 11:50:03 +01:00 · 2026-05-12 11:40:51 +01:00 · 2026-05-12 11:34:52 +01:00 · 2026-05-12 11:31:26 +01:00 · 2026-05-12 11:14:20 +01:00
7 changed files with 439 additions and 40 deletions
@@ -1,13 +1,2 @@
-aiohttp==3.10.1
-aiodns~=3.6.0
-pycares~=4.11.0
-anyio~=4.4
-lib~=4.0
-nltk~=3.9
-psutil~=6.0
-pycryptodome~=3.20
-Requests~=2.32
-transformers~=4.52
-utils==1.0.*
-hf_transfer>=0.1.9
 vastai-sdk>=0.3.0
+nltk==3.9.4 
@@ -2,10 +2,17 @@

 set -e -o pipefail

+# Check for force update flag
+FORCE_UPDATE=false
+if [ -f "/.force_update" ]; then
+    echo "Force update flag detected at /.force_update"
+    FORCE_UPDATE=true
+fi
+
 WORKSPACE_DIR="${WORKSPACE_DIR:-/workspace}"

 SERVER_DIR="$WORKSPACE_DIR/vast-pyworker"
-ENV_PATH="$WORKSPACE_DIR/worker-env"
+ENV_PATH="${ENV_PATH:-$WORKSPACE_DIR/worker-env}"
 DEBUG_LOG="$WORKSPACE_DIR/debug.log"
 PYWORKER_LOG="$WORKSPACE_DIR/pyworker.log"

@@ -47,29 +54,38 @@ JSON
 }

 function install_vastai_sdk() {
-    # If SDK_BRANCH is set, install vastai-sdk from the vast-sdk repo at that branch/tag/commit.
+    local uv_flags=()
+    if [ "${USE_SYSTEM_PYTHON:-}" = "true" ]; then
+        uv_flags+=(--system --break-system-packages)
+    fi
+    if [ "$FORCE_UPDATE" = true ]; then
+        uv_flags+=(--force-reinstall)
+        echo "Force reinstalling vastai"
+    fi
+
+    # If SDK_BRANCH is set, install vastai from the vast-cli repo at that branch/tag/commit.
    if [ -n "${SDK_BRANCH:-}" ]; then
        if [ -n "${SDK_VERSION:-}" ]; then
            echo "WARNING: Both SDK_BRANCH and SDK_VERSION are set; using SDK_BRANCH=${SDK_BRANCH}"
        fi
-        echo "Installing vastai-sdk from https://github.com/vast-ai/vast-sdk/ @ ${SDK_BRANCH}"
-        if ! uv pip install "vastai-sdk @ git+https://github.com/vast-ai/vast-sdk.git@${SDK_BRANCH}"; then
-            report_error_and_exit "Failed to install vastai-sdk from vast-ai/vast-sdk@${SDK_BRANCH}"
+        echo "Installing vastai from https://github.com/vast-ai/vast-cli/ @ ${SDK_BRANCH}"
+        if ! uv pip install "${uv_flags[@]}" "vastai @ git+https://github.com/vast-ai/vast-cli.git@${SDK_BRANCH}"; then
+            report_error_and_exit "Failed to install vastai from vast-ai/vast-cli@${SDK_BRANCH}"
        fi
        return 0
    fi

    if [ -n "${SDK_VERSION:-}" ]; then
-        echo "Installing vastai-sdk version ${SDK_VERSION}"
-        if ! uv pip install "vastai-sdk==${SDK_VERSION}"; then
-            report_error_and_exit "Failed to install vastai-sdk==${SDK_VERSION}"
+        echo "Installing vastai version ${SDK_VERSION}"
+        if ! uv pip install "${uv_flags[@]}" "vastai==${SDK_VERSION}"; then
+            report_error_and_exit "Failed to install vastai==${SDK_VERSION}"
        fi
        return 0
    fi

-    echo "Installing default vastai-sdk"
-    if ! uv pip install vastai-sdk; then
-        report_error_and_exit "Failed to install vastai-sdk"
+    echo "Installing default vastai"
+    if ! uv pip install "${uv_flags[@]}" vastai; then
+        report_error_and_exit "Failed to install vastai"
    fi
 }

@@ -90,7 +106,8 @@ echo_var DEBUG_LOG
 echo_var PYWORKER_LOG
 echo_var MODEL_LOG

-if [ -e "$MODEL_LOG" ]; then
+ROTATE_MODEL_LOG="${ROTATE_MODEL_LOG:-false}"
+if [ "$ROTATE_MODEL_LOG" = "true" ] && [ -e "$MODEL_LOG" ]; then
    echo "Rotating model log at $MODEL_LOG to $MODEL_LOG.old"
    if ! cat "$MODEL_LOG" >> "$MODEL_LOG.old"; then
        report_error_and_exit "Failed to rotate model log"
@@ -111,8 +128,21 @@ if ! grep -q "VAST" /etc/environment; then
    fi
 fi

-if [ ! -d "$ENV_PATH" ]
-then
+if [ "${USE_SYSTEM_PYTHON:-}" = "true" ]; then
+    echo "Using system Python: $(which python3)"
+    if ! which uv > /dev/null 2>&1; then
+        if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
+            report_error_and_exit "Failed to install uv package manager"
+        fi
+        if [[ -f ~/.local/bin/env ]]; then
+            if ! source ~/.local/bin/env; then
+                report_error_and_exit "Failed to source uv environment"
+            fi
+        fi
+    fi
+    install_vastai_sdk
+    touch ~/.no_auto_tmux
+elif [ ! -d "$ENV_PATH" ]; then
    echo "setting up venv"
    if ! which uv; then
        if ! curl -LsSf https://astral.sh/uv/install.sh | sh; then
@@ -131,12 +161,29 @@ then
        if ! git clone "${PYWORKER_REPO:-https://github.com/vast-ai/pyworker}" "$SERVER_DIR"; then
            report_error_and_exit "Failed to clone pyworker repository"
        fi
+    elif [ "$FORCE_UPDATE" = true ]; then
+        echo "Force updating pyworker repository"
+        if ! (cd "$SERVER_DIR" && git fetch --all); then
+            report_error_and_exit "Failed to fetch pyworker repository updates"
+        fi
    fi
    if [[ -n ${PYWORKER_REF:-} ]]; then
+        if [ "$FORCE_UPDATE" = true ]; then
+            echo "Force updating to pyworker reference: $PYWORKER_REF"
+            if ! (cd "$SERVER_DIR" && git checkout "$PYWORKER_REF" && git pull); then
+                report_error_and_exit "Failed to force update pyworker reference: $PYWORKER_REF"
+            fi
+        else
            if ! (cd "$SERVER_DIR" && git checkout "$PYWORKER_REF"); then
                report_error_and_exit "Failed to checkout pyworker reference: $PYWORKER_REF"
            fi
        fi
+    elif [ "$FORCE_UPDATE" = true ]; then
+        echo "Force updating pyworker to latest"
+        if ! (cd "$SERVER_DIR" && git pull); then
+            report_error_and_exit "Failed to pull latest pyworker changes"
+        fi
+    fi

    if ! uv venv --python-preference only-managed "$ENV_PATH" -p 3.10; then
        report_error_and_exit "Failed to create virtual environment"
@@ -161,11 +208,44 @@ else
            report_error_and_exit "Failed to source uv environment"
        fi
    fi
-    if ! source "$WORKSPACE_DIR/worker-env/bin/activate"; then
+    if ! source "$ENV_PATH/bin/activate"; then
        report_error_and_exit "Failed to activate existing virtual environment"
    fi
    echo "environment activated"
    echo "venv: $VIRTUAL_ENV"
+
+    # Handle force update for existing environment
+    if [ "$FORCE_UPDATE" = true ]; then
+        echo "Performing force update on existing environment"
+
+        if [[ -d $SERVER_DIR ]]; then
+            echo "Force updating pyworker repository"
+            if ! (cd "$SERVER_DIR" && git fetch --all); then
+                report_error_and_exit "Failed to fetch pyworker repository updates"
+            fi
+
+            if [[ -n ${PYWORKER_REF:-} ]]; then
+                echo "Force updating to pyworker reference: $PYWORKER_REF"
+                if ! (cd "$SERVER_DIR" && git checkout "$PYWORKER_REF" && git pull); then
+                    report_error_and_exit "Failed to force update pyworker reference: $PYWORKER_REF"
+                fi
+            else
+                echo "Force updating pyworker to latest"
+                if ! (cd "$SERVER_DIR" && git pull); then
+                    report_error_and_exit "Failed to pull latest pyworker changes"
+                fi
+            fi
+        fi
+
+        install_vastai_sdk
+    fi
+fi
+
+# Remove force update flag after successful update
+if [ "$FORCE_UPDATE" = true ]; then
+    echo "Removing force update flag"
+    rm -f "/.force_update"
+    echo "Force update completed successfully"
 fi

 if [ "$USE_SSL" = true ]; then
@@ -203,16 +283,51 @@ EOF
        report_error_and_exit "Failed to generate SSL certificate request"
    fi

-    if ! curl --header 'Content-Type: application/octet-stream' \
+    max_retries=5
+    retry_delay=2
+    for attempt in $(seq 1 "$max_retries"); do
+        http_code=$(curl -sS -o /etc/instance.crt -w '%{http_code}' \
+            --header 'Content-Type: application/octet-stream' \
            --data-binary @/etc/instance.csr \
-        -X \
-        POST "https://console.vast.ai/api/v0/sign_cert/?instance_id=$CONTAINER_ID" > /etc/instance.crt; then
-        report_error_and_exit "Failed to sign SSL certificate"
+            -X POST "https://console.vast.ai/api/v0/sign_cert/?instance_id=$CONTAINER_ID")
+        if [ "$http_code" -ge 200 ] && [ "$http_code" -lt 300 ]; then
+            break
        fi
+        echo "SSL cert signing attempt $attempt/$max_retries failed (HTTP $http_code)"
+        if [ "$attempt" -eq "$max_retries" ]; then
+            report_error_and_exit "Failed to sign SSL certificate after $max_retries attempts (HTTP $http_code)"
+        fi
+        sleep "$retry_delay"
+        retry_delay=$((retry_delay * 2))
+    done
 fi

 export REPORT_ADDR WORKER_PORT USE_SSL UNSECURED

+# ─── SDK Deployment Mode ───────────────────────────────────────────────
+if [ "$IS_DEPLOYMENT" = "true" ]; then
+    echo "=== SDK Deployment Mode ==="
+    echo "DEPLOYMENT_ID: $DEPLOYMENT_ID"
+
+    DEPLOY_DIR="/workspace/deployment"
+    mkdir -p "$DEPLOY_DIR"
+
+    VAST_API_BASE="${VAST_API_BASE:-https://console.vast.ai}"
+
+    # Download deployment code, retrying until the blob is available on S3.
+    # The s3_key exists in the DB as soon as the deployment is created, but the
+    # actual upload may still be in flight from the client side.
+
+    # Install SDK (uses the install_vastai_sdk function which supports SDK_BRANCH/SDK_VERSION)
+    install_vastai_sdk
+    # Run deployment in serve mode
+    export VAST_DEPLOYMENT_MODE=serve
+    echo "Starting deployment: python3 $DEPLOY_DIR/deployment.py"
+    serve-vast-deployment
+    exit $?
+fi
+# ─── End SDK Deployment Mode ───────────────────────────────────────────
+
 if ! cd "$SERVER_DIR"; then
    report_error_and_exit "Failed to cd into SERVER_DIR: $SERVER_DIR"
 fi
@@ -224,19 +339,19 @@ set +e
 PY_STATUS=1

 if [ -f "$SERVER_DIR/worker.py" ]; then
-    echo "trying worker.py"
+    echo "Running worker.py"
    python3 -m "worker" |& tee -a "$PYWORKER_LOG"
    PY_STATUS=${PIPESTATUS[0]}
 fi

 if [ "${PY_STATUS}" -ne 0 ] && [ -f "$SERVER_DIR/workers/$BACKEND/worker.py" ]; then
-    echo "trying workers.${BACKEND}.worker"
+    echo "Running workers.${BACKEND}.worker"
    python3 -m "workers.${BACKEND}.worker" |& tee -a "$PYWORKER_LOG"
    PY_STATUS=${PIPESTATUS[0]}
 fi

 if [ "${PY_STATUS}" -ne 0 ] && [ -f "$SERVER_DIR/workers/$BACKEND/server.py" ]; then
-    echo "trying workers.${BACKEND}.server"
+    echo "Running workers.${BACKEND}.server"
    python3 -m "workers.${BACKEND}.server" |& tee -a "$PYWORKER_LOG"
    PY_STATUS=${PIPESTATUS[0]}
 fi
@@ -250,4 +365,4 @@ if [ "${PY_STATUS}" -ne 0 ]; then
    report_error_and_exit "PyWorker exited with status ${PY_STATUS}"
 fi

-echo "launching PyWorker server done"
+echo "PyWorker bootstrap complete"
@@ -0,0 +1,88 @@
+# Null PyWorker
+
+Holds Vast Serverless reservations open without forwarding any work to a
+model. Use it when your real workload (a queue consumer in any language)
+runs as a separate process on the instance and you just want to drive
+Vast autoscaling: **one POST reserves a worker, one POST releases it.**
+
+## Use case
+
+You have a job queue on your own infrastructure (Redis, SQS, NATS, etc.)
+and a consumer (node, golang, python, a binary — anything) that pulls
+from it. You want one Vast worker per unit of in-flight work, scaling
+elastically from zero. The null PyWorker is the autoscaling driver; your
+consumer does the work.
+
+## How it works
+
+Reservations use the framework's session API. The SDK's
+`endpoint.session(...)` POSTs `/session/create` to reserve a worker;
+`session.close()` POSTs `/session/end` to release it. `max_sessions=1`
+means each worker holds exactly one reservation — the next reservation
+either lands on a free worker or triggers a scale-up.
+
+The PyWorker itself does nothing functional:
+
+- One trivial `/ping` route to satisfy the framework's benchmark
+  requirement (its `max_perf` is pinned to 100).
+- An internal `/release` endpoint on `127.0.0.1:18999` for the local
+  consumer to end the session without needing `session_auth`.
+
+## Endpoint parameters
+
+Tested working configuration:
+
+| Parameter | Value | Why |
+|---|---|---|
+| `target_util` | `1.0` | One session = one worker. Default `0.9` rounds up to an extra worker. |
+| `min_load` | `0` | Scale-to-zero floor. |
+| `max_queue_time` | `1` | Stop routing to an occupied worker after ~1s of implied queue. |
+| `target_queue_time` | `0.5` | Trigger scale-up promptly once anything queues. |
+| `inactivity_timeout` | `10` (seconds) | Permit scale-to-zero after 10s idle. |
+
+## API
+
+| Route | Where | Use |
+|---|---|---|
+| `POST /session/create` | endpoint, signed | Reserve a worker (`endpoint.session(...)`) |
+| `POST /session/end` | endpoint, signed | Release (`session.close()`) |
+| `POST /release` | `127.0.0.1:18999`, no auth | Local consumer release, no `session_auth` needed |
+
+## Healthcheck
+
+Default: stub on `127.0.0.1:18999/health` returning `200`. Set
+`BACKEND_HEALTH_URL=http://127.0.0.1:9090/health` (absolute URL) to point
+the framework at your queue consumer's health endpoint instead — if the
+consumer dies, the autoscaler sees the worker as broken.
+
+## Deploying
+
+1. Point `PYWORKER_REPO` at this repo (or your fork).
+2. Set `BACKEND=null` in the template.
+3. Run your queue consumer alongside the PyWorker. When it's done with
+   a unit of work:
+   ```bash
+   curl -X POST http://127.0.0.1:18999/release
+   ```
+
+## Client demo
+
+```bash
+# Single reservation, hold 180s
+python -m workers.null.client --endpoint <NAME> --instance alpha
+
+# Three concurrent reservations, started 30s apart, each held 360s
+python -m workers.null.client --endpoint <NAME> --instance alpha --count 3 --hold 360
+```
+
+Flags: `--count` (number of concurrent sessions, default 1), `--hold`
+(seconds each session is held, default 180), `--interval` (seconds
+between starts when `--count > 1`, default 30), `--cost` (cost reported
+at session-create, default 100 = `max_perf`), `--instance` (`prod` |
+`alpha` | `candidate` | `local`).
+
+## Environment variables
+
+- `BACKEND_HEALTH_URL` — absolute URL the framework healthchecks. Stub
+  is used when unset.
+- `NULL_CONTROL_PORT` — internal control server port. Defaults to `18999`.
@@ -0,0 +1,64 @@
+import argparse
+import asyncio
+import logging
+import os
+import sys
+
+from vastai import Serverless
+
+logging.basicConfig(
+    level=logging.INFO,
+    format="%(asctime)s[%(levelname)-5s] %(message)s",
+    datefmt="%Y-%m-%d %H:%M:%S",
+)
+log = logging.getLogger(__file__)
+
+
+async def reserve(client: Serverless, endpoint_name: str, hold: float, cost: int, label: str):
+    endpoint = await client.get_endpoint(name=endpoint_name)
+    async with await endpoint.session(cost=cost, lifetime=hold + 60) as s:
+        sid = s.session_id
+        log.info("[%s] %s open, holding %.0fs", label, sid, hold)
+        await asyncio.sleep(hold)
+    log.info("[%s] %s closed", label, sid)
+
+
+async def main_async():
+    p = argparse.ArgumentParser(description="Vast Null PyWorker demo client")
+    p.add_argument("--endpoint", default=os.environ.get("VAST_ENDPOINT", "null-prod"))
+    p.add_argument("--instance", choices=("prod", "alpha", "candidate", "local"),
+                   default=os.environ.get("VAST_INSTANCE", "prod"))
+    p.add_argument("--count", type=int, default=1,
+                   help="concurrent sessions to open (default: 1)")
+    p.add_argument("--interval", type=float, default=30.0,
+                   help="seconds between session starts when count>1 (default: 30)")
+    p.add_argument("--hold", type=float, default=180.0,
+                   help="seconds to hold each session (default: 180)")
+    p.add_argument("--cost", type=int, default=100,
+                   help="cost reported at session-create (default: 100)")
+    args = p.parse_args()
+
+    print(f"endpoint={args.endpoint} instance={args.instance} "
+          f"count={args.count} hold={args.hold}s cost={args.cost}")
+
+    try:
+        async with Serverless(instance=args.instance) as client:
+            tasks = []
+            for i in range(args.count):
+                label = f"res-{i+1}" if args.count > 1 else "reservation"
+                tasks.append(asyncio.create_task(
+                    reserve(client, args.endpoint, args.hold, args.cost, label),
+                    name=label,
+                ))
+                if i + 1 < args.count:
+                    await asyncio.sleep(args.interval)
+            await asyncio.gather(*tasks, return_exceptions=True)
+    except KeyboardInterrupt:
+        log.info("Interrupted")
+    except Exception as e:
+        log.error("Error: %s", e, exc_info=True)
+        sys.exit(1)
+
+
+if __name__ == "__main__":
+    asyncio.run(main_async())
@@ -0,0 +1,143 @@
+import asyncio
+import logging
+import os
+from contextlib import asynccontextmanager
+from urllib.parse import urlsplit
+
+from aiohttp import web
+
+from vastai import (
+    Worker,
+    WorkerConfig,
+    HandlerConfig,
+    BenchmarkConfig,
+    LogActionConfig,
+)
+
+log = logging.getLogger(__file__)
+
+TARGET_PERF = 100.0
+BENCHMARK_SENTINEL = "__null_worker_benchmark__"
+
+INTERNAL_HOST = "127.0.0.1"
+INTERNAL_PORT = int(os.environ.get("NULL_CONTROL_PORT", 18999))
+STUB_HEALTH_PATH = "/health"
+
+BACKEND_HEALTH_URL = os.environ.get("BACKEND_HEALTH_URL", "").strip()
+if BACKEND_HEALTH_URL:
+    _p = urlsplit(BACKEND_HEALTH_URL)
+    if not _p.scheme or not _p.hostname:
+        raise ValueError(f"BACKEND_HEALTH_URL must be absolute, got: {BACKEND_HEALTH_URL!r}")
+    HEALTH_BASE_URL = f"{_p.scheme}://{_p.hostname}"
+    HEALTH_PORT = _p.port or (443 if _p.scheme == "https" else 80)
+    HEALTH_PATH = _p.path or "/"
+    USE_STUB_HEALTH = False
+else:
+    HEALTH_BASE_URL = f"http://{INTERNAL_HOST}"
+    HEALTH_PORT = INTERNAL_PORT
+    HEALTH_PATH = STUB_HEALTH_PATH
+    USE_STUB_HEALTH = True
+
+
+_backend_ref: dict = {"backend": None}
+
+
+def _build_internal_app() -> web.Application:
+    app = web.Application()
+
+    async def release_handler(_request: web.Request) -> web.Response:
+        # Closes the singleton session. Uses name-mangled __close_session
+        # to bypass the session_auth check — safe because this server is
+        # bound to 127.0.0.1, and it spares the consumer from threading
+        # session_auth through its queue.
+        backend = _backend_ref.get("backend")
+        if backend is None:
+            return web.json_response({"released": False, "reason": "backend not ready"}, status=503)
+        sids = list(backend.sessions.keys())
+        if not sids:
+            return web.json_response({"released": False, "reason": "no active session"}, status=200)
+        closed = []
+        for sid in sids:
+            try:
+                if await backend._Backend__close_session(sid):
+                    closed.append(sid)
+            except Exception as e:
+                log.warning(f"Error closing session {sid}: {e}")
+        return web.json_response({"released": bool(closed), "session_ids": closed}, status=200)
+
+    app.router.add_post("/release", release_handler)
+
+    if USE_STUB_HEALTH:
+        async def stub_health(_request: web.Request) -> web.Response:
+            return web.Response(status=200, text="ok")
+        app.router.add_get(STUB_HEALTH_PATH, stub_health)
+
+    return app
+
+
+@asynccontextmanager
+async def null_lifecycle():
+    # Pin max_throughput to TARGET_PERF exactly — the framework's
+    # __run_benchmark short-circuits to float(file_contents) if this exists.
+    try:
+        with open(".has_benchmark", "w") as fh:
+            fh.write(str(int(TARGET_PERF)))
+    except OSError as e:
+        log.warning(f"Could not pin benchmark cache: {e}")
+
+    runner = web.AppRunner(_build_internal_app())
+    await runner.setup()
+    await web.TCPSite(runner, INTERNAL_HOST, INTERNAL_PORT).start()
+
+    log.info(
+        "Null pyworker control server: http://%s:%d (POST /release%s)",
+        INTERNAL_HOST,
+        INTERNAL_PORT,
+        f", GET {STUB_HEALTH_PATH}" if USE_STUB_HEALTH else "",
+    )
+    if not USE_STUB_HEALTH:
+        log.info("Framework healthcheck → %s", BACKEND_HEALTH_URL)
+
+    try:
+        yield
+    finally:
+        await runner.cleanup()
+
+
+async def ping(**params: object) -> dict:
+    # Exists only to satisfy the framework's "at least one handler with a
+    # BenchmarkConfig" requirement. Sleep 1s on the benchmark path as a
+    # fallback in case the .has_benchmark cache pin failed; otherwise the
+    # benchmark cache short-circuits and this never runs.
+    if params.get(BENCHMARK_SENTINEL):
+        await asyncio.sleep(1.0)
+        return {"ok": True, "benchmark": True}
+    return {"ok": True}
+
+
+worker_config = WorkerConfig(
+    model_server_url=HEALTH_BASE_URL,
+    model_server_port=HEALTH_PORT,
+    model_healthcheck_url=HEALTH_PATH,
+    lifecycle=null_lifecycle(),
+    max_sessions=1,
+    handlers=[
+        HandlerConfig(
+            route="/ping",
+            allow_parallel_requests=True,
+            remote_function=ping,
+            workload_calculator=lambda _payload: TARGET_PERF,
+            benchmark_config=BenchmarkConfig(
+                generator=lambda: {BENCHMARK_SENTINEL: True},
+                runs=1,
+                concurrency=1,
+                do_warmup=False,
+            ),
+        ),
+    ],
+    log_action_config=LogActionConfig(),
+)
+
+_worker = Worker(worker_config)
+_backend_ref["backend"] = _worker.backend
+_worker.run()
@@ -35,7 +35,7 @@ def benchmark_generator() -> dict:
    benchmark_data = {
        "inputs": prompt,
        "parameters": {
-            "max_new_tokens": 128,
+            "max_new_tokens": 500,
            "temperature": 0.7,
            "return_full_text": False
        }
Author	SHA1	Message	Date
Rob Ballantyne	a81d3febe7	Collapse null pyworker client to a single mode parameterized by --count Now that the session model means no HTTP connection is held during the reservation, the dichotomy between "single reserve" and "trapezoid demo" collapses — both are "open N sessions, each held for H seconds, started I seconds apart, close." Replace --reserve/--demo/--duration/--plateau with --count/--hold/--interval. --session-cost becomes --cost. Client is now 64 lines (down from 120). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 12:18:33 +01:00
Rob Ballantyne	913e3a8782	Simplify null pyworker code and docs Pass over all three files to drop verbose expository commentary that duplicated either the code or the README. Net: -284 lines. README now reads top-to-bottom in roughly the order someone would need the info: use case → how it works → endpoint params → API → healthcheck → deploy → demo. Endpoint params table uses the values actually tested on alpha (min_load=0, target_util=1, max_queue_time=1, target_queue_time=0.5, inactivity_timeout=10). Dropped the "known autoscaler quirk" section now that alpha addresses it; kept the --session-cost flag as a debugging knob. worker.py and client.py keep the same behavior but trim long block comments and multi-line docstrings the code didn't need. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 11:50:03 +01:00
Rob Ballantyne	47ad0ebe0a	Add --instance flag to null pyworker client Lets the demo target run-alpha.vast.ai (or candidate/local) without editing code. Defaults to prod; respects VAST_INSTANCE env var. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 11:40:51 +01:00
Rob Ballantyne	34fd21e76a	Revert default session cost to 100; document the over-provision as a workaround cost = max_perf = 100 is the intended steady-state semantics: one session = one worker, scaling elastically from zero. Reverting the default so the design reads correctly even where current autoscaler bugs make it misbehave (2→3 scale-up not firing reliably, scale-to-zero issues — fixes pending on the Vast side). README now describes the intended model first (clean unit occupancy, scale-to-zero via inactivity_timeout + min_load=0), then flags the known autoscaler quirk and presents --session-cost 200 as a temporary band-aid until the Vast fixes land. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 11:34:52 +01:00
Rob Ballantyne	1d2caaf554	Default null pyworker session cost to 2x max_perf Reporting cost == max_perf puts an occupied worker at exactly 100% utilization, which the autoscaler reads as "at target, no action." The 3rd session_create then 429s on both active workers and stalls in the global queue instead of triggering a cold-worker activation (observed: 1→2 active scales fine, 2→3 does not). Bumping cost to 2 * max_perf makes each session look like more than one worker's work, so the autoscaler always keeps an extra active worker hot. Slight over-provisioning, but the 3rd reservation lands directly on a free worker rather than queueing. Expose --session-cost on the client so the value can be swept without edits. README documents the trade-off. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 11:31:26 +01:00
Rob Ballantyne	01eff874d8	Correct queue-time guidance for null pyworker endpoints Earlier note claimed max_queue_time / target_queue_time were no-ops because the worker's internal wait_time property filters sessions out. That filter only affects per-worker rejection on a given handler — the autoscaler doesn't see the property and computes its own queue-time estimate from cur_load / max_perf, which does include sessions. With defaults around 30s, an occupied null worker (cur_load=100, max_perf=100, implied queue=1s) still looks "available" to the autoscaler, so a third reservation gets queued on an existing worker via repeated 429-retries instead of triggering scale-up. Fix: set max_queue_time = 0 and target_queue_time = 0 on the endpoint. Any in-flight load marks the worker "full" for routing, and any observed queue time triggers immediate scale-up. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 11:14:20 +01:00
Rob Ballantyne	d51f04a176	Await endpoint.session() in null pyworker client endpoint.session() forwards to start_endpoint_session, which is async def — so the call returns a coroutine, not a Session, despite the SDK's return-type annotation. Use 'async with await endpoint.session(...)' so the coroutine resolves to a Session before entering the context. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 11:07:32 +01:00
Rob Ballantyne	ef248ef695	Document endpoint scaling parameters for null pyworker Add a scaling-parameters section to the README covering target_util=1.0 (the critical one — the default 0.9 silently rounds up to one extra worker), min_load math, and why max_queue_time / target_queue_time don't matter here (sessions are filtered from wait_time so both signals stay at zero). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 11:06:04 +01:00
Rob Ballantyne	6a562a1376	Rewrite null pyworker on the framework session model Drop the held-/reserve approach in favour of the framework's session primitive (max_sessions=1 + /session/create). Sessions are excluded from the autoscaler's queue-wait math and don't suffer the cur_perf=0 degradation that a long-held request did, so this naturally produces the "one request comes in and you get a worker; release and it scales back down" model we were hand-rolling. Server side: - max_sessions=1; framework auto-registers /session/* routes - Drop custom /reserve handler, _active_reservation event, max_queue_ time=0.0, MAX_RESERVATION_SECONDS, _perf_heartbeat - Trivial /ping handler exists only to satisfy the framework's "at least one handler with BenchmarkConfig" requirement (and to give clients an extension/keepalive route) - /release on the internal control port is kept as a convenience for queue consumers that don't carry session_auth — calls the framework's __close_session via name-mangling, which bypasses the session_auth check but is fine for a localhost-only endpoint - Workload/perf back to 100 (conventional) Client side: - Uses endpoint.session(cost, lifetime) instead of POST /reserve - async with the SDK Session; close on exit posts /session/end with proper auth → 200 success in metrics - Demo and single modes both ride the same reserve() helper Sessions landed in vastai-sdk 0.4.2 (commit ec9ef59, 2026-01-20). Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 10:51:24 +01:00
Rob Ballantyne	6c2f194b28	Add perf heartbeat to keep null pyworker reporting peak throughput While a /reserve is held, no requests complete so workload_served stays at 0 each metrics tick. The autoscaler sees cur_perf=0 against max_perf=150, concludes the worker can't deliver claimed throughput, downgrades it, and gets cautious about scaling up — so additional /reserve requests pile up behind the held one instead of triggering a new worker. Add a 1Hz heartbeat coroutine that, while anything is in flight, sets workload_served back to TARGET_PERF (150) and flags update_pending. The metrics tick reads 150 and resets to 0; the heartbeat re-pins it before the next tick. Net effect: the autoscaler sees a saturated worker delivering at peak rate, which is the signal it needs to scale a new worker up rather than queue. The heartbeat needs the backend instance, which is only created inside Worker(...) — stash a reference in a module-level dict between Worker() and .run() so the lifecycle coroutine can reach it. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-12 10:35:18 +01:00
Rob Ballantyne	2aada7b210	Add --plateau to null pyworker demo (default 5min) Previously the first release fired only 30s after the third reservation started, so the autoscaler often hadn't even finished provisioning the third worker yet. Default plateau to 300s so all three workers are visibly running before scale-down begins; configurable via --plateau. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 18:26:31 +01:00
Rob Ballantyne	8df562e243	Standardize null pyworker load/perf on 150 Bump workload_calculator, benchmark cache value, and client cost from 100 to 150. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 18:17:57 +01:00
Rob Ballantyne	4eef5e22af	Pin null pyworker max_throughput to exactly 100 asyncio.sleep(1.0) takes slightly more than 1s due to event loop scheduling, so workload/time landed at ~99.x instead of 100. Pre-populate the framework's .has_benchmark cache file with "100" before the benchmark runs — __run_benchmark short-circuits to the cached value and skips the time-based calculation entirely. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 18:13:16 +01:00
Rob Ballantyne	9d969e376e	Standardize null pyworker load/perf on 100 Using 1 confused the serverless capacity math. Set workload_calculator, benchmark target throughput, and client cost all to 100 — the conventional default the rest of the system expects. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 18:09:16 +01:00
Rob Ballantyne	ef3f34a515	Restructure null pyworker --demo as a clean trapezoid Three reservations 30s apart, each with a 90s duration. They end one at a time, also 30s apart, then the client exits. Each reservation ends via its duration cap (200 success) rather than the previous "cancel one, leave two open" pattern that left two 499s pending. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 18:00:46 +01:00
Rob Ballantyne	147bf2597a	Set null pyworker client cost to 1 Match the server-side workload_calculator (1.0) so the autoscaler routing hint is consistent with what the worker reports. A null reservation is a unitless slot — no reason for client cost to be 100. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 17:47:19 +01:00
Rob Ballantyne	dc423e2999	Pin null pyworker benchmark to ~1.0 throughput The startup benchmark previously returned instantly, producing max_throughput around 339895. A null worker has no real throughput concept (each reservation is a unitless slot), so sleep 1s during the benchmark with workload=1 to record max_throughput ~= 1.0. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 17:22:45 +01:00
Rob Ballantyne	463f3de8ea	Add staggered --demo mode to null pyworker client Three concurrent /reserve calls 30s apart, then cancel the first to show the early-release path. The remaining two run until their duration cap. Useful for watching scale-up/scale-down behaviour in the autoscaler dashboard. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 17:08:44 +01:00
Rob Ballantyne	ed0db198c3	Reject queued /reserve immediately on busy null workers A held reservation runs for up to MAX_RESERVATION_SECONDS (default 1h), so queueing a second /reserve behind it makes no sense — the wait would dwarf any sane timeout. Set max_queue_time=0.0 so the framework rejects 429 as soon as another reservation is in flight, and serverless routes the request to a free worker or scales a new one up. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 17:05:02 +01:00
Rob Ballantyne	3668d948be	Simplify null pyworker README intro to serverless terminology Drop the "autoscaler provisions a worker if none is free" phrasing in favor of the simpler "request comes in and you get a worker; release and it scales back down." Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 17:02:41 +01:00
Rob Ballantyne	254ccdf181	Add /release control endpoint to null pyworker The held /reserve now waits on an asyncio.Event and resolves when the local queue consumer POSTs /release on the internal control port (127.0.0.1:18999 by default). This produces a 200 success in metrics instead of the 499 cancellation you got from disconnecting the client. The duration cap stays as a safety net for stuck consumers. The internal aiohttp server is now unconditional and hosts /release always; the stub /health route is added only when BACKEND_HEALTH_URL is unset. NULL_STUB_HEALTH_PORT is renamed to NULL_CONTROL_PORT to reflect the broader role. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 16:59:46 +01:00
Rob Ballantyne	89761b378a	Wire null pyworker healthcheck to a stub (and optional user URL) Adds an in-process aiohttp stub on 127.0.0.1:18999/health so the framework's periodic healthcheck has something live to talk to. Operators can override with BACKEND_HEALTH_URL to point at their queue consumer's /health endpoint, so the autoscaler marks the worker errored if the consumer dies. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 16:53:26 +01:00
Rob Ballantyne	18974873e5	Add null pyworker for queue-driven autoscaling A PyWorker that does not forward to any model server. POST /reserve holds the worker busy until the client disconnects (or the duration cap elapses), so users with their own job queue can drive Vast autoscaling without exposing inbound model traffic on the instance. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>	2026-05-11 16:48:52 +01:00
Lucas Armand	9bc9ba11c5	Increase TGI benchmark tokens to 500	2026-04-30 14:04:39 -07:00
LucasArmandVast	48fdc65e3d	Update to vastai package (#84 )	2026-04-14 10:41:31 -07:00
LucasArmandVast	2cd97315cd	Add nltk requirement for openai worker (#83 ) * Add nltk requirement for openai worker * pin version	2026-04-13 11:30:06 -07:00
Lucas Armand	83c31e25a9	Add force update detection	2026-03-31 13:46:22 -07:00
Lucas Armand	fbe1dca6fa	more env_path fixes	2026-03-30 16:28:51 -07:00
Lucas Armand	4c3120dbc5	allow override env_path	2026-03-30 16:25:01 -07:00
Lucas Armand	d7d9b915f6	allow break system packages	2026-03-30 16:09:17 -07:00
Lucas Armand	4660b337fb	Check for USE_SYSTEM_PYTHON	2026-03-30 14:46:38 -07:00
edgaratvast	7506ecb6b5	directly invoke one stop shop setup executable exported by vastai pip package for deployments (#82 )	2026-03-26 10:59:49 -07:00
LucasArmandVast	50633c5003	Update deployments script with retries. (#81 )	2026-03-23 14:58:32 -07:00
LucasArmandVast	2e8f18276f	Add beta deployments script (#80 )	2026-03-23 14:14:06 -07:00
Scott Darden	eba9c480eb	Merge pull request #79 from vast-ai/update-requirements Updated requirements to only require vastai-sdk	2026-01-14 12:07:33 -08:00
Lucas Armand	aaca1c9645	Updated requirements to only require vastai-sdk	2026-01-14 10:47:07 -08:00
LucasArmandVast	f319db6bd5	flag for model log rotate (#78 )	2026-01-12 17:03:18 -08:00
LucasArmandVast	4d786b4d17	SDK Versioning Improvements (#77 ) * Add SDK_BRANCH	2026-01-02 10:23:07 -08:00