Increase model wait time for vLLM

Merge pull request #66 from vast-ai/synthesis
PyWorker Error Handling
2025-12-03 12:38:52 -08:00 · 2025-11-25 16:02:26 -08:00 · 2025-11-25 16:01:23 -08:00 · 2025-11-24 15:24:06 -08:00 · 2025-11-24 15:22:23 -08:00 · 2025-11-24 15:02:33 -08:00
4 changed files with 95 additions and 31 deletions
@@ -30,7 +30,7 @@ from lib.data_types import (
    BenchmarkResult
 )

-VERSION = "0.2.0"
+VERSION = "0.2.1"

 MSG_HISTORY_LEN = 100
 log = logging.getLogger(__file__)
@@ -3,15 +3,17 @@ import logging
 from typing import List
 import ssl
 from asyncio import run, gather
-
+import asyncio

 from lib.backend import Backend
+from lib.metrics import Metrics
 from aiohttp import web

 log = logging.getLogger(__file__)


 def start_server(backend: Backend, routes: List[web.RouteDef], **kwargs):
+    try:
        log.debug("getting certificate...")
        use_ssl = os.environ.get("USE_SSL", "false") == "true"
        if use_ssl is True:
@@ -38,3 +40,21 @@ def start_server(backend: Backend, routes: List[web.RouteDef], **kwargs):
            await gather(site.start(), backend._start_tracking())

        run(main())
+
+    except Exception as e:
+        err_msg = f"PyWorker failed to launch: {e}"
+        log.error(err_msg)
+
+        async def beacon():
+            metrics = Metrics()
+            metrics._set_version(getattr(backend, "version", "0"))
+            metrics._set_mtoken(getattr(backend, "mtoken", ""))
+            try:
+                while True:
+                    metrics._model_errored(err_msg)
+                    await metrics._Metrics__send_metrics_and_reset()
+                    await asyncio.sleep(10)
+            finally:
+                await metrics.aclose()
+
+        run(beacon())
@@ -41,6 +41,14 @@ echo_var DEBUG_LOG
 echo_var PYWORKER_LOG
 echo_var MODEL_LOG

+# if instance is rebooted, we want to clear out the log file so pyworker doesn't read lines
+# from the run prior to reboot. past logs are saved in $MODEL_LOG.old for debugging only
+if [ -e "$MODEL_LOG" ]; then
+    echo "Rotating model log at $MODEL_LOG to $MODEL_LOG.old"
+    cat "$MODEL_LOG" >> "$MODEL_LOG.old" 
+    : > "$MODEL_LOG"
+fi
+
 # Populate /etc/environment with quoted values
 if ! grep -q "VAST" /etc/environment; then
    env -0 | grep -zEv "^(HOME=|SHLVL=)|CONDA" | while IFS= read -r -d '' line; do
@@ -124,9 +132,43 @@ cd "$SERVER_DIR"

 echo "launching PyWorker server"

-# if instance is rebooted, we want to clear out the log file so pyworker doesn't read lines
-# from the run prior to reboot. past logs are saved in $MODEL_LOG.old for debugging only
-[ -e "$MODEL_LOG" ] && cat "$MODEL_LOG" >> "$MODEL_LOG.old" && : > "$MODEL_LOG"
+set +e
+python3 -m "workers.$BACKEND.server" |& tee -a "$PYWORKER_LOG"
+PY_STATUS=${PIPESTATUS[0]}
+set -e
+
+if [ "${PY_STATUS}" -ne 0 ]; then
+  echo "PyWorker exited with status ${PY_STATUS}; notifying autoscaler..."
+  ERROR_MSG="PyWorker exited: code ${PY_STATUS}"
+  MTOKEN="${MASTER_TOKEN:-}"
+  VERSION="${PYWORKER_VERSION:-0}"
+
+  IFS=',' read -r -a REPORT_ADDRS <<< "${REPORT_ADDR}"
+  for addr in "${REPORT_ADDRS[@]}"; do
+    curl -sS -X POST -H 'Content-Type: application/json' \
+      -d "$(cat <<JSON
+{
+  "id": ${CONTAINER_ID:-0},
+  "mtoken": "${MTOKEN}",
+  "version": "${VERSION}",
+  "loadtime": 0,
+  "new_load": 0,
+  "cur_load": 0,
+  "rej_load": 0,
+  "max_perf": 0,
+  "cur_perf": 0,
+  "error_msg": "${ERROR_MSG}",
+  "num_requests_working": 0,
+  "num_requests_recieved": 0,
+  "additional_disk_usage": 0,
+  "working_request_idxs": [],
+  "cur_capacity": 0,
+  "max_capacity": 0,
+  "url": "${URL}"
+}
+JSON
+)" "${addr%/}/worker_status/" || true
+  done
+fi

-(python3 -m "workers.$BACKEND.server" |& tee -a "$PYWORKER_LOG") &
 echo "launching PyWorker server done"
@@ -11,6 +11,7 @@ MODEL_SERVER_START_LOG_MSG = [
    "llama runner started",  # Ollama
    '"message":"Connected","target":"text_generation_router"',  # TGI
    '"message":"Connected","target":"text_generation_router::server"',  # TGI
+    "main: model loaded" # llama.cpp
 ]

 MODEL_SERVER_ERROR_LOG_MSGS = [
@@ -34,6 +35,7 @@ backend = Backend(
    model_server_url=os.environ["MODEL_SERVER_URL"],
    model_log_file=os.environ["MODEL_LOG"],
    allow_parallel_requests=True,
+    max_wait_time=600.0,
    benchmark_handler=CompletionsHandler(benchmark_runs=3, benchmark_words=256),
    log_actions=[
        *[(LogAction.ModelLoaded, info_msg) for info_msg in MODEL_SERVER_START_LOG_MSG],
Author	SHA1	Message	Date
Lucas Armand	0bcd2219ea	Increase model wait time for vLLM	2025-12-03 12:38:52 -08:00
LucasArmandVast	0339b471c5	Merge pull request #66 from vast-ai/synthesis PyWorker Error Handling	2025-11-25 16:02:26 -08:00
Lucas Armand	e143162438	bumpy pyworker version	2025-11-25 16:01:23 -08:00
Lucas Armand	7986e51e9e	early errors	2025-11-24 15:24:06 -08:00
Lucas Armand	9c6ab78503	Move model log line	2025-11-24 15:22:23 -08:00
Lucas Armand	45e0c7d9ca	Move model log rotate to top	2025-11-24 15:02:33 -08:00
LucasArmandVast	7a792fd176	Merge pull request #64 from vast-ai/add-llama-log add llama log	2025-11-21 10:24:27 -08:00
Lucas Armand	e0449cb3c7	add llama log	2025-11-21 10:22:16 -08:00
Lucas Armand	a47c9d1ed0	remove test bugs	2025-11-11 18:13:46 -08:00
Lucas Armand	0b14562a63	dont exit on pyworker fail	2025-11-11 17:57:08 -08:00
Lucas Armand	de9b50abb9	use set +e	2025-11-11 17:53:36 -08:00
Lucas Armand	c510801723	fix	2025-11-11 17:49:34 -08:00
Lucas Armand	a12523b1d2	Added bad code to tgi server to test	2025-11-11 17:41:12 -08:00