remove test bugs

dont exit on pyworker fail
use set +e
2025-11-11 18:13:46 -08:00 · 2025-11-11 17:57:08 -08:00 · 2025-11-11 17:53:36 -08:00 · 2025-11-11 17:49:34 -08:00 · 2025-11-11 17:41:12 -08:00 · 2025-11-10 11:54:04 -08:00
7 changed files with 127 additions and 48 deletions
@@ -30,7 +30,7 @@ from lib.data_types import (
    BenchmarkResult
 )

-VERSION = "0.1.0"
+VERSION = "0.2.0"

 MSG_HISTORY_LEN = 100
 log = logging.getLogger(__file__)
@@ -66,10 +66,17 @@ class Backend:
    unsecured: bool = dataclasses.field(
        default_factory=lambda: bool(strtobool(os.environ.get("UNSECURED", "false"))),
    )
+    report_addr: str = dataclasses.field(
+        default_factory=lambda: os.environ.get("REPORT_ADDR", "https://run.vast.ai")
+    )
+    mtoken: str = dataclasses.field(
+        default_factory=lambda: os.environ.get("MASTER_TOKEN", "")
+    )

    def __post_init__(self):
        self.metrics = Metrics()
        self.metrics._set_version(self.version)
+        self.metrics._set_mtoken(self.mtoken)
        self._total_pubkey_fetch_errors = 0
        self._pubkey = self._fetch_pubkey()
        self.__start_healthcheck: bool = False
@@ -104,23 +111,19 @@ class Backend:

    #######################################Private#######################################
    def _fetch_pubkey(self):
-        command = ["curl", "-X", "GET", "https://run.vast.ai/pubkey/"]
-        result = subprocess.check_output(command, universal_newlines=True)
-        log.debug("public key:")
-        log.debug(result)
-        key = None
-        for _ in range(5):
-            try:
-                key = RSA.import_key(result)
-                break
-            except ValueError as e:
-                log.debug(f"Error downloading key: {e}")
-                time.sleep(15)
-        if key is None:
-            self._total_pubkey_fetch_errors += 1
-            if self._total_pubkey_fetch_errors >= MAX_PUBKEY_FETCH_ATTEMPTS:
-                self.backend_errored("Failed to get autoscaler pubkey")
-        return key
+        report_addr = self.report_addr.rstrip("/")
+        command = ["curl", "-X", "GET", f"{report_addr}/pubkey/"]
+        try:
+            result = subprocess.check_output(command, universal_newlines=True)
+            log.debug("public key:")
+            log.debug(result)
+            key = RSA.import_key(result)
+            if key is not None:
+                return key
+        except (ValueError , subprocess.CalledProcessError) as e:
+            log.debug(f"Error downloading key: {e}")
+        self.backend_errored("Failed to get autoscaler pubkey")
+       

    async def __handle_request(
        self,
@@ -286,6 +286,7 @@ class AutoScalerData:
    """Data that is reported to autoscaler"""

    id: int
+    mtoken: str
    version: str
    loadtime: float
    cur_load: float
@@ -28,6 +28,7 @@ def get_url() -> str:
@dataclass
 class Metrics:
    version: str = "0"
+    mtoken: str = ""
    last_metric_update: float = 0.0
    last_request_served: float = 0.0
    update_pending: bool = False
@@ -142,12 +143,16 @@ class Metrics:
    def _set_version(self, version: str) -> None:
        self.version = version

+    def _set_mtoken(self, mtoken: str) -> None:
+        self.mtoken = mtoken
+
    #######################################Private#######################################

    async def __send_delete_requests_and_reset(self):
        async def post(report_addr: str, idxs: list[int], success_flag: bool) -> bool:
            data = {
                "worker_id": self.id,
+                "mtoken": self.mtoken,
                "request_idxs": idxs,
                "success": success_flag,
            }
@@ -209,6 +214,7 @@ class Metrics:
        def compute_autoscaler_data() -> AutoScalerData:
            return AutoScalerData(
                id=self.id,
+                mtoken=self.mtoken,
                version=self.version,
                loadtime=(loadtime_snapshot or 0.0), 
                new_load=self.model_metrics.workload_processing,
@@ -228,17 +234,25 @@ class Metrics:

        async def send_data(report_addr: str) -> bool:
            data = compute_autoscaler_data()
-            full_path = report_addr.rstrip("/") + "/worker_status/"
+            log_data = asdict(data)
+            def obfuscate(secret: str) -> str:
+                if secret is None:
+                    return ""
+                return secret[:7] + "..." if len(secret) > 7 else ("*" * len(secret))
+            
+            log_data["mtoken"] = obfuscate(log_data.get("mtoken"))
            log.debug(
                "\n".join(
                    [
                        "#" * 60,
                        f"sending data to autoscaler",
-                        f"{json.dumps((asdict(data)), indent=2)}",
+                        f"{json.dumps(log_data, indent=2)}",
                        "#" * 60,
                    ]
                )
            )
+
+            full_path = report_addr.rstrip("/") + "/worker_status/"
            for attempt in range(1, 4):
                try:
                    session = await self.http()
@@ -3,38 +3,58 @@ import logging
 from typing import List
 import ssl
 from asyncio import run, gather
-
+import asyncio

 from lib.backend import Backend
+from lib.metrics import Metrics
 from aiohttp import web

 log = logging.getLogger(__file__)


 def start_server(backend: Backend, routes: List[web.RouteDef], **kwargs):
-    log.debug("getting certificate...")
-    use_ssl = os.environ.get("USE_SSL", "false") == "true"
-    if use_ssl is True:
-        ssl_context = ssl.create_default_context(ssl.Purpose.CLIENT_AUTH)
-        ssl_context.load_cert_chain(
-            certfile="/etc/instance.crt",
-            keyfile="/etc/instance.key",
-        )
-    else:
-        ssl_context = None
+    try:
+        log.debug("getting certificate...")
+        use_ssl = os.environ.get("USE_SSL", "false") == "true"
+        if use_ssl is True:
+            ssl_context = ssl.create_default_context(ssl.Purpose.CLIENT_AUTH)
+            ssl_context.load_cert_chain(
+                certfile="/etc/instance.crt",
+                keyfile="/etc/instance.key",
+            )
+        else:
+            ssl_context = None

-    async def main():
-        log.debug("starting server...")
-        app = web.Application()
-        app.add_routes(routes)
-        runner = web.AppRunner(app)
-        await runner.setup()
-        site = web.TCPSite(
-            runner,
-            ssl_context=ssl_context,
-            port=int(os.environ["WORKER_PORT"]),
-            **kwargs
-        )
-        await gather(site.start(), backend._start_tracking())
+        async def main():
+            log.debug("starting server...")
+            app = web.Application()
+            app.add_routes(routes)
+            runner = web.AppRunner(app)
+            await runner.setup()
+            site = web.TCPSite(
+                runner,
+                ssl_context=ssl_context,
+                port=int(os.environ["WORKER_PORT"]),
+                **kwargs
+            )
+            await gather(site.start(), backend._start_tracking())

-    run(main())
+        run(main())
+
+    except Exception as e:
+        err_msg = f"PyWorker failed to launch: {e}"
+        log.error(err_msg)
+
+        async def beacon():
+            metrics = Metrics()
+            metrics._set_version(getattr(backend, "version", "0"))
+            metrics._set_mtoken(getattr(backend, "mtoken", ""))
+            try:
+                while True:
+                    metrics._model_errored(err_msg)
+                    await metrics._Metrics__send_metrics_and_reset()
+                    await asyncio.sleep(10)
+            finally:
+                await metrics.aclose()
+
+        run(beacon())
@@ -9,7 +9,7 @@ ENV_PATH="$WORKSPACE_DIR/worker-env"
 DEBUG_LOG="$WORKSPACE_DIR/debug.log"
 PYWORKER_LOG="$WORKSPACE_DIR/pyworker.log"

-REPORT_ADDR="${REPORT_ADDR:-https://cloud.vast.ai/api/v0,https://run.vast.ai}"
+REPORT_ADDR="${REPORT_ADDR:-https://run.vast.ai}"
 USE_SSL="${USE_SSL:-true}"
 WORKER_PORT="${WORKER_PORT:-3000}"
 mkdir -p "$WORKSPACE_DIR"
@@ -128,5 +128,44 @@ echo "launching PyWorker server"
 # from the run prior to reboot. past logs are saved in $MODEL_LOG.old for debugging only
 [ -e "$MODEL_LOG" ] && cat "$MODEL_LOG" >> "$MODEL_LOG.old" && : > "$MODEL_LOG"

-(python3 -m "workers.$BACKEND.server" |& tee -a "$PYWORKER_LOG") &
-echo "launching PyWorker server done"
+
+set +e
+python3 -m "workers.$BACKEND.server" |& tee -a "$PYWORKER_LOG"
+PY_STATUS=${PIPESTATUS[0]}
+set -e
+
+if [ "${PY_STATUS}" -ne 0 ]; then
+  echo "PyWorker exited with status ${PY_STATUS}; notifying autoscaler..."
+  ERROR_MSG="PyWorker exited: code ${PY_STATUS}"
+  MTOKEN="${MASTER_TOKEN:-}"
+  VERSION="${PYWORKER_VERSION:-0}"
+
+  IFS=',' read -r -a REPORT_ADDRS <<< "${REPORT_ADDR}"
+  for addr in "${REPORT_ADDRS[@]}"; do
+    curl -sS -X POST -H 'Content-Type: application/json' \
+      -d "$(cat <<JSON
+{
+  "id": ${CONTAINER_ID:-0},
+  "mtoken": "${MTOKEN}",
+  "version": "${VERSION}",
+  "loadtime": 0,
+  "new_load": 0,
+  "cur_load": 0,
+  "rej_load": 0,
+  "max_perf": 0,
+  "cur_perf": 0,
+  "error_msg": "${ERROR_MSG}",
+  "num_requests_working": 0,
+  "num_requests_recieved": 0,
+  "additional_disk_usage": 0,
+  "working_request_idxs": [],
+  "cur_capacity": 0,
+  "max_capacity": 0,
+  "url": "${URL}"
+}
+JSON
+)" "${addr%/}/worker_status/" || true
+  done
+fi
+
+echo "launching PyWorker server done"
@@ -98,6 +98,7 @@ def call_text2image_workflow(
        endpoint=route_response["endpoint"],
        reqnum=route_response["reqnum"],
        url=route_response["url"],
+        request_idx=route_response["request_idx"],
    )
    
    # Build the payload for the worker request
@@ -82,6 +82,7 @@ def call_custom_workflow_for_sd3(
        endpoint=message["endpoint"],
        reqnum=message["reqnum"],
        url=message["url"],
+        request_idx=message["request_idx"],
    )
    workflow = {
        "3": {
Author	SHA1	Message	Date
Lucas Armand	a47c9d1ed0	remove test bugs	2025-11-11 18:13:46 -08:00
Lucas Armand	0b14562a63	dont exit on pyworker fail	2025-11-11 17:57:08 -08:00
Lucas Armand	de9b50abb9	use set +e	2025-11-11 17:53:36 -08:00
Lucas Armand	c510801723	fix	2025-11-11 17:49:34 -08:00
Lucas Armand	a12523b1d2	Added bad code to tgi server to test	2025-11-11 17:41:12 -08:00
LucasArmandVast	7db54f3bd7	Merge pull request #55 from vast-ai/use-mtoken Use mtoken	2025-11-10 11:54:04 -08:00
LucasArmandVast	d63a060202	Merge pull request #56 from vast-ai/obfuscate-mtoken Obfuscate mtoken in logs	2025-11-10 11:53:17 -08:00
Lucas Armand	c6521cb6d4	add ...	2025-11-07 10:10:35 -08:00
Lucas Armand	b7fe4ebb91	Obfuscate mtoken in logs	2025-11-07 10:02:39 -08:00
Lucas Armand	8ae7b74605	bump version to 0.2.0	2025-11-05 13:32:21 -08:00
Lucas Armand	106067d716	bump version to 0.1.1	2025-11-04 17:15:59 -08:00
Lucas Armand	f5134d4bf5	Fix spelling mistake	2025-11-04 16:59:39 -08:00
Lucas Armand	47e5460532	added mtoken	2025-11-04 15:55:14 -08:00
Colter-Downing	ec2ac0a21a	Merge pull request #52 from vast-ai/remove-sleeps-and-delays Remove sleeps and delays	2025-10-30 11:53:39 -07:00
Abiola Akinnubi	2cde573c56	Merge pull request #48 from vast-ai/comfy-request-idx Added request_idx to comfy auth_data	2025-10-30 11:27:35 -07:00
Abiola Akinnubi	b2e4a5db0c	Merge pull request #49 from vast-ai/unsecure_report_addr Added caller for REPORT_ADDR to backend.py to use the report add	2025-10-30 10:39:46 -07:00
Abiola Akinnubi	7437028cb2	Added caller for REPORT_ADDR to backend.py	2025-10-29 18:02:17 -07:00
edgaratvast	02c8307af7	remove redis pubsub from pyworker (#53 ) Co-authored-by: Edgar Lin <edgarlin2000@gmail.com>	2025-10-29 17:07:56 -07:00
Abiola Akinnubi	944f83fc03	Removed extra spaces from operator assignment	2025-10-28 21:03:52 +00:00
Abiola Akinnubi	f56bbc0ebe	Added request_idx to comfy auth_data	2025-10-27 03:17:06 +00:00