bump up version minor number

feat AUTO-695: add loaded_at attribute to AutoScalerData and Metrics classes
2025-11-14 18:07:17 -08:00 · 2025-11-14 17:07:06 -08:00
4 changed files with 32 additions and 96 deletions
@@ -146,6 +146,7 @@ class Metrics:
    def _set_mtoken(self, mtoken: str) -> None:
        self.mtoken = mtoken

+
    #######################################Private#######################################

    async def __send_delete_requests_and_reset(self):
@@ -280,7 +281,6 @@ class Metrics:

        if sent:
            # clear the one-shot loadtime only if we actually sent *this* value
-            self.system_metrics.reset(expected=loadtime_snapshot)
            self.update_pending = False
            self.model_metrics.reset()
            self.last_metric_update = time.time()
@@ -3,58 +3,38 @@ import logging
 from typing import List
 import ssl
 from asyncio import run, gather
-import asyncio
+

 from lib.backend import Backend
-from lib.metrics import Metrics
 from aiohttp import web

 log = logging.getLogger(__file__)


 def start_server(backend: Backend, routes: List[web.RouteDef], **kwargs):
-    try:
-        log.debug("getting certificate...")
-        use_ssl = os.environ.get("USE_SSL", "false") == "true"
-        if use_ssl is True:
-            ssl_context = ssl.create_default_context(ssl.Purpose.CLIENT_AUTH)
-            ssl_context.load_cert_chain(
-                certfile="/etc/instance.crt",
-                keyfile="/etc/instance.key",
-            )
-        else:
-            ssl_context = None
+    log.debug("getting certificate...")
+    use_ssl = os.environ.get("USE_SSL", "false") == "true"
+    if use_ssl is True:
+        ssl_context = ssl.create_default_context(ssl.Purpose.CLIENT_AUTH)
+        ssl_context.load_cert_chain(
+            certfile="/etc/instance.crt",
+            keyfile="/etc/instance.key",
+        )
+    else:
+        ssl_context = None

-        async def main():
-            log.debug("starting server...")
-            app = web.Application()
-            app.add_routes(routes)
-            runner = web.AppRunner(app)
-            await runner.setup()
-            site = web.TCPSite(
-                runner,
-                ssl_context=ssl_context,
-                port=int(os.environ["WORKER_PORT"]),
-                **kwargs
-            )
-            await gather(site.start(), backend._start_tracking())
+    async def main():
+        log.debug("starting server...")
+        app = web.Application()
+        app.add_routes(routes)
+        runner = web.AppRunner(app)
+        await runner.setup()
+        site = web.TCPSite(
+            runner,
+            ssl_context=ssl_context,
+            port=int(os.environ["WORKER_PORT"]),
+            **kwargs
+        )
+        await gather(site.start(), backend._start_tracking())

-        run(main())
-
-    except Exception as e:
-        err_msg = f"PyWorker failed to launch: {e}"
-        log.error(err_msg)
-
-        async def beacon():
-            metrics = Metrics()
-            metrics._set_version(getattr(backend, "version", "0"))
-            metrics._set_mtoken(getattr(backend, "mtoken", ""))
-            try:
-                while True:
-                    metrics._model_errored(err_msg)
-                    await metrics._Metrics__send_metrics_and_reset()
-                    await asyncio.sleep(10)
-            finally:
-                await metrics.aclose()
-
-        run(beacon())
+    run(main())
@@ -41,14 +41,6 @@ echo_var DEBUG_LOG
 echo_var PYWORKER_LOG
 echo_var MODEL_LOG

-# if instance is rebooted, we want to clear out the log file so pyworker doesn't read lines
-# from the run prior to reboot. past logs are saved in $MODEL_LOG.old for debugging only
-if [ -e "$MODEL_LOG" ]; then
-    echo "Rotating model log at $MODEL_LOG to $MODEL_LOG.old"
-    cat "$MODEL_LOG" >> "$MODEL_LOG.old" 
-    : > "$MODEL_LOG"
-fi
-
 # Populate /etc/environment with quoted values
 if ! grep -q "VAST" /etc/environment; then
    env -0 | grep -zEv "^(HOME=|SHLVL=)|CONDA" | while IFS= read -r -d '' line; do
@@ -132,43 +124,9 @@ cd "$SERVER_DIR"

 echo "launching PyWorker server"

-set +e
-python3 -m "workers.$BACKEND.server" |& tee -a "$PYWORKER_LOG"
-PY_STATUS=${PIPESTATUS[0]}
-set -e
-
-if [ "${PY_STATUS}" -ne 0 ]; then
-  echo "PyWorker exited with status ${PY_STATUS}; notifying autoscaler..."
-  ERROR_MSG="PyWorker exited: code ${PY_STATUS}"
-  MTOKEN="${MASTER_TOKEN:-}"
-  VERSION="${PYWORKER_VERSION:-0}"
-
-  IFS=',' read -r -a REPORT_ADDRS <<< "${REPORT_ADDR}"
-  for addr in "${REPORT_ADDRS[@]}"; do
-    curl -sS -X POST -H 'Content-Type: application/json' \
-      -d "$(cat <<JSON
-{
-  "id": ${CONTAINER_ID:-0},
-  "mtoken": "${MTOKEN}",
-  "version": "${VERSION}",
-  "loadtime": 0,
-  "new_load": 0,
-  "cur_load": 0,
-  "rej_load": 0,
-  "max_perf": 0,
-  "cur_perf": 0,
-  "error_msg": "${ERROR_MSG}",
-  "num_requests_working": 0,
-  "num_requests_recieved": 0,
-  "additional_disk_usage": 0,
-  "working_request_idxs": [],
-  "cur_capacity": 0,
-  "max_capacity": 0,
-  "url": "${URL}"
-}
-JSON
-)" "${addr%/}/worker_status/" || true
-  done
-fi
+# if instance is rebooted, we want to clear out the log file so pyworker doesn't read lines
+# from the run prior to reboot. past logs are saved in $MODEL_LOG.old for debugging only
+[ -e "$MODEL_LOG" ] && cat "$MODEL_LOG" >> "$MODEL_LOG.old" && : > "$MODEL_LOG"

+(python3 -m "workers.$BACKEND.server" |& tee -a "$PYWORKER_LOG") &
 echo "launching PyWorker server done"
@@ -11,7 +11,6 @@ MODEL_SERVER_START_LOG_MSG = [
    "llama runner started",  # Ollama
    '"message":"Connected","target":"text_generation_router"',  # TGI
    '"message":"Connected","target":"text_generation_router::server"',  # TGI
-    "main: model loaded" # llama.cpp
 ]

 MODEL_SERVER_ERROR_LOG_MSGS = [
@@ -35,7 +34,6 @@ backend = Backend(
    model_server_url=os.environ["MODEL_SERVER_URL"],
    model_log_file=os.environ["MODEL_LOG"],
    allow_parallel_requests=True,
-    max_wait_time=600.0,
    benchmark_handler=CompletionsHandler(benchmark_runs=3, benchmark_words=256),
    log_actions=[
        *[(LogAction.ModelLoaded, info_msg) for info_msg in MODEL_SERVER_START_LOG_MSG],
Author	SHA1	Message	Date
Abiola Akinnubi	74efc2cb42	bump up version minor number	2025-11-14 18:07:17 -08:00
Abiola Akinnubi	db3096bbaf	feat AUTO-695: add loaded_at attribute to AutoScalerData and Metrics classes	2025-11-14 17:07:06 -08:00