Skip to content

Commit 38fc96f

Browse files
committed
runtime/lava: add a liveness probe and fail fast on connect
Every device and queue query the scheduler makes against a lab that has gone away blocks for the full 30s request timeout, once per job and per platform. There is no way to ask "is this lab up?" without paying that cost on an endpoint that also does real work. Add LAVA.is_alive(), which polls /system/version/. lava-server answers it from a constant with no database access (SystemViewSet.version in lava_rest_app/v02/views.py) and the body is ~20 bytes, against ~16-78kB and a full device table scan for /devices/, so a caller can poll it regularly without adding any measurable load to the lab. Any HTTP answer counts as reachable, including the refusals: labs that return 401 or 403 to an anonymous or unprivileged request are up, and treating them as down would stop scheduling against them entirely. Only a transport failure or a 5xx is reported as unreachable. Also split the request timeout into a short connect timeout and the existing response timeout, so an unreachable host fails in seconds rather than after 30, and give Runtime a default is_alive() so callers can probe any runtime. Signed-off-by: Denys Fedoryshchenko <denys.f@collabora.com>
1 parent 4825869 commit 38fc96f

3 files changed

Lines changed: 146 additions & 3 deletions

File tree

‎kernelci/runtime/__init__.py‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -287,6 +287,15 @@ def get_job_id(self, job_object):
287287
def wait(self, job_object):
288288
"""Wait for a job to complete and get the exit status code"""
289289

290+
def is_alive(self):
291+
"""Check whether the runtime is reachable
292+
293+
Return a (alive, detail) tuple where *detail* describes the outcome
294+
for logging. Runtimes that have no cheap way of answering this, or
295+
that cannot become unreachable, report themselves as alive.
296+
"""
297+
return True, "liveness check not implemented"
298+
290299

291300
def get_runtime(
292301
config, user=None, token=None, custom_template_dir=None, kcictx=None

‎kernelci/runtime/lava.py‎

Lines changed: 51 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -346,6 +346,19 @@ class LAVA(Runtime):
346346
API_VERSION = "v0.2"
347347
RestAPIServer = namedtuple("RestAPIServer", ["url", "session"])
348348

349+
# Connecting is capped well below the response timeout so an unreachable
350+
# lab fails in seconds instead of tying up the caller for the full
351+
# request timeout on every single call.
352+
CONNECT_TIMEOUT = 5
353+
REQUEST_TIMEOUT = 30
354+
355+
# Liveness probe. lava-server answers /system/version/ from a constant
356+
# without touching the database (SystemViewSet.version in
357+
# lava_rest_app/v02/views.py) and the body is ~20 bytes, so this can be
358+
# polled regularly without adding any measurable load to the lab.
359+
LIVENESS_PATH = "system/version/"
360+
LIVENESS_TIMEOUT = 10
361+
349362
# LAVA supports 'high'/'medium'/'low' (100/50/0), but we define our own
350363
# values to allow scaling across labs with different priority ranges.
351364
PRIORITY_HIGHEST = 80
@@ -457,7 +470,9 @@ def wait(self, job_object):
457470
job_id = int(job_object)
458471
job_url = urljoin(self._server.url, "/".join(["jobs", str(job_id)]))
459472
while True:
460-
resp = self._server.session.get(job_url, timeout=30)
473+
resp = self._server.session.get(
474+
job_url, timeout=self._timeout()
475+
)
461476
resp.raise_for_status()
462477
data = resp.json()
463478
if data["state"] == "Finished":
@@ -477,8 +492,41 @@ def _connect(self):
477492
}
478493
return rest_api
479494

495+
def _timeout(self, read_timeout=None):
496+
"""Timeout tuple for requests: fail fast on connect, wait on read"""
497+
return (self.CONNECT_TIMEOUT, read_timeout or self.REQUEST_TIMEOUT)
498+
499+
def is_alive(self):
500+
"""Check that the LAVA instance is reachable
501+
502+
Any HTTP answer proves the instance is up, including the ones that
503+
refuse the request: 401 and 403 mean the token or the ACL is wrong,
504+
not that the lab is down. Only a transport failure or a server
505+
error counts as unreachable.
506+
"""
507+
if self._server.url is None:
508+
return True, "no server URL configured"
509+
url = urljoin(self._server.url, self.LIVENESS_PATH)
510+
try:
511+
resp = self._server.session.get(
512+
url, timeout=self._timeout(self.LIVENESS_TIMEOUT)
513+
)
514+
except requests.RequestException as exc:
515+
return False, str(exc)
516+
if resp.status_code >= 500:
517+
return False, f"HTTP {resp.status_code}"
518+
if resp.status_code != 200:
519+
return True, f"HTTP {resp.status_code} (reachable)"
520+
try:
521+
version = resp.json().get("version")
522+
except ValueError:
523+
version = None
524+
return True, f"version {version}" if version else "reachable"
525+
480526
def _get_response(self, url, params=None):
481-
resp = self._server.session.get(url, params=params, timeout=30)
527+
resp = self._server.session.get(
528+
url, params=params, timeout=self._timeout()
529+
)
482530
resp.raise_for_status()
483531
return resp.json()
484532

@@ -574,7 +622,7 @@ def _submit(self, job):
574622
jobs_url,
575623
json=job_data,
576624
allow_redirects=False,
577-
timeout=30,
625+
timeout=self._timeout(),
578626
)
579627
if resp.status_code >= 400:
580628
print(f"Error submitting job: {resp.status_code}, {resp.text}")

‎tests/test_runtime.py‎

Lines changed: 86 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -12,6 +12,7 @@
1212
from pathlib import Path
1313

1414
import pytest
15+
import requests
1516
import yaml
1617
from jinja2 import Environment, FileSystemLoader
1718
from jinja2.exceptions import TemplateRuntimeError
@@ -528,3 +529,88 @@ def test_compute_tuxrun_parameters_missing_branch():
528529
"""Nodes lacking kernel_revision.branch get an empty parameter set."""
529530
assert compute_tuxrun_parameters("fvp-aemva", {}) == {}
530531
assert compute_tuxrun_parameters("fvp-aemva", {"data": {}}) == {}
532+
533+
534+
class _LivenessSession:
535+
"""Session recording the liveness request and replaying a canned answer"""
536+
537+
def __init__(self, response=None, error=None):
538+
self.response = response
539+
self.error = error
540+
self.calls = []
541+
542+
def get(self, url, params=None, timeout=None):
543+
self.calls.append((url, timeout))
544+
if self.error:
545+
raise self.error
546+
return self.response
547+
548+
549+
def _liveness_lab(response=None, error=None):
550+
config = kernelci.config.load("tests/configs/lava-runtimes.yaml")
551+
runtime_config = config["runtimes"]["lab-min-12-max-40-new-runtime"]
552+
lab = kernelci.runtime.get_runtime(runtime_config)
553+
lab._server = types.SimpleNamespace(
554+
url="http://lava/api/v0.2/",
555+
session=_LivenessSession(response=response, error=error),
556+
)
557+
return lab
558+
559+
560+
def test_lava_is_alive_reports_the_version():
561+
"""A 200 from /system/version/ means the lab is up."""
562+
lab = _liveness_lab(_FakeResponse({"version": "2026.07"}))
563+
564+
alive, detail = lab.is_alive()
565+
566+
assert alive is True
567+
assert "2026.07" in detail
568+
url, timeout = lab._server.session.calls[0]
569+
assert url == "http://lava/api/v0.2/system/version/"
570+
# Connect fast, then allow the probe timeout for the answer.
571+
assert timeout == (
572+
kernelci.runtime.lava.LAVA.CONNECT_TIMEOUT,
573+
kernelci.runtime.lava.LAVA.LIVENESS_TIMEOUT,
574+
)
575+
576+
577+
def test_lava_is_alive_treats_forbidden_as_reachable():
578+
"""A lab that refuses the request has still answered it."""
579+
lab = _liveness_lab(_FakeResponse({}, status_code=403))
580+
581+
alive, detail = lab.is_alive()
582+
583+
assert alive is True
584+
assert "403" in detail
585+
586+
587+
def test_lava_is_alive_reports_server_errors_as_down():
588+
"""A 5xx means the instance cannot serve requests."""
589+
lab = _liveness_lab(_FakeResponse({}, status_code=502))
590+
591+
alive, detail = lab.is_alive()
592+
593+
assert alive is False
594+
assert "502" in detail
595+
596+
597+
def test_lava_is_alive_reports_transport_failures_as_down():
598+
"""An unreachable host is the case this probe exists for."""
599+
lab = _liveness_lab(
600+
error=requests.ConnectionError("Network is unreachable")
601+
)
602+
603+
alive, detail = lab.is_alive()
604+
605+
assert alive is False
606+
assert "Network is unreachable" in detail
607+
608+
609+
def test_lava_is_alive_without_server_url():
610+
"""A runtime storing jobs externally has no server to probe."""
611+
config = kernelci.config.load("tests/configs/lava-runtimes.yaml")
612+
runtime_config = config["runtimes"]["lab-min-12-max-40-new-runtime"]
613+
lab = kernelci.runtime.get_runtime(runtime_config)
614+
lab._server = types.SimpleNamespace(url=None, session=None)
615+
616+
assert lab.is_alive() == (True, "no server URL configured")

0 commit comments

Comments
 (0)