{"created": 1785264965.5436065, "duration": 5589.100587844849, "exitcode": 2, "root": "/workspace/source/test/e2e", "environment": {}, "summary": {"passed": 56, "failed": 5, "total": 61, "collected": 73}, "collectors": [{"nodeid": "explainer/test_art_explainer.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/explainer/test_art_explainer.py', 48, 'Skipped: ODH does not support art explainer at the moment')"}, {"nodeid": "predictor/test_grpc.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/predictor/test_grpc.py', 39, 'Skipped: Not testable in ODH at the moment')"}, {"nodeid": "predictor/test_torchserve.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/predictor/test_torchserve.py', 34, 'Skipped: ODH does not support torchserve at the moment')"}], "tests": [{"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-no-scheduler-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-no-scheduler-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "no_scheduler", "__wrapped__", "pytestmark", "router-no-scheduler-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.5644308509999973, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 239.412720026, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.039935275000061665, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_flow_control.py::test_flow_control_smoke[cluster_cpu-cluster_single_node-flow-control-utilization-detector]", "lineno": 46, "outcome": "passed", "keywords": ["test_flow_control_smoke[flow-control-utilization-detector]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "flow_control", "pytestmark", "flow-control-utilization-detector", "llminferenceservice", "llmisvc_core", "test_flow_control.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.5651865009995163, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 87.34817074000057, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.039442754999981844, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_flow_control.py::test_flow_control_smoke[cluster_cpu-cluster_single_node-flow-control-concurrency-detector]", "lineno": 46, "outcome": "passed", "keywords": ["test_flow_control_smoke[flow-control-concurrency-detector]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "flow_control", "pytestmark", "flow-control-concurrency-detector", "llminferenceservice", "llmisvc_core", "test_flow_control.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3113212870002826, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 75.71900022800037, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03403705599976092, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_gateway_section_name.py::test_gateway_section_name_propagation[cluster_single_node-cluster_cpu-with-section-name]", "lineno": 131, "outcome": "passed", "keywords": ["test_gateway_section_name_propagation[with-section-name]", "parametrize", "llmd_simulator", "cluster_single_node", "cluster_cpu", "pytestmark", "with-section-name", "llminferenceservice", "llmisvc_core", "test_gateway_section_name.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14422744500006957, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 10.794308358999842, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.030299594999632973, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_gateway_section_name.py::test_gateway_section_name_propagation[cluster_single_node-cluster_cpu-without-section-name]", "lineno": 131, "outcome": "passed", "keywords": ["test_gateway_section_name_propagation[without-section-name]", "parametrize", "llmd_simulator", "cluster_single_node", "cluster_cpu", "pytestmark", "without-section-name", "llminferenceservice", "llmisvc_core", "test_gateway_section_name.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15266185500058782, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 20.322219163000227, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.036511683999378874, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_enabled_requires_token[cluster_cpu-cluster_single_node-auth-enabled-default]", "lineno": 221, "outcome": "passed", "keywords": ["test_llm_auth_enabled_requires_token[auth-enabled-default]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-enabled-default", "llmisvc_core", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 1.1845244999994975, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 254.78334213400012, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.039375413000016124, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_multi_node-router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_multi_node", "__wrapped__", "pytestmark", "router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.44565704700016795, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 232.35918851799943, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03909298500002478, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_invalid_token_rejected[cluster_cpu-cluster_single_node-auth-invalid-token]", "lineno": 386, "outcome": "failed", "keywords": ["test_llm_auth_invalid_token_rejected[auth-invalid-token]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-invalid-token", "llmisvc_core", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3485569510003188, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 901.7819327739999, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T17:31:04Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T17:31:04Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T17:31:04Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-28T17:30:45Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T17:31:04Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T17:31:17Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T17:31:17Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T17:31:04Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_auth.py", "lineno": 440, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-single-cpu', 'model-fb-opt-125m'], prompt='KServe is a', service_name=...               {'name': 'model-fb-opt-125m-auth-invalid-20a12e51'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.llminferenceservice\n    @pytest.mark.auth\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"auth-invalid-token-test\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                ],\n                id=\"auth-invalid-token\",\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_auth_invalid_token_rejected(test_case: TestCase):  # noqa: F811\n        \"\"\"\n        Test that when auth is enabled:\n        - Requests with MALFORMED tokens are rejected\n        \"\"\"\n        inject_k8s_proxy()\n    \n        kserve_client = KServeClient(\n            config_file=os.environ.get(\"KUBECONFIG\", \"~/.kube/config\"),\n            client_configuration=client.Configuration(),\n        )\n    \n        service_name = test_case.llm_service.metadata.name\n        sa_name = f\"{service_name}-test-sa\"\n        ns = test_case.llm_service.metadata.namespace\n        test_failed = False\n    \n        # Enable auth for this test\n        if not test_case.llm_service.metadata.annotations:\n            test_case.llm_service.metadata.annotations = {}\n        test_case.llm_service.metadata.annotations[\n            \"security.opendatahub.io/enable-auth\"\n        ] = \"true\"\n    \n        try:\n            # Create LLMInferenceService\n            create_llmisvc(kserve_client, test_case.llm_service)\n>           wait_for_llm_isvc_ready(\n                kserve_client, test_case.llm_service, test_case.wait_timeout\n            )\n\nllmisvc/test_llm_auth.py:440: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7fc2591eca90>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...invali-efc9e473'},\n                       {'name': 'model-fb-opt-125m-auth-invalid-20a12e51'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-28T17:30:30.884470', start_time = 1785259830.8848135\nduration = 900.2377202510834, timestamp_end = '2026-07-28T17:45:31.122542'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fc2591eca90>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....-auth-invali-efc9e473'},\n                       {'name': 'model-fb-opt-125m-auth-invalid-20a12e51'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7fc258c50c20>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T17:31:04Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T17:31:04Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T17:31:04Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-28T17:30:45Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T17:31:04Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T17:31:17Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T17:31:17Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T17:31:04Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.002240597999843885, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-inline-config-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-inline-config-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-inline-config-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.6948626140001579, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 83.2576509140008, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.042264602000614104, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-model-qwen2.5-0.5b", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.30305829999997513, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 55.637499800000114, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.024328053000317595, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-configmap-ref-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.28039066800010914, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 62.52233558400076, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.07872886299992388, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-replicas-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-replicas-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-replicas-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3394550180000806, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 60.77174675099923, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03913888900024176, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-custom-template-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-custom-template-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-custom-template-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.30752922099964053, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 75.95374176799942, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04241746299976512, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3061212929997055, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 187.33734357999947, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03777293499933876, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3297442339999179, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 61.130034476999754, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03653867699995317, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-tokenizer-kvcache-workload-llmd-simulator-kvcache]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-tokenizer-kvcache-workload-llmd-simulator-kvcache]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-with-tokenizer-kvcache-workload-llmd-simulator-kvcache", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.32014254400019126, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 104.86097333399994, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0359048320005968, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3021259720007947, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 75.69591958199999, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.036186749000080454, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator0]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator0]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator0", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.23874806000003446, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 74.27638330799982, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04285748300026171, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator1]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator1]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "model_routing", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator1", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.41009906599992973, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 137.7707377959996, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.038981148999482684, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_disabled_no_token_required[cluster_cpu-cluster_single_node-auth-disabled]", "lineno": 523, "outcome": "passed", "keywords": ["test_llm_auth_disabled_no_token_required[auth-disabled]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-disabled", "llmisvc_core", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2408631289999903, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 150.95358872900033, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03757415600011882, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator2]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator2]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "model_routing", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator2", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.40162232700004097, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 132.66844822300027, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03811416099961207, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_hpa_deployment[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "lineno": 507, "outcome": "failed", "keywords": ["test_llm_autoscaling_hpa_deployment[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "parametrize", "autoscaling_hpa", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.9011546599995199, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 921.3044724970005, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T17:48:45Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T17:48:45Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T17:48:45Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-28T17:48:18Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T17:48:50Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T17:48:50Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T17:48:45Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-28T17:48:50Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T17:48:45Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 542, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 482, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_hpa\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-hpa\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-hpa-deploy\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_hpa_deployment(test_case: TestCase):\n        \"\"\"HPA + Deployment: HPA exists with WVA annotations; pods scale up under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:542: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fc258182f90>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:482: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7fc258182f90>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-hpa-4c186bcf'},\n                       {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-28T17:48:05.284533', start_time = 1785260885.2849283\nduration = 900.7014257907867, timestamp_end = '2026-07-28T18:03:05.986374'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fc258182f90>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-hpa-4c186bcf'},\n                       {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7fc253f751c0>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T17:48:45Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T17:48:45Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T17:48:45Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-28T17:48:18Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T17:48:50Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T17:48:50Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T17:48:45Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-28T17:48:50Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T17:48:45Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.0036560349999490427, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "model_routing", "lora", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2847342999993998, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 128.87363093000022, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.037598525999783305, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "model_routing", "lora", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2822875969995948, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 119.4587415040005, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03480999299972609, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-pvc]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-pvc]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-pvc", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 20.691606546999537, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 150.5959925499992, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03675922200000059, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-pd-cpu-model-pvc]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-pd-cpu-model-pvc]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-pd-cpu-model-pvc", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 20.472024205999332, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 199.51751744800004, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.05072393899990857, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_multi_node-router-managed-workload-simulated-dp-ep-cpu-model-pvc]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-pvc]", "parametrize", "asyncio", "cluster_cpu", "cluster_multi_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-simulated-dp-ep-cpu-model-pvc", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 30.527081142000497, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 146.48153457300032, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.033254573999329295, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_finalizer_added", "lineno": 191, "outcome": "passed", "keywords": ["test_config_finalizer_added", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15828071299984003, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 2.1056926100000055, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.031079897999916284, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_deletion_blocked_when_referenced", "lineno": 224, "outcome": "passed", "keywords": ["test_config_deletion_blocked_when_referenced", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17452435599989258, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 18.471372664000228, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.023517691000051855, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_keda_deployment[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda]", "lineno": 571, "outcome": "failed", "keywords": ["test_llm_autoscaling_keda_deployment[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda]", "parametrize", "autoscaling_keda", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.6645180879995678, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 920.2775075810005, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T18:03:55Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T18:03:55Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T18:03:55Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:03:36Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T18:04:20Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T18:04:20Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T18:03:55Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-28T18:04:20Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:03:55Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 606, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 482, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-keda'], pro...              {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_keda\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-keda\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-keda-deploy\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_keda_deployment(test_case: TestCase):\n        \"\"\"KEDA + Deployment: ScaledObject exists with WVA annotations; no HPA; pods scale up under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:606: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fc258b2c6d0>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-keda'], pro...              {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:482: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7fc258b2c6d0>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-ked-101f2a9d'},\n                       {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-28T18:03:27.005951', start_time = 1785261807.0062244\nduration = 900.011447429657, timestamp_end = '2026-07-28T18:18:27.017674'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fc258b2c6d0>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-ked-101f2a9d'},\n                       {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7fc253f75a80>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T18:03:55Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T18:03:55Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T18:03:55Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:03:36Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T18:04:20Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T18:04:20Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T18:03:55Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-28T18:04:20Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:03:55Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.0012520529999164864, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_deletion_allowed_when_unreferenced", "lineno": 314, "outcome": "passed", "keywords": ["test_config_deletion_allowed_when_unreferenced", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1414957080005479, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 4.088887834999696, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.030558274000213714, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_deletion_unblocked_after_service_deleted", "lineno": 349, "outcome": "passed", "keywords": ["test_config_deletion_unblocked_after_service_deleted", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14822154200010118, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 4.666955038999731, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.025685020000310033, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_well_known_config_deletion_prevented_by_webhook", "lineno": 431, "outcome": "passed", "keywords": ["test_well_known_config_deletion_prevented_by_webhook", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.03639841699987301, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.2447680980003497, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0001642340002945275, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_well_known_config_deletion_blocked_by_implicit_reference", "lineno": 465, "outcome": "passed", "keywords": ["test_well_known_config_deletion_blocked_by_implicit_reference", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1386717379991751, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 5.25145930600047, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.026410547000523366, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_v1alpha1_to_v1alpha2_conversion", "lineno": 211, "outcome": "passed", "keywords": ["test_v1alpha1_to_v1alpha2_conversion", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17112475199974142, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.3197383489996355, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.4061698469995463, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_v1alpha2_to_v1alpha1_conversion", "lineno": 302, "outcome": "passed", "keywords": ["test_v1alpha2_to_v1alpha1_conversion", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.22062305999952514, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.35662745799982076, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.1271354179998525, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_criticality_preservation_via_annotations", "lineno": 393, "outcome": "passed", "keywords": ["test_criticality_preservation_via_annotations", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1612300939996203, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.9012483779997638, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.32193195600029867, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_lora_criticality_preservation", "lineno": 530, "outcome": "passed", "keywords": ["test_lora_criticality_preservation", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14088264500060177, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.3354736680003043, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.5292543600007775, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_round_trip_conversion_preserves_fields", "lineno": 679, "outcome": "passed", "keywords": ["test_round_trip_conversion_preserves_fields", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17896558600023127, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.4886544029996003, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.33177224100018066, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_stop.py::test_llm_stop_feature[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 39, "outcome": "passed", "keywords": ["test_llm_stop_feature[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_stop.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.8986145489998307, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 301.8707415710005, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03937061399938102, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_lora_adapters.py::test_llm_with_lora_adapters[cluster_cpu-single-lora-adapter-hf]", "lineno": 209, "outcome": "passed", "keywords": ["test_llm_with_lora_adapters[single-lora-adapter-hf]", "parametrize", "cluster_cpu", "lora", "__wrapped__", "pytestmark", "single-lora-adapter-hf", "llminferenceservice", "llmisvc_core", "test_llm_lora_adapters.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14499326899931475, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 150.81111273900024, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04255582499990851, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_lora_adapters.py::test_llm_with_lora_adapters[cluster_cpu-multiple-lora-adapters]", "lineno": 209, "outcome": "passed", "keywords": ["test_llm_with_lora_adapters[multiple-lora-adapters]", "parametrize", "cluster_cpu", "lora", "__wrapped__", "pytestmark", "multiple-lora-adapters", "llminferenceservice", "llmisvc_core", "test_llm_lora_adapters.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2500667270005579, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 152.52126060699993, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04804131099990627, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_tls.py::test_llm_tls_resources[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 91, "outcome": "passed", "keywords": ["test_llm_tls_resources[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llmisvc_core", "test_llm_tls.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.6138609509998787, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 149.7347057729994, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03961164600059419, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_prestop_hook.py::test_prestop_hook[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 40, "outcome": "passed", "keywords": ["test_prestop_hook[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_prestop_hook.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.29265498600034334, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 229.0937554559996, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04286864500136289, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_hpa_lws[cluster_cpu-cluster_multi_node-router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-hpa]", "lineno": 635, "outcome": "failed", "keywords": ["test_llm_autoscaling_hpa_lws[router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-hpa]", "parametrize", "autoscaling_hpa", "cluster_cpu", "cluster_multi_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-hpa", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 1.054150461000063, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 926.3225229049995, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T18:19:20Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T18:19:20Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T18:19:07Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T18:19:37Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T18:19:37Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T18:19:21Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-28T18:19:37Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:19:20Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:19:21Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 670, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 482, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-hpa'], prompt='KSer...                {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_hpa\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-lws\",\n                        \"prometheus-scrape\",\n                        \"scaling-hpa\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-hpa-lws\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_multi_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_hpa_lws(test_case: TestCase):\n        \"\"\"HPA + LWS: HPA exists with WVA annotations; pods scale under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:670: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fc259f65a50>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-hpa'], prompt='KSer...                {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:482: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7fc259f65a50>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...ale-hpa-b29acdba'},\n                       {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-28T18:18:48.789897', start_time = 1785262728.7902377\nduration = 900.5734322071075, timestamp_end = '2026-07-28T18:33:49.363672'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fc259f65a50>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....autoscale-hpa-b29acdba'},\n                       {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7fc253f76700>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T18:19:20Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T18:19:20Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T18:19:07Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T18:19:37Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T18:19:37Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T18:19:21Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-28T18:19:37Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:19:20Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:19:21Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.0020802039998670807, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_rolling_upgrade.py::test_rolling_upgrade_coordination[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-model-fb-opt-125m]", "lineno": 40, "outcome": "passed", "keywords": ["test_rolling_upgrade_coordination[router-managed-workload-llmd-simulator-model-fb-opt-125m]", "parametrize", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_rolling_upgrade.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2672897960001137, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 81.38146512599997, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03646011999990151, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_storage_version_migration.py::TestStorageVersionMigration::test_storage_version_migration_after_simulated_upgrade", "lineno": 112, "outcome": "passed", "keywords": ["test_storage_version_migration_after_simulated_upgrade", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestStorageVersionMigration", "conversion", "test_storage_version_migration.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.18615827900066506, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 66.88495186, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 1.9801971199995023, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_leave_group", "lineno": 992, "outcome": "passed", "keywords": ["test_leave_group", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15271543900053075, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 46.7099886649994, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.025404958998478833, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_three_member_group", "lineno": 1038, "outcome": "passed", "keywords": ["test_three_member_group", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.13890906300002825, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 135.95413621599982, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03588998899977014, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_late_join", "lineno": 1089, "outcome": "passed", "keywords": ["test_late_join", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1848608559994318, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 95.57545631200082, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.038418438998633064, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_delete_at_nonzero_weight", "lineno": 1141, "outcome": "passed", "keywords": ["test_delete_at_nonzero_weight", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16727532100048847, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 66.51755835800031, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.022832024000308593, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_rollback", "lineno": 1183, "outcome": "passed", "keywords": ["test_rollback", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.12916229000074964, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 96.41305509900121, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03277436699863756, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_force_stop_route_owner", "lineno": 1254, "outcome": "passed", "keywords": ["test_force_stop_route_owner", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14421739800127398, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 88.86448345699864, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03189785600079631, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "custom_gateway", "__wrapped__", "pytestmark", "router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 1.131128749999334, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 122.17102395800066, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.027124745000037365, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2939042510006402, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 155.8485242550014, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.037335750999773154, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_keda_lws[cluster_cpu-cluster_multi_node-router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-keda]", "lineno": 693, "outcome": "failed", "keywords": ["test_llm_autoscaling_keda_lws[router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-keda]", "parametrize", "autoscaling_keda", "cluster_cpu", "cluster_multi_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-keda", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.31584606600154075, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 920.9089810950009, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T18:34:31Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T18:34:31Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T18:34:26Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T18:34:57Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T18:34:57Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T18:34:31Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-28T18:34:57Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:34:35Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:34:35Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 728, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 482, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-keda'], prompt='KSe...              {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_keda\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-lws\",\n                        \"prometheus-scrape\",\n                        \"scaling-keda\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-keda-lws\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_multi_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_keda_lws(test_case: TestCase):\n        \"\"\"KEDA + LWS: ScaledObject exists with WVA annotations; pods scale under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:728: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fc258161310>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-keda'], prompt='KSe...              {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:482: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7fc258161310>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-ked-231d315d'},\n                       {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-28T18:34:15.068504', start_time = 1785263655.0688283\nduration = 900.7306854724884, timestamp_end = '2026-07-28T18:49:15.799516'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fc258161310>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-ked-231d315d'},\n                       {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7fc253f774c0>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-28T18:34:31Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-28T18:34:31Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-28T18:34:26Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-28T18:34:57Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-28T18:34:57Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-28T18:34:31Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-28T18:34:57Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:34:35Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-28T18:34:35Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.0016409820000262698, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3157330859994545, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 153.0187003660012, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.030460381000011694, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "custom_gateway", "__wrapped__", "pytestmark", "router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 2.638119776000167, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 421.34205328999997, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.030468832999758888, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-pd-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.26991810700019414, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 226.56368300699978, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.022678046998407808, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.37623625500054914, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 204.37530213799982, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03445087899854116, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "custom_gateway", "__wrapped__", "pytestmark", "router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 2.1814352390010754, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 179.70597399899998, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0427944920011214, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}], "warnings": [{"message": "The event_loop fixture provided by pytest-asyncio has been redefined in\n/workspace/source/test/e2e/conftest.py:43\nReplacing the event_loop fixture with a custom implementation is deprecated\nand will lead to errors in the future.\nIf you want to request an asyncio event loop with a scope other than function\nscope, use the \"scope\" argument to the asyncio mark when marking the tests.\nIf you want to return different types of event loops, use the event_loop_policy\nfixture.\n", "category": "DeprecationWarning", "when": "runtest", "filename": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/pytest_asyncio/plugin.py", "lineno": 761}, {"message": "The test <Function test_flow_control_smoke[flow-control-utilization-detector]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_flow_control.py", "lineno": 47}, {"message": "The test <Function test_flow_control_smoke[flow-control-concurrency-detector]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_flow_control.py", "lineno": 47}, {"message": "The event_loop fixture provided by pytest-asyncio has been redefined in\n/workspace/source/test/e2e/conftest.py:43\nReplacing the event_loop fixture with a custom implementation is deprecated\nand will lead to errors in the future.\nIf you want to request an asyncio event loop with a scope other than function\nscope, use the \"scope\" argument to the asyncio mark when marking the tests.\nIf you want to return different types of event loops, use the event_loop_policy\nfixture.\n", "category": "DeprecationWarning", "when": "runtest", "filename": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/pytest_asyncio/plugin.py", "lineno": 761}, {"message": "The test <Function test_llm_inference_service[router-no-scheduler-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-inline-config-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-replicas-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-custom-template-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-tokenizer-kvcache-workload-llmd-simulator-kvcache]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator0]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator1]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator2]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-pd-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_stop_feature[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service_stop.py", "lineno": 40}, {"message": "The test <Function test_llm_tls_resources[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_tls.py", "lineno": 92}, {"message": "The test <Function test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-pd-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}]}