{"created": 1784661425.8583212, "duration": 5958.0040509700775, "exitcode": 2, "root": "/workspace/source/test/e2e", "environment": {}, "summary": {"passed": 47, "failed": 7, "error": 3, "total": 57, "collected": 60}, "collectors": [{"nodeid": "explainer/test_art_explainer.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/explainer/test_art_explainer.py', 38, 'Skipped: ODH does not support art explainer at the moment')"}, {"nodeid": "predictor/test_grpc.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/predictor/test_grpc.py', 35, 'Skipped: Not testable in ODH at the moment')"}, {"nodeid": "predictor/test_torchserve.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/predictor/test_torchserve.py', 34, 'Skipped: ODH does not support torchserve at the moment')"}], "tests": [{"nodeid": "llmisvc/test_flow_control.py::test_flow_control_smoke[cluster_cpu-cluster_single_node-flow-control-utilization-detector]", "lineno": 46, "outcome": "passed", "keywords": ["test_flow_control_smoke[flow-control-utilization-detector]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "flow_control", "pytestmark", "flow-control-utilization-detector", "llminferenceservice", "llmisvc_core", "test_flow_control.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.5245037040003808, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 80.92244299300364, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0313894179998897, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-custom-template-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-custom-template-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-custom-template-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.5312163630005671, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 79.81728536900482, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.042149279004661366, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.28997236599389, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 132.70973028401204, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03400454600341618, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_flow_control.py::test_flow_control_smoke[cluster_cpu-cluster_single_node-flow-control-concurrency-detector]", "lineno": 46, "outcome": "passed", "keywords": ["test_flow_control_smoke[flow-control-concurrency-detector]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "flow_control", "pytestmark", "flow-control-concurrency-detector", "llminferenceservice", "llmisvc_core", "test_flow_control.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.8877199120033765, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 69.60096432900173, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03933487299946137, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_gateway_section_name.py::test_gateway_section_name_propagation[cluster_single_node-cluster_cpu-with-section-name]", "lineno": 131, "outcome": "passed", "keywords": ["test_gateway_section_name_propagation[with-section-name]", "parametrize", "llmd_simulator", "cluster_single_node", "cluster_cpu", "pytestmark", "with-section-name", "llminferenceservice", "llmisvc_core", "test_gateway_section_name.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17428087499865796, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 10.359220141996047, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03462093399139121, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_gateway_section_name.py::test_gateway_section_name_propagation[cluster_single_node-cluster_cpu-without-section-name]", "lineno": 131, "outcome": "passed", "keywords": ["test_gateway_section_name_propagation[without-section-name]", "parametrize", "llmd_simulator", "cluster_single_node", "cluster_cpu", "pytestmark", "without-section-name", "llminferenceservice", "llmisvc_core", "test_gateway_section_name.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1634325799968792, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 19.418401955990703, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.037425276008434594, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_enabled_requires_token[cluster_cpu-cluster_single_node-auth-enabled-default]", "lineno": 221, "outcome": "passed", "keywords": ["test_llm_auth_enabled_requires_token[auth-enabled-default]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-enabled-default", "llmisvc_core", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.30358377000084147, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 206.03581132199906, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03991140199650545, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2963634780026041, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 59.05726008900092, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04045590799069032, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.30197008899995126, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 131.24753589200554, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03947237299871631, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_invalid_token_rejected[cluster_cpu-cluster_single_node-auth-invalid-token]", "lineno": 386, "outcome": "passed", "keywords": ["test_llm_auth_invalid_token_rejected[auth-invalid-token]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-invalid-token", "llmisvc_core", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.30078686999331694, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 174.89084004599135, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.035401355009526014, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator0]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator0]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator0", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3636894899973413, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 66.74067878699861, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03753905500343535, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator1]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator1]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "model_routing", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator1", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.49092480100807734, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 141.71060775099613, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03144287499890197, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_disabled_no_token_required[cluster_cpu-cluster_single_node-auth-disabled]", "lineno": 523, "outcome": "passed", "keywords": ["test_llm_auth_disabled_no_token_required[auth-disabled]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-disabled", "llmisvc_core", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.4708182970061898, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 157.31576771198888, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.040563399001257494, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator2]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator2]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "model_routing", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator2", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.5033924940071302, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 138.42994859200553, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03994920699915383, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_hpa_deployment[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "lineno": 505, "outcome": "failed", "keywords": ["test_llm_autoscaling_hpa_deployment[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "parametrize", "autoscaling_hpa", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3667619969928637, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 905.1696433699981, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T17:50:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/autoscale-hpa-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T17:50:02Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T17:50:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/autoscale-hpa-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T17:50:02Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T17:50:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/autoscale-hpa-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 540, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 480, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1249, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1260, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_hpa\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-hpa\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-hpa-deploy\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_hpa_deployment(test_case: TestCase):\n        \"\"\"HPA + Deployment: VA and HPA exist; pods scale up under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:540: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b3f39cd0>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:480: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f10b3f39cd0>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-hpa-4c186bcf'},\n                       {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-21T17:49:52.479816', start_time = 1784656192.4801464\nduration = 900.0293323993683, timestamp_end = '2026-07-21T18:04:52.509482'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b3f39cd0>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-hpa-4c186bcf'},\n                       {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(condition.get(\"type\"))\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1249: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f10b8062b60>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1260: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(condition.get(\"type\"))\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T17:50:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/autoscale-hpa-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T17:50:02Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T17:50:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/autoscale-hpa-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T17:50:02Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T17:50:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/autoscale-hpa-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1244: AssertionError"}, "teardown": {"duration": 0.04098579600395169, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "model_routing", "lora", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3042729520093417, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 142.31839161299285, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04765588299778756, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "model_routing", "lora", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3231892119947588, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 122.97012486800668, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03678356599994004, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-pvc]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-pvc]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-pvc", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 20.50957723299507, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 151.77882486599265, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.043060641997726634, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-pd-cpu-model-pvc]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-pd-cpu-model-pvc]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-pd-cpu-model-pvc", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 20.469131424004445, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 203.11637972800236, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.038000597996870056, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_multi_node-router-managed-workload-simulated-dp-ep-cpu-model-pvc]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-pvc]", "parametrize", "asyncio", "cluster_cpu", "cluster_multi_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-simulated-dp-ep-cpu-model-pvc", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 30.553725823003333, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 217.81956622299913, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04401021100056823, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_keda_deployment[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda]", "lineno": 569, "outcome": "failed", "keywords": ["test_llm_autoscaling_keda_deployment[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda]", "parametrize", "autoscaling_keda", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.4033646900061285, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 906.0911458019982, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:05:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-deployment-b2150d0b/autoscale-keda-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:05:02Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:05:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-deployment-b2150d0b/autoscale-keda-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:05:02Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:05:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-deployment-b2150d0b/autoscale-keda-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 604, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 480, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1249, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1260, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-keda'], pro...              {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_keda\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-keda\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-keda-deploy\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_keda_deployment(test_case: TestCase):\n        \"\"\"KEDA + Deployment: VA and ScaledObject exist; no HPA; pods scale up under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:604: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b80d2850>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-keda'], pro...              {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:480: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f10b80d2850>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-ked-101f2a9d'},\n                       {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-21T18:04:58.557132', start_time = 1784657098.557429\nduration = 900.4893100261688, timestamp_end = '2026-07-21T18:19:59.046754'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b80d2850>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-ked-101f2a9d'},\n                       {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(condition.get(\"type\"))\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1249: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f10b2db6ac0>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1260: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(condition.get(\"type\"))\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:05:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-deployment-b2150d0b/autoscale-keda-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:05:02Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:05:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-deployment-b2150d0b/autoscale-keda-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:05:02Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:05:02Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-deployment-b2150d0b/autoscale-keda-deploy-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1244: AssertionError"}, "teardown": {"duration": 0.038037046004319564, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_finalizer_added", "lineno": 191, "outcome": "passed", "keywords": ["test_config_finalizer_added", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16447829200478736, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 2.0690670199983288, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.031065876013599336, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_deletion_blocked_when_referenced", "lineno": 224, "outcome": "passed", "keywords": ["test_config_deletion_blocked_when_referenced", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1886418059875723, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 16.556599674004246, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.034072654001647606, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_deletion_allowed_when_unreferenced", "lineno": 314, "outcome": "passed", "keywords": ["test_config_deletion_allowed_when_unreferenced", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15621732200088445, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 4.43414566598949, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03553474199725315, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_deletion_unblocked_after_service_deleted", "lineno": 349, "outcome": "passed", "keywords": ["test_config_deletion_unblocked_after_service_deleted", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1767982400051551, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 4.2391237070114585, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.031266543999663554, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_well_known_config_deletion_prevented_by_webhook", "lineno": 431, "outcome": "passed", "keywords": ["test_well_known_config_deletion_prevented_by_webhook", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.0294641039945418, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.07632835399999749, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.00014931500481907278, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_well_known_config_deletion_blocked_by_implicit_reference", "lineno": 465, "outcome": "passed", "keywords": ["test_well_known_config_deletion_blocked_by_implicit_reference", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16016460000537336, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 14.496096816990757, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03386006300570443, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_v1alpha1_to_v1alpha2_conversion", "lineno": 210, "outcome": "passed", "keywords": ["test_v1alpha1_to_v1alpha2_conversion", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "llmisvc_core", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17301207801210694, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.31162975600454956, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.21457027200085577, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_v1alpha2_to_v1alpha1_conversion", "lineno": 301, "outcome": "passed", "keywords": ["test_v1alpha2_to_v1alpha1_conversion", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "llmisvc_core", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17341193099855445, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.38669982200372033, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.042411902002641, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_criticality_preservation_via_annotations", "lineno": 392, "outcome": "passed", "keywords": ["test_criticality_preservation_via_annotations", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "llmisvc_core", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1648764929996105, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.49157543999899644, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.6276891349989455, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_lora_criticality_preservation", "lineno": 529, "outcome": "passed", "keywords": ["test_lora_criticality_preservation", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "llmisvc_core", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15891214800649323, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.512979645995074, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.5330072899960214, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_round_trip_conversion_preserves_fields", "lineno": 678, "outcome": "passed", "keywords": ["test_round_trip_conversion_preserves_fields", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "llmisvc_core", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.18050008699356113, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.5856817940075416, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.23537853799643926, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_stop.py::test_llm_stop_feature[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 39, "outcome": "passed", "keywords": ["test_llm_stop_feature[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_stop.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.6035789280140307, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 293.79207378599676, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04288625299523119, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_lora_adapters.py::test_llm_with_lora_adapters[cluster_cpu-single-lora-adapter-hf]", "lineno": 209, "outcome": "passed", "keywords": ["test_llm_with_lora_adapters[single-lora-adapter-hf]", "parametrize", "cluster_cpu", "lora", "__wrapped__", "pytestmark", "single-lora-adapter-hf", "llminferenceservice", "llmisvc_core", "test_llm_lora_adapters.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17805046599823982, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 155.9560636440001, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03414082799281459, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_lora_adapters.py::test_llm_with_lora_adapters[cluster_cpu-multiple-lora-adapters]", "lineno": 209, "outcome": "passed", "keywords": ["test_llm_with_lora_adapters[multiple-lora-adapters]", "parametrize", "cluster_cpu", "lora", "__wrapped__", "pytestmark", "multiple-lora-adapters", "llminferenceservice", "llmisvc_core", "test_llm_lora_adapters.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16628043400123715, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 176.80055455799447, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.038029165007174015, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_tls.py::test_llm_tls_resources[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 91, "outcome": "passed", "keywords": ["test_llm_tls_resources[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llmisvc_core", "test_llm_tls.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2876391480094753, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 172.10991724701307, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04541190700547304, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_prestop_hook.py::test_prestop_hook[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 40, "outcome": "passed", "keywords": ["test_prestop_hook[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_prestop_hook.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.33178736800618935, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 224.22309282200877, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03600691500469111, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_hpa_lws[cluster_cpu-cluster_multi_node-router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-hpa]", "lineno": 633, "outcome": "failed", "keywords": ["test_llm_autoscaling_hpa_lws[router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-hpa]", "parametrize", "autoscaling_hpa", "cluster_cpu", "cluster_multi_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-hpa", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.33841812700848095, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 905.9676094030001, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:20:16Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/autoscale-hpa-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:20:16Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:20:16Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/autoscale-hpa-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:20:16Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:20:28Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:20:16Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/autoscale-hpa-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 668, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 480, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1249, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1260, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-hpa'], prompt='KSer...                {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_hpa\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-lws\",\n                        \"prometheus-scrape\",\n                        \"scaling-hpa\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-hpa-lws\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_multi_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_hpa_lws(test_case: TestCase):\n        \"\"\"HPA + LWS: VA and HPA exist; pods scale under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:668: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b3f72490>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-hpa'], prompt='KSer...                {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:480: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f10b3f72490>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...ale-hpa-b29acdba'},\n                       {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-21T18:20:05.062663', start_time = 1784658005.0629842\nduration = 900.6415681838989, timestamp_end = '2026-07-21T18:35:05.704556'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b3f72490>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....autoscale-hpa-b29acdba'},\n                       {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(condition.get(\"type\"))\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1249: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f10b2db7600>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1260: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(condition.get(\"type\"))\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:20:16Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/autoscale-hpa-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:20:16Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:20:16Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/autoscale-hpa-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:20:16Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:20:28Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:20:16Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/autoscale-hpa-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1244: AssertionError"}, "teardown": {"duration": 0.03582864100462757, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_rolling_upgrade.py::test_rolling_upgrade_coordination[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-model-fb-opt-125m]", "lineno": 40, "outcome": "passed", "keywords": ["test_rolling_upgrade_coordination[router-managed-workload-llmd-simulator-model-fb-opt-125m]", "parametrize", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_rolling_upgrade.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.28276039598858915, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 67.92516919100308, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04076478000206407, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_storage_version_migration.py::TestStorageVersionMigration::test_storage_version_migration_after_simulated_upgrade", "lineno": 111, "outcome": "passed", "keywords": ["test_storage_version_migration_after_simulated_upgrade", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "llmisvc_core", "TestStorageVersionMigration", "conversion", "test_storage_version_migration.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17489580099936575, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 62.946155308003654, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 1.9412395219987957, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "custom_gateway", "__wrapped__", "pytestmark", "router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 2.9302413429977605, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 157.60349579800095, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.02926285499415826, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-pd-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.36733930998889264, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 207.55488090400468, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03780316800111905, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.42713589800405316, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 209.4622082519927, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.038749707004171796, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_keda_lws[cluster_cpu-cluster_multi_node-router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-keda]", "lineno": 691, "outcome": "failed", "keywords": ["test_llm_autoscaling_keda_lws[router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-keda]", "parametrize", "autoscaling_keda", "cluster_cpu", "cluster_multi_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-keda", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.35127127899613697, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 905.7136699969997, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:35:30Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-lws-e541a132/autoscale-keda-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:35:30Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:35:30Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-lws-e541a132/autoscale-keda-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:35:30Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:35:52Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:35:30Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-lws-e541a132/autoscale-keda-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 726, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 480, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1249, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1260, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-keda'], prompt='KSe...              {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_keda\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-lws\",\n                        \"prometheus-scrape\",\n                        \"scaling-keda\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-keda-lws\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_multi_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_keda_lws(test_case: TestCase):\n        \"\"\"KEDA + LWS: VA and ScaledObject exist; pods scale under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:726: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b3f8b290>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-keda'], prompt='KSe...              {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:480: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f10b3f8b290>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-ked-231d315d'},\n                       {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-21T18:35:11.345886', start_time = 1784658911.3461716\nduration = 900.4803597927094, timestamp_end = '2026-07-21T18:50:11.826540'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b3f8b290>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-ked-231d315d'},\n                       {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(condition.get(\"type\"))\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1249: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f10b2db6840>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1260: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(condition.get(\"type\"))\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:35:30Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-lws-e541a132/autoscale-keda-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:35:30Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:35:30Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-lws-e541a132/autoscale-keda-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:35:30Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:35:52Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:35:30Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-keda-lws-e541a132/autoscale-keda-lws-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1244: AssertionError"}, "teardown": {"duration": 0.036003390996484086, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "custom_gateway", "__wrapped__", "pytestmark", "router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 2.9674595770047745, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 219.02730178101046, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.046844779993989505, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-no-scheduler-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-no-scheduler-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "no_scheduler", "__wrapped__", "pytestmark", "router-no-scheduler-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3261341449979227, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 159.26895494900236, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03625283599831164, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_multi_node-router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_multi_node", "__wrapped__", "pytestmark", "router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.42060516199853737, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 168.19041640599607, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04298343800473958, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-inline-config-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-inline-config-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-inline-config-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.36898161299177445, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 72.63369412900647, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04405472200596705, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-model-qwen2.5-0.5b", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2876234180002939, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 77.21955133799929, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04261218800093047, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-configmap-ref-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.35211095900740474, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 67.24346656900889, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.08704445000330452, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-replicas-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-replicas-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-replicas-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.26965130999451503, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 55.9590290789929, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.043094022999866866, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_update_hpa[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "lineno": 1102, "outcome": "failed", "keywords": ["test_llm_autoscaling_update_hpa[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "parametrize", "autoscaling_hpa", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.30514523599413224, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 905.4261815939972, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:49:04Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-update-hpa-efca8fe3/autoscale-update-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:49:04Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:49:04Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-update-hpa-efca8fe3/autoscale-update-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:49:04Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:49:04Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-update-hpa-efca8fe3/autoscale-update-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 1137, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 480, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1249, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1260, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-update-hp-9b5475bf'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_hpa\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-hpa\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-update-hpa\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_update_hpa(test_case: TestCase):\n        \"\"\"Patching maxReplicas should update the HPA; VA and HPA still exist.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:1137: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253a0ef650>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-update-hp-9b5475bf'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:480: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f253a0ef650>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-upd-080288c7'},\n                       {'name': 'scaling-hpa-autoscale-update-hp-9b5475bf'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-21T18:49:02.972712', start_time = 1784659742.9731867\nduration = 900.2921702861786, timestamp_end = '2026-07-21T19:04:03.265362'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253a0ef650>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-upd-080288c7'},\n                       {'name': 'scaling-hpa-autoscale-update-hp-9b5475bf'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(condition.get(\"type\"))\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1249: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f253abe8180>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1260: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(condition.get(\"type\"))\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:49:04Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-update-hpa-efca8fe3/autoscale-update-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:49:04Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:49:04Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-update-hpa-efca8fe3/autoscale-update-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:49:04Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:49:04Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-update-hpa-efca8fe3/autoscale-update-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1244: AssertionError"}, "teardown": {"duration": 0.03608509901096113, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_cleanup_hpa[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "lineno": 855, "outcome": "failed", "keywords": ["test_llm_autoscaling_cleanup_hpa[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "parametrize", "autoscaling_hpa", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3262698050093604, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 905.5400603449962, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:50:25Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/autoscale-cleanup-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:50:25Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:50:25Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/autoscale-cleanup-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:50:25Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:50:25Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/autoscale-cleanup-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 890, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 480, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1249, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1260, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1244, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-cleanup-h-aa1ae037'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_hpa\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-hpa\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-cleanup-hpa\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_cleanup_hpa(test_case: TestCase):\n        \"\"\"Removing scaling config should delete VA and HPA.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:890: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b3f18090>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-cleanup-h-aa1ae037'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:480: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f10b3f18090>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-cle-5a67f5d1'},\n                       {'name': 'scaling-hpa-autoscale-cleanup-h-aa1ae037'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-21T18:50:17.371359', start_time = 1784659817.3716135\nduration = 900.3978471755981, timestamp_end = '2026-07-21T19:05:17.769464'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f10b3f18090>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-cle-5a67f5d1'},\n                       {'name': 'scaling-hpa-autoscale-cleanup-h-aa1ae037'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(condition.get(\"type\"))\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1249: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f10b8068540>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1260: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(condition.get(\"type\"))\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready', 'RouterReady'}, expected {'WorkloadsReady', 'Ready', 'RouterReady'}, got [{'lastTransitionTime': '2026-07-21T18:50:25Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/autoscale-cleanup-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'severity': 'Info', 'status': 'False', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-21T18:50:25Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-21T18:50:25Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/autoscale-cleanup-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-21T18:50:25Z', 'status': 'Unknown', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-21T18:50:25Z', 'message': 'failed to reconcile main workload scaling: failed to reconcile main VA: failed to get v1alpha1.VariantAutoscaling e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/autoscale-cleanup-hpa-kserve-va: no matches for kind \"VariantAutoscaling\" in version \"llmd.ai/v1alpha1\"', 'reason': 'ScalingCRDNotFound', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1244: AssertionError"}, "teardown": {"duration": 0.03206000800128095, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_update_keda[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda]", "lineno": 1178, "outcome": "failed", "keywords": ["test_llm_autoscaling_update_keda[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda]", "parametrize", "autoscaling_keda", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3728388520103181, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 775.4402478310076, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1092, "message": "RuntimeError: \u274c Exception when calling CustomObjectsApi->get_namespaced_custom_object for LLMInferenceService: (500)\nReason: Internal Server Error\nHTTP response headers: HTTPHeaderDict({'Audit-Id': '07febcbe-3641-4256-891c-9e8040ea5d19', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:16:54 GMT', 'Content-Length': '264'})\nHTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"conversion webhook for serving.kserve.io/v1alpha2, Kind=LLMInferenceService failed: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/convert?timeout=30s\\\": EOF\",\"code\":500}"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 1213, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 480, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1249, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1260, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1219, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1092, "message": "RuntimeError"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253a096e50>\nname = 'autoscale-update-keda'\nnamespace = 'e2e-test-llm-autoscaling-update-keda-32b1adf1'\nversion = 'v1alpha1'\n\n    def get_llmisvc(\n        kserve_client: KServeClient,\n        name,\n        namespace,\n        version=constants.KSERVE_V1ALPHA1_VERSION,\n    ):\n        try:\n>           return kserve_client.api_instance.get_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICE,\n                name,\n            )\n\nllmisvc/test_llm_inference_service.py:1084: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253a543cd0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-autoscaling-update-keda-32b1adf1'\nplural = 'llminferenceservices', name = 'autoscale-update-keda'\nkwargs = {'_return_http_data_only': True}\n\n    def get_namespaced_custom_object(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: object\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1632: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253a543cd0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-autoscaling-update-keda-32b1adf1'\nplural = 'llminferenceservices', name = 'autoscale-update-keda'\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...], 'auth_settings': ['BearerToken'], 'body_params': None, ...}\nall_params = ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {'group': 'serving.kserve.io', 'name': 'autoscale-update-keda', 'namespace': 'e2e-test-llm-autoscaling-update-keda-32b1adf1', 'plural': 'llminferenceservices', ...}\nquery_params = []\n\n    def get_namespaced_custom_object_with_http_info(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(object, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'group',\n            'version',\n            'namespace',\n            'plural',\n            'name'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method get_namespaced_custom_object\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'group' is set\n        if self.api_client.client_side_validation and ('group' not in local_var_params or  # noqa: E501\n                                                        local_var_params['group'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `group` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'version' is set\n        if self.api_client.client_side_validation and ('version' not in local_var_params or  # noqa: E501\n                                                        local_var_params['version'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `version` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'namespace' is set\n        if self.api_client.client_side_validation and ('namespace' not in local_var_params or  # noqa: E501\n                                                        local_var_params['namespace'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `namespace` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'plural' is set\n        if self.api_client.client_side_validation and ('plural' not in local_var_params or  # noqa: E501\n                                                        local_var_params['plural'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `plural` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'name' is set\n        if self.api_client.client_side_validation and ('name' not in local_var_params or  # noqa: E501\n                                                        local_var_params['name'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `name` when calling `get_namespaced_custom_object`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n        if 'group' in local_var_params:\n            path_params['group'] = local_var_params['group']  # noqa: E501\n        if 'version' in local_var_params:\n            path_params['version'] = local_var_params['version']  # noqa: E501\n        if 'namespace' in local_var_params:\n            path_params['namespace'] = local_var_params['namespace']  # noqa: E501\n        if 'plural' in local_var_params:\n            path_params['plural'] = local_var_params['plural']  # noqa: E501\n        if 'name' in local_var_params:\n            path_params['name'] = local_var_params['name']  # noqa: E501\n    \n        query_params = []\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}', 'GET',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='object',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1739: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253a540710>\nresource_path = '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}'\nmethod = 'GET'\npath_params = {'group': 'serving.kserve.io', 'name': 'autoscale-update-keda', 'namespace': 'e2e-test-llm-autoscaling-update-keda-32b1adf1', 'plural': 'llminferenceservices', ...}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253a540710>\nresource_path = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-autoscaling-update-keda-32b1adf1/llminferenceservices/autoscale-update-keda'\nmethod = 'GET'\npath_params = [('group', 'serving.kserve.io'), ('version', 'v1alpha1'), ('namespace', 'e2e-test-llm-autoscaling-update-keda-32b1adf1'), ('plural', 'llminferenceservices'), ('name', 'autoscale-update-keda')]\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253a540710>\nmethod = 'GET'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-autoscaling-update-keda-32b1adf1/llminferenceservices/autoscale-update-keda'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = [], body = None, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n>           return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:373: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253a543b10>\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-autoscaling-update-keda-32b1adf1/llminferenceservices/autoscale-update-keda'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], _preload_content = True, _request_timeout = None\n\n    def GET(self, url, headers=None, query_params=None, _preload_content=True,\n            _request_timeout=None):\n>       return self.request(\"GET\", url,\n                            headers=headers,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            query_params=query_params)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:244: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253a543b10>\nmethod = 'GET'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-autoscaling-update-keda-32b1adf1/llminferenceservices/autoscale-update-keda'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'application/x-www-form-urlencoded':  # noqa: E501\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=False,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'multipart/form-data':\n                    # must del headers['Content-Type'], or the correct\n                    # Content-Type which generated by urllib3 will be\n                    # overwritten.\n                    del headers['Content-Type']\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=True,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                # Pass a `string` parameter directly in the body to support\n                # other content types than Json when `body` argument is\n                # provided in serialized form\n                elif isinstance(body, str) or isinstance(body, bytes):\n                    request_body = body\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                else:\n                    # Cannot generate the request from given parameters\n                    msg = \"\"\"Cannot prepare a request message for provided\n                             arguments. Please check that your arguments match\n                             declared content type.\"\"\"\n                    raise ApiException(status=0, reason=msg)\n            # For `GET`, `HEAD`\n            else:\n                r = self.pool_manager.request(method, url,\n                                              fields=query_params,\n                                              preload_content=_preload_content,\n                                              timeout=timeout,\n                                              headers=headers)\n        except urllib3.exceptions.SSLError as e:\n            msg = \"{0}\\n{1}\".format(type(e).__name__, str(e))\n            raise ApiException(status=0, reason=msg)\n    \n        if _preload_content:\n            r = RESTResponse(r)\n    \n            # In the python 3, the response.data is bytes.\n            # we need to decode it to string.\n            if six.PY3:\n                r.data = r.data.decode('utf8')\n    \n            # log response body\n            logger.debug(\"response body: %s\", r.data)\n    \n        if not 200 <= r.status <= 299:\n>           raise ApiException(http_resp=r)\nE           kubernetes.client.exceptions.ApiException: (500)\nE           Reason: Internal Server Error\nE           HTTP response headers: HTTPHeaderDict({'Audit-Id': '07febcbe-3641-4256-891c-9e8040ea5d19', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:16:54 GMT', 'Content-Length': '264'})\nE           HTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"conversion webhook for serving.kserve.io/v1alpha2, Kind=LLMInferenceService failed: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/convert?timeout=30s\\\": EOF\",\"code\":500}\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:238: ApiException\n\nThe above exception was the direct cause of the following exception:\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-keda'], pro...              {'name': 'scaling-keda-autoscale-update-k-a3916813'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_keda\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-keda\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-update-keda\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_update_keda(test_case: TestCase):\n        \"\"\"Patching maxReplicas should update the ScaledObject; VA and ScaledObject still exist.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:1213: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253a096e50>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-keda'], pro...              {'name': 'scaling-keda-autoscale-update-k-a3916813'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:480: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f253a096e50>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-upd-f4225611'},\n                       {'name': 'scaling-keda-autoscale-update-k-a3916813'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-21T19:04:09.059720', start_time = 1784660649.0600536\nduration = 765.287600517273, timestamp_end = '2026-07-21T19:16:54.347661'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253a096e50>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-upd-f4225611'},\n                       {'name': 'scaling-keda-autoscale-update-k-a3916813'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(condition.get(\"type\"))\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1249: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f253b3fee80>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1260: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n>       out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n\nllmisvc/test_llm_inference_service.py:1219: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253a096e50>\nname = 'autoscale-update-keda'\nnamespace = 'e2e-test-llm-autoscaling-update-keda-32b1adf1'\nversion = 'v1alpha1'\n\n    def get_llmisvc(\n        kserve_client: KServeClient,\n        name,\n        namespace,\n        version=constants.KSERVE_V1ALPHA1_VERSION,\n    ):\n        try:\n            return kserve_client.api_instance.get_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICE,\n                name,\n            )\n        except client.rest.ApiException as e:\n>           raise RuntimeError(\n                f\"\u274c Exception when calling CustomObjectsApi->\"\n                f\"get_namespaced_custom_object for LLMInferenceService: {e}\"\n            ) from e\nE           RuntimeError: \u274c Exception when calling CustomObjectsApi->get_namespaced_custom_object for LLMInferenceService: (500)\nE           Reason: Internal Server Error\nE           HTTP response headers: HTTPHeaderDict({'Audit-Id': '07febcbe-3641-4256-891c-9e8040ea5d19', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:16:54 GMT', 'Content-Length': '264'})\nE           HTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"conversion webhook for serving.kserve.io/v1alpha2, Kind=LLMInferenceService failed: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/convert?timeout=30s\\\": EOF\",\"code\":500}\n\nllmisvc/test_llm_inference_service.py:1092: RuntimeError"}, "teardown": {"duration": 0.0498255090060411, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]", "lineno": 242, "outcome": "error", "keywords": ["test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "custom_gateway", "__wrapped__", "pytestmark", "router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.24886913599038962, "outcome": "failed", "crash": {"path": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 238, "message": "kubernetes.client.exceptions.ApiException: (500)\nReason: Internal Server Error\nHTTP response headers: HTTPHeaderDict({'Audit-Id': 'ae8ce573-f003-4686-9276-df1b4af37e51', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:17:04 GMT', 'Content-Length': '701'})\nHTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"Internal error occurred: failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\",\"reason\":\"InternalError\",\"details\":{\"causes\":[{\"message\":\"failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\"}]},\"code\":500}"}, "traceback": [{"path": "common/gateway_proxy_istio.py", "lineno": 183, "message": ""}, {"path": "llmisvc/fixtures.py", "lineno": 1547, "message": ""}, {"path": "llmisvc/fixtures.py", "lineno": 1509, "message": ""}, {"path": "llmisvc/fixtures.py", "lineno": 1683, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py", "lineno": 231, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py", "lineno": 354, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 348, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 180, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 391, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 279, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 238, "message": "ApiException"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253b42ecd0>\nllm_config = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-with-ga...outer': {'gateway': {'refs': [{'name': 'router-gateway-1', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}]}}}}\nnamespace = 'e2e-test-llm-inference-service-6f0aa991'\n\n    def _create_or_update_llmisvc_config(kserve_client, llm_config, namespace=None):\n        \"\"\"Create or update an LLMInferenceServiceConfig resource.\"\"\"\n        version = llm_config[\"apiVersion\"].split(\"/\")[1]\n    \n        if namespace is None:\n            namespace = llm_config.get(\"metadata\", {}).get(\"namespace\", \"default\")\n    \n        name = llm_config.get(\"metadata\", {}).get(\"name\")\n        if not name:\n            raise ValueError(\"LLMInferenceServiceConfig must have a name in metadata\")\n    \n        logger.info(f\"Checking LLMInferenceServiceConfig {name} in namespace {namespace}\")\n    \n        try:\n>           existing_config = kserve_client.api_instance.get_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                name,\n            )\n\nllmisvc/fixtures.py:1657: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253b42ffd0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-6f0aa991'\nplural = 'llminferenceserviceconfigs'\nname = 'router-with-gateway-ref-llmisvc-aa92462b'\nkwargs = {'_return_http_data_only': True}\n\n    def get_namespaced_custom_object(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: object\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1632: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253b42ffd0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-6f0aa991'\nplural = 'llminferenceserviceconfigs'\nname = 'router-with-gateway-ref-llmisvc-aa92462b'\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...], 'auth_settings': ['BearerToken'], 'body_params': None, ...}\nall_params = ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {'group': 'serving.kserve.io', 'name': 'router-with-gateway-ref-llmisvc-aa92462b', 'namespace': 'e2e-test-llm-inference-service-6f0aa991', 'plural': 'llminferenceserviceconfigs', ...}\nquery_params = []\n\n    def get_namespaced_custom_object_with_http_info(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(object, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'group',\n            'version',\n            'namespace',\n            'plural',\n            'name'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method get_namespaced_custom_object\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'group' is set\n        if self.api_client.client_side_validation and ('group' not in local_var_params or  # noqa: E501\n                                                        local_var_params['group'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `group` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'version' is set\n        if self.api_client.client_side_validation and ('version' not in local_var_params or  # noqa: E501\n                                                        local_var_params['version'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `version` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'namespace' is set\n        if self.api_client.client_side_validation and ('namespace' not in local_var_params or  # noqa: E501\n                                                        local_var_params['namespace'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `namespace` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'plural' is set\n        if self.api_client.client_side_validation and ('plural' not in local_var_params or  # noqa: E501\n                                                        local_var_params['plural'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `plural` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'name' is set\n        if self.api_client.client_side_validation and ('name' not in local_var_params or  # noqa: E501\n                                                        local_var_params['name'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `name` when calling `get_namespaced_custom_object`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n        if 'group' in local_var_params:\n            path_params['group'] = local_var_params['group']  # noqa: E501\n        if 'version' in local_var_params:\n            path_params['version'] = local_var_params['version']  # noqa: E501\n        if 'namespace' in local_var_params:\n            path_params['namespace'] = local_var_params['namespace']  # noqa: E501\n        if 'plural' in local_var_params:\n            path_params['plural'] = local_var_params['plural']  # noqa: E501\n        if 'name' in local_var_params:\n            path_params['name'] = local_var_params['name']  # noqa: E501\n    \n        query_params = []\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}', 'GET',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='object',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1739: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b42fdd0>\nresource_path = '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}'\nmethod = 'GET'\npath_params = {'group': 'serving.kserve.io', 'name': 'router-with-gateway-ref-llmisvc-aa92462b', 'namespace': 'e2e-test-llm-inference-service-6f0aa991', 'plural': 'llminferenceserviceconfigs', ...}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b42fdd0>\nresource_path = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-6f0aa991/llminferenceserviceconfigs/router-with-gateway-ref-llmisvc-aa92462b'\nmethod = 'GET'\npath_params = [('group', 'serving.kserve.io'), ('version', 'v1alpha1'), ('namespace', 'e2e-test-llm-inference-service-6f0aa991'), ('plural', 'llminferenceserviceconfigs'), ('name', 'router-with-gateway-ref-llmisvc-aa92462b')]\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b42fdd0>\nmethod = 'GET'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1a...namespaces/e2e-test-llm-inference-service-6f0aa991/llminferenceserviceconfigs/router-with-gateway-ref-llmisvc-aa92462b'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = [], body = None, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n>           return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:373: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253b42dd90>\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1a...namespaces/e2e-test-llm-inference-service-6f0aa991/llminferenceserviceconfigs/router-with-gateway-ref-llmisvc-aa92462b'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], _preload_content = True, _request_timeout = None\n\n    def GET(self, url, headers=None, query_params=None, _preload_content=True,\n            _request_timeout=None):\n>       return self.request(\"GET\", url,\n                            headers=headers,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            query_params=query_params)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:244: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253b42dd90>\nmethod = 'GET'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1a...namespaces/e2e-test-llm-inference-service-6f0aa991/llminferenceserviceconfigs/router-with-gateway-ref-llmisvc-aa92462b'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'application/x-www-form-urlencoded':  # noqa: E501\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=False,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'multipart/form-data':\n                    # must del headers['Content-Type'], or the correct\n                    # Content-Type which generated by urllib3 will be\n                    # overwritten.\n                    del headers['Content-Type']\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=True,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                # Pass a `string` parameter directly in the body to support\n                # other content types than Json when `body` argument is\n                # provided in serialized form\n                elif isinstance(body, str) or isinstance(body, bytes):\n                    request_body = body\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                else:\n                    # Cannot generate the request from given parameters\n                    msg = \"\"\"Cannot prepare a request message for provided\n                             arguments. Please check that your arguments match\n                             declared content type.\"\"\"\n                    raise ApiException(status=0, reason=msg)\n            # For `GET`, `HEAD`\n            else:\n                r = self.pool_manager.request(method, url,\n                                              fields=query_params,\n                                              preload_content=_preload_content,\n                                              timeout=timeout,\n                                              headers=headers)\n        except urllib3.exceptions.SSLError as e:\n            msg = \"{0}\\n{1}\".format(type(e).__name__, str(e))\n            raise ApiException(status=0, reason=msg)\n    \n        if _preload_content:\n            r = RESTResponse(r)\n    \n            # In the python 3, the response.data is bytes.\n            # we need to decode it to string.\n            if six.PY3:\n                r.data = r.data.decode('utf8')\n    \n            # log response body\n            logger.debug(\"response body: %s\", r.data)\n    \n        if not 200 <= r.status <= 299:\n>           raise ApiException(http_resp=r)\nE           kubernetes.client.exceptions.ApiException: (404)\nE           Reason: Not Found\nE           HTTP response headers: HTTPHeaderDict({'Audit-Id': 'c454168b-ce94-4b48-96c0-50977371963f', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:17:04 GMT', 'Content-Length': '338'})\nE           HTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"llminferenceserviceconfigs.serving.kserve.io \\\"router-with-gateway-ref-llmisvc-aa92462b\\\" not found\",\"reason\":\"NotFound\",\"details\":{\"name\":\"router-with-gateway-ref-llmisvc-aa92462b\",\"group\":\"serving.kserve.io\",\"kind\":\"llminferenceserviceconfigs\"},\"code\":404}\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:238: ApiException\n\nDuring handling of the above exception, another exception occurred:\n\nrequest = <SubRequest 'ensure_gateway_proxy_memory' for <Function test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]>>\n\n    @pytest.fixture(autouse=True)\n    def ensure_gateway_proxy_memory(request):\n        \"\"\"After test setup creates gateways, patch them for proxy memory.\"\"\"\n        if not GATEWAY_PROXY_MEMORY:\n            return\n    \n        # Let test_case (llmisvc) create gateways first\n    \n        if \"test_case\" in request.fixturenames:\n>           request.getfixturevalue(\"test_case\")\n\ncommon/gateway_proxy_istio.py:183: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nrequest = <SubRequest 'test_case' for <Function test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]>>\ntest_namespace = 'e2e-test-llm-inference-service-6f0aa991'\n\n    @pytest.fixture(scope=\"function\")\n    def test_case(request, test_namespace):\n        tc = request.param\n        ns = test_namespace\n    \n        inject_k8s_proxy()\n    \n        kserve_client = KServeClient(\n            config_file=os.environ.get(\"KUBECONFIG\", \"~/.kube/config\"),\n            client_configuration=client.Configuration(),\n        )\n    \n        tc.namespace = ns\n        for peer in tc.peers:\n            peer.namespace = ns\n    \n        for func in tc.before_test:\n            func(tc)\n    \n>       _setup_test_case_service(kserve_client, tc, request.node.name, namespace=ns)\n\nllmisvc/fixtures.py:1547: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253b42ecd0>\ntc = TestCase(base_refs=['router-with-gateway-ref', 'router-with-managed-route', 'model-fb-opt-125m', 'workload-llmd-simula...est=[<function <lambda> at 0x7f254064fba0>], after_test=[], peers=[], llm_service=None, model_name='facebook/opt-125m')\ntest_node_name = 'test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]'\nnamespace = 'e2e-test-llm-inference-service-6f0aa991', peer_index = None\n\n    def _setup_test_case_service(\n        kserve_client, tc, test_node_name, namespace, peer_index=None\n    ):\n        \"\"\"Create LLMInferenceServiceConfigs and build the LLMInferenceService for a TestCase.\n    \n        Returns a list of created config names for cleanup tracking.\n        \"\"\"\n        missing_refs = [\n            ref for ref in tc.base_refs if ref not in LLMINFERENCESERVICE_CONFIGS\n        ]\n        if missing_refs:\n            raise ValueError(\n                f\"Missing base_refs in LLMINFERENCESERVICE_CONFIGS: {missing_refs}\"\n            )\n        if not tc.service_name:\n            suffix = f\"-peer-{peer_index}\" if peer_index is not None else \"\"\n            tc.service_name = generate_service_name(test_node_name + suffix, tc.base_refs)\n        if tc.model_name == \"default/model\":\n            tc.model_name = _get_model_name_from_configs(tc.base_refs)\n        elif \"{namespace}\" in tc.model_name:\n            tc.model_name = tc.model_name.format(namespace=namespace)\n    \n        created_configs = []\n        unique_base_refs = []\n        for base_ref in tc.base_refs:\n            unique_config_name = generate_k8s_safe_suffix(base_ref, [tc.service_name])\n            unique_base_refs.append(unique_config_name)\n    \n            config = LLMINFERENCESERVICE_CONFIGS[base_ref]\n            spec = config(namespace) if callable(config) else copy.deepcopy(config)\n    \n            unique_config_body = {\n                \"apiVersion\": \"serving.kserve.io/v1alpha1\",\n                \"kind\": \"LLMInferenceServiceConfig\",\n                \"metadata\": {\n                    \"name\": unique_config_name,\n                    \"namespace\": namespace,\n                },\n                \"spec\": spec,\n            }\n    \n>           _create_or_update_llmisvc_config(kserve_client, unique_config_body, namespace)\n\nllmisvc/fixtures.py:1509: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253b42ecd0>\nllm_config = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-with-ga...outer': {'gateway': {'refs': [{'name': 'router-gateway-1', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}]}}}}\nnamespace = 'e2e-test-llm-inference-service-6f0aa991'\n\n    def _create_or_update_llmisvc_config(kserve_client, llm_config, namespace=None):\n        \"\"\"Create or update an LLMInferenceServiceConfig resource.\"\"\"\n        version = llm_config[\"apiVersion\"].split(\"/\")[1]\n    \n        if namespace is None:\n            namespace = llm_config.get(\"metadata\", {}).get(\"namespace\", \"default\")\n    \n        name = llm_config.get(\"metadata\", {}).get(\"name\")\n        if not name:\n            raise ValueError(\"LLMInferenceServiceConfig must have a name in metadata\")\n    \n        logger.info(f\"Checking LLMInferenceServiceConfig {name} in namespace {namespace}\")\n    \n        try:\n            existing_config = kserve_client.api_instance.get_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                name,\n            )\n    \n            llm_config[\"metadata\"] = existing_config[\"metadata\"]\n    \n            outputs = kserve_client.api_instance.replace_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                name,\n                llm_config,\n            )\n            logger.info(f\"\u2713 Successfully updated LLMInferenceServiceConfig {name}\")\n            return outputs\n    \n        except client.rest.ApiException as e:\n            if e.status == 404:  # Not found - create it\n                logger.info(\n                    f\"Resource not found, creating LLMInferenceServiceConfig {name}\"\n                )\n>               outputs = kserve_client.api_instance.create_namespaced_custom_object(\n                    constants.KSERVE_GROUP,\n                    version,\n                    namespace,\n                    KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                    llm_config,\n                )\n\nllmisvc/fixtures.py:1683: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253b42ffd0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-6f0aa991'\nplural = 'llminferenceserviceconfigs'\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-with-ga...outer': {'gateway': {'refs': [{'name': 'router-gateway-1', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}]}}}}\nkwargs = {'_return_http_data_only': True}\n\n    def create_namespaced_custom_object(self, group, version, namespace, plural, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespaced_custom_object  # noqa: E501\n    \n        Creates a namespace scoped Custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespaced_custom_object(group, version, namespace, plural, body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: The custom resource's group name (required)\n        :param str version: The custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: The custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param object body: The JSON schema of the Resource to create. (required)\n        :param str pretty: If 'true', then the output is pretty printed.\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered. (optional)\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: object\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.create_namespaced_custom_object_with_http_info(group, version, namespace, plural, body, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:231: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253b42ffd0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-6f0aa991'\nplural = 'llminferenceserviceconfigs'\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-with-ga...outer': {'gateway': {'refs': [{'name': 'router-gateway-1', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}]}}}}\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['group', 'version', 'namespace', 'plural', 'body', 'pretty', ...], 'au...2b', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}, 'spec': {'router': {'gateway': {'refs': [{...}]}}}}, ...}\nall_params = ['group', 'version', 'namespace', 'plural', 'body', 'pretty', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {'group': 'serving.kserve.io', 'namespace': 'e2e-test-llm-inference-service-6f0aa991', 'plural': 'llminferenceserviceconfigs', 'version': 'v1alpha1'}\nquery_params = []\n\n    def create_namespaced_custom_object_with_http_info(self, group, version, namespace, plural, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespaced_custom_object  # noqa: E501\n    \n        Creates a namespace scoped Custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespaced_custom_object_with_http_info(group, version, namespace, plural, body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: The custom resource's group name (required)\n        :param str version: The custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: The custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param object body: The JSON schema of the Resource to create. (required)\n        :param str pretty: If 'true', then the output is pretty printed.\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered. (optional)\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(object, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'group',\n            'version',\n            'namespace',\n            'plural',\n            'body',\n            'pretty',\n            'dry_run',\n            'field_manager',\n            'field_validation'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method create_namespaced_custom_object\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'group' is set\n        if self.api_client.client_side_validation and ('group' not in local_var_params or  # noqa: E501\n                                                        local_var_params['group'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `group` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'version' is set\n        if self.api_client.client_side_validation and ('version' not in local_var_params or  # noqa: E501\n                                                        local_var_params['version'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `version` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'namespace' is set\n        if self.api_client.client_side_validation and ('namespace' not in local_var_params or  # noqa: E501\n                                                        local_var_params['namespace'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `namespace` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'plural' is set\n        if self.api_client.client_side_validation and ('plural' not in local_var_params or  # noqa: E501\n                                                        local_var_params['plural'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `plural` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'body' is set\n        if self.api_client.client_side_validation and ('body' not in local_var_params or  # noqa: E501\n                                                        local_var_params['body'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `body` when calling `create_namespaced_custom_object`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n        if 'group' in local_var_params:\n            path_params['group'] = local_var_params['group']  # noqa: E501\n        if 'version' in local_var_params:\n            path_params['version'] = local_var_params['version']  # noqa: E501\n        if 'namespace' in local_var_params:\n            path_params['namespace'] = local_var_params['namespace']  # noqa: E501\n        if 'plural' in local_var_params:\n            path_params['plural'] = local_var_params['plural']  # noqa: E501\n    \n        query_params = []\n        if 'pretty' in local_var_params and local_var_params['pretty'] is not None:  # noqa: E501\n            query_params.append(('pretty', local_var_params['pretty']))  # noqa: E501\n        if 'dry_run' in local_var_params and local_var_params['dry_run'] is not None:  # noqa: E501\n            query_params.append(('dryRun', local_var_params['dry_run']))  # noqa: E501\n        if 'field_manager' in local_var_params and local_var_params['field_manager'] is not None:  # noqa: E501\n            query_params.append(('fieldManager', local_var_params['field_manager']))  # noqa: E501\n        if 'field_validation' in local_var_params and local_var_params['field_validation'] is not None:  # noqa: E501\n            query_params.append(('fieldValidation', local_var_params['field_validation']))  # noqa: E501\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        if 'body' in local_var_params:\n            body_params = local_var_params['body']\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/apis/{group}/{version}/namespaces/{namespace}/{plural}', 'POST',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='object',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:354: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b42fdd0>\nresource_path = '/apis/{group}/{version}/namespaces/{namespace}/{plural}'\nmethod = 'POST'\npath_params = {'group': 'serving.kserve.io', 'namespace': 'e2e-test-llm-inference-service-6f0aa991', 'plural': 'llminferenceserviceconfigs', 'version': 'v1alpha1'}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-with-ga...outer': {'gateway': {'refs': [{'name': 'router-gateway-1', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}]}}}}\npost_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b42fdd0>\nresource_path = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-6f0aa991/llminferenceserviceconfigs'\nmethod = 'POST'\npath_params = [('group', 'serving.kserve.io'), ('version', 'v1alpha1'), ('namespace', 'e2e-test-llm-inference-service-6f0aa991'), ('plural', 'llminferenceserviceconfigs')]\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-with-ga...outer': {'gateway': {'refs': [{'name': 'router-gateway-1', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}]}}}}\npost_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b42fdd0>\nmethod = 'POST'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-6f0aa991/llminferenceserviceconfigs'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = []\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-with-ga...outer': {'gateway': {'refs': [{'name': 'router-gateway-1', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}]}}}}\n_preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n            return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n                                        headers=headers)\n        elif method == \"HEAD\":\n            return self.rest_client.HEAD(url,\n                                         query_params=query_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n                                         headers=headers)\n        elif method == \"OPTIONS\":\n            return self.rest_client.OPTIONS(url,\n                                            query_params=query_params,\n                                            headers=headers,\n                                            _preload_content=_preload_content,\n                                            _request_timeout=_request_timeout)\n        elif method == \"POST\":\n>           return self.rest_client.POST(url,\n                                         query_params=query_params,\n                                         headers=headers,\n                                         post_params=post_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:391: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253b42dd90>\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-6f0aa991/llminferenceserviceconfigs'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], post_params = []\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-with-ga...outer': {'gateway': {'refs': [{'name': 'router-gateway-1', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}]}}}}\n_preload_content = True, _request_timeout = None\n\n    def POST(self, url, headers=None, query_params=None, post_params=None,\n             body=None, _preload_content=True, _request_timeout=None):\n>       return self.request(\"POST\", url,\n                            headers=headers,\n                            query_params=query_params,\n                            post_params=post_params,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            body=body)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:279: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253b42dd90>\nmethod = 'POST'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-6f0aa991/llminferenceserviceconfigs'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-with-ga...outer': {'gateway': {'refs': [{'name': 'router-gateway-1', 'namespace': 'e2e-test-llm-inference-service-6f0aa991'}]}}}}\npost_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'application/x-www-form-urlencoded':  # noqa: E501\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=False,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'multipart/form-data':\n                    # must del headers['Content-Type'], or the correct\n                    # Content-Type which generated by urllib3 will be\n                    # overwritten.\n                    del headers['Content-Type']\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=True,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                # Pass a `string` parameter directly in the body to support\n                # other content types than Json when `body` argument is\n                # provided in serialized form\n                elif isinstance(body, str) or isinstance(body, bytes):\n                    request_body = body\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                else:\n                    # Cannot generate the request from given parameters\n                    msg = \"\"\"Cannot prepare a request message for provided\n                             arguments. Please check that your arguments match\n                             declared content type.\"\"\"\n                    raise ApiException(status=0, reason=msg)\n            # For `GET`, `HEAD`\n            else:\n                r = self.pool_manager.request(method, url,\n                                              fields=query_params,\n                                              preload_content=_preload_content,\n                                              timeout=timeout,\n                                              headers=headers)\n        except urllib3.exceptions.SSLError as e:\n            msg = \"{0}\\n{1}\".format(type(e).__name__, str(e))\n            raise ApiException(status=0, reason=msg)\n    \n        if _preload_content:\n            r = RESTResponse(r)\n    \n            # In the python 3, the response.data is bytes.\n            # we need to decode it to string.\n            if six.PY3:\n                r.data = r.data.decode('utf8')\n    \n            # log response body\n            logger.debug(\"response body: %s\", r.data)\n    \n        if not 200 <= r.status <= 299:\n>           raise ApiException(http_resp=r)\nE           kubernetes.client.exceptions.ApiException: (500)\nE           Reason: Internal Server Error\nE           HTTP response headers: HTTPHeaderDict({'Audit-Id': 'ae8ce573-f003-4686-9276-df1b4af37e51', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:17:04 GMT', 'Content-Length': '701'})\nE           HTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"Internal error occurred: failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\",\"reason\":\"InternalError\",\"details\":{\"causes\":[{\"message\":\"failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\"}]},\"code\":500}\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:238: ApiException"}, "teardown": {"duration": 0.0313226939906599, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "error", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16644821100635454, "outcome": "failed", "crash": {"path": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 238, "message": "kubernetes.client.exceptions.ApiException: (500)\nReason: Internal Server Error\nHTTP response headers: HTTPHeaderDict({'Audit-Id': 'f7923bb0-c720-428a-9a66-a87e921bfe40', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:17:05 GMT', 'Content-Length': '701'})\nHTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"Internal error occurred: failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\",\"reason\":\"InternalError\",\"details\":{\"causes\":[{\"message\":\"failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\"}]},\"code\":500}"}, "traceback": [{"path": "common/gateway_proxy_istio.py", "lineno": 183, "message": ""}, {"path": "llmisvc/fixtures.py", "lineno": 1547, "message": ""}, {"path": "llmisvc/fixtures.py", "lineno": 1509, "message": ""}, {"path": "llmisvc/fixtures.py", "lineno": 1683, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py", "lineno": 231, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py", "lineno": 354, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 348, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 180, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 391, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 279, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 238, "message": "ApiException"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253b316b10>\nllm_config = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-managed...rvice-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [{...}]}}}}}\nnamespace = 'e2e-test-llm-inference-service-079cb970'\n\n    def _create_or_update_llmisvc_config(kserve_client, llm_config, namespace=None):\n        \"\"\"Create or update an LLMInferenceServiceConfig resource.\"\"\"\n        version = llm_config[\"apiVersion\"].split(\"/\")[1]\n    \n        if namespace is None:\n            namespace = llm_config.get(\"metadata\", {}).get(\"namespace\", \"default\")\n    \n        name = llm_config.get(\"metadata\", {}).get(\"name\")\n        if not name:\n            raise ValueError(\"LLMInferenceServiceConfig must have a name in metadata\")\n    \n        logger.info(f\"Checking LLMInferenceServiceConfig {name} in namespace {namespace}\")\n    \n        try:\n>           existing_config = kserve_client.api_instance.get_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                name,\n            )\n\nllmisvc/fixtures.py:1657: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253b485590>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-079cb970'\nplural = 'llminferenceserviceconfigs'\nname = 'router-managed-llmisvc-model-fb-aee408e0'\nkwargs = {'_return_http_data_only': True}\n\n    def get_namespaced_custom_object(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: object\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1632: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253b485590>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-079cb970'\nplural = 'llminferenceserviceconfigs'\nname = 'router-managed-llmisvc-model-fb-aee408e0'\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...], 'auth_settings': ['BearerToken'], 'body_params': None, ...}\nall_params = ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {'group': 'serving.kserve.io', 'name': 'router-managed-llmisvc-model-fb-aee408e0', 'namespace': 'e2e-test-llm-inference-service-079cb970', 'plural': 'llminferenceserviceconfigs', ...}\nquery_params = []\n\n    def get_namespaced_custom_object_with_http_info(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(object, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'group',\n            'version',\n            'namespace',\n            'plural',\n            'name'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method get_namespaced_custom_object\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'group' is set\n        if self.api_client.client_side_validation and ('group' not in local_var_params or  # noqa: E501\n                                                        local_var_params['group'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `group` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'version' is set\n        if self.api_client.client_side_validation and ('version' not in local_var_params or  # noqa: E501\n                                                        local_var_params['version'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `version` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'namespace' is set\n        if self.api_client.client_side_validation and ('namespace' not in local_var_params or  # noqa: E501\n                                                        local_var_params['namespace'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `namespace` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'plural' is set\n        if self.api_client.client_side_validation and ('plural' not in local_var_params or  # noqa: E501\n                                                        local_var_params['plural'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `plural` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'name' is set\n        if self.api_client.client_side_validation and ('name' not in local_var_params or  # noqa: E501\n                                                        local_var_params['name'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `name` when calling `get_namespaced_custom_object`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n        if 'group' in local_var_params:\n            path_params['group'] = local_var_params['group']  # noqa: E501\n        if 'version' in local_var_params:\n            path_params['version'] = local_var_params['version']  # noqa: E501\n        if 'namespace' in local_var_params:\n            path_params['namespace'] = local_var_params['namespace']  # noqa: E501\n        if 'plural' in local_var_params:\n            path_params['plural'] = local_var_params['plural']  # noqa: E501\n        if 'name' in local_var_params:\n            path_params['name'] = local_var_params['name']  # noqa: E501\n    \n        query_params = []\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}', 'GET',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='object',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1739: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b484690>\nresource_path = '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}'\nmethod = 'GET'\npath_params = {'group': 'serving.kserve.io', 'name': 'router-managed-llmisvc-model-fb-aee408e0', 'namespace': 'e2e-test-llm-inference-service-079cb970', 'plural': 'llminferenceserviceconfigs', ...}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b484690>\nresource_path = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-079cb970/llminferenceserviceconfigs/router-managed-llmisvc-model-fb-aee408e0'\nmethod = 'GET'\npath_params = [('group', 'serving.kserve.io'), ('version', 'v1alpha1'), ('namespace', 'e2e-test-llm-inference-service-079cb970'), ('plural', 'llminferenceserviceconfigs'), ('name', 'router-managed-llmisvc-model-fb-aee408e0')]\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b484690>\nmethod = 'GET'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1a...namespaces/e2e-test-llm-inference-service-079cb970/llminferenceserviceconfigs/router-managed-llmisvc-model-fb-aee408e0'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = [], body = None, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n>           return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:373: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253b486510>\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1a...namespaces/e2e-test-llm-inference-service-079cb970/llminferenceserviceconfigs/router-managed-llmisvc-model-fb-aee408e0'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], _preload_content = True, _request_timeout = None\n\n    def GET(self, url, headers=None, query_params=None, _preload_content=True,\n            _request_timeout=None):\n>       return self.request(\"GET\", url,\n                            headers=headers,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            query_params=query_params)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:244: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253b486510>\nmethod = 'GET'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1a...namespaces/e2e-test-llm-inference-service-079cb970/llminferenceserviceconfigs/router-managed-llmisvc-model-fb-aee408e0'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'application/x-www-form-urlencoded':  # noqa: E501\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=False,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'multipart/form-data':\n                    # must del headers['Content-Type'], or the correct\n                    # Content-Type which generated by urllib3 will be\n                    # overwritten.\n                    del headers['Content-Type']\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=True,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                # Pass a `string` parameter directly in the body to support\n                # other content types than Json when `body` argument is\n                # provided in serialized form\n                elif isinstance(body, str) or isinstance(body, bytes):\n                    request_body = body\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                else:\n                    # Cannot generate the request from given parameters\n                    msg = \"\"\"Cannot prepare a request message for provided\n                             arguments. Please check that your arguments match\n                             declared content type.\"\"\"\n                    raise ApiException(status=0, reason=msg)\n            # For `GET`, `HEAD`\n            else:\n                r = self.pool_manager.request(method, url,\n                                              fields=query_params,\n                                              preload_content=_preload_content,\n                                              timeout=timeout,\n                                              headers=headers)\n        except urllib3.exceptions.SSLError as e:\n            msg = \"{0}\\n{1}\".format(type(e).__name__, str(e))\n            raise ApiException(status=0, reason=msg)\n    \n        if _preload_content:\n            r = RESTResponse(r)\n    \n            # In the python 3, the response.data is bytes.\n            # we need to decode it to string.\n            if six.PY3:\n                r.data = r.data.decode('utf8')\n    \n            # log response body\n            logger.debug(\"response body: %s\", r.data)\n    \n        if not 200 <= r.status <= 299:\n>           raise ApiException(http_resp=r)\nE           kubernetes.client.exceptions.ApiException: (404)\nE           Reason: Not Found\nE           HTTP response headers: HTTPHeaderDict({'Audit-Id': '1084c458-94ea-45c8-bfe6-e21a352b8a25', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:17:05 GMT', 'Content-Length': '338'})\nE           HTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"llminferenceserviceconfigs.serving.kserve.io \\\"router-managed-llmisvc-model-fb-aee408e0\\\" not found\",\"reason\":\"NotFound\",\"details\":{\"name\":\"router-managed-llmisvc-model-fb-aee408e0\",\"group\":\"serving.kserve.io\",\"kind\":\"llminferenceserviceconfigs\"},\"code\":404}\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:238: ApiException\n\nDuring handling of the above exception, another exception occurred:\n\nrequest = <SubRequest 'ensure_gateway_proxy_memory' for <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]>>\n\n    @pytest.fixture(autouse=True)\n    def ensure_gateway_proxy_memory(request):\n        \"\"\"After test setup creates gateways, patch them for proxy memory.\"\"\"\n        if not GATEWAY_PROXY_MEMORY:\n            return\n    \n        # Let test_case (llmisvc) create gateways first\n    \n        if \"test_case\" in request.fixturenames:\n>           request.getfixturevalue(\"test_case\")\n\ncommon/gateway_proxy_istio.py:183: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nrequest = <SubRequest 'test_case' for <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]>>\ntest_namespace = 'e2e-test-llm-inference-service-079cb970'\n\n    @pytest.fixture(scope=\"function\")\n    def test_case(request, test_namespace):\n        tc = request.param\n        ns = test_namespace\n    \n        inject_k8s_proxy()\n    \n        kserve_client = KServeClient(\n            config_file=os.environ.get(\"KUBECONFIG\", \"~/.kube/config\"),\n            client_configuration=client.Configuration(),\n        )\n    \n        tc.namespace = ns\n        for peer in tc.peers:\n            peer.namespace = ns\n    \n        for func in tc.before_test:\n            func(tc)\n    \n>       _setup_test_case_service(kserve_client, tc, request.node.name, namespace=ns)\n\nllmisvc/fixtures.py:1547: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253b316b10>\ntc = TestCase(base_refs=['router-managed', 'workload-single-cpu', 'model-fb-opt-125m'], prompt='KServe is a', service_name=...inference-service-079cb970', before_test=[], after_test=[], peers=[], llm_service=None, model_name='facebook/opt-125m')\ntest_node_name = 'test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]'\nnamespace = 'e2e-test-llm-inference-service-079cb970', peer_index = None\n\n    def _setup_test_case_service(\n        kserve_client, tc, test_node_name, namespace, peer_index=None\n    ):\n        \"\"\"Create LLMInferenceServiceConfigs and build the LLMInferenceService for a TestCase.\n    \n        Returns a list of created config names for cleanup tracking.\n        \"\"\"\n        missing_refs = [\n            ref for ref in tc.base_refs if ref not in LLMINFERENCESERVICE_CONFIGS\n        ]\n        if missing_refs:\n            raise ValueError(\n                f\"Missing base_refs in LLMINFERENCESERVICE_CONFIGS: {missing_refs}\"\n            )\n        if not tc.service_name:\n            suffix = f\"-peer-{peer_index}\" if peer_index is not None else \"\"\n            tc.service_name = generate_service_name(test_node_name + suffix, tc.base_refs)\n        if tc.model_name == \"default/model\":\n            tc.model_name = _get_model_name_from_configs(tc.base_refs)\n        elif \"{namespace}\" in tc.model_name:\n            tc.model_name = tc.model_name.format(namespace=namespace)\n    \n        created_configs = []\n        unique_base_refs = []\n        for base_ref in tc.base_refs:\n            unique_config_name = generate_k8s_safe_suffix(base_ref, [tc.service_name])\n            unique_base_refs.append(unique_config_name)\n    \n            config = LLMINFERENCESERVICE_CONFIGS[base_ref]\n            spec = config(namespace) if callable(config) else copy.deepcopy(config)\n    \n            unique_config_body = {\n                \"apiVersion\": \"serving.kserve.io/v1alpha1\",\n                \"kind\": \"LLMInferenceServiceConfig\",\n                \"metadata\": {\n                    \"name\": unique_config_name,\n                    \"namespace\": namespace,\n                },\n                \"spec\": spec,\n            }\n    \n>           _create_or_update_llmisvc_config(kserve_client, unique_config_body, namespace)\n\nllmisvc/fixtures.py:1509: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253b316b10>\nllm_config = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-managed...rvice-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [{...}]}}}}}\nnamespace = 'e2e-test-llm-inference-service-079cb970'\n\n    def _create_or_update_llmisvc_config(kserve_client, llm_config, namespace=None):\n        \"\"\"Create or update an LLMInferenceServiceConfig resource.\"\"\"\n        version = llm_config[\"apiVersion\"].split(\"/\")[1]\n    \n        if namespace is None:\n            namespace = llm_config.get(\"metadata\", {}).get(\"namespace\", \"default\")\n    \n        name = llm_config.get(\"metadata\", {}).get(\"name\")\n        if not name:\n            raise ValueError(\"LLMInferenceServiceConfig must have a name in metadata\")\n    \n        logger.info(f\"Checking LLMInferenceServiceConfig {name} in namespace {namespace}\")\n    \n        try:\n            existing_config = kserve_client.api_instance.get_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                name,\n            )\n    \n            llm_config[\"metadata\"] = existing_config[\"metadata\"]\n    \n            outputs = kserve_client.api_instance.replace_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                name,\n                llm_config,\n            )\n            logger.info(f\"\u2713 Successfully updated LLMInferenceServiceConfig {name}\")\n            return outputs\n    \n        except client.rest.ApiException as e:\n            if e.status == 404:  # Not found - create it\n                logger.info(\n                    f\"Resource not found, creating LLMInferenceServiceConfig {name}\"\n                )\n>               outputs = kserve_client.api_instance.create_namespaced_custom_object(\n                    constants.KSERVE_GROUP,\n                    version,\n                    namespace,\n                    KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                    llm_config,\n                )\n\nllmisvc/fixtures.py:1683: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253b485590>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-079cb970'\nplural = 'llminferenceserviceconfigs'\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-managed...rvice-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [{...}]}}}}}\nkwargs = {'_return_http_data_only': True}\n\n    def create_namespaced_custom_object(self, group, version, namespace, plural, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespaced_custom_object  # noqa: E501\n    \n        Creates a namespace scoped Custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespaced_custom_object(group, version, namespace, plural, body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: The custom resource's group name (required)\n        :param str version: The custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: The custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param object body: The JSON schema of the Resource to create. (required)\n        :param str pretty: If 'true', then the output is pretty printed.\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered. (optional)\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: object\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.create_namespaced_custom_object_with_http_info(group, version, namespace, plural, body, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:231: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253b485590>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-079cb970'\nplural = 'llminferenceserviceconfigs'\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-managed...rvice-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [{...}]}}}}}\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['group', 'version', 'namespace', 'plural', 'body', 'pretty', ...], 'au...e-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [...]}}}}}, ...}\nall_params = ['group', 'version', 'namespace', 'plural', 'body', 'pretty', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {'group': 'serving.kserve.io', 'namespace': 'e2e-test-llm-inference-service-079cb970', 'plural': 'llminferenceserviceconfigs', 'version': 'v1alpha1'}\nquery_params = []\n\n    def create_namespaced_custom_object_with_http_info(self, group, version, namespace, plural, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespaced_custom_object  # noqa: E501\n    \n        Creates a namespace scoped Custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespaced_custom_object_with_http_info(group, version, namespace, plural, body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: The custom resource's group name (required)\n        :param str version: The custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: The custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param object body: The JSON schema of the Resource to create. (required)\n        :param str pretty: If 'true', then the output is pretty printed.\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered. (optional)\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(object, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'group',\n            'version',\n            'namespace',\n            'plural',\n            'body',\n            'pretty',\n            'dry_run',\n            'field_manager',\n            'field_validation'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method create_namespaced_custom_object\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'group' is set\n        if self.api_client.client_side_validation and ('group' not in local_var_params or  # noqa: E501\n                                                        local_var_params['group'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `group` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'version' is set\n        if self.api_client.client_side_validation and ('version' not in local_var_params or  # noqa: E501\n                                                        local_var_params['version'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `version` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'namespace' is set\n        if self.api_client.client_side_validation and ('namespace' not in local_var_params or  # noqa: E501\n                                                        local_var_params['namespace'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `namespace` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'plural' is set\n        if self.api_client.client_side_validation and ('plural' not in local_var_params or  # noqa: E501\n                                                        local_var_params['plural'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `plural` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'body' is set\n        if self.api_client.client_side_validation and ('body' not in local_var_params or  # noqa: E501\n                                                        local_var_params['body'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `body` when calling `create_namespaced_custom_object`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n        if 'group' in local_var_params:\n            path_params['group'] = local_var_params['group']  # noqa: E501\n        if 'version' in local_var_params:\n            path_params['version'] = local_var_params['version']  # noqa: E501\n        if 'namespace' in local_var_params:\n            path_params['namespace'] = local_var_params['namespace']  # noqa: E501\n        if 'plural' in local_var_params:\n            path_params['plural'] = local_var_params['plural']  # noqa: E501\n    \n        query_params = []\n        if 'pretty' in local_var_params and local_var_params['pretty'] is not None:  # noqa: E501\n            query_params.append(('pretty', local_var_params['pretty']))  # noqa: E501\n        if 'dry_run' in local_var_params and local_var_params['dry_run'] is not None:  # noqa: E501\n            query_params.append(('dryRun', local_var_params['dry_run']))  # noqa: E501\n        if 'field_manager' in local_var_params and local_var_params['field_manager'] is not None:  # noqa: E501\n            query_params.append(('fieldManager', local_var_params['field_manager']))  # noqa: E501\n        if 'field_validation' in local_var_params and local_var_params['field_validation'] is not None:  # noqa: E501\n            query_params.append(('fieldValidation', local_var_params['field_validation']))  # noqa: E501\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        if 'body' in local_var_params:\n            body_params = local_var_params['body']\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/apis/{group}/{version}/namespaces/{namespace}/{plural}', 'POST',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='object',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:354: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b484690>\nresource_path = '/apis/{group}/{version}/namespaces/{namespace}/{plural}'\nmethod = 'POST'\npath_params = {'group': 'serving.kserve.io', 'namespace': 'e2e-test-llm-inference-service-079cb970', 'plural': 'llminferenceserviceconfigs', 'version': 'v1alpha1'}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-managed...rvice-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [{...}]}}}}}\npost_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b484690>\nresource_path = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-079cb970/llminferenceserviceconfigs'\nmethod = 'POST'\npath_params = [('group', 'serving.kserve.io'), ('version', 'v1alpha1'), ('namespace', 'e2e-test-llm-inference-service-079cb970'), ('plural', 'llminferenceserviceconfigs')]\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-managed...rvice-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [{...}]}}}}}\npost_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253b484690>\nmethod = 'POST'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-079cb970/llminferenceserviceconfigs'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = []\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-managed...rvice-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [{...}]}}}}}\n_preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n            return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n                                        headers=headers)\n        elif method == \"HEAD\":\n            return self.rest_client.HEAD(url,\n                                         query_params=query_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n                                         headers=headers)\n        elif method == \"OPTIONS\":\n            return self.rest_client.OPTIONS(url,\n                                            query_params=query_params,\n                                            headers=headers,\n                                            _preload_content=_preload_content,\n                                            _request_timeout=_request_timeout)\n        elif method == \"POST\":\n>           return self.rest_client.POST(url,\n                                         query_params=query_params,\n                                         headers=headers,\n                                         post_params=post_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:391: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253b486510>\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-079cb970/llminferenceserviceconfigs'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], post_params = []\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-managed...rvice-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [{...}]}}}}}\n_preload_content = True, _request_timeout = None\n\n    def POST(self, url, headers=None, query_params=None, post_params=None,\n             body=None, _preload_content=True, _request_timeout=None):\n>       return self.request(\"POST\", url,\n                            headers=headers,\n                            query_params=query_params,\n                            post_params=post_params,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            body=body)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:279: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253b486510>\nmethod = 'POST'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-079cb970/llminferenceserviceconfigs'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-managed...rvice-079cb970'}, 'spec': {'router': {'gateway': {}, 'route': {}, 'scheduler': {'template': {'containers': [{...}]}}}}}\npost_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'application/x-www-form-urlencoded':  # noqa: E501\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=False,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'multipart/form-data':\n                    # must del headers['Content-Type'], or the correct\n                    # Content-Type which generated by urllib3 will be\n                    # overwritten.\n                    del headers['Content-Type']\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=True,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                # Pass a `string` parameter directly in the body to support\n                # other content types than Json when `body` argument is\n                # provided in serialized form\n                elif isinstance(body, str) or isinstance(body, bytes):\n                    request_body = body\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                else:\n                    # Cannot generate the request from given parameters\n                    msg = \"\"\"Cannot prepare a request message for provided\n                             arguments. Please check that your arguments match\n                             declared content type.\"\"\"\n                    raise ApiException(status=0, reason=msg)\n            # For `GET`, `HEAD`\n            else:\n                r = self.pool_manager.request(method, url,\n                                              fields=query_params,\n                                              preload_content=_preload_content,\n                                              timeout=timeout,\n                                              headers=headers)\n        except urllib3.exceptions.SSLError as e:\n            msg = \"{0}\\n{1}\".format(type(e).__name__, str(e))\n            raise ApiException(status=0, reason=msg)\n    \n        if _preload_content:\n            r = RESTResponse(r)\n    \n            # In the python 3, the response.data is bytes.\n            # we need to decode it to string.\n            if six.PY3:\n                r.data = r.data.decode('utf8')\n    \n            # log response body\n            logger.debug(\"response body: %s\", r.data)\n    \n        if not 200 <= r.status <= 299:\n>           raise ApiException(http_resp=r)\nE           kubernetes.client.exceptions.ApiException: (500)\nE           Reason: Internal Server Error\nE           HTTP response headers: HTTPHeaderDict({'Audit-Id': 'f7923bb0-c720-428a-9a66-a87e921bfe40', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:17:05 GMT', 'Content-Length': '701'})\nE           HTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"Internal error occurred: failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\",\"reason\":\"InternalError\",\"details\":{\"causes\":[{\"message\":\"failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\"}]},\"code\":500}\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:238: ApiException"}, "teardown": {"duration": 0.022536486998433247, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "error", "keywords": ["test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16409869500785135, "outcome": "failed", "crash": {"path": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 238, "message": "kubernetes.client.exceptions.ApiException: (500)\nReason: Internal Server Error\nHTTP response headers: HTTPHeaderDict({'Audit-Id': '946260ef-a27a-454c-b45e-9cd5789f52d0', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:17:05 GMT', 'Content-Length': '701'})\nHTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"Internal error occurred: failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\",\"reason\":\"InternalError\",\"details\":{\"causes\":[{\"message\":\"failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\"}]},\"code\":500}"}, "traceback": [{"path": "common/gateway_proxy_istio.py", "lineno": 183, "message": ""}, {"path": "llmisvc/fixtures.py", "lineno": 1547, "message": ""}, {"path": "llmisvc/fixtures.py", "lineno": 1509, "message": ""}, {"path": "llmisvc/fixtures.py", "lineno": 1683, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py", "lineno": 231, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py", "lineno": 354, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 348, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 180, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 391, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 279, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 238, "message": "ApiException"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253a697590>\nllm_config = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-custom-...t-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {'rules': [...]}}}}}}\nnamespace = 'e2e-test-llm-inference-service-caba0b4a'\n\n    def _create_or_update_llmisvc_config(kserve_client, llm_config, namespace=None):\n        \"\"\"Create or update an LLMInferenceServiceConfig resource.\"\"\"\n        version = llm_config[\"apiVersion\"].split(\"/\")[1]\n    \n        if namespace is None:\n            namespace = llm_config.get(\"metadata\", {}).get(\"namespace\", \"default\")\n    \n        name = llm_config.get(\"metadata\", {}).get(\"name\")\n        if not name:\n            raise ValueError(\"LLMInferenceServiceConfig must have a name in metadata\")\n    \n        logger.info(f\"Checking LLMInferenceServiceConfig {name} in namespace {namespace}\")\n    \n        try:\n>           existing_config = kserve_client.api_instance.get_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                name,\n            )\n\nllmisvc/fixtures.py:1657: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253abb14d0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-caba0b4a'\nplural = 'llminferenceserviceconfigs'\nname = 'router-custom-route-timeout-cus-3c5d4892'\nkwargs = {'_return_http_data_only': True}\n\n    def get_namespaced_custom_object(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: object\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1632: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253abb14d0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-caba0b4a'\nplural = 'llminferenceserviceconfigs'\nname = 'router-custom-route-timeout-cus-3c5d4892'\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...], 'auth_settings': ['BearerToken'], 'body_params': None, ...}\nall_params = ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {'group': 'serving.kserve.io', 'name': 'router-custom-route-timeout-cus-3c5d4892', 'namespace': 'e2e-test-llm-inference-service-caba0b4a', 'plural': 'llminferenceserviceconfigs', ...}\nquery_params = []\n\n    def get_namespaced_custom_object_with_http_info(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(object, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'group',\n            'version',\n            'namespace',\n            'plural',\n            'name'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method get_namespaced_custom_object\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'group' is set\n        if self.api_client.client_side_validation and ('group' not in local_var_params or  # noqa: E501\n                                                        local_var_params['group'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `group` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'version' is set\n        if self.api_client.client_side_validation and ('version' not in local_var_params or  # noqa: E501\n                                                        local_var_params['version'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `version` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'namespace' is set\n        if self.api_client.client_side_validation and ('namespace' not in local_var_params or  # noqa: E501\n                                                        local_var_params['namespace'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `namespace` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'plural' is set\n        if self.api_client.client_side_validation and ('plural' not in local_var_params or  # noqa: E501\n                                                        local_var_params['plural'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `plural` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'name' is set\n        if self.api_client.client_side_validation and ('name' not in local_var_params or  # noqa: E501\n                                                        local_var_params['name'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `name` when calling `get_namespaced_custom_object`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n        if 'group' in local_var_params:\n            path_params['group'] = local_var_params['group']  # noqa: E501\n        if 'version' in local_var_params:\n            path_params['version'] = local_var_params['version']  # noqa: E501\n        if 'namespace' in local_var_params:\n            path_params['namespace'] = local_var_params['namespace']  # noqa: E501\n        if 'plural' in local_var_params:\n            path_params['plural'] = local_var_params['plural']  # noqa: E501\n        if 'name' in local_var_params:\n            path_params['name'] = local_var_params['name']  # noqa: E501\n    \n        query_params = []\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}', 'GET',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='object',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1739: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253a74b3d0>\nresource_path = '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}'\nmethod = 'GET'\npath_params = {'group': 'serving.kserve.io', 'name': 'router-custom-route-timeout-cus-3c5d4892', 'namespace': 'e2e-test-llm-inference-service-caba0b4a', 'plural': 'llminferenceserviceconfigs', ...}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253a74b3d0>\nresource_path = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-caba0b4a/llminferenceserviceconfigs/router-custom-route-timeout-cus-3c5d4892'\nmethod = 'GET'\npath_params = [('group', 'serving.kserve.io'), ('version', 'v1alpha1'), ('namespace', 'e2e-test-llm-inference-service-caba0b4a'), ('plural', 'llminferenceserviceconfigs'), ('name', 'router-custom-route-timeout-cus-3c5d4892')]\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253a74b3d0>\nmethod = 'GET'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1a...namespaces/e2e-test-llm-inference-service-caba0b4a/llminferenceserviceconfigs/router-custom-route-timeout-cus-3c5d4892'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = [], body = None, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n>           return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:373: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253a749390>\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1a...namespaces/e2e-test-llm-inference-service-caba0b4a/llminferenceserviceconfigs/router-custom-route-timeout-cus-3c5d4892'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], _preload_content = True, _request_timeout = None\n\n    def GET(self, url, headers=None, query_params=None, _preload_content=True,\n            _request_timeout=None):\n>       return self.request(\"GET\", url,\n                            headers=headers,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            query_params=query_params)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:244: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253a749390>\nmethod = 'GET'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1a...namespaces/e2e-test-llm-inference-service-caba0b4a/llminferenceserviceconfigs/router-custom-route-timeout-cus-3c5d4892'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'application/x-www-form-urlencoded':  # noqa: E501\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=False,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'multipart/form-data':\n                    # must del headers['Content-Type'], or the correct\n                    # Content-Type which generated by urllib3 will be\n                    # overwritten.\n                    del headers['Content-Type']\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=True,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                # Pass a `string` parameter directly in the body to support\n                # other content types than Json when `body` argument is\n                # provided in serialized form\n                elif isinstance(body, str) or isinstance(body, bytes):\n                    request_body = body\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                else:\n                    # Cannot generate the request from given parameters\n                    msg = \"\"\"Cannot prepare a request message for provided\n                             arguments. Please check that your arguments match\n                             declared content type.\"\"\"\n                    raise ApiException(status=0, reason=msg)\n            # For `GET`, `HEAD`\n            else:\n                r = self.pool_manager.request(method, url,\n                                              fields=query_params,\n                                              preload_content=_preload_content,\n                                              timeout=timeout,\n                                              headers=headers)\n        except urllib3.exceptions.SSLError as e:\n            msg = \"{0}\\n{1}\".format(type(e).__name__, str(e))\n            raise ApiException(status=0, reason=msg)\n    \n        if _preload_content:\n            r = RESTResponse(r)\n    \n            # In the python 3, the response.data is bytes.\n            # we need to decode it to string.\n            if six.PY3:\n                r.data = r.data.decode('utf8')\n    \n            # log response body\n            logger.debug(\"response body: %s\", r.data)\n    \n        if not 200 <= r.status <= 299:\n>           raise ApiException(http_resp=r)\nE           kubernetes.client.exceptions.ApiException: (404)\nE           Reason: Not Found\nE           HTTP response headers: HTTPHeaderDict({'Audit-Id': 'ce555b46-67ea-479c-9eaf-96b17e96e67a', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:17:05 GMT', 'Content-Length': '338'})\nE           HTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"llminferenceserviceconfigs.serving.kserve.io \\\"router-custom-route-timeout-cus-3c5d4892\\\" not found\",\"reason\":\"NotFound\",\"details\":{\"name\":\"router-custom-route-timeout-cus-3c5d4892\",\"group\":\"serving.kserve.io\",\"kind\":\"llminferenceserviceconfigs\"},\"code\":404}\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:238: ApiException\n\nDuring handling of the above exception, another exception occurred:\n\nrequest = <SubRequest 'ensure_gateway_proxy_memory' for <Function test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]>>\n\n    @pytest.fixture(autouse=True)\n    def ensure_gateway_proxy_memory(request):\n        \"\"\"After test setup creates gateways, patch them for proxy memory.\"\"\"\n        if not GATEWAY_PROXY_MEMORY:\n            return\n    \n        # Let test_case (llmisvc) create gateways first\n    \n        if \"test_case\" in request.fixturenames:\n>           request.getfixturevalue(\"test_case\")\n\ncommon/gateway_proxy_istio.py:183: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nrequest = <SubRequest 'test_case' for <Function test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]>>\ntest_namespace = 'e2e-test-llm-inference-service-caba0b4a'\n\n    @pytest.fixture(scope=\"function\")\n    def test_case(request, test_namespace):\n        tc = request.param\n        ns = test_namespace\n    \n        inject_k8s_proxy()\n    \n        kserve_client = KServeClient(\n            config_file=os.environ.get(\"KUBECONFIG\", \"~/.kube/config\"),\n            client_configuration=client.Configuration(),\n        )\n    \n        tc.namespace = ns\n        for peer in tc.peers:\n            peer.namespace = ns\n    \n        for func in tc.before_test:\n            func(tc)\n    \n>       _setup_test_case_service(kserve_client, tc, request.node.name, namespace=ns)\n\nllmisvc/fixtures.py:1547: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253a697590>\ntc = TestCase(base_refs=['router-custom-route-timeout', 'scheduler-managed', 'workload-single-cpu', 'model-fb-opt-125m'], p...inference-service-caba0b4a', before_test=[], after_test=[], peers=[], llm_service=None, model_name='facebook/opt-125m')\ntest_node_name = 'test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]'\nnamespace = 'e2e-test-llm-inference-service-caba0b4a', peer_index = None\n\n    def _setup_test_case_service(\n        kserve_client, tc, test_node_name, namespace, peer_index=None\n    ):\n        \"\"\"Create LLMInferenceServiceConfigs and build the LLMInferenceService for a TestCase.\n    \n        Returns a list of created config names for cleanup tracking.\n        \"\"\"\n        missing_refs = [\n            ref for ref in tc.base_refs if ref not in LLMINFERENCESERVICE_CONFIGS\n        ]\n        if missing_refs:\n            raise ValueError(\n                f\"Missing base_refs in LLMINFERENCESERVICE_CONFIGS: {missing_refs}\"\n            )\n        if not tc.service_name:\n            suffix = f\"-peer-{peer_index}\" if peer_index is not None else \"\"\n            tc.service_name = generate_service_name(test_node_name + suffix, tc.base_refs)\n        if tc.model_name == \"default/model\":\n            tc.model_name = _get_model_name_from_configs(tc.base_refs)\n        elif \"{namespace}\" in tc.model_name:\n            tc.model_name = tc.model_name.format(namespace=namespace)\n    \n        created_configs = []\n        unique_base_refs = []\n        for base_ref in tc.base_refs:\n            unique_config_name = generate_k8s_safe_suffix(base_ref, [tc.service_name])\n            unique_base_refs.append(unique_config_name)\n    \n            config = LLMINFERENCESERVICE_CONFIGS[base_ref]\n            spec = config(namespace) if callable(config) else copy.deepcopy(config)\n    \n            unique_config_body = {\n                \"apiVersion\": \"serving.kserve.io/v1alpha1\",\n                \"kind\": \"LLMInferenceServiceConfig\",\n                \"metadata\": {\n                    \"name\": unique_config_name,\n                    \"namespace\": namespace,\n                },\n                \"spec\": spec,\n            }\n    \n>           _create_or_update_llmisvc_config(kserve_client, unique_config_body, namespace)\n\nllmisvc/fixtures.py:1509: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f253a697590>\nllm_config = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-custom-...t-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {'rules': [...]}}}}}}\nnamespace = 'e2e-test-llm-inference-service-caba0b4a'\n\n    def _create_or_update_llmisvc_config(kserve_client, llm_config, namespace=None):\n        \"\"\"Create or update an LLMInferenceServiceConfig resource.\"\"\"\n        version = llm_config[\"apiVersion\"].split(\"/\")[1]\n    \n        if namespace is None:\n            namespace = llm_config.get(\"metadata\", {}).get(\"namespace\", \"default\")\n    \n        name = llm_config.get(\"metadata\", {}).get(\"name\")\n        if not name:\n            raise ValueError(\"LLMInferenceServiceConfig must have a name in metadata\")\n    \n        logger.info(f\"Checking LLMInferenceServiceConfig {name} in namespace {namespace}\")\n    \n        try:\n            existing_config = kserve_client.api_instance.get_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                name,\n            )\n    \n            llm_config[\"metadata\"] = existing_config[\"metadata\"]\n    \n            outputs = kserve_client.api_instance.replace_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                name,\n                llm_config,\n            )\n            logger.info(f\"\u2713 Successfully updated LLMInferenceServiceConfig {name}\")\n            return outputs\n    \n        except client.rest.ApiException as e:\n            if e.status == 404:  # Not found - create it\n                logger.info(\n                    f\"Resource not found, creating LLMInferenceServiceConfig {name}\"\n                )\n>               outputs = kserve_client.api_instance.create_namespaced_custom_object(\n                    constants.KSERVE_GROUP,\n                    version,\n                    namespace,\n                    KSERVE_PLURAL_LLMINFERENCESERVICECONFIG,\n                    llm_config,\n                )\n\nllmisvc/fixtures.py:1683: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253abb14d0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-caba0b4a'\nplural = 'llminferenceserviceconfigs'\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-custom-...t-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {'rules': [...]}}}}}}\nkwargs = {'_return_http_data_only': True}\n\n    def create_namespaced_custom_object(self, group, version, namespace, plural, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespaced_custom_object  # noqa: E501\n    \n        Creates a namespace scoped Custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespaced_custom_object(group, version, namespace, plural, body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: The custom resource's group name (required)\n        :param str version: The custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: The custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param object body: The JSON schema of the Resource to create. (required)\n        :param str pretty: If 'true', then the output is pretty printed.\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered. (optional)\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: object\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.create_namespaced_custom_object_with_http_info(group, version, namespace, plural, body, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:231: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f253abb14d0>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-caba0b4a'\nplural = 'llminferenceserviceconfigs'\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-custom-...t-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {'rules': [...]}}}}}}\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['group', 'version', 'namespace', 'plural', 'body', 'pretty', ...], 'au...e-test-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {...}}}}}}, ...}\nall_params = ['group', 'version', 'namespace', 'plural', 'body', 'pretty', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {'group': 'serving.kserve.io', 'namespace': 'e2e-test-llm-inference-service-caba0b4a', 'plural': 'llminferenceserviceconfigs', 'version': 'v1alpha1'}\nquery_params = []\n\n    def create_namespaced_custom_object_with_http_info(self, group, version, namespace, plural, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespaced_custom_object  # noqa: E501\n    \n        Creates a namespace scoped Custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespaced_custom_object_with_http_info(group, version, namespace, plural, body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: The custom resource's group name (required)\n        :param str version: The custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: The custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param object body: The JSON schema of the Resource to create. (required)\n        :param str pretty: If 'true', then the output is pretty printed.\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered. (optional)\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(object, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'group',\n            'version',\n            'namespace',\n            'plural',\n            'body',\n            'pretty',\n            'dry_run',\n            'field_manager',\n            'field_validation'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method create_namespaced_custom_object\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'group' is set\n        if self.api_client.client_side_validation and ('group' not in local_var_params or  # noqa: E501\n                                                        local_var_params['group'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `group` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'version' is set\n        if self.api_client.client_side_validation and ('version' not in local_var_params or  # noqa: E501\n                                                        local_var_params['version'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `version` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'namespace' is set\n        if self.api_client.client_side_validation and ('namespace' not in local_var_params or  # noqa: E501\n                                                        local_var_params['namespace'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `namespace` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'plural' is set\n        if self.api_client.client_side_validation and ('plural' not in local_var_params or  # noqa: E501\n                                                        local_var_params['plural'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `plural` when calling `create_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'body' is set\n        if self.api_client.client_side_validation and ('body' not in local_var_params or  # noqa: E501\n                                                        local_var_params['body'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `body` when calling `create_namespaced_custom_object`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n        if 'group' in local_var_params:\n            path_params['group'] = local_var_params['group']  # noqa: E501\n        if 'version' in local_var_params:\n            path_params['version'] = local_var_params['version']  # noqa: E501\n        if 'namespace' in local_var_params:\n            path_params['namespace'] = local_var_params['namespace']  # noqa: E501\n        if 'plural' in local_var_params:\n            path_params['plural'] = local_var_params['plural']  # noqa: E501\n    \n        query_params = []\n        if 'pretty' in local_var_params and local_var_params['pretty'] is not None:  # noqa: E501\n            query_params.append(('pretty', local_var_params['pretty']))  # noqa: E501\n        if 'dry_run' in local_var_params and local_var_params['dry_run'] is not None:  # noqa: E501\n            query_params.append(('dryRun', local_var_params['dry_run']))  # noqa: E501\n        if 'field_manager' in local_var_params and local_var_params['field_manager'] is not None:  # noqa: E501\n            query_params.append(('fieldManager', local_var_params['field_manager']))  # noqa: E501\n        if 'field_validation' in local_var_params and local_var_params['field_validation'] is not None:  # noqa: E501\n            query_params.append(('fieldValidation', local_var_params['field_validation']))  # noqa: E501\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        if 'body' in local_var_params:\n            body_params = local_var_params['body']\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/apis/{group}/{version}/namespaces/{namespace}/{plural}', 'POST',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='object',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:354: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253a74b3d0>\nresource_path = '/apis/{group}/{version}/namespaces/{namespace}/{plural}'\nmethod = 'POST'\npath_params = {'group': 'serving.kserve.io', 'namespace': 'e2e-test-llm-inference-service-caba0b4a', 'plural': 'llminferenceserviceconfigs', 'version': 'v1alpha1'}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-custom-...t-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {'rules': [...]}}}}}}\npost_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253a74b3d0>\nresource_path = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-caba0b4a/llminferenceserviceconfigs'\nmethod = 'POST'\npath_params = [('group', 'serving.kserve.io'), ('version', 'v1alpha1'), ('namespace', 'e2e-test-llm-inference-service-caba0b4a'), ('plural', 'llminferenceserviceconfigs')]\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-custom-...t-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {'rules': [...]}}}}}}\npost_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f253a74b3d0>\nmethod = 'POST'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-caba0b4a/llminferenceserviceconfigs'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = []\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-custom-...t-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {'rules': [...]}}}}}}\n_preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n            return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n                                        headers=headers)\n        elif method == \"HEAD\":\n            return self.rest_client.HEAD(url,\n                                         query_params=query_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n                                         headers=headers)\n        elif method == \"OPTIONS\":\n            return self.rest_client.OPTIONS(url,\n                                            query_params=query_params,\n                                            headers=headers,\n                                            _preload_content=_preload_content,\n                                            _request_timeout=_request_timeout)\n        elif method == \"POST\":\n>           return self.rest_client.POST(url,\n                                         query_params=query_params,\n                                         headers=headers,\n                                         post_params=post_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:391: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253a749390>\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-caba0b4a/llminferenceserviceconfigs'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], post_params = []\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-custom-...t-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {'rules': [...]}}}}}}\n_preload_content = True, _request_timeout = None\n\n    def POST(self, url, headers=None, query_params=None, post_params=None,\n             body=None, _preload_content=True, _request_timeout=None):\n>       return self.request(\"POST\", url,\n                            headers=headers,\n                            query_params=query_params,\n                            post_params=post_params,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            body=body)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:279: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f253a749390>\nmethod = 'POST'\nurl = 'https://a418b12a6f56347aaa0711817377ef1a-d45d7be9e6592c03.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-caba0b4a/llminferenceserviceconfigs'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'apiVersion': 'serving.kserve.io/v1alpha1', 'kind': 'LLMInferenceServiceConfig', 'metadata': {'name': 'router-custom-...t-llm-inference-service-caba0b4a'}, 'spec': {'router': {'gateway': {}, 'route': {'http': {'spec': {'rules': [...]}}}}}}\npost_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'application/x-www-form-urlencoded':  # noqa: E501\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=False,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'multipart/form-data':\n                    # must del headers['Content-Type'], or the correct\n                    # Content-Type which generated by urllib3 will be\n                    # overwritten.\n                    del headers['Content-Type']\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=True,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                # Pass a `string` parameter directly in the body to support\n                # other content types than Json when `body` argument is\n                # provided in serialized form\n                elif isinstance(body, str) or isinstance(body, bytes):\n                    request_body = body\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                else:\n                    # Cannot generate the request from given parameters\n                    msg = \"\"\"Cannot prepare a request message for provided\n                             arguments. Please check that your arguments match\n                             declared content type.\"\"\"\n                    raise ApiException(status=0, reason=msg)\n            # For `GET`, `HEAD`\n            else:\n                r = self.pool_manager.request(method, url,\n                                              fields=query_params,\n                                              preload_content=_preload_content,\n                                              timeout=timeout,\n                                              headers=headers)\n        except urllib3.exceptions.SSLError as e:\n            msg = \"{0}\\n{1}\".format(type(e).__name__, str(e))\n            raise ApiException(status=0, reason=msg)\n    \n        if _preload_content:\n            r = RESTResponse(r)\n    \n            # In the python 3, the response.data is bytes.\n            # we need to decode it to string.\n            if six.PY3:\n                r.data = r.data.decode('utf8')\n    \n            # log response body\n            logger.debug(\"response body: %s\", r.data)\n    \n        if not 200 <= r.status <= 299:\n>           raise ApiException(http_resp=r)\nE           kubernetes.client.exceptions.ApiException: (500)\nE           Reason: Internal Server Error\nE           HTTP response headers: HTTPHeaderDict({'Audit-Id': '946260ef-a27a-454c-b45e-9cd5789f52d0', 'Cache-Control': 'no-cache, private', 'Content-Type': 'application/json', 'Strict-Transport-Security': 'max-age=31536000; includeSubDomains; preload', 'X-Kubernetes-Pf-Flowschema-Uid': 'd7b39bf7-9eb1-4272-9a71-ca6aab602d13', 'X-Kubernetes-Pf-Prioritylevel-Uid': 'd1a12026-da65-40c9-ab91-a3a487e7d758', 'Date': 'Tue, 21 Jul 2026 19:17:05 GMT', 'Content-Length': '701'})\nE           HTTP response body: {\"kind\":\"Status\",\"apiVersion\":\"v1\",\"metadata\":{},\"status\":\"Failure\",\"message\":\"Internal error occurred: failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\",\"reason\":\"InternalError\",\"details\":{\"causes\":[{\"message\":\"failed calling webhook \\\"llminferenceserviceconfig.kserve-webhook-server.v1alpha1.validator\\\": failed to call webhook: Post \\\"https://llmisvc-webhook-server-service.kserve.svc:443/validate-serving-kserve-io-v1alpha1-llminferenceserviceconfig?timeout=10s\\\": EOF\"}]},\"code\":500}\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:238: ApiException"}, "teardown": {"duration": 0.029108541988534853, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}], "warnings": [{"message": "The event_loop fixture provided by pytest-asyncio has been redefined in\n/workspace/source/test/e2e/conftest.py:43\nReplacing the event_loop fixture with a custom implementation is deprecated\nand will lead to errors in the future.\nIf you want to request an asyncio event loop with a scope other than function\nscope, use the \"scope\" argument to the asyncio mark when marking the tests.\nIf you want to return different types of event loops, use the event_loop_policy\nfixture.\n", "category": "DeprecationWarning", "when": "runtest", "filename": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/pytest_asyncio/plugin.py", "lineno": 761}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-custom-template-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The event_loop fixture provided by pytest-asyncio has been redefined in\n/workspace/source/test/e2e/conftest.py:43\nReplacing the event_loop fixture with a custom implementation is deprecated\nand will lead to errors in the future.\nIf you want to request an asyncio event loop with a scope other than function\nscope, use the \"scope\" argument to the asyncio mark when marking the tests.\nIf you want to return different types of event loops, use the event_loop_policy\nfixture.\n", "category": "DeprecationWarning", "when": "runtest", "filename": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/pytest_asyncio/plugin.py", "lineno": 761}, {"message": "The test <Function test_flow_control_smoke[flow-control-utilization-detector]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_flow_control.py", "lineno": 47}, {"message": "The test <Function test_flow_control_smoke[flow-control-concurrency-detector]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_flow_control.py", "lineno": 47}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator0]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator1]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator2]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-pd-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_stop_feature[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service_stop.py", "lineno": 40}, {"message": "The test <Function test_llm_tls_resources[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_tls.py", "lineno": 92}, {"message": "The test <Function test_llm_inference_service[router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-pd-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-no-scheduler-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-inline-config-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-replicas-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}]}