{"created": 1785438589.3261333, "duration": 6982.877428293228, "exitcode": 2, "root": "/workspace/source/test/e2e", "environment": {}, "summary": {"passed": 52, "failed": 7, "error": 3, "total": 62, "collected": 73}, "collectors": [{"nodeid": "explainer/test_art_explainer.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/explainer/test_art_explainer.py', 48, 'Skipped: ODH does not support art explainer at the moment')"}, {"nodeid": "predictor/test_grpc.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/predictor/test_grpc.py', 39, 'Skipped: Not testable in ODH at the moment')"}, {"nodeid": "predictor/test_torchserve.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/predictor/test_torchserve.py', 34, 'Skipped: ODH does not support torchserve at the moment')"}], "tests": [{"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-no-scheduler-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-no-scheduler-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "no_scheduler", "__wrapped__", "pytestmark", "router-no-scheduler-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.4354416869900888, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 203.24956921700505, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03138154999760445, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_flow_control.py::test_flow_control_smoke[cluster_cpu-cluster_single_node-flow-control-utilization-detector]", "lineno": 46, "outcome": "passed", "keywords": ["test_flow_control_smoke[flow-control-utilization-detector]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "flow_control", "pytestmark", "flow-control-utilization-detector", "llminferenceservice", "llmisvc_core", "test_flow_control.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.5372299979935633, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 109.0082651309931, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03816451599413995, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_flow_control.py::test_flow_control_smoke[cluster_cpu-cluster_single_node-flow-control-concurrency-detector]", "lineno": 46, "outcome": "passed", "keywords": ["test_flow_control_smoke[flow-control-concurrency-detector]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "flow_control", "pytestmark", "flow-control-concurrency-detector", "llminferenceservice", "llmisvc_core", "test_flow_control.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.32666339199931826, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 60.552706110000145, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0404390760086244, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_gateway_section_name.py::test_gateway_section_name_propagation[cluster_single_node-cluster_cpu-with-section-name]", "lineno": 131, "outcome": "passed", "keywords": ["test_gateway_section_name_propagation[with-section-name]", "parametrize", "llmd_simulator", "cluster_single_node", "cluster_cpu", "pytestmark", "with-section-name", "llminferenceservice", "llmisvc_core", "test_gateway_section_name.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15231062800739892, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 12.220785931000137, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03493137100304011, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_gateway_section_name.py::test_gateway_section_name_propagation[cluster_single_node-cluster_cpu-without-section-name]", "lineno": 131, "outcome": "passed", "keywords": ["test_gateway_section_name_propagation[without-section-name]", "parametrize", "llmd_simulator", "cluster_single_node", "cluster_cpu", "pytestmark", "without-section-name", "llminferenceservice", "llmisvc_core", "test_gateway_section_name.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14327879900520202, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 17.810131108999485, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03381769001134671, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_enabled_requires_token[cluster_cpu-cluster_single_node-auth-enabled-default]", "lineno": 221, "outcome": "passed", "keywords": ["test_llm_auth_enabled_requires_token[auth-enabled-default]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-enabled-default", "llmisvc_core", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.620671171011054, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 180.29552913599764, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03703664199565537, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_multi_node-router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_multi_node", "__wrapped__", "pytestmark", "router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.48353932600002736, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 212.4256221120013, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.039594168003532104, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_invalid_token_rejected[cluster_cpu-cluster_single_node-auth-invalid-token]", "lineno": 386, "outcome": "passed", "keywords": ["test_llm_auth_invalid_token_rejected[auth-invalid-token]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-invalid-token", "llmisvc_core", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.4524311889981618, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 153.11440580700582, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03225067300081719, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-inline-config-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-inline-config-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-inline-config-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.5433704820025014, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 72.39487070300675, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04173647999414243, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-model-qwen2.5-0.5b", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.279114707998815, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 56.434232962012175, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03570480798953213, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_disabled_no_token_required[cluster_cpu-cluster_single_node-auth-disabled]", "lineno": 523, "outcome": "passed", "keywords": ["test_llm_auth_disabled_no_token_required[auth-disabled]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-disabled", "llmisvc_core", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2645888469996862, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 150.51132763799978, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.036595262004993856, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-configmap-ref-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.6999557570088655, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 67.14051767199999, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.05647370401129592, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-replicas-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-replicas-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-replicas-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.4225935939903138, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 60.071266205995926, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.037157012993702665, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-custom-template-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-custom-template-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-custom-template-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.28050981399428565, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 59.25698345899582, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.044388037000317127, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_hpa_deployment[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "lineno": 507, "outcome": "failed", "keywords": ["test_llm_autoscaling_hpa_deployment[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "parametrize", "autoscaling_hpa", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 1.042825343000004, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 920.8543529370072, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T17:25:21Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T17:25:21Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T17:25:21Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:25:08Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T17:25:39Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T17:25:39Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T17:25:22Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T17:25:39Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:25:22Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 542, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 482, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_hpa\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-hpa\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-hpa-deploy\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_hpa_deployment(test_case: TestCase):\n        \"\"\"HPA + Deployment: HPA exists with WVA annotations; pods scale up under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:542: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24da7247d0>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:482: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f24da7247d0>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-hpa-4c186bcf'},\n                       {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-30T17:24:56.360388', start_time = 1785432296.3606734\nduration = 900.5952887535095, timestamp_end = '2026-07-30T17:39:56.955973'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24da7247d0>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-hpa-4c186bcf'},\n                       {'name': 'scaling-hpa-autoscale-hpa-deplo-347a3180'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f24da72e3e0>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T17:25:21Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T17:25:21Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T17:25:21Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:25:08Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T17:25:39Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T17:25:39Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T17:25:22Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T17:25:39Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:25:22Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-deployment-9ba6f3f4/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-deploy-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.0026009969878941774, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3020293580048019, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 180.66623990000517, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.023995034003746696, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3877332830015803, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 82.97180070600007, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.037075959000503644, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-tokenizer-kvcache-workload-llmd-simulator-kvcache]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-tokenizer-kvcache-workload-llmd-simulator-kvcache]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-with-tokenizer-kvcache-workload-llmd-simulator-kvcache", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3955009159981273, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 103.50094674099819, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.038951862006797455, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3840968450094806, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 65.02144141199824, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03642714599845931, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator0]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator0]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator0", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.24825902500015218, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 68.59104588899936, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.044241274998057634, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator1]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator1]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "model_routing", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator1", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.6905845959990984, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 126.28031142400869, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.039050968989613466, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator2]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator2]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "model_routing", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator2", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.43123031199502293, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 118.31190905299445, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.040537203996791504, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "model_routing", "lora", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.34194466499320697, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 140.65414688599412, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03790804599702824, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_keda_deployment[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda]", "lineno": 571, "outcome": "failed", "keywords": ["test_llm_autoscaling_keda_deployment[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda]", "parametrize", "autoscaling_keda", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-keda", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3601102399989031, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 935.720951610012, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T17:40:59Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T17:40:59Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T17:40:59Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:40:44Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T17:41:15Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T17:41:15Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T17:40:59Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T17:41:15Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:40:59Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 606, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 482, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-keda'], pro...              {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_keda\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-keda\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-keda-deploy\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_keda_deployment(test_case: TestCase):\n        \"\"\"KEDA + Deployment: ScaledObject exists with WVA annotations; no HPA; pods scale up under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:606: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24dad53c90>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-keda'], pro...              {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:482: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f24dad53c90>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-ked-101f2a9d'},\n                       {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-30T17:40:17.764980', start_time = 1785433217.765324\nduration = 900.3194184303284, timestamp_end = '2026-07-30T17:55:18.084762'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24dad53c90>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-ked-101f2a9d'},\n                       {'name': 'scaling-keda-autoscale-keda-dep-1ac84077'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f24da561a80>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T17:40:59Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T17:40:59Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T17:40:59Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:40:44Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T17:41:15Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T17:41:15Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T17:40:59Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T17:41:15Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:40:59Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.002007123999646865, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "model_routing", "lora", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.6912232549948385, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 120.135010907994, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04496038499928545, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-pvc]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-pvc]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-pvc", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 20.61847385400324, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 156.72274211600597, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04559327599417884, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-pd-cpu-model-pvc]", "lineno": 242, "outcome": "failed", "keywords": ["test_llm_inference_service[router-managed-workload-pd-cpu-model-pvc]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-pd-cpu-model-pvc", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 10.58501321201038, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 902.6766124749993, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-30T17:46:30Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T17:46:30Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T17:48:23Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:46:30Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'severity': 'Info', 'status': 'False', 'type': 'PrefillWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:46:16Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T17:46:30Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T17:46:52Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T17:46:52Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:46:30Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_inference_service.py", "lineno": 866, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-pd-cpu', 'model-pvc'], prompt='KServe is a', service_name='llmisvc-mod...              {'name': 'model-pvc-llmisvc-model-pvc-rou-49c1f027'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.asyncio(loop_scope=\"session\")\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-with-gateway-ref\",\n                        \"router-with-managed-route\",\n                        \"model-fb-opt-125m\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/completions\",\n                    prompt=\"KServe is a\",\n                    payload_formatter=completions_payload,\n                    response_assertion=create_response_assertion(with_field=\"choices\"),\n                    expected_gateway=\"router-gateway-1\",\n                    before_test=[\n                        lambda tc: create_router_resources(\n                            gateways=[\n                                make_router_gateway(\n                                    \"router-gateway-1\",\n                                    tc.namespace,\n                                ),\n                            ],\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                    pytest.mark.custom_gateway,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    payload_formatter=completions_payload,\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-custom-route-timeout\",\n                        \"scheduler-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"custom-route-timeout-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-with-refs\",\n                        \"scheduler-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"router-with-refs-test\",\n                    expected_gateway=\"router-gateway-1\",\n                    before_test=[\n                        lambda tc: create_router_resources(\n                            gateways=[\n                                make_router_gateway(\n                                    \"router-gateway-1\",\n                                    tc.namespace,\n                                ),\n                            ],\n                            routes=[\n                                make_router_main_route(\n                                    \"router-route-1\",\n                                    tc.namespace,\n                                    \"router-gateway-1\",\n                                    \"router-with-refs-test\",\n                                ),\n                                make_router_health_route(\n                                    \"router-route-2\",\n                                    tc.namespace,\n                                    \"router-gateway-1\",\n                                    \"router-with-refs-test\",\n                                ),\n                            ],\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.custom_gateway,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\"router-managed\", \"workload-pd-cpu\", \"model-fb-opt-125m\"],\n                    prompt=\"You are an expert in Kubernetes-native machine learning serving platforms, with deep knowledge of the KServe project. \"\n                    \"Explain the challenges of serving large-scale models, GPU scheduling, and how KServe integrates with capabilities like multi-model serving. \"\n                    \"Provide a detailed comparison with open source alternatives, focusing on operational trade-offs.\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-custom-route-timeout-pd\",\n                        \"scheduler-managed\",\n                        \"workload-pd-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"You are an expert in Kubernetes-native machine learning serving platforms, with deep knowledge of the KServe project. \"\n                    \"Explain the challenges of serving large-scale models, GPU scheduling, and how KServe integrates with capabilities like multi-model serving. \"\n                    \"Provide a detailed comparison with open source alternatives, focusing on operational trade-offs.\",\n                    service_name=\"custom-route-timeout-pd-test\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-with-refs-pd\",\n                        \"scheduler-managed\",\n                        \"workload-pd-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"You are an expert in Kubernetes-native machine learning serving platforms, with deep knowledge of the KServe project. \"\n                    \"Explain the challenges of serving large-scale models, GPU scheduling, and how KServe integrates with capabilities like multi-model serving. \"\n                    \"Provide a detailed comparison with open source alternatives, focusing on operational trade-offs.\",\n                    service_name=\"router-with-refs-pd-test\",\n                    response_assertion=assert_200_with_choices,\n                    expected_gateway=\"router-gateway-2\",\n                    before_test=[\n                        lambda tc: create_router_resources(\n                            gateways=[\n                                make_router_gateway(\n                                    \"router-gateway-2\",\n                                    tc.namespace,\n                                ),\n                            ],\n                            routes=[\n                                make_router_main_route(\n                                    \"router-route-3\",\n                                    tc.namespace,\n                                    \"router-gateway-2\",\n                                    \"router-with-refs-pd-test\",\n                                ),\n                                make_router_health_route(\n                                    \"router-route-4\",\n                                    tc.namespace,\n                                    \"router-gateway-2\",\n                                    \"router-with-refs-pd-test\",\n                                ),\n                            ],\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.custom_gateway,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-dp-ep-gpu\",\n                        \"workload-dp-ep-prefill-gpu\",\n                        \"model-deepseek-v2-lite\",\n                    ],\n                    prompt=\"Delve into the multifaceted implications of a fully disaggregated cloud architecture, specifically \"\n                    \"where the compute plane (P) and the data plane (D) are independently deployed and managed for a \"\n                    \"geographically distributed, high-throughput, low-latency microservices ecosystem. Beyond the \"\n                    \"fundamental challenges of network latency and data consistency, elaborate on the advanced \"\n                    \"considerations and trade-offs inherent in such a setup: 1. Network Architecture and Protocols: \"\n                    \"How would the network fabric and underlying protocols (e.g., RDMA, custom transport layers) need to \"\n                    \"evolve to support optimal performance and minimize inter-plane communication overhead, especially for \"\n                    \"synchronous operations? Discuss the role of network programmability (e.g., SDN, P4) in dynamically \"\n                    \"optimizing routing and traffic flow between P and D. 2. Advanced Data Consistency and Durability: \"\n                    \"Explore sophisticated data consistency models (e.g., causal consistency, strong eventual consistency) \"\n                    \"and their applicability in balancing performance and data integrity across a globally distributed data plane. \"\n                    \"Detail strategies for ensuring data durability and fault tolerance, including multi-region replication, \"\n                    \"intelligent partitioning, and recovery mechanisms in the event of partial or full plane failures. \"\n                    \"3. Dynamic Resource Orchestration and Cost Optimization: Analyze how an orchestration layer would intelligently \"\n                    \"manage the independent scaling of compute (P) and data (D) resources, considering fluctuating workloads, \"\n                    \"cost efficiency, and performance targets (e.g., using predictive analytics for resource provisioning). \"\n                    \"Discuss mechanisms for dynamically reallocating compute nodes to different data partitions based on \"\n                    \"workload patterns and data locality, potentially involving live migration strategies. \"\n                    \"4. Security and Compliance in a Distributed Landscape: Address the enhanced security perimeter \"\n                    \"challenges, including securing communication channels between P and D (encryption in transit, mutual TLS), \"\n                    \"fine-grained access control to data at rest and in motion, and identity management across disaggregated \"\n                    \"components. Discuss how such an architecture impacts compliance with regulatory frameworks (e.g., GDPR, HIPAA) \"\n                    \"concerning data sovereignty, privacy, and auditability. 5. Operational Complexity and Observability: \"\n                    \"Examine the increased complexity in monitoring, logging, and tracing across highly decoupled compute and \"\n                    \"data planes. What specialized tooling and practices (e.g., distributed tracing with OpenTelemetry, advanced AIOps) \"\n                    \"would be essential? How would incident response and troubleshooting differ in this disaggregated environment \"\n                    \"compared to traditional integrated systems? Consider the challenges of pinpointing root causes across \"\n                    \"independent failures. 6. Real-world Applicability and Future Trends: Identify specific industries \"\n                    \"or use cases (e.g., high-frequency trading, IoT edge processing, large language model inference) \"\n                    \"where the benefits of P/D disaggregation would strongly outweigh its complexities. \"\n                    \"Conclude by speculating on emerging technologies or paradigms (e.g., serverless compute functions \"\n                    \"directly interacting with object storage, in-memory disaggregation) that could further drive or \"\n                    \"transform P/D disaggregation in cloud computing.\",\n                    max_tokens=2000,\n                ),\n                marks=[\n                    pytest.mark.cluster_gpu,\n                    pytest.mark.cluster_nvidia,\n                    pytest.mark.cluster_nvidia_roce,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-no-scheduler\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"What is KServe?\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.no_scheduler,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-simulated-dp-ep-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"This test simulates DP+EP that can run on CPU, the idea is to test the LWS-based deployment, \"\n                    \"but without the resources requirements for DP+EP (GPUs and ROCe/IB).\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_multi_node],\n            ),\n            # Scheduler config tests\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-inline-config\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-inline-config-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            # Chat completions endpoint coverage\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                        \"model-qwen2.5-0.5b\",\n                    ],\n                    model_name=\"Qwen/Qwen2.5-0.5B-Instruct\",\n                    endpoint=\"/v1/chat/completions\",\n                    prompt=\"What is KServe?\",\n                    payload_formatter=chat_completions_payload,\n                    response_assertion=create_response_assertion(with_field=\"choices\"),\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-configmap-ref\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-configmap-ref-test\",\n                    before_test=[\n                        lambda tc: create_scheduler_configmap(namespace=tc.namespace)\n                    ],\n                    after_test=[\n                        lambda tc: delete_scheduler_configmap(namespace=tc.namespace)\n                    ],\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-replicas\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-ha-replicas-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-custom-template\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-custom-template-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            # Scheduler v0.6 \u2192 v0.7 migration tests.\n            # Deploy v0.6-style configs and verify the controller migrates them\n            # so the v0.7 scheduler boots successfully.\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-v06-pd-config-migration\",\n                        \"workload-llmd-simulator-pd\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-v06-pd-migration-test\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-v06-nonzero-threshold-migration\",\n                        \"workload-llmd-simulator-pd\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-v06-threshold-migration-test\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Standalone tokenizer \u2014 clean path: token-producer in inline config\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-tokenizer-kvcache\",\n                        \"workload-llmd-simulator-kvcache\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"tokenizer-clean-path-test\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Standalone tokenizer \u2014 migration path: legacy precise-prefix-cache-scorer\n            # triggers auto-provisioned tokenizer without explicit tokenizer:{} field\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-precise-prefix-cache-inline-config\",\n                        \"workload-llmd-simulator-kvcache\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"tokenizer-migration-path-test\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Models endpoint coverage\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/models\",\n                    response_assertion=create_response_assertion(with_field=\"data\"),\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 /v1/completions\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/completions\",\n                    prompt=\"KServe is a\",\n                    payload_formatter=completions_payload,\n                    response_assertion=assert_model_field_matches(\"facebook/opt-125m\"),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/facebook/opt-125m\",\n                    },\n                    peers=[\n                        TestCase(\n                            base_refs=[\n                                \"router-managed\",\n                                \"workload-llmd-simulator\",\n                                \"model-qwen2.5-0.5b\",\n                            ],\n                            endpoint=\"/v1/completions\",\n                            prompt=\"KServe is a\",\n                            payload_formatter=completions_payload,\n                            response_assertion=assert_model_field_matches(\n                                \"Qwen/Qwen2.5-0.5B-Instruct\"\n                            ),\n                            url_getter=get_model_routing_url,\n                            extra_headers={\n                                MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/Qwen/Qwen2.5-0.5B-Instruct\",\n                            },\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                    pytest.mark.model_routing,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 /v1/chat/completions\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/chat/completions\",\n                    prompt=\"What is KServe?\",\n                    payload_formatter=chat_completions_payload,\n                    response_assertion=assert_model_field_matches(\"facebook/opt-125m\"),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/facebook/opt-125m\",\n                    },\n                    peers=[\n                        TestCase(\n                            base_refs=[\n                                \"router-managed\",\n                                \"workload-llmd-simulator\",\n                                \"model-qwen2.5-0.5b\",\n                            ],\n                            endpoint=\"/v1/chat/completions\",\n                            prompt=\"What is KServe?\",\n                            payload_formatter=chat_completions_payload,\n                            response_assertion=assert_model_field_matches(\n                                \"Qwen/Qwen2.5-0.5B-Instruct\"\n                            ),\n                            url_getter=get_model_routing_url,\n                            extra_headers={\n                                MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/Qwen/Qwen2.5-0.5B-Instruct\",\n                            },\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                    pytest.mark.model_routing,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 LoRA adapter\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m-with-lora-hf\",\n                    ],\n                    endpoint=\"/v1/completions\",\n                    prompt=\"KServe is a\",\n                    model_name=\"publishers/{namespace}/models/lora-adapter-1\",\n                    payload_formatter=completions_payload,\n                    response_assertion=assert_model_field_matches(\n                        \"publishers/{namespace}/models/lora-adapter-1\"\n                    ),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/lora-adapter-1\",\n                    },\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.model_routing,\n                    pytest.mark.lora,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 /v1/models (base + LoRA)\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m-with-lora-hf\",\n                    ],\n                    endpoint=\"/v1/models\",\n                    response_assertion=assert_models_contains(\n                        \"facebook/opt-125m\",\n                        \"publishers/{namespace}/models/facebook/opt-125m\",\n                        \"lora-adapter-1\",\n                        \"publishers/{namespace}/models/lora-adapter-1\",\n                    ),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/facebook/opt-125m\",\n                    },\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.model_routing,\n                    pytest.mark.lora,\n                ],\n            ),\n            # PVC storage tests -- validate direct PVC volume mount with real vLLM serving\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-pvc\",\n                    ],\n                    prompt=\"KServe is a\",\n                    response_assertion=assert_200_with_choices,\n                    before_test=[lambda tc: ensure_pvc_with_model(namespace=tc.namespace)],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.pvc_storage,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-pd-cpu\",\n                        \"model-pvc\",\n                    ],\n                    prompt=\"KServe is a\",\n                    response_assertion=assert_200_with_choices,\n                    before_test=[lambda tc: ensure_pvc_with_model(namespace=tc.namespace)],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.pvc_storage,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-simulated-dp-ep-cpu\",\n                        \"model-pvc\",\n                    ],\n                    prompt=\"KServe is a\",\n                    before_test=[lambda tc: ensure_pvc_with_model(namespace=tc.namespace)],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_multi_node,\n                    pytest.mark.pvc_storage,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_inference_service(test_case: TestCase):  # noqa: F811\n        inject_k8s_proxy()\n    \n        kserve_client = KServeClient(\n            config_file=os.environ.get(\"KUBECONFIG\", \"~/.kube/config\"),\n            client_configuration=client.Configuration(),\n        )\n    \n        service_name = test_case.llm_service.metadata.name\n        prefix = test_case.log_prefix\n    \n        test_failed = False\n        try:\n            print(f\"{prefix} Creating LLMInferenceService {service_name}\")\n            create_llmisvc(kserve_client, test_case.llm_service)\n            print(f\"{prefix} Waiting for LLMInferenceService {service_name} to be ready\")\n>           wait_for_llm_isvc_ready(\n                kserve_client, test_case.llm_service, test_case.wait_timeout\n            )\n\nllmisvc/test_llm_inference_service.py:866: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f9b18f32b10>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...del-p-9d807ba3'},\n                       {'name': 'model-pvc-llmisvc-model-pvc-rou-49c1f027'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-30T17:45:41.765322', start_time = 1785433541.7657423\nduration = 900.7897028923035, timestamp_end = '2026-07-30T18:00:42.555448'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f9b18f32b10>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....svc-model-p-9d807ba3'},\n                       {'name': 'model-pvc-llmisvc-model-pvc-rou-49c1f027'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f9b13c99940>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'WorkloadsReady', 'Ready'}, expected {'WorkloadsReady', 'RouterReady', 'Ready'}, got [{'lastTransitionTime': '2026-07-30T17:46:30Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T17:46:30Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T17:48:23Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:46:30Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'severity': 'Info', 'status': 'False', 'type': 'PrefillWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:46:16Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T17:46:30Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T17:46:52Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T17:46:52Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:46:30Z', 'message': 'Deployment does not have minimum availability.', 'reason': 'MinimumReplicasUnavailable', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.0016625869902782142, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_hpa_lws[cluster_cpu-cluster_multi_node-router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-hpa]", "lineno": 635, "outcome": "failed", "keywords": ["test_llm_autoscaling_hpa_lws[router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-hpa]", "parametrize", "autoscaling_hpa", "cluster_cpu", "cluster_multi_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-hpa", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 1.224956370992004, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 931.4064315010037, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T17:56:35Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T17:56:35Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T17:56:22Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T17:57:00Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T17:57:00Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T17:56:44Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T17:57:00Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:56:35Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:56:44Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 670, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 482, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-hpa'], prompt='KSer...                {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_hpa\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-lws\",\n                        \"prometheus-scrape\",\n                        \"scaling-hpa\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-hpa-lws\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_multi_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_hpa_lws(test_case: TestCase):\n        \"\"\"HPA + LWS: HPA exists with WVA annotations; pods scale under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:670: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24da5ac0d0>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-hpa'], prompt='KSer...                {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:482: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f24da5ac0d0>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...ale-hpa-b29acdba'},\n                       {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-30T17:55:55.053485', start_time = 1785434155.0538018\nduration = 900.6009018421173, timestamp_end = '2026-07-30T18:10:55.654707'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24da5ac0d0>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....autoscale-hpa-b29acdba'},\n                       {'name': 'scaling-hpa-autoscale-hpa-lws-b344a3ff'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f24da5616c0>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T17:56:35Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T17:56:35Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T17:56:22Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T17:57:00Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T17:57:00Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T17:56:44Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T17:57:00Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:56:35Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T17:56:44Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-hpa-lws-d4cbcfd2/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-hpa-lws-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.001329049002379179, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_multi_node-router-managed-workload-simulated-dp-ep-cpu-model-pvc]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-pvc]", "parametrize", "asyncio", "cluster_cpu", "cluster_multi_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-simulated-dp-ep-cpu-model-pvc", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 21.191737447006744, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 200.79476422500738, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04059104100451805, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_finalizer_added", "lineno": 191, "outcome": "passed", "keywords": ["test_config_finalizer_added", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14334826699632686, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 2.184826062002685, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.032792810001410544, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_deletion_blocked_when_referenced", "lineno": 224, "outcome": "passed", "keywords": ["test_config_deletion_blocked_when_referenced", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15948505098640453, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 4.524898071002099, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.030825976995402016, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_deletion_allowed_when_unreferenced", "lineno": 314, "outcome": "passed", "keywords": ["test_config_deletion_allowed_when_unreferenced", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14616196900897194, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 4.260436947006383, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.025326732997200452, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_config_deletion_unblocked_after_service_deleted", "lineno": 349, "outcome": "passed", "keywords": ["test_config_deletion_unblocked_after_service_deleted", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16333277900412213, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 4.654377856000792, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.02748563000932336, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_well_known_config_deletion_prevented_by_webhook", "lineno": 431, "outcome": "passed", "keywords": ["test_well_known_config_deletion_prevented_by_webhook", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.02905265399022028, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.14369460099260323, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.00016596300702076405, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_config_deletion.py::test_well_known_config_deletion_blocked_by_implicit_reference", "lineno": 465, "outcome": "passed", "keywords": ["test_well_known_config_deletion_blocked_by_implicit_reference", "cluster_single_node", "cluster_cpu", "__wrapped__", "pytestmark", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_config_deletion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1392597240046598, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 4.8607371440011775, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.024258620003820397, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_v1alpha1_to_v1alpha2_conversion", "lineno": 211, "outcome": "passed", "keywords": ["test_v1alpha1_to_v1alpha2_conversion", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15952581599412952, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.6139662670029793, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.42969607100530993, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_v1alpha2_to_v1alpha1_conversion", "lineno": 302, "outcome": "passed", "keywords": ["test_v1alpha2_to_v1alpha1_conversion", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17171390299336053, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.5906490869965637, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.1343432470021071, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_criticality_preservation_via_annotations", "lineno": 393, "outcome": "passed", "keywords": ["test_criticality_preservation_via_annotations", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17201005600509234, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.6926582409942057, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.5381828870013123, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_lora_criticality_preservation", "lineno": 530, "outcome": "passed", "keywords": ["test_lora_criticality_preservation", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15811193399713375, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.4185706589923939, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.4228329829929862, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_round_trip_conversion_preserves_fields", "lineno": 679, "outcome": "passed", "keywords": ["test_round_trip_conversion_preserves_fields", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestLLMInferenceServiceConversion", "conversion", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17948210699250922, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.5798961230029818, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.8289017299975967, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_stop.py::test_llm_stop_feature[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 39, "outcome": "passed", "keywords": ["test_llm_stop_feature[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service_stop.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.9047131650004303, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 347.3257051880064, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04731664899736643, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_lora_adapters.py::test_llm_with_lora_adapters[cluster_cpu-single-lora-adapter-hf]", "lineno": 209, "outcome": "passed", "keywords": ["test_llm_with_lora_adapters[single-lora-adapter-hf]", "parametrize", "cluster_cpu", "lora", "__wrapped__", "pytestmark", "single-lora-adapter-hf", "llminferenceservice", "llmisvc_core", "test_llm_lora_adapters.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1500415909977164, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 172.54077654800494, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.036384581006132066, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_keda_lws[cluster_cpu-cluster_multi_node-router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-keda]", "lineno": 693, "outcome": "failed", "keywords": ["test_llm_autoscaling_keda_lws[router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-keda]", "parametrize", "autoscaling_keda", "cluster_cpu", "cluster_multi_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-lws-prometheus-scrape-scaling-keda", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.5707835819921456, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 930.8988420109963, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T18:12:21Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T18:12:21Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T18:12:08Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T18:12:51Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T18:12:51Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T18:12:21Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T18:12:51Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T18:12:21Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T18:12:21Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 728, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 482, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-keda'], prompt='KSe...              {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_keda\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-lws\",\n                        \"prometheus-scrape\",\n                        \"scaling-keda\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-keda-lws\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_multi_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_keda_lws(test_case: TestCase):\n        \"\"\"KEDA + LWS: ScaledObject exists with WVA annotations; pods scale under load.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:728: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24d9a8dd10>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-lws', 'prometheus-scrape', 'scaling-keda'], prompt='KSe...              {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:482: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f24d9a8dd10>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-ked-231d315d'},\n                       {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-30T18:11:26.779046', start_time = 1785435086.7793698\nduration = 900.3845915794373, timestamp_end = '2026-07-30T18:26:27.163973'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24d9a8dd10>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-ked-231d315d'},\n                       {'name': 'scaling-keda-autoscale-keda-lws-1337f511'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f24da5634c0>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T18:12:21Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T18:12:21Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T18:12:08Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T18:12:51Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T18:12:51Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T18:12:21Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T18:12:51Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T18:12:21Z', 'severity': 'Info', 'status': 'True', 'type': 'WorkerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T18:12:21Z', 'message': 'failed to ensure HPA is correctly created for ScaledObject: error parsing prometheus metadata: error parsing prometheus metadata: bearer token=<empty> is required when bearer auth is enabled', 'reason': 'ScaledObjectCheckFailed', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.0022704599978169426, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_lora_adapters.py::test_llm_with_lora_adapters[cluster_cpu-multiple-lora-adapters]", "lineno": 209, "outcome": "passed", "keywords": ["test_llm_with_lora_adapters[multiple-lora-adapters]", "parametrize", "cluster_cpu", "lora", "__wrapped__", "pytestmark", "multiple-lora-adapters", "llminferenceservice", "llmisvc_core", "test_llm_lora_adapters.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16335302899824455, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 168.4040102029976, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.039692915001069196, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_tls.py::test_llm_tls_resources[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 91, "outcome": "passed", "keywords": ["test_llm_tls_resources[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llmisvc_core", "test_llm_tls.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.7868108159891563, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 180.2104863590066, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.044647962000453845, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_prestop_hook.py::test_prestop_hook[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 40, "outcome": "passed", "keywords": ["test_prestop_hook[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_prestop_hook.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 1.0511573539988603, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 194.24929584999336, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04137721601000521, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_rolling_upgrade.py::test_rolling_upgrade_coordination[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-model-fb-opt-125m]", "lineno": 40, "outcome": "passed", "keywords": ["test_rolling_upgrade_coordination[router-managed-workload-llmd-simulator-model-fb-opt-125m]", "parametrize", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "llmd_simulator", "test_rolling_upgrade.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 1.0015363410057034, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 148.1956324769999, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04217021699878387, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_storage_version_migration.py::TestStorageVersionMigration::test_storage_version_migration_after_simulated_upgrade", "lineno": 112, "outcome": "passed", "keywords": ["test_storage_version_migration_after_simulated_upgrade", "cluster_single_node", "cluster_cpu", "pytestmark", "llminferenceservice", "TestStorageVersionMigration", "conversion", "test_storage_version_migration.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1748475439962931, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 67.1480283090059, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 2.011450613004854, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_leave_group", "lineno": 992, "outcome": "passed", "keywords": ["test_leave_group", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1602944740006933, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 105.62164363000193, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.05189799499930814, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_autoscaling_wva.py::test_llm_autoscaling_cleanup_hpa[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "lineno": 857, "outcome": "failed", "keywords": ["test_llm_autoscaling_cleanup_hpa[router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa]", "parametrize", "autoscaling_hpa", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-no-replicas-prometheus-scrape-scaling-hpa", "llminferenceservice", "llmisvc_autoscaling", "autoscaling_wva", "test_llm_autoscaling_wva.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 1.1594325709884288, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 946.2370474630006, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T18:27:47Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T18:27:47Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T18:27:47Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-30T18:27:26Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T18:28:04Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-cleanup-hpa-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T18:28:04Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T18:27:47Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-cleanup-hpa-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T18:28:04Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T18:27:47Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-cleanup-hpa-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]"}, "traceback": [{"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 892, "message": ""}, {"path": "llmisvc/test_llm_autoscaling_wva.py", "lineno": 482, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1376, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1371, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-cleanup-h-aa1ae037'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.autoscaling_hpa\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator-no-replicas\",\n                        \"prometheus-scrape\",\n                        \"scaling-hpa\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"autoscale-cleanup-hpa\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_autoscaling_cleanup_hpa(test_case: TestCase):\n        \"\"\"Removing scaling config should delete HPA.\"\"\"\n        inject_k8s_proxy()\n        kserve_client = _new_kserve_client()\n        service_name = test_case.llm_service.metadata.name\n        ns = test_case.namespace\n    \n        try:\n>           _create_and_wait(kserve_client, test_case)\n\nllmisvc/test_llm_autoscaling_wva.py:892: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24d9c06390>\ntest_case = TestCase(base_refs=['router-managed', 'workload-llmd-simulator-no-replicas', 'prometheus-scrape', 'scaling-hpa'], prom...              {'name': 'scaling-hpa-autoscale-cleanup-h-aa1ae037'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    def _create_and_wait(kserve_client, test_case):\n        \"\"\"Create LLMISVC and wait for it to be ready.\"\"\"\n        create_llmisvc(kserve_client, test_case.llm_service)\n>       wait_for_llm_isvc_ready(\n            kserve_client, test_case.llm_service, test_case.wait_timeout\n        )\n\nllmisvc/test_llm_autoscaling_wva.py:482: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f24d9c06390>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...e-cle-5a67f5d1'},\n                       {'name': 'scaling-hpa-autoscale-cleanup-h-aa1ae037'}]},\n 'status': None}, 900)\nkwargs = {}, func_name = 'wait_for_llm_isvc_ready'\ntimestamp_start = '2026-07-30T18:26:59.154479', start_time = 1785436019.1547492\nduration = 900.5441663265228, timestamp_end = '2026-07-30T18:41:59.698918'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f24d9c06390>\ngiven = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....toscale-cle-5a67f5d1'},\n                       {'name': 'scaling-hpa-autoscale-cleanup-h-aa1ae037'}]},\n 'status': None}\ntimeout_seconds = 900\n\n    @log_execution\n    def wait_for_llm_isvc_ready(\n        kserve_client: KServeClient,\n        given: V1alpha1LLMInferenceService,\n        timeout_seconds: int = 900,\n    ) -> str:\n        def assert_llm_isvc_ready():\n            out = get_llmisvc(\n                kserve_client,\n                given.metadata.name,\n                given.metadata.namespace,\n                given.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in out:\n                raise AssertionError(\"No status found in LLM inference service\")\n    \n            status = out[\"status\"]\n            if \"conditions\" not in status:\n                raise AssertionError(\"No conditions found in status\")\n    \n            expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n            got_true_conditions = set()\n            all_condition_types = set()\n    \n            conditions = status[\"conditions\"]\n    \n            for condition in conditions:\n                ctype = condition.get(\"type\")\n                all_condition_types.add(ctype)\n                if condition.get(\"status\") == \"True\":\n                    got_true_conditions.add(ctype)\n    \n            # When TokenizerReady is present, it must also be True\n            if \"TokenizerReady\" in all_condition_types:\n                expected_true_conditions.add(\"TokenizerReady\")\n    \n            missing_conditions = expected_true_conditions - got_true_conditions\n            if missing_conditions:\n                raise AssertionError(\n                    f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n                )\n            return True\n    \n>       return wait_for(assert_llm_isvc_ready, timeout=timeout_seconds, interval=1.0)\n\nllmisvc/test_llm_inference_service.py:1376: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_llm_isvc_ready.<locals>.assert_llm_isvc_ready at 0x7f24da561c60>\ntimeout = 900, interval = 1.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def assert_llm_isvc_ready():\n        out = get_llmisvc(\n            kserve_client,\n            given.metadata.name,\n            given.metadata.namespace,\n            given.api_version.split(\"/\")[1],\n        )\n    \n        if \"status\" not in out:\n            raise AssertionError(\"No status found in LLM inference service\")\n    \n        status = out[\"status\"]\n        if \"conditions\" not in status:\n            raise AssertionError(\"No conditions found in status\")\n    \n        expected_true_conditions = {\"Ready\", \"WorkloadsReady\", \"RouterReady\"}\n        got_true_conditions = set()\n        all_condition_types = set()\n    \n        conditions = status[\"conditions\"]\n    \n        for condition in conditions:\n            ctype = condition.get(\"type\")\n            all_condition_types.add(ctype)\n            if condition.get(\"status\") == \"True\":\n                got_true_conditions.add(ctype)\n    \n        # When TokenizerReady is present, it must also be True\n        if \"TokenizerReady\" in all_condition_types:\n            expected_true_conditions.add(\"TokenizerReady\")\n    \n        missing_conditions = expected_true_conditions - got_true_conditions\n        if missing_conditions:\n>           raise AssertionError(\n                f\"Missing true conditions: {missing_conditions}, expected {expected_true_conditions}, got {conditions}\"\n            )\nE           AssertionError: Missing true conditions: {'Ready', 'WorkloadsReady'}, expected {'Ready', 'RouterReady', 'WorkloadsReady'}, got [{'lastTransitionTime': '2026-07-30T18:27:47Z', 'severity': 'Info', 'status': 'True', 'type': 'HTTPRoutesReady'}, {'lastTransitionTime': '2026-07-30T18:27:47Z', 'severity': 'Info', 'status': 'True', 'type': 'InferencePoolReady'}, {'lastTransitionTime': '2026-07-30T18:27:47Z', 'severity': 'Info', 'status': 'True', 'type': 'MainWorkloadReady'}, {'lastTransitionTime': '2026-07-30T18:27:26Z', 'severity': 'Info', 'status': 'True', 'type': 'PresetsCombined'}, {'lastTransitionTime': '2026-07-30T18:28:04Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-cleanup-hpa-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'Ready'}, {'lastTransitionTime': '2026-07-30T18:28:04Z', 'status': 'True', 'type': 'RouterReady'}, {'lastTransitionTime': '2026-07-30T18:27:47Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-cleanup-hpa-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'severity': 'Info', 'status': 'False', 'type': 'ScalingReady'}, {'lastTransitionTime': '2026-07-30T18:28:04Z', 'severity': 'Info', 'status': 'True', 'type': 'SchedulerWorkloadReady'}, {'lastTransitionTime': '2026-07-30T18:27:47Z', 'message': 'the HPA was unable to compute the replica count: unable to get external metric e2e-test-llm-autoscaling-cleanup-hpa-a41f0e40/wva_desired_replicas/&LabelSelector{MatchLabels:map[string]string{variant_name: autoscale-cleanup-hpa-kserve-hpa,},MatchExpressions:[]LabelSelectorRequirement{},}: unable to fetch metrics from external metrics API: scaledObject name is not specified', 'reason': 'FailedGetExternalMetric', 'status': 'False', 'type': 'WorkloadsReady'}]\n\nllmisvc/test_llm_inference_service.py:1371: AssertionError"}, "teardown": {"duration": 0.0016700170090189204, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_three_member_group", "lineno": 1038, "outcome": "passed", "keywords": ["test_three_member_group", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1665156939998269, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 141.20566847900045, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03307767599471845, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_late_join", "lineno": 1089, "outcome": "passed", "keywords": ["test_late_join", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14812365399848204, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 150.81741410901304, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0347756339906482, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_delete_at_nonzero_weight", "lineno": 1141, "outcome": "passed", "keywords": ["test_delete_at_nonzero_weight", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16100240200466942, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 144.72424202800903, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03820388700114563, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_rollback", "lineno": 1183, "outcome": "passed", "keywords": ["test_rollback", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15163906299858354, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 148.20102136800415, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.031175456999335438, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_canary_lifecycle.py::TestCanaryLifecycle::test_force_stop_route_owner", "lineno": 1254, "outcome": "passed", "keywords": ["test_force_stop_route_owner", "llminferenceservice", "llmisvc_core", "TestCanaryLifecycle", "cluster_cpu", "traffic", "test_llm_canary_lifecycle.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15385597800195683, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 118.60854426100559, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03561890299897641, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "llmd_simulator", "custom_gateway", "__wrapped__", "pytestmark", "router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 2.966874813006143, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 197.8404608819983, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.03785607099416666, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.722529084989219, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 164.99776702499366, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.039937788009410724, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "passed", "keywords": ["test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.8511809249903308, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 180.70783342099458, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04146186700381804, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "failed", "keywords": ["test_llm_inference_service[router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "custom_gateway", "__wrapped__", "pytestmark", "router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 2.850609072993393, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 1246.776653125984, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1232, "message": "AssertionError: \u274c Failed to get service URL: \u274c Failed to get URL for LLM inference service router-with-refs-test: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))"}, "traceback": [{"path": "llmisvc/test_llm_inference_service.py", "lineno": 870, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1284, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1387, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1232, "message": "AssertionError"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b13bac450>\n\n    def _new_conn(self) -> socket.socket:\n        \"\"\"Establish a socket connection and set nodelay settings on it.\n    \n        :return: New socket connection.\n        \"\"\"\n        try:\n>           sock = connection.create_connection(\n                (self._dns_host, self.port),\n                self.timeout,\n                source_address=self.source_address,\n                socket_options=self.socket_options,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:204: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\naddress = ('a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', 6443)\ntimeout = None, source_address = None, socket_options = [(6, 1, 1)]\n\n    def create_connection(\n        address: tuple[str, int],\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        source_address: tuple[str, int] | None = None,\n        socket_options: _TYPE_SOCKET_OPTIONS | None = None,\n    ) -> socket.socket:\n        \"\"\"Connect to *address* and return the socket object.\n    \n        Convenience function.  Connect to *address* (a 2-tuple ``(host,\n        port)``) and return the socket object.  Passing the optional\n        *timeout* parameter will set the timeout on the socket instance\n        before attempting to connect.  If no *timeout* is supplied, the\n        global default timeout setting returned by :func:`socket.getdefaulttimeout`\n        is used.  If *source_address* is set it must be a tuple of (host, port)\n        for the socket to bind as a source address before making the connection.\n        An host of '' or port 0 tells the OS to use the default.\n        \"\"\"\n    \n        host, port = address\n        if host.startswith(\"[\"):\n            host = host.strip(\"[]\")\n        err = None\n    \n        # Using the value from allowed_gai_family() in the context of getaddrinfo lets\n        # us select whether to work with IPv4 DNS records, IPv6 records, or both.\n        # The original create_connection function always returns all records.\n        family = allowed_gai_family()\n    \n        try:\n            host.encode(\"idna\")\n        except UnicodeError:\n            raise LocationParseError(f\"'{host}', label empty or too long\") from None\n    \n>       for res in socket.getaddrinfo(host, port, family, socket.SOCK_STREAM):\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/connection.py:60: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nhost = 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com'\nport = 6443, family = <AddressFamily.AF_UNSPEC: 0>\ntype = <SocketKind.SOCK_STREAM: 1>, proto = 0, flags = 0\n\n    def getaddrinfo(host, port, family=0, type=0, proto=0, flags=0):\n        \"\"\"Resolve host and port into list of address info entries.\n    \n        Translate the host/port argument into a sequence of 5-tuples that contain\n        all the necessary arguments for creating a socket connected to that service.\n        host is a domain name, a string representation of an IPv4/v6 address or\n        None. port is a string service name such as 'http', a numeric port number or\n        None. By passing None as the value of host and port, you can pass NULL to\n        the underlying C API.\n    \n        The family, type and proto arguments can be optionally specified in order to\n        narrow the list of addresses returned. Passing zero as a value for each of\n        these arguments selects the full range of results.\n        \"\"\"\n        # We override this function since we want to translate the numeric family\n        # and socket type values to enum constants.\n        addrlist = []\n>       for res in _socket.getaddrinfo(host, port, family, type, proto, flags):\nE       socket.gaierror: [Errno -2] Name or service not known\n\n/usr/lib64/python3.11/socket.py:974: gaierror\n\nThe above exception was the direct cause of the following exception:\n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\nmethod = 'GET'\nurl = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nbody = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False, err = None\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n>           response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:788: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b13bac450>\nmethod = 'GET'\nurl = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nbody = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\ntimeout = Timeout(connect=None, read=None, total=None), chunked = False\nresponse_conn = None, preload_content = True, decode_content = True\nenforce_content_length = True\n\n    def _make_request(\n        self,\n        conn: BaseHTTPConnection,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | None = None,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        chunked: bool = False,\n        response_conn: BaseHTTPConnection | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        enforce_content_length: bool = True,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Perform a request on a given urllib connection object taken from our\n        pool.\n    \n        :param conn:\n            a connection from one of our connection pools\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            Pass ``None`` to retry until you receive a response. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param response_conn:\n            Set this to ``None`` if you will handle releasing the connection or\n            set the connection to have the response release it.\n    \n        :param preload_content:\n          If True, the response's body will be preloaded during construction.\n    \n        :param decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param enforce_content_length:\n            Enforce content length checking. Body returned by server must match\n            value of Content-Length header, if present. Otherwise, raise error.\n        \"\"\"\n        self.num_requests += 1\n    \n        timeout_obj = self._get_timeout(timeout)\n        timeout_obj.start_connect()\n        conn.timeout = Timeout.resolve_default_timeout(timeout_obj.connect_timeout)\n    \n        try:\n            # Trigger any extra validation we need to do.\n            try:\n                self._validate_conn(conn)\n            except (SocketTimeout, BaseSSLError) as e:\n                self._raise_timeout(err=e, url=url, timeout_value=conn.timeout)\n                raise\n    \n        # _validate_conn() starts the connection to an HTTPS proxy\n        # so we need to wrap errors with 'ProxyError' here too.\n        except (\n            OSError,\n            NewConnectionError,\n            TimeoutError,\n            BaseSSLError,\n            CertificateError,\n            SSLError,\n        ) as e:\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            # If the connection didn't successfully connect to it's proxy\n            # then there\n            if isinstance(\n                new_e, (OSError, NewConnectionError, TimeoutError, SSLError)\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n>           raise new_e\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:488: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b13bac450>\nmethod = 'GET'\nurl = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nbody = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\ntimeout = Timeout(connect=None, read=None, total=None), chunked = False\nresponse_conn = None, preload_content = True, decode_content = True\nenforce_content_length = True\n\n    def _make_request(\n        self,\n        conn: BaseHTTPConnection,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | None = None,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        chunked: bool = False,\n        response_conn: BaseHTTPConnection | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        enforce_content_length: bool = True,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Perform a request on a given urllib connection object taken from our\n        pool.\n    \n        :param conn:\n            a connection from one of our connection pools\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            Pass ``None`` to retry until you receive a response. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param response_conn:\n            Set this to ``None`` if you will handle releasing the connection or\n            set the connection to have the response release it.\n    \n        :param preload_content:\n          If True, the response's body will be preloaded during construction.\n    \n        :param decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param enforce_content_length:\n            Enforce content length checking. Body returned by server must match\n            value of Content-Length header, if present. Otherwise, raise error.\n        \"\"\"\n        self.num_requests += 1\n    \n        timeout_obj = self._get_timeout(timeout)\n        timeout_obj.start_connect()\n        conn.timeout = Timeout.resolve_default_timeout(timeout_obj.connect_timeout)\n    \n        try:\n            # Trigger any extra validation we need to do.\n            try:\n>               self._validate_conn(conn)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:464: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b13bac450>\n\n    def _validate_conn(self, conn: BaseHTTPConnection) -> None:\n        \"\"\"\n        Called right before a request is made, after the socket is created.\n        \"\"\"\n        super()._validate_conn(conn)\n    \n        # Force connect early to allow us to validate the connection.\n        if conn.is_closed:\n>           conn.connect()\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:1106: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b13bac450>\n\n    def connect(self) -> None:\n        # Today we don't need to be doing this step before the /actual/ socket\n        # connection, however in the future we'll need to decide whether to\n        # create a new socket or re-use an existing \"shared\" socket as a part\n        # of the HTTP/2 handshake dance.\n        if self._tunnel_host is not None and self._tunnel_port is not None:\n            probe_http2_host = self._tunnel_host\n            probe_http2_port = self._tunnel_port\n        else:\n            probe_http2_host = self.host\n            probe_http2_port = self.port\n    \n        # Check if the target origin supports HTTP/2.\n        # If the value comes back as 'None' it means that the current thread\n        # is probing for HTTP/2 support. Otherwise, we're waiting for another\n        # probe to complete, or we get a value right away.\n        target_supports_http2: bool | None\n        if \"h2\" in ssl_.ALPN_PROTOCOLS:\n            target_supports_http2 = http2_probe.acquire_and_get(\n                host=probe_http2_host, port=probe_http2_port\n            )\n        else:\n            # If HTTP/2 isn't going to be offered it doesn't matter if\n            # the target supports HTTP/2. Don't want to make a probe.\n            target_supports_http2 = False\n    \n        if self._connect_callback is not None:\n            self._connect_callback(\n                \"before connect\",\n                thread_id=threading.get_ident(),\n                target_supports_http2=target_supports_http2,\n            )\n    \n        try:\n            sock: socket.socket | ssl.SSLSocket\n>           self.sock = sock = self._new_conn()\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:759: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b13bac450>\n\n    def _new_conn(self) -> socket.socket:\n        \"\"\"Establish a socket connection and set nodelay settings on it.\n    \n        :return: New socket connection.\n        \"\"\"\n        try:\n            sock = connection.create_connection(\n                (self._dns_host, self.port),\n                self.timeout,\n                source_address=self.source_address,\n                socket_options=self.socket_options,\n            )\n        except socket.gaierror as e:\n>           raise NameResolutionError(self.host, self, e) from e\nE           urllib3.exceptions.NameResolutionError: HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:211: NameResolutionError\n\nThe above exception was the direct cause of the following exception:\n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f9b18f301d0>\nllm_isvc = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....router-with-ec5d4bfa'},\n                       {'name': 'model-fb-opt-125m-router-with-r-6d64416a'}]},\n 'status': None}\n\n    @log_execution\n    def get_llm_service_url(\n        kserve_client: KServeClient, llm_isvc: V1alpha1LLMInferenceService\n    ):\n        service_name = llm_isvc.metadata.name\n    \n        try:\n>           llm_isvc = get_llmisvc(\n                kserve_client,\n                llm_isvc.metadata.name,\n                llm_isvc.metadata.namespace,\n                llm_isvc.api_version.split(\"/\")[1],\n            )\n\nllmisvc/test_llm_inference_service.py:1296: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f9b18f301d0>\nname = 'router-with-refs-test'\nnamespace = 'e2e-test-llm-inference-service-028f7809', version = 'v1alpha1'\n\n    def get_llmisvc(\n        kserve_client: KServeClient,\n        name,\n        namespace,\n        version=constants.KSERVE_V1ALPHA1_VERSION,\n    ):\n        try:\n>           return kserve_client.api_instance.get_namespaced_custom_object(\n                constants.KSERVE_GROUP,\n                version,\n                namespace,\n                KSERVE_PLURAL_LLMINFERENCESERVICE,\n                name,\n            )\n\nllmisvc/test_llm_inference_service.py:1204: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f9b12013550>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-028f7809'\nplural = 'llminferenceservices', name = 'router-with-refs-test'\nkwargs = {'_return_http_data_only': True}\n\n    def get_namespaced_custom_object(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: object\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1632: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.custom_objects_api.CustomObjectsApi object at 0x7f9b12013550>\ngroup = 'serving.kserve.io', version = 'v1alpha1'\nnamespace = 'e2e-test-llm-inference-service-028f7809'\nplural = 'llminferenceservices', name = 'router-with-refs-test'\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...], 'auth_settings': ['BearerToken'], 'body_params': None, ...}\nall_params = ['group', 'version', 'namespace', 'plural', 'name', 'async_req', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {'group': 'serving.kserve.io', 'name': 'router-with-refs-test', 'namespace': 'e2e-test-llm-inference-service-028f7809', 'plural': 'llminferenceservices', ...}\nquery_params = []\n\n    def get_namespaced_custom_object_with_http_info(self, group, version, namespace, plural, name, **kwargs):  # noqa: E501\n        \"\"\"get_namespaced_custom_object  # noqa: E501\n    \n        Returns a namespace scoped custom object  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.get_namespaced_custom_object_with_http_info(group, version, namespace, plural, name, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param str group: the custom resource's group (required)\n        :param str version: the custom resource's version (required)\n        :param str namespace: The custom resource's namespace (required)\n        :param str plural: the custom resource's plural name. For TPRs this would be lowercase plural kind. (required)\n        :param str name: the custom object's name (required)\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(object, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'group',\n            'version',\n            'namespace',\n            'plural',\n            'name'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method get_namespaced_custom_object\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'group' is set\n        if self.api_client.client_side_validation and ('group' not in local_var_params or  # noqa: E501\n                                                        local_var_params['group'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `group` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'version' is set\n        if self.api_client.client_side_validation and ('version' not in local_var_params or  # noqa: E501\n                                                        local_var_params['version'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `version` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'namespace' is set\n        if self.api_client.client_side_validation and ('namespace' not in local_var_params or  # noqa: E501\n                                                        local_var_params['namespace'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `namespace` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'plural' is set\n        if self.api_client.client_side_validation and ('plural' not in local_var_params or  # noqa: E501\n                                                        local_var_params['plural'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `plural` when calling `get_namespaced_custom_object`\")  # noqa: E501\n        # verify the required parameter 'name' is set\n        if self.api_client.client_side_validation and ('name' not in local_var_params or  # noqa: E501\n                                                        local_var_params['name'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `name` when calling `get_namespaced_custom_object`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n        if 'group' in local_var_params:\n            path_params['group'] = local_var_params['group']  # noqa: E501\n        if 'version' in local_var_params:\n            path_params['version'] = local_var_params['version']  # noqa: E501\n        if 'namespace' in local_var_params:\n            path_params['namespace'] = local_var_params['namespace']  # noqa: E501\n        if 'plural' in local_var_params:\n            path_params['plural'] = local_var_params['plural']  # noqa: E501\n        if 'name' in local_var_params:\n            path_params['name'] = local_var_params['name']  # noqa: E501\n    \n        query_params = []\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}', 'GET',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='object',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/custom_objects_api.py:1739: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b12012a90>\nresource_path = '/apis/{group}/{version}/namespaces/{namespace}/{plural}/{name}'\nmethod = 'GET'\npath_params = {'group': 'serving.kserve.io', 'name': 'router-with-refs-test', 'namespace': 'e2e-test-llm-inference-service-028f7809', 'plural': 'llminferenceservices', ...}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b12012a90>\nresource_path = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nmethod = 'GET'\npath_params = [('group', 'serving.kserve.io'), ('version', 'v1alpha1'), ('namespace', 'e2e-test-llm-inference-service-028f7809'), ('plural', 'llminferenceservices'), ('name', 'router-with-refs-test')]\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = [], files = {}, response_type = 'object'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b12012a90>\nmethod = 'GET'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = [], body = None, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n>           return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:373: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f9b12010c10>\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], _preload_content = True, _request_timeout = None\n\n    def GET(self, url, headers=None, query_params=None, _preload_content=True,\n            _request_timeout=None):\n>       return self.request(\"GET\", url,\n                            headers=headers,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            query_params=query_params)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:244: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f9b12010c10>\nmethod = 'GET'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = None, post_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'application/x-www-form-urlencoded':  # noqa: E501\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=False,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                elif headers['Content-Type'] == 'multipart/form-data':\n                    # must del headers['Content-Type'], or the correct\n                    # Content-Type which generated by urllib3 will be\n                    # overwritten.\n                    del headers['Content-Type']\n                    r = self.pool_manager.request(\n                        method, url,\n                        fields=post_params,\n                        encode_multipart=True,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                # Pass a `string` parameter directly in the body to support\n                # other content types than Json when `body` argument is\n                # provided in serialized form\n                elif isinstance(body, str) or isinstance(body, bytes):\n                    request_body = body\n                    r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n                        headers=headers)\n                else:\n                    # Cannot generate the request from given parameters\n                    msg = \"\"\"Cannot prepare a request message for provided\n                             arguments. Please check that your arguments match\n                             declared content type.\"\"\"\n                    raise ApiException(status=0, reason=msg)\n            # For `GET`, `HEAD`\n            else:\n>               r = self.pool_manager.request(method, url,\n                                              fields=query_params,\n                                              preload_content=_preload_content,\n                                              timeout=timeout,\n                                              headers=headers)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:217: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b12011890>\nmethod = 'GET'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nbody = None, fields = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\njson = None, urlopen_kw = {'preload_content': True, 'timeout': None}\n\n    def request(\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        fields: _TYPE_FIELDS | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        json: typing.Any | None = None,\n        **urlopen_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Make a request using :meth:`urlopen` with the appropriate encoding of\n        ``fields`` based on the ``method`` used.\n    \n        This is a convenience method that requires the least amount of manual\n        effort. It can be used in most situations, while still having the\n        option to drop down to more specific methods when necessary, such as\n        :meth:`request_encode_url`, :meth:`request_encode_body`,\n        or even the lowest level :meth:`urlopen`.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param fields:\n            Data to encode and send in the URL or request body, depending on ``method``.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param json:\n            Data to encode and send as JSON with UTF-encoded in the request body.\n            The ``\"Content-Type\"`` header will be set to ``\"application/json\"``\n            unless specified otherwise.\n        \"\"\"\n        method = method.upper()\n    \n        if json is not None and body is not None:\n            raise TypeError(\n                \"request got values for both 'body' and 'json' parameters which are mutually exclusive\"\n            )\n    \n        if json is not None:\n            if headers is None:\n                headers = self.headers\n    \n            if not (\"content-type\" in map(str.lower, headers.keys())):\n                headers = HTTPHeaderDict(headers)\n                headers[\"Content-Type\"] = \"application/json\"\n    \n            body = _json.dumps(json, separators=(\",\", \":\"), ensure_ascii=False).encode(\n                \"utf-8\"\n            )\n    \n        if body is not None:\n            urlopen_kw[\"body\"] = body\n    \n        if method in self._encode_url_methods:\n>           return self.request_encode_url(\n                method,\n                url,\n                fields=fields,  # type: ignore[arg-type]\n                headers=headers,\n                **urlopen_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py:135: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b12011890>\nmethod = 'GET'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nfields = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nurlopen_kw = {'preload_content': True, 'timeout': None}\nextra_kw = {'headers': {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}, 'preload_content': True, 'timeout': None}\n\n    def request_encode_url(\n        self,\n        method: str,\n        url: str,\n        fields: _TYPE_ENCODE_URL_FIELDS | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        **urlopen_kw: str,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Make a request using :meth:`urlopen` with the ``fields`` encoded in\n        the url. This is useful for request methods like GET, HEAD, DELETE, etc.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param fields:\n            Data to encode and send in the URL.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n        \"\"\"\n        if headers is None:\n            headers = self.headers\n    \n        extra_kw: dict[str, typing.Any] = {\"headers\": headers}\n        extra_kw.update(urlopen_kw)\n    \n        if fields:\n            url += \"?\" + urlencode(fields)\n    \n>       return self.urlopen(method, url, **extra_kw)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py:182: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b12011890>\nmethod = 'GET'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nredirect = True\nkw = {'assert_same_host': False, 'headers': {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}, 'preload_content': True, 'redirect': False, ...}\nu = Url(scheme='https', auth=None, host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', p...espaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test', query=None, fragment=None)\nconn = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\n\n    def urlopen(  # type: ignore[override]\n        self, method: str, url: str, redirect: bool = True, **kw: typing.Any\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Same as :meth:`urllib3.HTTPConnectionPool.urlopen`\n        with custom cross-host redirect logic and only sends the request-uri\n        portion of the ``url``.\n    \n        The given ``url`` parameter must be absolute, such that an appropriate\n        :class:`urllib3.connectionpool.ConnectionPool` can be chosen for it.\n        \"\"\"\n        u = parse_url(url)\n    \n        if u.scheme is None:\n            warnings.warn(\n                \"URLs without a scheme (ie 'https://') are deprecated and will raise an error \"\n                \"in urllib3 v3.0. To avoid this FutureWarning ensure all URLs \"\n                \"start with 'https://' or 'http://'. Read more in this issue: \"\n                \"https://github.com/urllib3/urllib3/issues/2920\",\n                category=FutureWarning,\n                stacklevel=2,\n            )\n    \n        conn = self.connection_from_host(u.host, port=u.port, scheme=u.scheme)\n    \n        kw[\"assert_same_host\"] = False\n        kw[\"redirect\"] = False\n    \n        if \"headers\" not in kw:\n            kw[\"headers\"] = self.headers\n    \n        if self._proxy_requires_url_absolute_form(u):\n            response = conn.urlopen(method, url, **kw)\n        else:\n>           response = conn.urlopen(method, u.request_uri, **kw)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/poolmanager.py:457: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\nmethod = 'GET'\nurl = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nbody = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nretries = Retry(total=2, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = ConnectionResetError(104, 'Connection reset by peer'), clean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\nmethod = 'GET'\nurl = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nbody = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nretries = Retry(total=1, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = ConnectTimeoutError(<HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...on to a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com timed out. (connect timeout=None)')\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\nmethod = 'GET'\nurl = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nbody = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\nmethod = 'GET'\nurl = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nbody = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False, err = None\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n>           retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:842: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nmethod = 'GET'\nurl = '/apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test'\nresponse = None\nerror = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\n_pool = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b13b535d0>\n_stacktrace = <traceback object at 0x7f9b13bae880>\n\n    def increment(\n        self,\n        method: str | None = None,\n        url: str | None = None,\n        response: BaseHTTPResponse | None = None,\n        error: Exception | None = None,\n        _pool: ConnectionPool | None = None,\n        _stacktrace: TracebackType | None = None,\n    ) -> Self:\n        \"\"\"Return a new Retry object with incremented retry counters.\n    \n        :param response: A response object, or None, if the server did not\n            return a response.\n        :type response: :class:`~urllib3.response.BaseHTTPResponse`\n        :param Exception error: An error encountered during the request, or\n            None if the response was received successfully.\n    \n        :return: A new ``Retry`` object.\n        \"\"\"\n        if self.total is False and error:\n            # Disabled, indicate to re-raise the error.\n            raise reraise(type(error), error, _stacktrace)\n    \n        total = self.total\n        if total is not None:\n            total -= 1\n    \n        connect = self.connect\n        read = self.read\n        redirect = self.redirect\n        status_count = self.status\n        other = self.other\n        cause = \"unknown\"\n        status = None\n        redirect_location = None\n    \n        if error and self._is_connection_error(error):\n            # Connect retry?\n            if connect is False:\n                raise reraise(type(error), error, _stacktrace)\n            elif connect is not None:\n                connect -= 1\n    \n        elif error and self._is_read_error(error):\n            # Read retry?\n            if read is False or method is None or not self._is_method_retryable(method):\n                raise reraise(type(error), error, _stacktrace)\n            elif read is not None:\n                read -= 1\n    \n        elif error:\n            # Other retry?\n            if other is not None:\n                other -= 1\n    \n        elif response and response.get_redirect_location():\n            # Redirect retry?\n            if redirect is not None:\n                redirect -= 1\n            cause = \"too many redirects\"\n            response_redirect_location = response.get_redirect_location()\n            if response_redirect_location:\n                redirect_location = response_redirect_location\n            status = response.status\n    \n        else:\n            # Incrementing because of a server error like a 500 in\n            # status_forcelist and the given method is in the allowed_methods\n            cause = ResponseError.GENERIC_ERROR\n            if response and response.status:\n                if status_count is not None:\n                    status_count -= 1\n                cause = ResponseError.SPECIFIC_ERROR.format(status_code=response.status)\n                status = response.status\n    \n        history = self.history + (\n            RequestHistory(method, url, error, status, redirect_location),\n        )\n    \n        new_retry = self.new(\n            total=total,\n            connect=connect,\n            read=read,\n            redirect=redirect,\n            status=status_count,\n            other=other,\n            history=history,\n        )\n    \n        if new_retry.is_exhausted():\n            reason = error or ResponseError(cause)\n>           raise MaxRetryError(_pool, url, reason) from reason  # type: ignore[arg-type]\nE           urllib3.exceptions.MaxRetryError: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py:543: MaxRetryError\n\nThe above exception was the direct cause of the following exception:\n\n    def get_successful_response():\n        try:\n            if test_case.url_getter:\n                service_url = test_case.url_getter(kserve_client, test_case.llm_service)\n            else:\n>               service_url = get_llm_service_url(kserve_client, test_case.llm_service)\n\nllmisvc/test_llm_inference_service.py:1230: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f9b18f301d0>, {'api_version': 'serving.kserve.io/v1alpha1',\n 'kin...outer-with-ec5d4bfa'},\n                       {'name': 'model-fb-opt-125m-router-with-r-6d64416a'}]},\n 'status': None})\nkwargs = {}, func_name = 'get_llm_service_url'\ntimestamp_start = '2026-07-30T19:03:10.801647', start_time = 1785438190.802384\nduration = 392.7594265937805, timestamp_end = '2026-07-30T19:09:43.561814'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f9b18f301d0>\nllm_isvc = {'api_version': 'serving.kserve.io/v1alpha1',\n 'kind': 'LLMInferenceService',\n 'metadata': {'annotations': {'security....router-with-ec5d4bfa'},\n                       {'name': 'model-fb-opt-125m-router-with-r-6d64416a'}]},\n 'status': None}\n\n    @log_execution\n    def get_llm_service_url(\n        kserve_client: KServeClient, llm_isvc: V1alpha1LLMInferenceService\n    ):\n        service_name = llm_isvc.metadata.name\n    \n        try:\n            llm_isvc = get_llmisvc(\n                kserve_client,\n                llm_isvc.metadata.name,\n                llm_isvc.metadata.namespace,\n                llm_isvc.api_version.split(\"/\")[1],\n            )\n    \n            if \"status\" not in llm_isvc:\n                raise ValueError(\n                    f\"\u274c No status found in LLM inference service {service_name} status: {llm_isvc}\"\n                )\n    \n            status = llm_isvc[\"status\"]\n    \n            if \"url\" in status and status[\"url\"]:\n                return status[\"url\"]\n    \n            if (\n                \"addresses\" in status\n                and status[\"addresses\"]\n                and len(status[\"addresses\"]) > 0\n            ):\n                first_address = status[\"addresses\"][0]\n                if \"url\" in first_address:\n                    return first_address[\"url\"]\n    \n            raise ValueError(\n                f\"\u274c No URL found in LLM inference service {service_name} status\"\n            )\n    \n        except Exception as e:\n>           raise ValueError(\n                f\"\u274c Failed to get URL for LLM inference service {service_name}: {e}\"\n            ) from e\nE           ValueError: \u274c Failed to get URL for LLM inference service router-with-refs-test: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))\n\nllmisvc/test_llm_inference_service.py:1327: ValueError\n\nThe above exception was the direct cause of the following exception:\n\ntest_case = TestCase(base_refs=['router-with-refs', 'scheduler-managed', 'workload-single-cpu', 'model-fb-opt-125m'], prompt='KSer...              {'name': 'model-fb-opt-125m-router-with-r-6d64416a'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.asyncio(loop_scope=\"session\")\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-with-gateway-ref\",\n                        \"router-with-managed-route\",\n                        \"model-fb-opt-125m\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/completions\",\n                    prompt=\"KServe is a\",\n                    payload_formatter=completions_payload,\n                    response_assertion=create_response_assertion(with_field=\"choices\"),\n                    expected_gateway=\"router-gateway-1\",\n                    before_test=[\n                        lambda tc: create_router_resources(\n                            gateways=[\n                                make_router_gateway(\n                                    \"router-gateway-1\",\n                                    tc.namespace,\n                                ),\n                            ],\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                    pytest.mark.custom_gateway,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    payload_formatter=completions_payload,\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-custom-route-timeout\",\n                        \"scheduler-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"custom-route-timeout-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-with-refs\",\n                        \"scheduler-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"router-with-refs-test\",\n                    expected_gateway=\"router-gateway-1\",\n                    before_test=[\n                        lambda tc: create_router_resources(\n                            gateways=[\n                                make_router_gateway(\n                                    \"router-gateway-1\",\n                                    tc.namespace,\n                                ),\n                            ],\n                            routes=[\n                                make_router_main_route(\n                                    \"router-route-1\",\n                                    tc.namespace,\n                                    \"router-gateway-1\",\n                                    \"router-with-refs-test\",\n                                ),\n                                make_router_health_route(\n                                    \"router-route-2\",\n                                    tc.namespace,\n                                    \"router-gateway-1\",\n                                    \"router-with-refs-test\",\n                                ),\n                            ],\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.custom_gateway,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\"router-managed\", \"workload-pd-cpu\", \"model-fb-opt-125m\"],\n                    prompt=\"You are an expert in Kubernetes-native machine learning serving platforms, with deep knowledge of the KServe project. \"\n                    \"Explain the challenges of serving large-scale models, GPU scheduling, and how KServe integrates with capabilities like multi-model serving. \"\n                    \"Provide a detailed comparison with open source alternatives, focusing on operational trade-offs.\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-custom-route-timeout-pd\",\n                        \"scheduler-managed\",\n                        \"workload-pd-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"You are an expert in Kubernetes-native machine learning serving platforms, with deep knowledge of the KServe project. \"\n                    \"Explain the challenges of serving large-scale models, GPU scheduling, and how KServe integrates with capabilities like multi-model serving. \"\n                    \"Provide a detailed comparison with open source alternatives, focusing on operational trade-offs.\",\n                    service_name=\"custom-route-timeout-pd-test\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-with-refs-pd\",\n                        \"scheduler-managed\",\n                        \"workload-pd-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"You are an expert in Kubernetes-native machine learning serving platforms, with deep knowledge of the KServe project. \"\n                    \"Explain the challenges of serving large-scale models, GPU scheduling, and how KServe integrates with capabilities like multi-model serving. \"\n                    \"Provide a detailed comparison with open source alternatives, focusing on operational trade-offs.\",\n                    service_name=\"router-with-refs-pd-test\",\n                    response_assertion=assert_200_with_choices,\n                    expected_gateway=\"router-gateway-2\",\n                    before_test=[\n                        lambda tc: create_router_resources(\n                            gateways=[\n                                make_router_gateway(\n                                    \"router-gateway-2\",\n                                    tc.namespace,\n                                ),\n                            ],\n                            routes=[\n                                make_router_main_route(\n                                    \"router-route-3\",\n                                    tc.namespace,\n                                    \"router-gateway-2\",\n                                    \"router-with-refs-pd-test\",\n                                ),\n                                make_router_health_route(\n                                    \"router-route-4\",\n                                    tc.namespace,\n                                    \"router-gateway-2\",\n                                    \"router-with-refs-pd-test\",\n                                ),\n                            ],\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.custom_gateway,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-dp-ep-gpu\",\n                        \"workload-dp-ep-prefill-gpu\",\n                        \"model-deepseek-v2-lite\",\n                    ],\n                    prompt=\"Delve into the multifaceted implications of a fully disaggregated cloud architecture, specifically \"\n                    \"where the compute plane (P) and the data plane (D) are independently deployed and managed for a \"\n                    \"geographically distributed, high-throughput, low-latency microservices ecosystem. Beyond the \"\n                    \"fundamental challenges of network latency and data consistency, elaborate on the advanced \"\n                    \"considerations and trade-offs inherent in such a setup: 1. Network Architecture and Protocols: \"\n                    \"How would the network fabric and underlying protocols (e.g., RDMA, custom transport layers) need to \"\n                    \"evolve to support optimal performance and minimize inter-plane communication overhead, especially for \"\n                    \"synchronous operations? Discuss the role of network programmability (e.g., SDN, P4) in dynamically \"\n                    \"optimizing routing and traffic flow between P and D. 2. Advanced Data Consistency and Durability: \"\n                    \"Explore sophisticated data consistency models (e.g., causal consistency, strong eventual consistency) \"\n                    \"and their applicability in balancing performance and data integrity across a globally distributed data plane. \"\n                    \"Detail strategies for ensuring data durability and fault tolerance, including multi-region replication, \"\n                    \"intelligent partitioning, and recovery mechanisms in the event of partial or full plane failures. \"\n                    \"3. Dynamic Resource Orchestration and Cost Optimization: Analyze how an orchestration layer would intelligently \"\n                    \"manage the independent scaling of compute (P) and data (D) resources, considering fluctuating workloads, \"\n                    \"cost efficiency, and performance targets (e.g., using predictive analytics for resource provisioning). \"\n                    \"Discuss mechanisms for dynamically reallocating compute nodes to different data partitions based on \"\n                    \"workload patterns and data locality, potentially involving live migration strategies. \"\n                    \"4. Security and Compliance in a Distributed Landscape: Address the enhanced security perimeter \"\n                    \"challenges, including securing communication channels between P and D (encryption in transit, mutual TLS), \"\n                    \"fine-grained access control to data at rest and in motion, and identity management across disaggregated \"\n                    \"components. Discuss how such an architecture impacts compliance with regulatory frameworks (e.g., GDPR, HIPAA) \"\n                    \"concerning data sovereignty, privacy, and auditability. 5. Operational Complexity and Observability: \"\n                    \"Examine the increased complexity in monitoring, logging, and tracing across highly decoupled compute and \"\n                    \"data planes. What specialized tooling and practices (e.g., distributed tracing with OpenTelemetry, advanced AIOps) \"\n                    \"would be essential? How would incident response and troubleshooting differ in this disaggregated environment \"\n                    \"compared to traditional integrated systems? Consider the challenges of pinpointing root causes across \"\n                    \"independent failures. 6. Real-world Applicability and Future Trends: Identify specific industries \"\n                    \"or use cases (e.g., high-frequency trading, IoT edge processing, large language model inference) \"\n                    \"where the benefits of P/D disaggregation would strongly outweigh its complexities. \"\n                    \"Conclude by speculating on emerging technologies or paradigms (e.g., serverless compute functions \"\n                    \"directly interacting with object storage, in-memory disaggregation) that could further drive or \"\n                    \"transform P/D disaggregation in cloud computing.\",\n                    max_tokens=2000,\n                ),\n                marks=[\n                    pytest.mark.cluster_gpu,\n                    pytest.mark.cluster_nvidia,\n                    pytest.mark.cluster_nvidia_roce,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-no-scheduler\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"What is KServe?\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.no_scheduler,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-simulated-dp-ep-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"This test simulates DP+EP that can run on CPU, the idea is to test the LWS-based deployment, \"\n                    \"but without the resources requirements for DP+EP (GPUs and ROCe/IB).\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_multi_node],\n            ),\n            # Scheduler config tests\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-inline-config\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-inline-config-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            # Chat completions endpoint coverage\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                        \"model-qwen2.5-0.5b\",\n                    ],\n                    model_name=\"Qwen/Qwen2.5-0.5B-Instruct\",\n                    endpoint=\"/v1/chat/completions\",\n                    prompt=\"What is KServe?\",\n                    payload_formatter=chat_completions_payload,\n                    response_assertion=create_response_assertion(with_field=\"choices\"),\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-configmap-ref\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-configmap-ref-test\",\n                    before_test=[\n                        lambda tc: create_scheduler_configmap(namespace=tc.namespace)\n                    ],\n                    after_test=[\n                        lambda tc: delete_scheduler_configmap(namespace=tc.namespace)\n                    ],\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-replicas\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-ha-replicas-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-custom-template\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-custom-template-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            # Scheduler v0.6 \u2192 v0.7 migration tests.\n            # Deploy v0.6-style configs and verify the controller migrates them\n            # so the v0.7 scheduler boots successfully.\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-v06-pd-config-migration\",\n                        \"workload-llmd-simulator-pd\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-v06-pd-migration-test\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-v06-nonzero-threshold-migration\",\n                        \"workload-llmd-simulator-pd\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-v06-threshold-migration-test\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Standalone tokenizer \u2014 clean path: token-producer in inline config\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-tokenizer-kvcache\",\n                        \"workload-llmd-simulator-kvcache\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"tokenizer-clean-path-test\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Standalone tokenizer \u2014 migration path: legacy precise-prefix-cache-scorer\n            # triggers auto-provisioned tokenizer without explicit tokenizer:{} field\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-precise-prefix-cache-inline-config\",\n                        \"workload-llmd-simulator-kvcache\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"tokenizer-migration-path-test\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Models endpoint coverage\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/models\",\n                    response_assertion=create_response_assertion(with_field=\"data\"),\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 /v1/completions\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/completions\",\n                    prompt=\"KServe is a\",\n                    payload_formatter=completions_payload,\n                    response_assertion=assert_model_field_matches(\"facebook/opt-125m\"),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/facebook/opt-125m\",\n                    },\n                    peers=[\n                        TestCase(\n                            base_refs=[\n                                \"router-managed\",\n                                \"workload-llmd-simulator\",\n                                \"model-qwen2.5-0.5b\",\n                            ],\n                            endpoint=\"/v1/completions\",\n                            prompt=\"KServe is a\",\n                            payload_formatter=completions_payload,\n                            response_assertion=assert_model_field_matches(\n                                \"Qwen/Qwen2.5-0.5B-Instruct\"\n                            ),\n                            url_getter=get_model_routing_url,\n                            extra_headers={\n                                MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/Qwen/Qwen2.5-0.5B-Instruct\",\n                            },\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                    pytest.mark.model_routing,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 /v1/chat/completions\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/chat/completions\",\n                    prompt=\"What is KServe?\",\n                    payload_formatter=chat_completions_payload,\n                    response_assertion=assert_model_field_matches(\"facebook/opt-125m\"),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/facebook/opt-125m\",\n                    },\n                    peers=[\n                        TestCase(\n                            base_refs=[\n                                \"router-managed\",\n                                \"workload-llmd-simulator\",\n                                \"model-qwen2.5-0.5b\",\n                            ],\n                            endpoint=\"/v1/chat/completions\",\n                            prompt=\"What is KServe?\",\n                            payload_formatter=chat_completions_payload,\n                            response_assertion=assert_model_field_matches(\n                                \"Qwen/Qwen2.5-0.5B-Instruct\"\n                            ),\n                            url_getter=get_model_routing_url,\n                            extra_headers={\n                                MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/Qwen/Qwen2.5-0.5B-Instruct\",\n                            },\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                    pytest.mark.model_routing,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 LoRA adapter\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m-with-lora-hf\",\n                    ],\n                    endpoint=\"/v1/completions\",\n                    prompt=\"KServe is a\",\n                    model_name=\"publishers/{namespace}/models/lora-adapter-1\",\n                    payload_formatter=completions_payload,\n                    response_assertion=assert_model_field_matches(\n                        \"publishers/{namespace}/models/lora-adapter-1\"\n                    ),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/lora-adapter-1\",\n                    },\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.model_routing,\n                    pytest.mark.lora,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 /v1/models (base + LoRA)\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m-with-lora-hf\",\n                    ],\n                    endpoint=\"/v1/models\",\n                    response_assertion=assert_models_contains(\n                        \"facebook/opt-125m\",\n                        \"publishers/{namespace}/models/facebook/opt-125m\",\n                        \"lora-adapter-1\",\n                        \"publishers/{namespace}/models/lora-adapter-1\",\n                    ),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: \"publishers/{namespace}/models/facebook/opt-125m\",\n                    },\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.model_routing,\n                    pytest.mark.lora,\n                ],\n            ),\n            # PVC storage tests -- validate direct PVC volume mount with real vLLM serving\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-pvc\",\n                    ],\n                    prompt=\"KServe is a\",\n                    response_assertion=assert_200_with_choices,\n                    before_test=[lambda tc: ensure_pvc_with_model(namespace=tc.namespace)],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.pvc_storage,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-pd-cpu\",\n                        \"model-pvc\",\n                    ],\n                    prompt=\"KServe is a\",\n                    response_assertion=assert_200_with_choices,\n                    before_test=[lambda tc: ensure_pvc_with_model(namespace=tc.namespace)],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.pvc_storage,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-simulated-dp-ep-cpu\",\n                        \"model-pvc\",\n                    ],\n                    prompt=\"KServe is a\",\n                    before_test=[lambda tc: ensure_pvc_with_model(namespace=tc.namespace)],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_multi_node,\n                    pytest.mark.pvc_storage,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_inference_service(test_case: TestCase):  # noqa: F811\n        inject_k8s_proxy()\n    \n        kserve_client = KServeClient(\n            config_file=os.environ.get(\"KUBECONFIG\", \"~/.kube/config\"),\n            client_configuration=client.Configuration(),\n        )\n    \n        service_name = test_case.llm_service.metadata.name\n        prefix = test_case.log_prefix\n    \n        test_failed = False\n        try:\n            print(f\"{prefix} Creating LLMInferenceService {service_name}\")\n            create_llmisvc(kserve_client, test_case.llm_service)\n            print(f\"{prefix} Waiting for LLMInferenceService {service_name} to be ready\")\n            wait_for_llm_isvc_ready(\n                kserve_client, test_case.llm_service, test_case.wait_timeout\n            )\n            print(f\"{prefix} Waiting for model response from {service_name}\")\n>           wait_for_model_response(\n                kserve_client,\n                test_case,\n                test_case.wait_timeout,\n                extra_headers=test_case.extra_headers,\n            )\n\nllmisvc/test_llm_inference_service.py:870: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7f9b18f301d0>, TestCase(base_refs=['router-with-refs', 'scheduler-...        {'name': 'model-fb-opt-125m-router-with-r-6d64416a'}]},\n 'status': None}, model_name='facebook/opt-125m'), 900)\nkwargs = {'extra_headers': None}, func_name = 'wait_for_model_response'\ntimestamp_start = '2026-07-30T18:51:28.654173', start_time = 1785437488.6544714\nduration = 1094.907615661621, timestamp_end = '2026-07-30T19:09:43.562088'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7f9b18f301d0>\ntest_case = TestCase(base_refs=['router-with-refs', 'scheduler-managed', 'workload-single-cpu', 'model-fb-opt-125m'], prompt='KSer...              {'name': 'model-fb-opt-125m-router-with-r-6d64416a'}]},\n 'status': None}, model_name='facebook/opt-125m')\ntimeout_seconds = 900, extra_headers = None\n\n    @log_execution\n    def wait_for_model_response(\n        kserve_client: KServeClient,\n        test_case: TestCase,  # noqa: F811\n        timeout_seconds: int = 900,\n        extra_headers: Optional[Dict[str, str]] = None,\n    ) -> str:\n        def get_successful_response():\n            try:\n                if test_case.url_getter:\n                    service_url = test_case.url_getter(kserve_client, test_case.llm_service)\n                else:\n                    service_url = get_llm_service_url(kserve_client, test_case.llm_service)\n            except Exception as e:\n                raise AssertionError(f\"\u274c Failed to get service URL: {e}\") from e\n    \n            model_url = service_url + test_case.endpoint\n    \n            headers = {\"Content-Type\": \"application/json\"}\n            ns = test_case.namespace or \"\"\n            resolved_headers = (\n                {k: v.format(namespace=ns) for k, v in extra_headers.items()}\n                if extra_headers\n                else {}\n            )\n            headers.update(resolved_headers)\n            if test_case.payload_formatter is not None:\n                test_payload = test_case.payload_formatter(test_case)\n            elif test_case.prompt is not None:\n                test_payload = {\n                    \"model\": resolved_headers.get(\n                        MODEL_ROUTING_HEADER, test_case.model_name\n                    ),\n                    \"prompt\": test_case.prompt,\n                    \"max_tokens\": test_case.max_tokens,\n                }\n            else:\n                test_payload = None\n    \n            logger.info(f\"Calling LLM service at {model_url} with payload {test_payload}\")\n            try:\n                if test_payload is not None:\n                    response = post_with_retry(\n                        model_url,\n                        headers=headers,\n                        json_data=test_payload,\n                        timeout=test_case.response_timeout,\n                    )\n                else:\n                    response = get_with_retry(\n                        model_url,\n                        headers=headers,\n                        timeout=test_case.response_timeout,\n                    )\n            except Exception as e:\n                logger.error(f\"\u274c Failed to call model: {e}\")\n                raise AssertionError(f\"\u274c Failed to call model: {e}\") from e\n    \n            logger.info(f\"Model response is {response.status_code}: {response.text[:500]}\")\n    \n            if 200 <= response.status_code < 300:\n                return response\n            raise AssertionError(\n                f\"Service returned {response.status_code}: {response.text}\"\n            )\n    \n>       response = wait_for(get_successful_response, timeout=timeout_seconds, interval=5.0)\n\nllmisvc/test_llm_inference_service.py:1284: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_model_response.<locals>.get_successful_response at 0x7f9b13c33740>\ntimeout = 900, interval = 5.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1387: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def get_successful_response():\n        try:\n            if test_case.url_getter:\n                service_url = test_case.url_getter(kserve_client, test_case.llm_service)\n            else:\n                service_url = get_llm_service_url(kserve_client, test_case.llm_service)\n        except Exception as e:\n>           raise AssertionError(f\"\u274c Failed to get service URL: {e}\") from e\nE           AssertionError: \u274c Failed to get service URL: \u274c Failed to get URL for LLM inference service router-with-refs-test: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /apis/serving.kserve.io/v1alpha1/namespaces/e2e-test-llm-inference-service-028f7809/llminferenceservices/router-with-refs-test (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))\n\nllmisvc/test_llm_inference_service.py:1232: AssertionError"}, "teardown": {"duration": 0.002288808987941593, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "error", "keywords": ["test_llm_inference_service[router-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-pd-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.028685352997854352, "outcome": "failed", "crash": {"path": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py", "lineno": 543, "message": "urllib3.exceptions.MaxRetryError: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /api/v1/namespaces (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))"}, "traceback": [{"path": "common/gateway_proxy_istio.py", "lineno": 183, "message": ""}, {"path": "llmisvc/conftest.py", "lineno": 159, "message": ""}, {"path": "llmisvc/namespace.py", "lineno": 81, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py", "lineno": 6363, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py", "lineno": 6454, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 348, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 180, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 391, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 279, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 172, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py", "lineno": 143, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py", "lineno": 278, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/poolmanager.py", "lineno": 457, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 872, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 872, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 872, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 842, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py", "lineno": 543, "message": "MaxRetryError"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b1325e110>\n\n    def _new_conn(self) -> socket.socket:\n        \"\"\"Establish a socket connection and set nodelay settings on it.\n    \n        :return: New socket connection.\n        \"\"\"\n        try:\n>           sock = connection.create_connection(\n                (self._dns_host, self.port),\n                self.timeout,\n                source_address=self.source_address,\n                socket_options=self.socket_options,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:204: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\naddress = ('a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', 6443)\ntimeout = None, source_address = None, socket_options = [(6, 1, 1)]\n\n    def create_connection(\n        address: tuple[str, int],\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        source_address: tuple[str, int] | None = None,\n        socket_options: _TYPE_SOCKET_OPTIONS | None = None,\n    ) -> socket.socket:\n        \"\"\"Connect to *address* and return the socket object.\n    \n        Convenience function.  Connect to *address* (a 2-tuple ``(host,\n        port)``) and return the socket object.  Passing the optional\n        *timeout* parameter will set the timeout on the socket instance\n        before attempting to connect.  If no *timeout* is supplied, the\n        global default timeout setting returned by :func:`socket.getdefaulttimeout`\n        is used.  If *source_address* is set it must be a tuple of (host, port)\n        for the socket to bind as a source address before making the connection.\n        An host of '' or port 0 tells the OS to use the default.\n        \"\"\"\n    \n        host, port = address\n        if host.startswith(\"[\"):\n            host = host.strip(\"[]\")\n        err = None\n    \n        # Using the value from allowed_gai_family() in the context of getaddrinfo lets\n        # us select whether to work with IPv4 DNS records, IPv6 records, or both.\n        # The original create_connection function always returns all records.\n        family = allowed_gai_family()\n    \n        try:\n            host.encode(\"idna\")\n        except UnicodeError:\n            raise LocationParseError(f\"'{host}', label empty or too long\") from None\n    \n>       for res in socket.getaddrinfo(host, port, family, socket.SOCK_STREAM):\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/connection.py:60: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nhost = 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com'\nport = 6443, family = <AddressFamily.AF_UNSPEC: 0>\ntype = <SocketKind.SOCK_STREAM: 1>, proto = 0, flags = 0\n\n    def getaddrinfo(host, port, family=0, type=0, proto=0, flags=0):\n        \"\"\"Resolve host and port into list of address info entries.\n    \n        Translate the host/port argument into a sequence of 5-tuples that contain\n        all the necessary arguments for creating a socket connected to that service.\n        host is a domain name, a string representation of an IPv4/v6 address or\n        None. port is a string service name such as 'http', a numeric port number or\n        None. By passing None as the value of host and port, you can pass NULL to\n        the underlying C API.\n    \n        The family, type and proto arguments can be optionally specified in order to\n        narrow the list of addresses returned. Passing zero as a value for each of\n        these arguments selects the full range of results.\n        \"\"\"\n        # We override this function since we want to translate the numeric family\n        # and socket type values to enum constants.\n        addrlist = []\n>       for res in _socket.getaddrinfo(host, port, family, type, proto, flags):\nE       socket.gaierror: [Errno -2] Name or service not known\n\n/usr/lib64/python3.11/socket.py:974: gaierror\n\nThe above exception was the direct cause of the following exception:\n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False, err = None\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n>           response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:788: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b1325e110>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\ntimeout = Timeout(connect=None, read=None, total=None), chunked = False\nresponse_conn = None, preload_content = True, decode_content = True\nenforce_content_length = True\n\n    def _make_request(\n        self,\n        conn: BaseHTTPConnection,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | None = None,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        chunked: bool = False,\n        response_conn: BaseHTTPConnection | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        enforce_content_length: bool = True,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Perform a request on a given urllib connection object taken from our\n        pool.\n    \n        :param conn:\n            a connection from one of our connection pools\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            Pass ``None`` to retry until you receive a response. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param response_conn:\n            Set this to ``None`` if you will handle releasing the connection or\n            set the connection to have the response release it.\n    \n        :param preload_content:\n          If True, the response's body will be preloaded during construction.\n    \n        :param decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param enforce_content_length:\n            Enforce content length checking. Body returned by server must match\n            value of Content-Length header, if present. Otherwise, raise error.\n        \"\"\"\n        self.num_requests += 1\n    \n        timeout_obj = self._get_timeout(timeout)\n        timeout_obj.start_connect()\n        conn.timeout = Timeout.resolve_default_timeout(timeout_obj.connect_timeout)\n    \n        try:\n            # Trigger any extra validation we need to do.\n            try:\n                self._validate_conn(conn)\n            except (SocketTimeout, BaseSSLError) as e:\n                self._raise_timeout(err=e, url=url, timeout_value=conn.timeout)\n                raise\n    \n        # _validate_conn() starts the connection to an HTTPS proxy\n        # so we need to wrap errors with 'ProxyError' here too.\n        except (\n            OSError,\n            NewConnectionError,\n            TimeoutError,\n            BaseSSLError,\n            CertificateError,\n            SSLError,\n        ) as e:\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            # If the connection didn't successfully connect to it's proxy\n            # then there\n            if isinstance(\n                new_e, (OSError, NewConnectionError, TimeoutError, SSLError)\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n>           raise new_e\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:488: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b1325e110>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\ntimeout = Timeout(connect=None, read=None, total=None), chunked = False\nresponse_conn = None, preload_content = True, decode_content = True\nenforce_content_length = True\n\n    def _make_request(\n        self,\n        conn: BaseHTTPConnection,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | None = None,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        chunked: bool = False,\n        response_conn: BaseHTTPConnection | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        enforce_content_length: bool = True,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Perform a request on a given urllib connection object taken from our\n        pool.\n    \n        :param conn:\n            a connection from one of our connection pools\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            Pass ``None`` to retry until you receive a response. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param response_conn:\n            Set this to ``None`` if you will handle releasing the connection or\n            set the connection to have the response release it.\n    \n        :param preload_content:\n          If True, the response's body will be preloaded during construction.\n    \n        :param decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param enforce_content_length:\n            Enforce content length checking. Body returned by server must match\n            value of Content-Length header, if present. Otherwise, raise error.\n        \"\"\"\n        self.num_requests += 1\n    \n        timeout_obj = self._get_timeout(timeout)\n        timeout_obj.start_connect()\n        conn.timeout = Timeout.resolve_default_timeout(timeout_obj.connect_timeout)\n    \n        try:\n            # Trigger any extra validation we need to do.\n            try:\n>               self._validate_conn(conn)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:464: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b1325e110>\n\n    def _validate_conn(self, conn: BaseHTTPConnection) -> None:\n        \"\"\"\n        Called right before a request is made, after the socket is created.\n        \"\"\"\n        super()._validate_conn(conn)\n    \n        # Force connect early to allow us to validate the connection.\n        if conn.is_closed:\n>           conn.connect()\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:1106: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b1325e110>\n\n    def connect(self) -> None:\n        # Today we don't need to be doing this step before the /actual/ socket\n        # connection, however in the future we'll need to decide whether to\n        # create a new socket or re-use an existing \"shared\" socket as a part\n        # of the HTTP/2 handshake dance.\n        if self._tunnel_host is not None and self._tunnel_port is not None:\n            probe_http2_host = self._tunnel_host\n            probe_http2_port = self._tunnel_port\n        else:\n            probe_http2_host = self.host\n            probe_http2_port = self.port\n    \n        # Check if the target origin supports HTTP/2.\n        # If the value comes back as 'None' it means that the current thread\n        # is probing for HTTP/2 support. Otherwise, we're waiting for another\n        # probe to complete, or we get a value right away.\n        target_supports_http2: bool | None\n        if \"h2\" in ssl_.ALPN_PROTOCOLS:\n            target_supports_http2 = http2_probe.acquire_and_get(\n                host=probe_http2_host, port=probe_http2_port\n            )\n        else:\n            # If HTTP/2 isn't going to be offered it doesn't matter if\n            # the target supports HTTP/2. Don't want to make a probe.\n            target_supports_http2 = False\n    \n        if self._connect_callback is not None:\n            self._connect_callback(\n                \"before connect\",\n                thread_id=threading.get_ident(),\n                target_supports_http2=target_supports_http2,\n            )\n    \n        try:\n            sock: socket.socket | ssl.SSLSocket\n>           self.sock = sock = self._new_conn()\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:759: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b1325e110>\n\n    def _new_conn(self) -> socket.socket:\n        \"\"\"Establish a socket connection and set nodelay settings on it.\n    \n        :return: New socket connection.\n        \"\"\"\n        try:\n            sock = connection.create_connection(\n                (self._dns_host, self.port),\n                self.timeout,\n                source_address=self.source_address,\n                socket_options=self.socket_options,\n            )\n        except socket.gaierror as e:\n>           raise NameResolutionError(self.host, self, e) from e\nE           urllib3.exceptions.NameResolutionError: HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:211: NameResolutionError\n\nThe above exception was the direct cause of the following exception:\n\nrequest = <SubRequest 'ensure_gateway_proxy_memory' for <Function test_llm_inference_service[router-managed-workload-pd-cpu-model-fb-opt-125m]>>\n\n    @pytest.fixture(autouse=True)\n    def ensure_gateway_proxy_memory(request):\n        \"\"\"After test setup creates gateways, patch them for proxy memory.\"\"\"\n        if not GATEWAY_PROXY_MEMORY:\n            return\n    \n        # Let test_case (llmisvc) create gateways first\n    \n        if \"test_case\" in request.fixturenames:\n>           request.getfixturevalue(\"test_case\")\n\ncommon/gateway_proxy_istio.py:183: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nrequest = <SubRequest 'test_namespace' for <Function test_llm_inference_service[router-managed-workload-pd-cpu-model-fb-opt-125m]>>\n\n    @pytest.fixture(scope=\"function\")\n    def test_namespace(request):\n        \"\"\"Create a per-test namespace with secrets, clean up after the test.\"\"\"\n        inject_k8s_proxy()\n        ns = generate_namespace_name(request.node.name)\n>       create_test_namespace(ns)\n\nllmisvc/conftest.py:159: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nnamespace = 'e2e-test-llm-inference-service-62876e81'\n\n    def create_test_namespace(namespace: str) -> None:\n        \"\"\"Create a labeled namespace for a single test.\"\"\"\n        core_v1 = client.CoreV1Api()\n        ns = client.V1Namespace(\n            metadata=client.V1ObjectMeta(\n                name=namespace,\n                labels={\n                    TEST_NAMESPACE_LABEL_KEY: TEST_NAMESPACE_LABEL_VALUE,\n                },\n            )\n        )\n        try:\n>           core_v1.create_namespace(ns)\n\nllmisvc/namespace.py:81: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.core_v1_api.CoreV1Api object at 0x7f9b13ab5690>\nbody = {'api_version': None,\n 'kind': None,\n 'metadata': {'annotations': None,\n              'creation_timestamp': None,\n    ... 'resource_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}\nkwargs = {'_return_http_data_only': True}\n\n    def create_namespace(self, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespace  # noqa: E501\n    \n        create a Namespace  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespace(body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param V1Namespace body: (required)\n        :param str pretty: If 'true', then the output is pretty printed. Defaults to 'false' unless the user-agent indicates a browser or command-line HTTP tool (curl and wget).\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: V1Namespace\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.create_namespace_with_http_info(body, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py:6363: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.core_v1_api.CoreV1Api object at 0x7f9b13ab5690>\nbody = {'api_version': None,\n 'kind': None,\n 'metadata': {'annotations': None,\n              'creation_timestamp': None,\n    ... 'resource_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['body', 'pretty', 'dry_run', 'field_manager', 'field_validation', 'asy...urce_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}, ...}\nall_params = ['body', 'pretty', 'dry_run', 'field_manager', 'field_validation', 'async_req', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {}, query_params = []\n\n    def create_namespace_with_http_info(self, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespace  # noqa: E501\n    \n        create a Namespace  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespace_with_http_info(body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param V1Namespace body: (required)\n        :param str pretty: If 'true', then the output is pretty printed. Defaults to 'false' unless the user-agent indicates a browser or command-line HTTP tool (curl and wget).\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered.\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(V1Namespace, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'body',\n            'pretty',\n            'dry_run',\n            'field_manager',\n            'field_validation'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method create_namespace\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'body' is set\n        if self.api_client.client_side_validation and ('body' not in local_var_params or  # noqa: E501\n                                                        local_var_params['body'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `body` when calling `create_namespace`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n    \n        query_params = []\n        if 'pretty' in local_var_params and local_var_params['pretty'] is not None:  # noqa: E501\n            query_params.append(('pretty', local_var_params['pretty']))  # noqa: E501\n        if 'dry_run' in local_var_params and local_var_params['dry_run'] is not None:  # noqa: E501\n            query_params.append(('dryRun', local_var_params['dry_run']))  # noqa: E501\n        if 'field_manager' in local_var_params and local_var_params['field_manager'] is not None:  # noqa: E501\n            query_params.append(('fieldManager', local_var_params['field_manager']))  # noqa: E501\n        if 'field_validation' in local_var_params and local_var_params['field_validation'] is not None:  # noqa: E501\n            query_params.append(('fieldValidation', local_var_params['field_validation']))  # noqa: E501\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        if 'body' in local_var_params:\n            body_params = local_var_params['body']\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json', 'application/yaml', 'application/vnd.kubernetes.protobuf', 'application/cbor'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/api/v1/namespaces', 'POST',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='V1Namespace',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py:6454: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b13ab4e50>\nresource_path = '/api/v1/namespaces', method = 'POST', path_params = {}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'api_version': None,\n 'kind': None,\n 'metadata': {'annotations': None,\n              'creation_timestamp': None,\n    ... 'resource_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}\npost_params = [], files = {}, response_type = 'V1Namespace'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b13ab4e50>\nresource_path = '/api/v1/namespaces', method = 'POST', path_params = {}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-62876e81'}}\npost_params = [], files = {}, response_type = 'V1Namespace'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b13ab4e50>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = []\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-62876e81'}}\n_preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n            return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n                                        headers=headers)\n        elif method == \"HEAD\":\n            return self.rest_client.HEAD(url,\n                                         query_params=query_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n                                         headers=headers)\n        elif method == \"OPTIONS\":\n            return self.rest_client.OPTIONS(url,\n                                            query_params=query_params,\n                                            headers=headers,\n                                            _preload_content=_preload_content,\n                                            _request_timeout=_request_timeout)\n        elif method == \"POST\":\n>           return self.rest_client.POST(url,\n                                         query_params=query_params,\n                                         headers=headers,\n                                         post_params=post_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:391: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f9b13ab5f10>\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], post_params = []\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-62876e81'}}\n_preload_content = True, _request_timeout = None\n\n    def POST(self, url, headers=None, query_params=None, post_params=None,\n             body=None, _preload_content=True, _request_timeout=None):\n>       return self.request(\"POST\", url,\n                            headers=headers,\n                            query_params=query_params,\n                            post_params=post_params,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            body=body)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:279: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f9b13ab5f10>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-62876e81'}}\npost_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n>                   r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:172: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b13ab4090>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}'\nfields = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\njson = None\nurlopen_kw = {'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}', 'preload_content': True, 'timeout': None}\n\n    def request(\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        fields: _TYPE_FIELDS | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        json: typing.Any | None = None,\n        **urlopen_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Make a request using :meth:`urlopen` with the appropriate encoding of\n        ``fields`` based on the ``method`` used.\n    \n        This is a convenience method that requires the least amount of manual\n        effort. It can be used in most situations, while still having the\n        option to drop down to more specific methods when necessary, such as\n        :meth:`request_encode_url`, :meth:`request_encode_body`,\n        or even the lowest level :meth:`urlopen`.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param fields:\n            Data to encode and send in the URL or request body, depending on ``method``.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param json:\n            Data to encode and send as JSON with UTF-encoded in the request body.\n            The ``\"Content-Type\"`` header will be set to ``\"application/json\"``\n            unless specified otherwise.\n        \"\"\"\n        method = method.upper()\n    \n        if json is not None and body is not None:\n            raise TypeError(\n                \"request got values for both 'body' and 'json' parameters which are mutually exclusive\"\n            )\n    \n        if json is not None:\n            if headers is None:\n                headers = self.headers\n    \n            if not (\"content-type\" in map(str.lower, headers.keys())):\n                headers = HTTPHeaderDict(headers)\n                headers[\"Content-Type\"] = \"application/json\"\n    \n            body = _json.dumps(json, separators=(\",\", \":\"), ensure_ascii=False).encode(\n                \"utf-8\"\n            )\n    \n        if body is not None:\n            urlopen_kw[\"body\"] = body\n    \n        if method in self._encode_url_methods:\n            return self.request_encode_url(\n                method,\n                url,\n                fields=fields,  # type: ignore[arg-type]\n                headers=headers,\n                **urlopen_kw,\n            )\n        else:\n>           return self.request_encode_body(\n                method, url, fields=fields, headers=headers, **urlopen_kw\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py:143: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b13ab4090>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nfields = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nencode_multipart = True, multipart_boundary = None\nurlopen_kw = {'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}', 'preload_content': True, 'timeout': None}\nextra_kw = {'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}'...nt': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'}), 'preload_content': True, 'timeout': None}\n\n    def request_encode_body(\n        self,\n        method: str,\n        url: str,\n        fields: _TYPE_FIELDS | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        encode_multipart: bool = True,\n        multipart_boundary: str | None = None,\n        **urlopen_kw: str,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Make a request using :meth:`urlopen` with the ``fields`` encoded in\n        the body. This is useful for request methods like POST, PUT, PATCH, etc.\n    \n        When ``encode_multipart=True`` (default), then\n        :func:`urllib3.encode_multipart_formdata` is used to encode\n        the payload with the appropriate content type. Otherwise\n        :func:`urllib.parse.urlencode` is used with the\n        'application/x-www-form-urlencoded' content type.\n    \n        Multipart encoding must be used when posting files, and it's reasonably\n        safe to use it in other times too. However, it may break request\n        signing, such as with OAuth.\n    \n        Supports an optional ``fields`` parameter of key/value strings AND\n        key/filetuple. A filetuple is a (filename, data, MIME type) tuple where\n        the MIME type is optional. For example::\n    \n            fields = {\n                'foo': 'bar',\n                'fakefile': ('foofile.txt', 'contents of foofile'),\n                'realfile': ('barfile.txt', open('realfile').read()),\n                'typedfile': ('bazfile.bin', open('bazfile').read(),\n                              'image/jpeg'),\n                'nonamefile': 'contents of nonamefile field',\n            }\n    \n        When uploading a file, providing a filename (the first parameter of the\n        tuple) is optional but recommended to best mimic behavior of browsers.\n    \n        Note that if ``headers`` are supplied, the 'Content-Type' header will\n        be overwritten because it depends on the dynamic random boundary string\n        which is used to compose the body of the request. The random boundary\n        string can be explicitly set with the ``multipart_boundary`` parameter.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param fields:\n            Data to encode and send in the request body.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param encode_multipart:\n            If True, encode the ``fields`` using the multipart/form-data MIME\n            format.\n    \n        :param multipart_boundary:\n            If not specified, then a random boundary will be generated using\n            :func:`urllib3.filepost.choose_boundary`.\n        \"\"\"\n        if headers is None:\n            headers = self.headers\n    \n        extra_kw: dict[str, typing.Any] = {\"headers\": HTTPHeaderDict(headers)}\n        body: bytes | str\n    \n        if fields:\n            if \"body\" in urlopen_kw:\n                raise TypeError(\n                    \"request got values for both 'fields' and 'body', can only specify one.\"\n                )\n    \n            if encode_multipart:\n                body, content_type = encode_multipart_formdata(\n                    fields, boundary=multipart_boundary\n                )\n            else:\n                body, content_type = (\n                    urlencode(fields),  # type: ignore[arg-type]\n                    \"application/x-www-form-urlencoded\",\n                )\n    \n            extra_kw[\"body\"] = body\n            extra_kw[\"headers\"].setdefault(\"Content-Type\", content_type)\n    \n        extra_kw.update(urlopen_kw)\n    \n>       return self.urlopen(method, url, **extra_kw)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py:278: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b13ab4090>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nredirect = True\nkw = {'assert_same_host': False, 'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inf...', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'}), 'preload_content': True, ...}\nu = Url(scheme='https', auth=None, host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443, path='/api/v1/namespaces', query=None, fragment=None)\nconn = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\n\n    def urlopen(  # type: ignore[override]\n        self, method: str, url: str, redirect: bool = True, **kw: typing.Any\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Same as :meth:`urllib3.HTTPConnectionPool.urlopen`\n        with custom cross-host redirect logic and only sends the request-uri\n        portion of the ``url``.\n    \n        The given ``url`` parameter must be absolute, such that an appropriate\n        :class:`urllib3.connectionpool.ConnectionPool` can be chosen for it.\n        \"\"\"\n        u = parse_url(url)\n    \n        if u.scheme is None:\n            warnings.warn(\n                \"URLs without a scheme (ie 'https://') are deprecated and will raise an error \"\n                \"in urllib3 v3.0. To avoid this FutureWarning ensure all URLs \"\n                \"start with 'https://' or 'http://'. Read more in this issue: \"\n                \"https://github.com/urllib3/urllib3/issues/2920\",\n                category=FutureWarning,\n                stacklevel=2,\n            )\n    \n        conn = self.connection_from_host(u.host, port=u.port, scheme=u.scheme)\n    \n        kw[\"assert_same_host\"] = False\n        kw[\"redirect\"] = False\n    \n        if \"headers\" not in kw:\n            kw[\"headers\"] = self.headers\n    \n        if self._proxy_requires_url_absolute_form(u):\n            response = conn.urlopen(method, url, **kw)\n        else:\n>           response = conn.urlopen(method, u.request_uri, **kw)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/poolmanager.py:457: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=2, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=1, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-62876e81\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False, err = None\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n>           retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:842: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nmethod = 'POST', url = '/api/v1/namespaces', response = None\nerror = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\n_pool = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b127ca650>\n_stacktrace = <traceback object at 0x7f9b1325e180>\n\n    def increment(\n        self,\n        method: str | None = None,\n        url: str | None = None,\n        response: BaseHTTPResponse | None = None,\n        error: Exception | None = None,\n        _pool: ConnectionPool | None = None,\n        _stacktrace: TracebackType | None = None,\n    ) -> Self:\n        \"\"\"Return a new Retry object with incremented retry counters.\n    \n        :param response: A response object, or None, if the server did not\n            return a response.\n        :type response: :class:`~urllib3.response.BaseHTTPResponse`\n        :param Exception error: An error encountered during the request, or\n            None if the response was received successfully.\n    \n        :return: A new ``Retry`` object.\n        \"\"\"\n        if self.total is False and error:\n            # Disabled, indicate to re-raise the error.\n            raise reraise(type(error), error, _stacktrace)\n    \n        total = self.total\n        if total is not None:\n            total -= 1\n    \n        connect = self.connect\n        read = self.read\n        redirect = self.redirect\n        status_count = self.status\n        other = self.other\n        cause = \"unknown\"\n        status = None\n        redirect_location = None\n    \n        if error and self._is_connection_error(error):\n            # Connect retry?\n            if connect is False:\n                raise reraise(type(error), error, _stacktrace)\n            elif connect is not None:\n                connect -= 1\n    \n        elif error and self._is_read_error(error):\n            # Read retry?\n            if read is False or method is None or not self._is_method_retryable(method):\n                raise reraise(type(error), error, _stacktrace)\n            elif read is not None:\n                read -= 1\n    \n        elif error:\n            # Other retry?\n            if other is not None:\n                other -= 1\n    \n        elif response and response.get_redirect_location():\n            # Redirect retry?\n            if redirect is not None:\n                redirect -= 1\n            cause = \"too many redirects\"\n            response_redirect_location = response.get_redirect_location()\n            if response_redirect_location:\n                redirect_location = response_redirect_location\n            status = response.status\n    \n        else:\n            # Incrementing because of a server error like a 500 in\n            # status_forcelist and the given method is in the allowed_methods\n            cause = ResponseError.GENERIC_ERROR\n            if response and response.status:\n                if status_count is not None:\n                    status_count -= 1\n                cause = ResponseError.SPECIFIC_ERROR.format(status_code=response.status)\n                status = response.status\n    \n        history = self.history + (\n            RequestHistory(method, url, error, status, redirect_location),\n        )\n    \n        new_retry = self.new(\n            total=total,\n            connect=connect,\n            read=read,\n            redirect=redirect,\n            status=status_count,\n            other=other,\n            history=history,\n        )\n    \n        if new_retry.is_exhausted():\n            reason = error or ResponseError(cause)\n>           raise MaxRetryError(_pool, url, reason) from reason  # type: ignore[arg-type]\nE           urllib3.exceptions.MaxRetryError: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /api/v1/namespaces (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py:543: MaxRetryError"}, "teardown": {"duration": 0.0002850159944500774, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "error", "keywords": ["test_llm_inference_service[router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.02852108998922631, "outcome": "failed", "crash": {"path": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py", "lineno": 543, "message": "urllib3.exceptions.MaxRetryError: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /api/v1/namespaces (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))"}, "traceback": [{"path": "common/gateway_proxy_istio.py", "lineno": 183, "message": ""}, {"path": "llmisvc/conftest.py", "lineno": 159, "message": ""}, {"path": "llmisvc/namespace.py", "lineno": 81, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py", "lineno": 6363, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py", "lineno": 6454, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 348, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 180, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 391, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 279, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 172, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py", "lineno": 143, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py", "lineno": 278, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/poolmanager.py", "lineno": 457, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 872, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 872, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 872, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 842, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py", "lineno": 543, "message": "MaxRetryError"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b127e8610>\n\n    def _new_conn(self) -> socket.socket:\n        \"\"\"Establish a socket connection and set nodelay settings on it.\n    \n        :return: New socket connection.\n        \"\"\"\n        try:\n>           sock = connection.create_connection(\n                (self._dns_host, self.port),\n                self.timeout,\n                source_address=self.source_address,\n                socket_options=self.socket_options,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:204: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\naddress = ('a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', 6443)\ntimeout = None, source_address = None, socket_options = [(6, 1, 1)]\n\n    def create_connection(\n        address: tuple[str, int],\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        source_address: tuple[str, int] | None = None,\n        socket_options: _TYPE_SOCKET_OPTIONS | None = None,\n    ) -> socket.socket:\n        \"\"\"Connect to *address* and return the socket object.\n    \n        Convenience function.  Connect to *address* (a 2-tuple ``(host,\n        port)``) and return the socket object.  Passing the optional\n        *timeout* parameter will set the timeout on the socket instance\n        before attempting to connect.  If no *timeout* is supplied, the\n        global default timeout setting returned by :func:`socket.getdefaulttimeout`\n        is used.  If *source_address* is set it must be a tuple of (host, port)\n        for the socket to bind as a source address before making the connection.\n        An host of '' or port 0 tells the OS to use the default.\n        \"\"\"\n    \n        host, port = address\n        if host.startswith(\"[\"):\n            host = host.strip(\"[]\")\n        err = None\n    \n        # Using the value from allowed_gai_family() in the context of getaddrinfo lets\n        # us select whether to work with IPv4 DNS records, IPv6 records, or both.\n        # The original create_connection function always returns all records.\n        family = allowed_gai_family()\n    \n        try:\n            host.encode(\"idna\")\n        except UnicodeError:\n            raise LocationParseError(f\"'{host}', label empty or too long\") from None\n    \n>       for res in socket.getaddrinfo(host, port, family, socket.SOCK_STREAM):\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/connection.py:60: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nhost = 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com'\nport = 6443, family = <AddressFamily.AF_UNSPEC: 0>\ntype = <SocketKind.SOCK_STREAM: 1>, proto = 0, flags = 0\n\n    def getaddrinfo(host, port, family=0, type=0, proto=0, flags=0):\n        \"\"\"Resolve host and port into list of address info entries.\n    \n        Translate the host/port argument into a sequence of 5-tuples that contain\n        all the necessary arguments for creating a socket connected to that service.\n        host is a domain name, a string representation of an IPv4/v6 address or\n        None. port is a string service name such as 'http', a numeric port number or\n        None. By passing None as the value of host and port, you can pass NULL to\n        the underlying C API.\n    \n        The family, type and proto arguments can be optionally specified in order to\n        narrow the list of addresses returned. Passing zero as a value for each of\n        these arguments selects the full range of results.\n        \"\"\"\n        # We override this function since we want to translate the numeric family\n        # and socket type values to enum constants.\n        addrlist = []\n>       for res in _socket.getaddrinfo(host, port, family, type, proto, flags):\nE       socket.gaierror: [Errno -2] Name or service not known\n\n/usr/lib64/python3.11/socket.py:974: gaierror\n\nThe above exception was the direct cause of the following exception:\n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False, err = None\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n>           response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:788: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b127e8610>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\ntimeout = Timeout(connect=None, read=None, total=None), chunked = False\nresponse_conn = None, preload_content = True, decode_content = True\nenforce_content_length = True\n\n    def _make_request(\n        self,\n        conn: BaseHTTPConnection,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | None = None,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        chunked: bool = False,\n        response_conn: BaseHTTPConnection | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        enforce_content_length: bool = True,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Perform a request on a given urllib connection object taken from our\n        pool.\n    \n        :param conn:\n            a connection from one of our connection pools\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            Pass ``None`` to retry until you receive a response. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param response_conn:\n            Set this to ``None`` if you will handle releasing the connection or\n            set the connection to have the response release it.\n    \n        :param preload_content:\n          If True, the response's body will be preloaded during construction.\n    \n        :param decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param enforce_content_length:\n            Enforce content length checking. Body returned by server must match\n            value of Content-Length header, if present. Otherwise, raise error.\n        \"\"\"\n        self.num_requests += 1\n    \n        timeout_obj = self._get_timeout(timeout)\n        timeout_obj.start_connect()\n        conn.timeout = Timeout.resolve_default_timeout(timeout_obj.connect_timeout)\n    \n        try:\n            # Trigger any extra validation we need to do.\n            try:\n                self._validate_conn(conn)\n            except (SocketTimeout, BaseSSLError) as e:\n                self._raise_timeout(err=e, url=url, timeout_value=conn.timeout)\n                raise\n    \n        # _validate_conn() starts the connection to an HTTPS proxy\n        # so we need to wrap errors with 'ProxyError' here too.\n        except (\n            OSError,\n            NewConnectionError,\n            TimeoutError,\n            BaseSSLError,\n            CertificateError,\n            SSLError,\n        ) as e:\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            # If the connection didn't successfully connect to it's proxy\n            # then there\n            if isinstance(\n                new_e, (OSError, NewConnectionError, TimeoutError, SSLError)\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n>           raise new_e\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:488: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b127e8610>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\ntimeout = Timeout(connect=None, read=None, total=None), chunked = False\nresponse_conn = None, preload_content = True, decode_content = True\nenforce_content_length = True\n\n    def _make_request(\n        self,\n        conn: BaseHTTPConnection,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | None = None,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        chunked: bool = False,\n        response_conn: BaseHTTPConnection | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        enforce_content_length: bool = True,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Perform a request on a given urllib connection object taken from our\n        pool.\n    \n        :param conn:\n            a connection from one of our connection pools\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            Pass ``None`` to retry until you receive a response. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param response_conn:\n            Set this to ``None`` if you will handle releasing the connection or\n            set the connection to have the response release it.\n    \n        :param preload_content:\n          If True, the response's body will be preloaded during construction.\n    \n        :param decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param enforce_content_length:\n            Enforce content length checking. Body returned by server must match\n            value of Content-Length header, if present. Otherwise, raise error.\n        \"\"\"\n        self.num_requests += 1\n    \n        timeout_obj = self._get_timeout(timeout)\n        timeout_obj.start_connect()\n        conn.timeout = Timeout.resolve_default_timeout(timeout_obj.connect_timeout)\n    \n        try:\n            # Trigger any extra validation we need to do.\n            try:\n>               self._validate_conn(conn)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:464: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b127e8610>\n\n    def _validate_conn(self, conn: BaseHTTPConnection) -> None:\n        \"\"\"\n        Called right before a request is made, after the socket is created.\n        \"\"\"\n        super()._validate_conn(conn)\n    \n        # Force connect early to allow us to validate the connection.\n        if conn.is_closed:\n>           conn.connect()\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:1106: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b127e8610>\n\n    def connect(self) -> None:\n        # Today we don't need to be doing this step before the /actual/ socket\n        # connection, however in the future we'll need to decide whether to\n        # create a new socket or re-use an existing \"shared\" socket as a part\n        # of the HTTP/2 handshake dance.\n        if self._tunnel_host is not None and self._tunnel_port is not None:\n            probe_http2_host = self._tunnel_host\n            probe_http2_port = self._tunnel_port\n        else:\n            probe_http2_host = self.host\n            probe_http2_port = self.port\n    \n        # Check if the target origin supports HTTP/2.\n        # If the value comes back as 'None' it means that the current thread\n        # is probing for HTTP/2 support. Otherwise, we're waiting for another\n        # probe to complete, or we get a value right away.\n        target_supports_http2: bool | None\n        if \"h2\" in ssl_.ALPN_PROTOCOLS:\n            target_supports_http2 = http2_probe.acquire_and_get(\n                host=probe_http2_host, port=probe_http2_port\n            )\n        else:\n            # If HTTP/2 isn't going to be offered it doesn't matter if\n            # the target supports HTTP/2. Don't want to make a probe.\n            target_supports_http2 = False\n    \n        if self._connect_callback is not None:\n            self._connect_callback(\n                \"before connect\",\n                thread_id=threading.get_ident(),\n                target_supports_http2=target_supports_http2,\n            )\n    \n        try:\n            sock: socket.socket | ssl.SSLSocket\n>           self.sock = sock = self._new_conn()\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:759: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b127e8610>\n\n    def _new_conn(self) -> socket.socket:\n        \"\"\"Establish a socket connection and set nodelay settings on it.\n    \n        :return: New socket connection.\n        \"\"\"\n        try:\n            sock = connection.create_connection(\n                (self._dns_host, self.port),\n                self.timeout,\n                source_address=self.source_address,\n                socket_options=self.socket_options,\n            )\n        except socket.gaierror as e:\n>           raise NameResolutionError(self.host, self, e) from e\nE           urllib3.exceptions.NameResolutionError: HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:211: NameResolutionError\n\nThe above exception was the direct cause of the following exception:\n\nrequest = <SubRequest 'ensure_gateway_proxy_memory' for <Function test_llm_inference_service[router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]>>\n\n    @pytest.fixture(autouse=True)\n    def ensure_gateway_proxy_memory(request):\n        \"\"\"After test setup creates gateways, patch them for proxy memory.\"\"\"\n        if not GATEWAY_PROXY_MEMORY:\n            return\n    \n        # Let test_case (llmisvc) create gateways first\n    \n        if \"test_case\" in request.fixturenames:\n>           request.getfixturevalue(\"test_case\")\n\ncommon/gateway_proxy_istio.py:183: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nrequest = <SubRequest 'test_namespace' for <Function test_llm_inference_service[router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]>>\n\n    @pytest.fixture(scope=\"function\")\n    def test_namespace(request):\n        \"\"\"Create a per-test namespace with secrets, clean up after the test.\"\"\"\n        inject_k8s_proxy()\n        ns = generate_namespace_name(request.node.name)\n>       create_test_namespace(ns)\n\nllmisvc/conftest.py:159: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nnamespace = 'e2e-test-llm-inference-service-d73d44f4'\n\n    def create_test_namespace(namespace: str) -> None:\n        \"\"\"Create a labeled namespace for a single test.\"\"\"\n        core_v1 = client.CoreV1Api()\n        ns = client.V1Namespace(\n            metadata=client.V1ObjectMeta(\n                name=namespace,\n                labels={\n                    TEST_NAMESPACE_LABEL_KEY: TEST_NAMESPACE_LABEL_VALUE,\n                },\n            )\n        )\n        try:\n>           core_v1.create_namespace(ns)\n\nllmisvc/namespace.py:81: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.core_v1_api.CoreV1Api object at 0x7f9b182d3490>\nbody = {'api_version': None,\n 'kind': None,\n 'metadata': {'annotations': None,\n              'creation_timestamp': None,\n    ... 'resource_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}\nkwargs = {'_return_http_data_only': True}\n\n    def create_namespace(self, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespace  # noqa: E501\n    \n        create a Namespace  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespace(body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param V1Namespace body: (required)\n        :param str pretty: If 'true', then the output is pretty printed. Defaults to 'false' unless the user-agent indicates a browser or command-line HTTP tool (curl and wget).\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: V1Namespace\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.create_namespace_with_http_info(body, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py:6363: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.core_v1_api.CoreV1Api object at 0x7f9b182d3490>\nbody = {'api_version': None,\n 'kind': None,\n 'metadata': {'annotations': None,\n              'creation_timestamp': None,\n    ... 'resource_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['body', 'pretty', 'dry_run', 'field_manager', 'field_validation', 'asy...urce_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}, ...}\nall_params = ['body', 'pretty', 'dry_run', 'field_manager', 'field_validation', 'async_req', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {}, query_params = []\n\n    def create_namespace_with_http_info(self, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespace  # noqa: E501\n    \n        create a Namespace  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespace_with_http_info(body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param V1Namespace body: (required)\n        :param str pretty: If 'true', then the output is pretty printed. Defaults to 'false' unless the user-agent indicates a browser or command-line HTTP tool (curl and wget).\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered.\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(V1Namespace, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'body',\n            'pretty',\n            'dry_run',\n            'field_manager',\n            'field_validation'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method create_namespace\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'body' is set\n        if self.api_client.client_side_validation and ('body' not in local_var_params or  # noqa: E501\n                                                        local_var_params['body'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `body` when calling `create_namespace`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n    \n        query_params = []\n        if 'pretty' in local_var_params and local_var_params['pretty'] is not None:  # noqa: E501\n            query_params.append(('pretty', local_var_params['pretty']))  # noqa: E501\n        if 'dry_run' in local_var_params and local_var_params['dry_run'] is not None:  # noqa: E501\n            query_params.append(('dryRun', local_var_params['dry_run']))  # noqa: E501\n        if 'field_manager' in local_var_params and local_var_params['field_manager'] is not None:  # noqa: E501\n            query_params.append(('fieldManager', local_var_params['field_manager']))  # noqa: E501\n        if 'field_validation' in local_var_params and local_var_params['field_validation'] is not None:  # noqa: E501\n            query_params.append(('fieldValidation', local_var_params['field_validation']))  # noqa: E501\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        if 'body' in local_var_params:\n            body_params = local_var_params['body']\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json', 'application/yaml', 'application/vnd.kubernetes.protobuf', 'application/cbor'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/api/v1/namespaces', 'POST',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='V1Namespace',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py:6454: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b1277fa50>\nresource_path = '/api/v1/namespaces', method = 'POST', path_params = {}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'api_version': None,\n 'kind': None,\n 'metadata': {'annotations': None,\n              'creation_timestamp': None,\n    ... 'resource_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}\npost_params = [], files = {}, response_type = 'V1Namespace'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b1277fa50>\nresource_path = '/api/v1/namespaces', method = 'POST', path_params = {}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-d73d44f4'}}\npost_params = [], files = {}, response_type = 'V1Namespace'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b1277fa50>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = []\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-d73d44f4'}}\n_preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n            return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n                                        headers=headers)\n        elif method == \"HEAD\":\n            return self.rest_client.HEAD(url,\n                                         query_params=query_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n                                         headers=headers)\n        elif method == \"OPTIONS\":\n            return self.rest_client.OPTIONS(url,\n                                            query_params=query_params,\n                                            headers=headers,\n                                            _preload_content=_preload_content,\n                                            _request_timeout=_request_timeout)\n        elif method == \"POST\":\n>           return self.rest_client.POST(url,\n                                         query_params=query_params,\n                                         headers=headers,\n                                         post_params=post_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:391: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f9b1277f250>\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], post_params = []\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-d73d44f4'}}\n_preload_content = True, _request_timeout = None\n\n    def POST(self, url, headers=None, query_params=None, post_params=None,\n             body=None, _preload_content=True, _request_timeout=None):\n>       return self.request(\"POST\", url,\n                            headers=headers,\n                            query_params=query_params,\n                            post_params=post_params,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            body=body)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:279: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f9b1277f250>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-d73d44f4'}}\npost_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n>                   r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:172: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b1277f050>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}'\nfields = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\njson = None\nurlopen_kw = {'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}', 'preload_content': True, 'timeout': None}\n\n    def request(\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        fields: _TYPE_FIELDS | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        json: typing.Any | None = None,\n        **urlopen_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Make a request using :meth:`urlopen` with the appropriate encoding of\n        ``fields`` based on the ``method`` used.\n    \n        This is a convenience method that requires the least amount of manual\n        effort. It can be used in most situations, while still having the\n        option to drop down to more specific methods when necessary, such as\n        :meth:`request_encode_url`, :meth:`request_encode_body`,\n        or even the lowest level :meth:`urlopen`.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param fields:\n            Data to encode and send in the URL or request body, depending on ``method``.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param json:\n            Data to encode and send as JSON with UTF-encoded in the request body.\n            The ``\"Content-Type\"`` header will be set to ``\"application/json\"``\n            unless specified otherwise.\n        \"\"\"\n        method = method.upper()\n    \n        if json is not None and body is not None:\n            raise TypeError(\n                \"request got values for both 'body' and 'json' parameters which are mutually exclusive\"\n            )\n    \n        if json is not None:\n            if headers is None:\n                headers = self.headers\n    \n            if not (\"content-type\" in map(str.lower, headers.keys())):\n                headers = HTTPHeaderDict(headers)\n                headers[\"Content-Type\"] = \"application/json\"\n    \n            body = _json.dumps(json, separators=(\",\", \":\"), ensure_ascii=False).encode(\n                \"utf-8\"\n            )\n    \n        if body is not None:\n            urlopen_kw[\"body\"] = body\n    \n        if method in self._encode_url_methods:\n            return self.request_encode_url(\n                method,\n                url,\n                fields=fields,  # type: ignore[arg-type]\n                headers=headers,\n                **urlopen_kw,\n            )\n        else:\n>           return self.request_encode_body(\n                method, url, fields=fields, headers=headers, **urlopen_kw\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py:143: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b1277f050>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nfields = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nencode_multipart = True, multipart_boundary = None\nurlopen_kw = {'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}', 'preload_content': True, 'timeout': None}\nextra_kw = {'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}'...nt': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'}), 'preload_content': True, 'timeout': None}\n\n    def request_encode_body(\n        self,\n        method: str,\n        url: str,\n        fields: _TYPE_FIELDS | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        encode_multipart: bool = True,\n        multipart_boundary: str | None = None,\n        **urlopen_kw: str,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Make a request using :meth:`urlopen` with the ``fields`` encoded in\n        the body. This is useful for request methods like POST, PUT, PATCH, etc.\n    \n        When ``encode_multipart=True`` (default), then\n        :func:`urllib3.encode_multipart_formdata` is used to encode\n        the payload with the appropriate content type. Otherwise\n        :func:`urllib.parse.urlencode` is used with the\n        'application/x-www-form-urlencoded' content type.\n    \n        Multipart encoding must be used when posting files, and it's reasonably\n        safe to use it in other times too. However, it may break request\n        signing, such as with OAuth.\n    \n        Supports an optional ``fields`` parameter of key/value strings AND\n        key/filetuple. A filetuple is a (filename, data, MIME type) tuple where\n        the MIME type is optional. For example::\n    \n            fields = {\n                'foo': 'bar',\n                'fakefile': ('foofile.txt', 'contents of foofile'),\n                'realfile': ('barfile.txt', open('realfile').read()),\n                'typedfile': ('bazfile.bin', open('bazfile').read(),\n                              'image/jpeg'),\n                'nonamefile': 'contents of nonamefile field',\n            }\n    \n        When uploading a file, providing a filename (the first parameter of the\n        tuple) is optional but recommended to best mimic behavior of browsers.\n    \n        Note that if ``headers`` are supplied, the 'Content-Type' header will\n        be overwritten because it depends on the dynamic random boundary string\n        which is used to compose the body of the request. The random boundary\n        string can be explicitly set with the ``multipart_boundary`` parameter.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param fields:\n            Data to encode and send in the request body.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param encode_multipart:\n            If True, encode the ``fields`` using the multipart/form-data MIME\n            format.\n    \n        :param multipart_boundary:\n            If not specified, then a random boundary will be generated using\n            :func:`urllib3.filepost.choose_boundary`.\n        \"\"\"\n        if headers is None:\n            headers = self.headers\n    \n        extra_kw: dict[str, typing.Any] = {\"headers\": HTTPHeaderDict(headers)}\n        body: bytes | str\n    \n        if fields:\n            if \"body\" in urlopen_kw:\n                raise TypeError(\n                    \"request got values for both 'fields' and 'body', can only specify one.\"\n                )\n    \n            if encode_multipart:\n                body, content_type = encode_multipart_formdata(\n                    fields, boundary=multipart_boundary\n                )\n            else:\n                body, content_type = (\n                    urlencode(fields),  # type: ignore[arg-type]\n                    \"application/x-www-form-urlencoded\",\n                )\n    \n            extra_kw[\"body\"] = body\n            extra_kw[\"headers\"].setdefault(\"Content-Type\", content_type)\n    \n        extra_kw.update(urlopen_kw)\n    \n>       return self.urlopen(method, url, **extra_kw)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py:278: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b1277f050>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nredirect = True\nkw = {'assert_same_host': False, 'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inf...', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'}), 'preload_content': True, ...}\nu = Url(scheme='https', auth=None, host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443, path='/api/v1/namespaces', query=None, fragment=None)\nconn = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\n\n    def urlopen(  # type: ignore[override]\n        self, method: str, url: str, redirect: bool = True, **kw: typing.Any\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Same as :meth:`urllib3.HTTPConnectionPool.urlopen`\n        with custom cross-host redirect logic and only sends the request-uri\n        portion of the ``url``.\n    \n        The given ``url`` parameter must be absolute, such that an appropriate\n        :class:`urllib3.connectionpool.ConnectionPool` can be chosen for it.\n        \"\"\"\n        u = parse_url(url)\n    \n        if u.scheme is None:\n            warnings.warn(\n                \"URLs without a scheme (ie 'https://') are deprecated and will raise an error \"\n                \"in urllib3 v3.0. To avoid this FutureWarning ensure all URLs \"\n                \"start with 'https://' or 'http://'. Read more in this issue: \"\n                \"https://github.com/urllib3/urllib3/issues/2920\",\n                category=FutureWarning,\n                stacklevel=2,\n            )\n    \n        conn = self.connection_from_host(u.host, port=u.port, scheme=u.scheme)\n    \n        kw[\"assert_same_host\"] = False\n        kw[\"redirect\"] = False\n    \n        if \"headers\" not in kw:\n            kw[\"headers\"] = self.headers\n    \n        if self._proxy_requires_url_absolute_form(u):\n            response = conn.urlopen(method, url, **kw)\n        else:\n>           response = conn.urlopen(method, u.request_uri, **kw)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/poolmanager.py:457: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=2, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=1, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-d73d44f4\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False, err = None\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n>           retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:842: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nmethod = 'POST', url = '/api/v1/namespaces', response = None\nerror = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\n_pool = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b136c3d50>\n_stacktrace = <traceback object at 0x7f9b127ebf80>\n\n    def increment(\n        self,\n        method: str | None = None,\n        url: str | None = None,\n        response: BaseHTTPResponse | None = None,\n        error: Exception | None = None,\n        _pool: ConnectionPool | None = None,\n        _stacktrace: TracebackType | None = None,\n    ) -> Self:\n        \"\"\"Return a new Retry object with incremented retry counters.\n    \n        :param response: A response object, or None, if the server did not\n            return a response.\n        :type response: :class:`~urllib3.response.BaseHTTPResponse`\n        :param Exception error: An error encountered during the request, or\n            None if the response was received successfully.\n    \n        :return: A new ``Retry`` object.\n        \"\"\"\n        if self.total is False and error:\n            # Disabled, indicate to re-raise the error.\n            raise reraise(type(error), error, _stacktrace)\n    \n        total = self.total\n        if total is not None:\n            total -= 1\n    \n        connect = self.connect\n        read = self.read\n        redirect = self.redirect\n        status_count = self.status\n        other = self.other\n        cause = \"unknown\"\n        status = None\n        redirect_location = None\n    \n        if error and self._is_connection_error(error):\n            # Connect retry?\n            if connect is False:\n                raise reraise(type(error), error, _stacktrace)\n            elif connect is not None:\n                connect -= 1\n    \n        elif error and self._is_read_error(error):\n            # Read retry?\n            if read is False or method is None or not self._is_method_retryable(method):\n                raise reraise(type(error), error, _stacktrace)\n            elif read is not None:\n                read -= 1\n    \n        elif error:\n            # Other retry?\n            if other is not None:\n                other -= 1\n    \n        elif response and response.get_redirect_location():\n            # Redirect retry?\n            if redirect is not None:\n                redirect -= 1\n            cause = \"too many redirects\"\n            response_redirect_location = response.get_redirect_location()\n            if response_redirect_location:\n                redirect_location = response_redirect_location\n            status = response.status\n    \n        else:\n            # Incrementing because of a server error like a 500 in\n            # status_forcelist and the given method is in the allowed_methods\n            cause = ResponseError.GENERIC_ERROR\n            if response and response.status:\n                if status_count is not None:\n                    status_count -= 1\n                cause = ResponseError.SPECIFIC_ERROR.format(status_code=response.status)\n                status = response.status\n    \n        history = self.history + (\n            RequestHistory(method, url, error, status, redirect_location),\n        )\n    \n        new_retry = self.new(\n            total=total,\n            connect=connect,\n            read=read,\n            redirect=redirect,\n            status=status_count,\n            other=other,\n            history=history,\n        )\n    \n        if new_retry.is_exhausted():\n            reason = error or ResponseError(cause)\n>           raise MaxRetryError(_pool, url, reason) from reason  # type: ignore[arg-type]\nE           urllib3.exceptions.MaxRetryError: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /api/v1/namespaces (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py:543: MaxRetryError"}, "teardown": {"duration": 0.00027221598429605365, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 242, "outcome": "error", "keywords": ["test_llm_inference_service[router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "cluster_cpu", "cluster_single_node", "custom_gateway", "__wrapped__", "pytestmark", "router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m", "llminferenceservice", "llmisvc_core", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.031345651019364595, "outcome": "failed", "crash": {"path": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py", "lineno": 543, "message": "urllib3.exceptions.MaxRetryError: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /api/v1/namespaces (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))"}, "traceback": [{"path": "common/gateway_proxy_istio.py", "lineno": 183, "message": ""}, {"path": "llmisvc/conftest.py", "lineno": 159, "message": ""}, {"path": "llmisvc/namespace.py", "lineno": 81, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py", "lineno": 6363, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py", "lineno": 6454, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 348, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 180, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py", "lineno": 391, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 279, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py", "lineno": 172, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py", "lineno": 143, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py", "lineno": 278, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/poolmanager.py", "lineno": 457, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 872, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 872, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 872, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py", "lineno": 842, "message": ""}, {"path": "../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py", "lineno": 543, "message": "MaxRetryError"}], "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b12a65950>\n\n    def _new_conn(self) -> socket.socket:\n        \"\"\"Establish a socket connection and set nodelay settings on it.\n    \n        :return: New socket connection.\n        \"\"\"\n        try:\n>           sock = connection.create_connection(\n                (self._dns_host, self.port),\n                self.timeout,\n                source_address=self.source_address,\n                socket_options=self.socket_options,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:204: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\naddress = ('a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', 6443)\ntimeout = None, source_address = None, socket_options = [(6, 1, 1)]\n\n    def create_connection(\n        address: tuple[str, int],\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        source_address: tuple[str, int] | None = None,\n        socket_options: _TYPE_SOCKET_OPTIONS | None = None,\n    ) -> socket.socket:\n        \"\"\"Connect to *address* and return the socket object.\n    \n        Convenience function.  Connect to *address* (a 2-tuple ``(host,\n        port)``) and return the socket object.  Passing the optional\n        *timeout* parameter will set the timeout on the socket instance\n        before attempting to connect.  If no *timeout* is supplied, the\n        global default timeout setting returned by :func:`socket.getdefaulttimeout`\n        is used.  If *source_address* is set it must be a tuple of (host, port)\n        for the socket to bind as a source address before making the connection.\n        An host of '' or port 0 tells the OS to use the default.\n        \"\"\"\n    \n        host, port = address\n        if host.startswith(\"[\"):\n            host = host.strip(\"[]\")\n        err = None\n    \n        # Using the value from allowed_gai_family() in the context of getaddrinfo lets\n        # us select whether to work with IPv4 DNS records, IPv6 records, or both.\n        # The original create_connection function always returns all records.\n        family = allowed_gai_family()\n    \n        try:\n            host.encode(\"idna\")\n        except UnicodeError:\n            raise LocationParseError(f\"'{host}', label empty or too long\") from None\n    \n>       for res in socket.getaddrinfo(host, port, family, socket.SOCK_STREAM):\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/connection.py:60: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nhost = 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com'\nport = 6443, family = <AddressFamily.AF_UNSPEC: 0>\ntype = <SocketKind.SOCK_STREAM: 1>, proto = 0, flags = 0\n\n    def getaddrinfo(host, port, family=0, type=0, proto=0, flags=0):\n        \"\"\"Resolve host and port into list of address info entries.\n    \n        Translate the host/port argument into a sequence of 5-tuples that contain\n        all the necessary arguments for creating a socket connected to that service.\n        host is a domain name, a string representation of an IPv4/v6 address or\n        None. port is a string service name such as 'http', a numeric port number or\n        None. By passing None as the value of host and port, you can pass NULL to\n        the underlying C API.\n    \n        The family, type and proto arguments can be optionally specified in order to\n        narrow the list of addresses returned. Passing zero as a value for each of\n        these arguments selects the full range of results.\n        \"\"\"\n        # We override this function since we want to translate the numeric family\n        # and socket type values to enum constants.\n        addrlist = []\n>       for res in _socket.getaddrinfo(host, port, family, type, proto, flags):\nE       socket.gaierror: [Errno -2] Name or service not known\n\n/usr/lib64/python3.11/socket.py:974: gaierror\n\nThe above exception was the direct cause of the following exception:\n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False, err = None\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n>           response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:788: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b12a65950>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\ntimeout = Timeout(connect=None, read=None, total=None), chunked = False\nresponse_conn = None, preload_content = True, decode_content = True\nenforce_content_length = True\n\n    def _make_request(\n        self,\n        conn: BaseHTTPConnection,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | None = None,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        chunked: bool = False,\n        response_conn: BaseHTTPConnection | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        enforce_content_length: bool = True,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Perform a request on a given urllib connection object taken from our\n        pool.\n    \n        :param conn:\n            a connection from one of our connection pools\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            Pass ``None`` to retry until you receive a response. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param response_conn:\n            Set this to ``None`` if you will handle releasing the connection or\n            set the connection to have the response release it.\n    \n        :param preload_content:\n          If True, the response's body will be preloaded during construction.\n    \n        :param decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param enforce_content_length:\n            Enforce content length checking. Body returned by server must match\n            value of Content-Length header, if present. Otherwise, raise error.\n        \"\"\"\n        self.num_requests += 1\n    \n        timeout_obj = self._get_timeout(timeout)\n        timeout_obj.start_connect()\n        conn.timeout = Timeout.resolve_default_timeout(timeout_obj.connect_timeout)\n    \n        try:\n            # Trigger any extra validation we need to do.\n            try:\n                self._validate_conn(conn)\n            except (SocketTimeout, BaseSSLError) as e:\n                self._raise_timeout(err=e, url=url, timeout_value=conn.timeout)\n                raise\n    \n        # _validate_conn() starts the connection to an HTTPS proxy\n        # so we need to wrap errors with 'ProxyError' here too.\n        except (\n            OSError,\n            NewConnectionError,\n            TimeoutError,\n            BaseSSLError,\n            CertificateError,\n            SSLError,\n        ) as e:\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            # If the connection didn't successfully connect to it's proxy\n            # then there\n            if isinstance(\n                new_e, (OSError, NewConnectionError, TimeoutError, SSLError)\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n>           raise new_e\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:488: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b12a65950>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\ntimeout = Timeout(connect=None, read=None, total=None), chunked = False\nresponse_conn = None, preload_content = True, decode_content = True\nenforce_content_length = True\n\n    def _make_request(\n        self,\n        conn: BaseHTTPConnection,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | None = None,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        chunked: bool = False,\n        response_conn: BaseHTTPConnection | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        enforce_content_length: bool = True,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Perform a request on a given urllib connection object taken from our\n        pool.\n    \n        :param conn:\n            a connection from one of our connection pools\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            Pass ``None`` to retry until you receive a response. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param response_conn:\n            Set this to ``None`` if you will handle releasing the connection or\n            set the connection to have the response release it.\n    \n        :param preload_content:\n          If True, the response's body will be preloaded during construction.\n    \n        :param decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param enforce_content_length:\n            Enforce content length checking. Body returned by server must match\n            value of Content-Length header, if present. Otherwise, raise error.\n        \"\"\"\n        self.num_requests += 1\n    \n        timeout_obj = self._get_timeout(timeout)\n        timeout_obj.start_connect()\n        conn.timeout = Timeout.resolve_default_timeout(timeout_obj.connect_timeout)\n    \n        try:\n            # Trigger any extra validation we need to do.\n            try:\n>               self._validate_conn(conn)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:464: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\nconn = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b12a65950>\n\n    def _validate_conn(self, conn: BaseHTTPConnection) -> None:\n        \"\"\"\n        Called right before a request is made, after the socket is created.\n        \"\"\"\n        super()._validate_conn(conn)\n    \n        # Force connect early to allow us to validate the connection.\n        if conn.is_closed:\n>           conn.connect()\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:1106: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b12a65950>\n\n    def connect(self) -> None:\n        # Today we don't need to be doing this step before the /actual/ socket\n        # connection, however in the future we'll need to decide whether to\n        # create a new socket or re-use an existing \"shared\" socket as a part\n        # of the HTTP/2 handshake dance.\n        if self._tunnel_host is not None and self._tunnel_port is not None:\n            probe_http2_host = self._tunnel_host\n            probe_http2_port = self._tunnel_port\n        else:\n            probe_http2_host = self.host\n            probe_http2_port = self.port\n    \n        # Check if the target origin supports HTTP/2.\n        # If the value comes back as 'None' it means that the current thread\n        # is probing for HTTP/2 support. Otherwise, we're waiting for another\n        # probe to complete, or we get a value right away.\n        target_supports_http2: bool | None\n        if \"h2\" in ssl_.ALPN_PROTOCOLS:\n            target_supports_http2 = http2_probe.acquire_and_get(\n                host=probe_http2_host, port=probe_http2_port\n            )\n        else:\n            # If HTTP/2 isn't going to be offered it doesn't matter if\n            # the target supports HTTP/2. Don't want to make a probe.\n            target_supports_http2 = False\n    \n        if self._connect_callback is not None:\n            self._connect_callback(\n                \"before connect\",\n                thread_id=threading.get_ident(),\n                target_supports_http2=target_supports_http2,\n            )\n    \n        try:\n            sock: socket.socket | ssl.SSLSocket\n>           self.sock = sock = self._new_conn()\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:759: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443) at 0x7f9b12a65950>\n\n    def _new_conn(self) -> socket.socket:\n        \"\"\"Establish a socket connection and set nodelay settings on it.\n    \n        :return: New socket connection.\n        \"\"\"\n        try:\n            sock = connection.create_connection(\n                (self._dns_host, self.port),\n                self.timeout,\n                source_address=self.source_address,\n                socket_options=self.socket_options,\n            )\n        except socket.gaierror as e:\n>           raise NameResolutionError(self.host, self, e) from e\nE           urllib3.exceptions.NameResolutionError: HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connection.py:211: NameResolutionError\n\nThe above exception was the direct cause of the following exception:\n\nrequest = <SubRequest 'ensure_gateway_proxy_memory' for <Function test_llm_inference_service[router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]>>\n\n    @pytest.fixture(autouse=True)\n    def ensure_gateway_proxy_memory(request):\n        \"\"\"After test setup creates gateways, patch them for proxy memory.\"\"\"\n        if not GATEWAY_PROXY_MEMORY:\n            return\n    \n        # Let test_case (llmisvc) create gateways first\n    \n        if \"test_case\" in request.fixturenames:\n>           request.getfixturevalue(\"test_case\")\n\ncommon/gateway_proxy_istio.py:183: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nrequest = <SubRequest 'test_namespace' for <Function test_llm_inference_service[router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]>>\n\n    @pytest.fixture(scope=\"function\")\n    def test_namespace(request):\n        \"\"\"Create a per-test namespace with secrets, clean up after the test.\"\"\"\n        inject_k8s_proxy()\n        ns = generate_namespace_name(request.node.name)\n>       create_test_namespace(ns)\n\nllmisvc/conftest.py:159: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nnamespace = 'e2e-test-llm-inference-service-b5dd93e6'\n\n    def create_test_namespace(namespace: str) -> None:\n        \"\"\"Create a labeled namespace for a single test.\"\"\"\n        core_v1 = client.CoreV1Api()\n        ns = client.V1Namespace(\n            metadata=client.V1ObjectMeta(\n                name=namespace,\n                labels={\n                    TEST_NAMESPACE_LABEL_KEY: TEST_NAMESPACE_LABEL_VALUE,\n                },\n            )\n        )\n        try:\n>           core_v1.create_namespace(ns)\n\nllmisvc/namespace.py:81: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.core_v1_api.CoreV1Api object at 0x7f9b1300f210>\nbody = {'api_version': None,\n 'kind': None,\n 'metadata': {'annotations': None,\n              'creation_timestamp': None,\n    ... 'resource_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}\nkwargs = {'_return_http_data_only': True}\n\n    def create_namespace(self, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespace  # noqa: E501\n    \n        create a Namespace  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespace(body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param V1Namespace body: (required)\n        :param str pretty: If 'true', then the output is pretty printed. Defaults to 'false' unless the user-agent indicates a browser or command-line HTTP tool (curl and wget).\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: V1Namespace\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n        kwargs['_return_http_data_only'] = True\n>       return self.create_namespace_with_http_info(body, **kwargs)  # noqa: E501\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py:6363: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api.core_v1_api.CoreV1Api object at 0x7f9b1300f210>\nbody = {'api_version': None,\n 'kind': None,\n 'metadata': {'annotations': None,\n              'creation_timestamp': None,\n    ... 'resource_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}\nkwargs = {'_return_http_data_only': True}\nlocal_var_params = {'_return_http_data_only': True, 'all_params': ['body', 'pretty', 'dry_run', 'field_manager', 'field_validation', 'asy...urce_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}, ...}\nall_params = ['body', 'pretty', 'dry_run', 'field_manager', 'field_validation', 'async_req', ...]\nkey = '_return_http_data_only', val = True, collection_formats = {}\npath_params = {}, query_params = []\n\n    def create_namespace_with_http_info(self, body, **kwargs):  # noqa: E501\n        \"\"\"create_namespace  # noqa: E501\n    \n        create a Namespace  # noqa: E501\n        This method makes a synchronous HTTP request by default. To make an\n        asynchronous HTTP request, please pass async_req=True\n        >>> thread = api.create_namespace_with_http_info(body, async_req=True)\n        >>> result = thread.get()\n    \n        :param async_req bool: execute request asynchronously\n        :param V1Namespace body: (required)\n        :param str pretty: If 'true', then the output is pretty printed. Defaults to 'false' unless the user-agent indicates a browser or command-line HTTP tool (curl and wget).\n        :param str dry_run: When present, indicates that modifications should not be persisted. An invalid or unrecognized dryRun directive will result in an error response and no further processing of the request. Valid values are: - All: all dry run stages will be processed\n        :param str field_manager: fieldManager is a name associated with the actor or entity that is making these changes. The value must be less than or 128 characters long, and only contain printable characters, as defined by https://golang.org/pkg/unicode/#IsPrint.\n        :param str field_validation: fieldValidation instructs the server on how to handle objects in the request (POST/PUT/PATCH) containing unknown or duplicate fields. Valid values are: - Ignore: This will ignore any unknown fields that are silently dropped from the object, and will ignore all but the last duplicate field that the decoder encounters. This is the default behavior prior to v1.23. - Warn: This will send a warning via the standard warning response header for each unknown field that is dropped from the object, and for each duplicate field that is encountered. The request will still succeed if there are no other errors, and will only persist the last of any duplicate fields. This is the default in v1.23+ - Strict: This will fail the request with a BadRequest error if any unknown fields would be dropped from the object, or if any duplicate fields are present. The error returned from the server will contain all unknown and duplicate fields encountered.\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return: tuple(V1Namespace, status_code(int), headers(HTTPHeaderDict))\n                 If the method is called asynchronously,\n                 returns the request thread.\n        \"\"\"\n    \n        local_var_params = locals()\n    \n        all_params = [\n            'body',\n            'pretty',\n            'dry_run',\n            'field_manager',\n            'field_validation'\n        ]\n        all_params.extend(\n            [\n                'async_req',\n                '_return_http_data_only',\n                '_preload_content',\n                '_request_timeout'\n            ]\n        )\n    \n        for key, val in six.iteritems(local_var_params['kwargs']):\n            if key not in all_params:\n                raise ApiTypeError(\n                    \"Got an unexpected keyword argument '%s'\"\n                    \" to method create_namespace\" % key\n                )\n            local_var_params[key] = val\n        del local_var_params['kwargs']\n        # verify the required parameter 'body' is set\n        if self.api_client.client_side_validation and ('body' not in local_var_params or  # noqa: E501\n                                                        local_var_params['body'] is None):  # noqa: E501\n            raise ApiValueError(\"Missing the required parameter `body` when calling `create_namespace`\")  # noqa: E501\n    \n        collection_formats = {}\n    \n        path_params = {}\n    \n        query_params = []\n        if 'pretty' in local_var_params and local_var_params['pretty'] is not None:  # noqa: E501\n            query_params.append(('pretty', local_var_params['pretty']))  # noqa: E501\n        if 'dry_run' in local_var_params and local_var_params['dry_run'] is not None:  # noqa: E501\n            query_params.append(('dryRun', local_var_params['dry_run']))  # noqa: E501\n        if 'field_manager' in local_var_params and local_var_params['field_manager'] is not None:  # noqa: E501\n            query_params.append(('fieldManager', local_var_params['field_manager']))  # noqa: E501\n        if 'field_validation' in local_var_params and local_var_params['field_validation'] is not None:  # noqa: E501\n            query_params.append(('fieldValidation', local_var_params['field_validation']))  # noqa: E501\n    \n        header_params = {}\n    \n        form_params = []\n        local_var_files = {}\n    \n        body_params = None\n        if 'body' in local_var_params:\n            body_params = local_var_params['body']\n        # HTTP header `Accept`\n        header_params['Accept'] = self.api_client.select_header_accept(\n            ['application/json', 'application/yaml', 'application/vnd.kubernetes.protobuf', 'application/cbor'])  # noqa: E501\n    \n        # Authentication setting\n        auth_settings = ['BearerToken']  # noqa: E501\n    \n>       return self.api_client.call_api(\n            '/api/v1/namespaces', 'POST',\n            path_params,\n            query_params,\n            header_params,\n            body=body_params,\n            post_params=form_params,\n            files=local_var_files,\n            response_type='V1Namespace',  # noqa: E501\n            auth_settings=auth_settings,\n            async_req=local_var_params.get('async_req'),\n            _return_http_data_only=local_var_params.get('_return_http_data_only'),  # noqa: E501\n            _preload_content=local_var_params.get('_preload_content', True),\n            _request_timeout=local_var_params.get('_request_timeout'),\n            collection_formats=collection_formats)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api/core_v1_api.py:6454: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b1300e950>\nresource_path = '/api/v1/namespaces', method = 'POST', path_params = {}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'api_version': None,\n 'kind': None,\n 'metadata': {'annotations': None,\n              'creation_timestamp': None,\n    ... 'resource_version': None,\n              'self_link': None,\n              'uid': None},\n 'spec': None,\n 'status': None}\npost_params = [], files = {}, response_type = 'V1Namespace'\nauth_settings = ['BearerToken'], async_req = None, _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def call_api(self, resource_path, method,\n                 path_params=None, query_params=None, header_params=None,\n                 body=None, post_params=None, files=None,\n                 response_type=None, auth_settings=None, async_req=None,\n                 _return_http_data_only=None, collection_formats=None,\n                 _preload_content=True, _request_timeout=None, _host=None):\n        \"\"\"Makes the HTTP request (synchronous) and returns deserialized data.\n    \n        To make an async_req request, set the async_req parameter.\n    \n        :param resource_path: Path to method endpoint.\n        :param method: Method to call.\n        :param path_params: Path parameters in the url.\n        :param query_params: Query parameters in the url.\n        :param header_params: Header parameters to be\n            placed in the request header.\n        :param body: Request body.\n        :param post_params dict: Request post form parameters,\n            for `application/x-www-form-urlencoded`, `multipart/form-data`.\n        :param auth_settings list: Auth Settings names for the request.\n        :param response: Response data type.\n        :param files dict: key -> filename, value -> filepath,\n            for `multipart/form-data`.\n        :param async_req bool: execute request asynchronously\n        :param _return_http_data_only: response data without head status code\n                                       and headers\n        :param collection_formats: dict of collection formats for path, query,\n            header, and post parameters.\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        :return:\n            If async_req parameter is True,\n            the request will be called asynchronously.\n            The method will return the request thread.\n            If parameter async_req is False or missing,\n            then the method will return the response directly.\n        \"\"\"\n        if not async_req:\n>           return self.__call_api(resource_path, method,\n                                   path_params, query_params, header_params,\n                                   body, post_params, files,\n                                   response_type, auth_settings,\n                                   _return_http_data_only, collection_formats,\n                                   _preload_content, _request_timeout, _host)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:348: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b1300e950>\nresource_path = '/api/v1/namespaces', method = 'POST', path_params = {}\nquery_params = []\nheader_params = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-b5dd93e6'}}\npost_params = [], files = {}, response_type = 'V1Namespace'\nauth_settings = ['BearerToken'], _return_http_data_only = True\ncollection_formats = {}, _preload_content = True, _request_timeout = None\n_host = None\n\n    def __call_api(\n            self, resource_path, method, path_params=None,\n            query_params=None, header_params=None, body=None, post_params=None,\n            files=None, response_type=None, auth_settings=None,\n            _return_http_data_only=None, collection_formats=None,\n            _preload_content=True, _request_timeout=None, _host=None):\n    \n        config = self.configuration\n    \n        # header parameters\n        header_params = header_params or {}\n        header_params.update(self.default_headers)\n        if self.cookie:\n            header_params['Cookie'] = self.cookie\n        if header_params:\n            header_params = self.sanitize_for_serialization(header_params)\n            header_params = dict(self.parameters_to_tuples(header_params,\n                                                           collection_formats))\n    \n        # path parameters\n        if path_params:\n            path_params = self.sanitize_for_serialization(path_params)\n            path_params = self.parameters_to_tuples(path_params,\n                                                    collection_formats)\n            for k, v in path_params:\n                # specified safe chars, encode everything\n                resource_path = resource_path.replace(\n                    '{%s}' % k,\n                    quote(str(v), safe=config.safe_chars_for_path_param)\n                )\n    \n        # query parameters\n        if query_params:\n            query_params = self.sanitize_for_serialization(query_params)\n            query_params = self.parameters_to_tuples(query_params,\n                                                     collection_formats)\n    \n        # post parameters\n        if post_params or files:\n            post_params = post_params if post_params else []\n            post_params = self.sanitize_for_serialization(post_params)\n            post_params = self.parameters_to_tuples(post_params,\n                                                    collection_formats)\n            post_params.extend(self.files_parameters(files))\n    \n        # auth setting\n        self.update_params_for_auth(header_params, query_params, auth_settings)\n    \n        # body\n        if body:\n            body = self.sanitize_for_serialization(body)\n    \n        # request url\n        if _host is None:\n            url = self.configuration.host + resource_path\n        else:\n            # use server/host defined in path or operation instead\n            url = _host + resource_path\n    \n        # perform request and return response\n>       response_data = self.request(\n            method, url, query_params=query_params, headers=header_params,\n            post_params=post_params, body=body,\n            _preload_content=_preload_content,\n            _request_timeout=_request_timeout)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:180: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.api_client.ApiClient object at 0x7f9b1300e950>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\npost_params = []\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-b5dd93e6'}}\n_preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                post_params=None, body=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Makes the HTTP request using RESTClient.\"\"\"\n        if method == \"GET\":\n            return self.rest_client.GET(url,\n                                        query_params=query_params,\n                                        _preload_content=_preload_content,\n                                        _request_timeout=_request_timeout,\n                                        headers=headers)\n        elif method == \"HEAD\":\n            return self.rest_client.HEAD(url,\n                                         query_params=query_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n                                         headers=headers)\n        elif method == \"OPTIONS\":\n            return self.rest_client.OPTIONS(url,\n                                            query_params=query_params,\n                                            headers=headers,\n                                            _preload_content=_preload_content,\n                                            _request_timeout=_request_timeout)\n        elif method == \"POST\":\n>           return self.rest_client.POST(url,\n                                         query_params=query_params,\n                                         headers=headers,\n                                         post_params=post_params,\n                                         _preload_content=_preload_content,\n                                         _request_timeout=_request_timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/api_client.py:391: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f9b1300e9d0>\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nquery_params = [], post_params = []\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-b5dd93e6'}}\n_preload_content = True, _request_timeout = None\n\n    def POST(self, url, headers=None, query_params=None, post_params=None,\n             body=None, _preload_content=True, _request_timeout=None):\n>       return self.request(\"POST\", url,\n                            headers=headers,\n                            query_params=query_params,\n                            post_params=post_params,\n                            _preload_content=_preload_content,\n                            _request_timeout=_request_timeout,\n                            body=body)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:279: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <kubernetes.client.rest.RESTClientObject object at 0x7f9b1300e9d0>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nquery_params = []\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nbody = {'metadata': {'labels': {'kserve.io/e2e-test': 'true'}, 'name': 'e2e-test-llm-inference-service-b5dd93e6'}}\npost_params = {}, _preload_content = True, _request_timeout = None\n\n    def request(self, method, url, query_params=None, headers=None,\n                body=None, post_params=None, _preload_content=True,\n                _request_timeout=None):\n        \"\"\"Perform requests.\n    \n        :param method: http request method\n        :param url: http request url\n        :param query_params: query parameters in the url\n        :param headers: http request headers\n        :param body: request json body, for `application/json`\n        :param post_params: request post parameters,\n                            `application/x-www-form-urlencoded`\n                            and `multipart/form-data`\n        :param _preload_content: if False, the urllib3.HTTPResponse object will\n                                 be returned without reading/decoding response\n                                 data. Default is True.\n        :param _request_timeout: timeout setting for this request. If one\n                                 number provided, it will be total request\n                                 timeout. It can also be a pair (tuple) of\n                                 (connection, read) timeouts.\n        \"\"\"\n        method = method.upper()\n        assert method in ['GET', 'HEAD', 'DELETE', 'POST', 'PUT',\n                          'PATCH', 'OPTIONS']\n    \n        if post_params and body:\n            raise ApiValueError(\n                \"body parameter cannot be used with post_params parameter.\"\n            )\n    \n        post_params = post_params or {}\n        headers = headers or {}\n    \n        timeout = None\n        if _request_timeout:\n            if isinstance(_request_timeout, (int, ) if six.PY3 else (int, long)):  # noqa: E501,F821\n                timeout = urllib3.Timeout(total=_request_timeout)\n            elif (isinstance(_request_timeout, tuple) and\n                  len(_request_timeout) == 2):\n                timeout = urllib3.Timeout(\n                    connect=_request_timeout[0], read=_request_timeout[1])\n    \n        if 'Content-Type' not in headers:\n            headers['Content-Type'] = 'application/json'\n    \n        try:\n            # For `POST`, `PUT`, `PATCH`, `OPTIONS`, `DELETE`\n            if method in ['POST', 'PUT', 'PATCH', 'OPTIONS', 'DELETE']:\n                if query_params:\n                    url += '?' + urlencode(query_params)\n                if (re.search('json', headers['Content-Type'], re.IGNORECASE) or\n                        headers['Content-Type'] == 'application/apply-patch+yaml'):\n                    if headers['Content-Type'] == 'application/json-patch+json':\n                        if not isinstance(body, list):\n                            headers['Content-Type'] = \\\n                                'application/strategic-merge-patch+json'\n                    request_body = None\n                    if body is not None:\n                        request_body = json.dumps(body)\n>                   r = self.pool_manager.request(\n                        method, url,\n                        body=request_body,\n                        preload_content=_preload_content,\n                        timeout=timeout,\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/kubernetes/client/rest.py:172: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b1300d010>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}'\nfields = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\njson = None\nurlopen_kw = {'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}', 'preload_content': True, 'timeout': None}\n\n    def request(\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        fields: _TYPE_FIELDS | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        json: typing.Any | None = None,\n        **urlopen_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Make a request using :meth:`urlopen` with the appropriate encoding of\n        ``fields`` based on the ``method`` used.\n    \n        This is a convenience method that requires the least amount of manual\n        effort. It can be used in most situations, while still having the\n        option to drop down to more specific methods when necessary, such as\n        :meth:`request_encode_url`, :meth:`request_encode_body`,\n        or even the lowest level :meth:`urlopen`.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param fields:\n            Data to encode and send in the URL or request body, depending on ``method``.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param json:\n            Data to encode and send as JSON with UTF-encoded in the request body.\n            The ``\"Content-Type\"`` header will be set to ``\"application/json\"``\n            unless specified otherwise.\n        \"\"\"\n        method = method.upper()\n    \n        if json is not None and body is not None:\n            raise TypeError(\n                \"request got values for both 'body' and 'json' parameters which are mutually exclusive\"\n            )\n    \n        if json is not None:\n            if headers is None:\n                headers = self.headers\n    \n            if not (\"content-type\" in map(str.lower, headers.keys())):\n                headers = HTTPHeaderDict(headers)\n                headers[\"Content-Type\"] = \"application/json\"\n    \n            body = _json.dumps(json, separators=(\",\", \":\"), ensure_ascii=False).encode(\n                \"utf-8\"\n            )\n    \n        if body is not None:\n            urlopen_kw[\"body\"] = body\n    \n        if method in self._encode_url_methods:\n            return self.request_encode_url(\n                method,\n                url,\n                fields=fields,  # type: ignore[arg-type]\n                headers=headers,\n                **urlopen_kw,\n            )\n        else:\n>           return self.request_encode_body(\n                method, url, fields=fields, headers=headers, **urlopen_kw\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py:143: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b1300d010>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nfields = None\nheaders = {'Accept': 'application/json', 'Content-Type': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python'}\nencode_multipart = True, multipart_boundary = None\nurlopen_kw = {'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}', 'preload_content': True, 'timeout': None}\nextra_kw = {'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}'...nt': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'}), 'preload_content': True, 'timeout': None}\n\n    def request_encode_body(\n        self,\n        method: str,\n        url: str,\n        fields: _TYPE_FIELDS | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        encode_multipart: bool = True,\n        multipart_boundary: str | None = None,\n        **urlopen_kw: str,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Make a request using :meth:`urlopen` with the ``fields`` encoded in\n        the body. This is useful for request methods like POST, PUT, PATCH, etc.\n    \n        When ``encode_multipart=True`` (default), then\n        :func:`urllib3.encode_multipart_formdata` is used to encode\n        the payload with the appropriate content type. Otherwise\n        :func:`urllib.parse.urlencode` is used with the\n        'application/x-www-form-urlencoded' content type.\n    \n        Multipart encoding must be used when posting files, and it's reasonably\n        safe to use it in other times too. However, it may break request\n        signing, such as with OAuth.\n    \n        Supports an optional ``fields`` parameter of key/value strings AND\n        key/filetuple. A filetuple is a (filename, data, MIME type) tuple where\n        the MIME type is optional. For example::\n    \n            fields = {\n                'foo': 'bar',\n                'fakefile': ('foofile.txt', 'contents of foofile'),\n                'realfile': ('barfile.txt', open('realfile').read()),\n                'typedfile': ('bazfile.bin', open('bazfile').read(),\n                              'image/jpeg'),\n                'nonamefile': 'contents of nonamefile field',\n            }\n    \n        When uploading a file, providing a filename (the first parameter of the\n        tuple) is optional but recommended to best mimic behavior of browsers.\n    \n        Note that if ``headers`` are supplied, the 'Content-Type' header will\n        be overwritten because it depends on the dynamic random boundary string\n        which is used to compose the body of the request. The random boundary\n        string can be explicitly set with the ``multipart_boundary`` parameter.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param fields:\n            Data to encode and send in the request body.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param encode_multipart:\n            If True, encode the ``fields`` using the multipart/form-data MIME\n            format.\n    \n        :param multipart_boundary:\n            If not specified, then a random boundary will be generated using\n            :func:`urllib3.filepost.choose_boundary`.\n        \"\"\"\n        if headers is None:\n            headers = self.headers\n    \n        extra_kw: dict[str, typing.Any] = {\"headers\": HTTPHeaderDict(headers)}\n        body: bytes | str\n    \n        if fields:\n            if \"body\" in urlopen_kw:\n                raise TypeError(\n                    \"request got values for both 'fields' and 'body', can only specify one.\"\n                )\n    \n            if encode_multipart:\n                body, content_type = encode_multipart_formdata(\n                    fields, boundary=multipart_boundary\n                )\n            else:\n                body, content_type = (\n                    urlencode(fields),  # type: ignore[arg-type]\n                    \"application/x-www-form-urlencoded\",\n                )\n    \n            extra_kw[\"body\"] = body\n            extra_kw[\"headers\"].setdefault(\"Content-Type\", content_type)\n    \n        extra_kw.update(urlopen_kw)\n    \n>       return self.urlopen(method, url, **extra_kw)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/_request_methods.py:278: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.poolmanager.PoolManager object at 0x7f9b1300d010>\nmethod = 'POST'\nurl = 'https://a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com:6443/api/v1/namespaces'\nredirect = True\nkw = {'assert_same_host': False, 'body': '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inf...', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'}), 'preload_content': True, ...}\nu = Url(scheme='https', auth=None, host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443, path='/api/v1/namespaces', query=None, fragment=None)\nconn = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\n\n    def urlopen(  # type: ignore[override]\n        self, method: str, url: str, redirect: bool = True, **kw: typing.Any\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Same as :meth:`urllib3.HTTPConnectionPool.urlopen`\n        with custom cross-host redirect logic and only sends the request-uri\n        portion of the ``url``.\n    \n        The given ``url`` parameter must be absolute, such that an appropriate\n        :class:`urllib3.connectionpool.ConnectionPool` can be chosen for it.\n        \"\"\"\n        u = parse_url(url)\n    \n        if u.scheme is None:\n            warnings.warn(\n                \"URLs without a scheme (ie 'https://') are deprecated and will raise an error \"\n                \"in urllib3 v3.0. To avoid this FutureWarning ensure all URLs \"\n                \"start with 'https://' or 'http://'. Read more in this issue: \"\n                \"https://github.com/urllib3/urllib3/issues/2920\",\n                category=FutureWarning,\n                stacklevel=2,\n            )\n    \n        conn = self.connection_from_host(u.host, port=u.port, scheme=u.scheme)\n    \n        kw[\"assert_same_host\"] = False\n        kw[\"redirect\"] = False\n    \n        if \"headers\" not in kw:\n            kw[\"headers\"] = self.headers\n    \n        if self._proxy_requires_url_absolute_form(u):\n            response = conn.urlopen(method, url, **kw)\n        else:\n>           response = conn.urlopen(method, u.request_uri, **kw)\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/poolmanager.py:457: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=2, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=1, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False\nerr = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n            retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n            retries.sleep()\n    \n            # Keep track of the error for the retry warning.\n            err = e\n    \n        finally:\n            if not clean_exit:\n                # We hit some kind of exception, handled or otherwise. We need\n                # to throw the connection away unless explicitly told not to.\n                # Close the connection, set the variable to None, and make sure\n                # we put the None back in the pool to avoid leaking it.\n                if conn:\n                    conn.close()\n                    conn = None\n                release_this_conn = True\n    \n            if release_this_conn:\n                # Put the connection back to be reused. If the connection is\n                # expired then it will be None, which will get replaced with a\n                # fresh connection during _get_conn.\n                self._put_conn(conn)\n    \n        if not conn:\n            # Try again\n            log.warning(\n                \"Retrying (%r) after connection broken by '%r': %s\", retries, err, url\n            )\n>           return self.urlopen(\n                method,\n                url,\n                body,\n                headers,\n                retries,\n                redirect,\n                assert_same_host,\n                timeout=timeout,\n                pool_timeout=pool_timeout,\n                release_conn=release_conn,\n                chunked=chunked,\n                body_pos=body_pos,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:872: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\nmethod = 'POST', url = '/api/v1/namespaces'\nbody = '{\"metadata\": {\"labels\": {\"kserve.io/e2e-test\": \"true\"}, \"name\": \"e2e-test-llm-inference-service-b5dd93e6\"}}'\nheaders = HTTPHeaderDict({'Accept': 'application/json', 'User-Agent': 'OpenAPI-Generator/32.0.1/python', 'Content-Type': 'application/json'})\nretries = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nredirect = False, assert_same_host = False, timeout = None, pool_timeout = None\nrelease_conn = True, chunked = False, body_pos = None, preload_content = True\ndecode_content = True, response_kw = {}, destination_scheme = None, conn = None\nrelease_this_conn = True, http_tunnel_required = False, err = None\nclean_exit = False\n\n    def urlopen(  # type: ignore[override]\n        self,\n        method: str,\n        url: str,\n        body: _TYPE_BODY | None = None,\n        headers: typing.Mapping[str, str] | None = None,\n        retries: Retry | bool | int | None = None,\n        redirect: bool = True,\n        assert_same_host: bool = True,\n        timeout: _TYPE_TIMEOUT = _DEFAULT_TIMEOUT,\n        pool_timeout: int | None = None,\n        release_conn: bool | None = None,\n        chunked: bool = False,\n        body_pos: _TYPE_BODY_POSITION | None = None,\n        preload_content: bool = True,\n        decode_content: bool = True,\n        **response_kw: typing.Any,\n    ) -> BaseHTTPResponse:\n        \"\"\"\n        Get a connection from the pool and perform an HTTP request. This is the\n        lowest level call for making a request, so you'll need to specify all\n        the raw details.\n    \n        .. note::\n    \n           More commonly, it's appropriate to use a convenience method\n           such as :meth:`request`.\n    \n        .. note::\n    \n           `release_conn` will only behave as expected if\n           `preload_content=False` because we want to make\n           `preload_content=False` the default behaviour someday soon without\n           breaking backwards compatibility.\n    \n        :param method:\n            HTTP request method (such as GET, POST, PUT, etc.)\n    \n        :param url:\n            The URL to perform the request on.\n    \n        :param body:\n            Data to send in the request body, either :class:`str`, :class:`bytes`,\n            an iterable of :class:`str`/:class:`bytes`, or a file-like object.\n    \n        :param headers:\n            Dictionary of custom headers to send, such as User-Agent,\n            If-None-Match, etc. If None, pool headers are used. If provided,\n            these headers completely replace any pool-specific headers.\n    \n        :param retries:\n            Configure the number of retries to allow before raising a\n            :class:`~urllib3.exceptions.MaxRetryError` exception.\n    \n            If ``None`` (default) will retry 3 times, see ``Retry.DEFAULT``. Pass a\n            :class:`~urllib3.util.retry.Retry` object for fine-grained control\n            over different types of retries.\n            Pass an integer number to retry connection errors that many times,\n            but no other types of errors. Pass zero to never retry.\n    \n            If ``False``, then retries are disabled and any exception is raised\n            immediately. Also, instead of raising a MaxRetryError on redirects,\n            the redirect response will be returned.\n    \n        :type retries: :class:`~urllib3.util.retry.Retry`, False, or an int.\n    \n        :param redirect:\n            If True, automatically handle redirects (status codes 301, 302,\n            303, 307, 308). Each redirect counts as a retry. Disabling retries\n            will disable redirect, too.\n    \n        :param assert_same_host:\n            If ``True``, will make sure that the host of the pool requests is\n            consistent else will raise HostChangedError. When ``False``, you can\n            use the pool on an HTTP proxy and request foreign hosts.\n    \n        :param timeout:\n            If specified, overrides the default timeout for this one\n            request. It may be a float (in seconds) or an instance of\n            :class:`urllib3.util.Timeout`.\n    \n        :param pool_timeout:\n            If set and the pool is set to block=True, then this method will\n            block for ``pool_timeout`` seconds and raise EmptyPoolError if no\n            connection is available within the time period.\n    \n        :param bool preload_content:\n            If True, the response's body will be preloaded into memory.\n    \n        :param bool decode_content:\n            If True, will attempt to decode the body based on the\n            'content-encoding' header.\n    \n        :param release_conn:\n            If False, then the urlopen call will not release the connection\n            back into the pool once a response is received (but will release if\n            you read the entire contents of the response such as when\n            `preload_content=True`). This is useful if you're not preloading\n            the response's content immediately. You will need to call\n            ``r.release_conn()`` on the response ``r`` to return the connection\n            back into the pool. If None, it takes the value of ``preload_content``\n            which defaults to ``True``.\n    \n        :param bool chunked:\n            If True, urllib3 will send the body using chunked transfer\n            encoding. Otherwise, urllib3 will send the body using the standard\n            content-length form. Defaults to False.\n    \n        :param int body_pos:\n            Position to seek to in file-like body in the event of a retry or\n            redirect. Typically this won't need to be set because urllib3 will\n            auto-populate the value when needed.\n        \"\"\"\n        # Ensure that the URL we're connecting to is properly encoded\n        if url.startswith(\"/\"):\n            # URLs starting with / are inherently schemeless.\n            url = to_str(_encode_target(url))\n            destination_scheme = None\n        else:\n            parsed_url = parse_url(url)\n            destination_scheme = parsed_url.scheme\n            url = to_str(parsed_url.url)\n    \n        if headers is None:\n            headers = self.headers\n    \n        if not isinstance(retries, Retry):\n            retries = Retry.from_int(retries, redirect=redirect, default=self.retries)\n    \n        if release_conn is None:\n            release_conn = preload_content\n    \n        # Check host\n        if assert_same_host and not self.is_same_host(url):\n            raise HostChangedError(self, url, retries)\n    \n        conn = None\n    \n        # Track whether `conn` needs to be released before\n        # returning/raising/recursing. Update this variable if necessary, and\n        # leave `release_conn` constant throughout the function. That way, if\n        # the function recurses, the original value of `release_conn` will be\n        # passed down into the recursive call, and its value will be respected.\n        #\n        # See issue #651 [1] for details.\n        #\n        # [1] <https://github.com/urllib3/urllib3/issues/651>\n        release_this_conn = release_conn\n    \n        http_tunnel_required = connection_requires_http_tunnel(\n            self.proxy, self.proxy_config, destination_scheme\n        )\n    \n        # Merge the proxy headers. Only done when not using HTTP CONNECT. We\n        # have to copy the headers dict so we can safely change it without those\n        # changes being reflected in anyone else's copy.\n        if not http_tunnel_required:\n            headers = headers.copy()  # type: ignore[attr-defined]\n            headers.update(self.proxy_headers)  # type: ignore[union-attr]\n    \n        # Must keep the exception bound to a separate variable or else Python 3\n        # complains about UnboundLocalError.\n        err = None\n    \n        # Keep track of whether we cleanly exited the except block. This\n        # ensures we do proper cleanup in finally.\n        clean_exit = False\n    \n        # Rewind body position, if needed. Record current position\n        # for future rewinds in the event of a redirect/retry.\n        body_pos = set_file_position(body, body_pos)\n    \n        try:\n            # Request a connection from the queue.\n            timeout_obj = self._get_timeout(timeout)\n            conn = self._get_conn(timeout=pool_timeout)\n    \n            conn.timeout = timeout_obj.connect_timeout  # type: ignore[assignment]\n    \n            # Is this a closed/new connection that requires CONNECT tunnelling?\n            if self.proxy is not None and http_tunnel_required and conn.is_closed:\n                try:\n                    self._prepare_proxy(conn)\n                except (BaseSSLError, OSError, SocketTimeout) as e:\n                    self._raise_timeout(\n                        err=e, url=self.proxy.url, timeout_value=conn.timeout\n                    )\n                    raise\n    \n            # If we're going to release the connection in ``finally:``, then\n            # the response doesn't need to know about the connection. Otherwise\n            # it will also try to release it and we'll have a double-release\n            # mess.\n            response_conn = conn if not release_conn else None\n    \n            # Make the request on the HTTPConnection object\n            response = self._make_request(\n                conn,\n                method,\n                url,\n                timeout=timeout_obj,\n                body=body,\n                headers=headers,\n                chunked=chunked,\n                retries=retries,\n                response_conn=response_conn,\n                preload_content=preload_content,\n                decode_content=decode_content,\n                **response_kw,\n            )\n    \n            # Everything went great!\n            clean_exit = True\n    \n        except EmptyPoolError:\n            # Didn't get a connection from the pool, no need to clean up\n            clean_exit = True\n            release_this_conn = False\n            raise\n    \n        except (\n            TimeoutError,\n            HTTPException,\n            OSError,\n            ProtocolError,\n            BaseSSLError,\n            SSLError,\n            CertificateError,\n            ProxyError,\n        ) as e:\n            # Discard the connection for these exceptions. It will be\n            # replaced during the next _get_conn() call.\n            clean_exit = False\n            new_e: Exception = e\n            if isinstance(e, (BaseSSLError, CertificateError)):\n                new_e = SSLError(e)\n            if isinstance(\n                new_e,\n                (\n                    OSError,\n                    NewConnectionError,\n                    TimeoutError,\n                    SSLError,\n                    HTTPException,\n                ),\n            ) and (conn and conn.proxy and not conn.has_connected_to_proxy):\n                new_e = _wrap_proxy_error(new_e, conn.proxy.scheme)\n            elif isinstance(new_e, (OSError, HTTPException)):\n                new_e = ProtocolError(\"Connection aborted.\", new_e)\n    \n>           retries = retries.increment(\n                method, url, error=new_e, _pool=self, _stacktrace=sys.exc_info()[2]\n            )\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/connectionpool.py:842: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nself = Retry(total=0, connect=None, read=None, redirect=None, status=None)\nmethod = 'POST', url = '/api/v1/namespaces', response = None\nerror = NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.c...a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\")\n_pool = <urllib3.connectionpool.HTTPSConnectionPool object at 0x7f9b1300de50>\n_stacktrace = <traceback object at 0x7f9b12a661c0>\n\n    def increment(\n        self,\n        method: str | None = None,\n        url: str | None = None,\n        response: BaseHTTPResponse | None = None,\n        error: Exception | None = None,\n        _pool: ConnectionPool | None = None,\n        _stacktrace: TracebackType | None = None,\n    ) -> Self:\n        \"\"\"Return a new Retry object with incremented retry counters.\n    \n        :param response: A response object, or None, if the server did not\n            return a response.\n        :type response: :class:`~urllib3.response.BaseHTTPResponse`\n        :param Exception error: An error encountered during the request, or\n            None if the response was received successfully.\n    \n        :return: A new ``Retry`` object.\n        \"\"\"\n        if self.total is False and error:\n            # Disabled, indicate to re-raise the error.\n            raise reraise(type(error), error, _stacktrace)\n    \n        total = self.total\n        if total is not None:\n            total -= 1\n    \n        connect = self.connect\n        read = self.read\n        redirect = self.redirect\n        status_count = self.status\n        other = self.other\n        cause = \"unknown\"\n        status = None\n        redirect_location = None\n    \n        if error and self._is_connection_error(error):\n            # Connect retry?\n            if connect is False:\n                raise reraise(type(error), error, _stacktrace)\n            elif connect is not None:\n                connect -= 1\n    \n        elif error and self._is_read_error(error):\n            # Read retry?\n            if read is False or method is None or not self._is_method_retryable(method):\n                raise reraise(type(error), error, _stacktrace)\n            elif read is not None:\n                read -= 1\n    \n        elif error:\n            # Other retry?\n            if other is not None:\n                other -= 1\n    \n        elif response and response.get_redirect_location():\n            # Redirect retry?\n            if redirect is not None:\n                redirect -= 1\n            cause = \"too many redirects\"\n            response_redirect_location = response.get_redirect_location()\n            if response_redirect_location:\n                redirect_location = response_redirect_location\n            status = response.status\n    \n        else:\n            # Incrementing because of a server error like a 500 in\n            # status_forcelist and the given method is in the allowed_methods\n            cause = ResponseError.GENERIC_ERROR\n            if response and response.status:\n                if status_count is not None:\n                    status_count -= 1\n                cause = ResponseError.SPECIFIC_ERROR.format(status_code=response.status)\n                status = response.status\n    \n        history = self.history + (\n            RequestHistory(method, url, error, status, redirect_location),\n        )\n    \n        new_retry = self.new(\n            total=total,\n            connect=connect,\n            read=read,\n            redirect=redirect,\n            status=status_count,\n            other=other,\n            history=history,\n        )\n    \n        if new_retry.is_exhausted():\n            reason = error or ResponseError(cause)\n>           raise MaxRetryError(_pool, url, reason) from reason  # type: ignore[arg-type]\nE           urllib3.exceptions.MaxRetryError: HTTPSConnectionPool(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Max retries exceeded with url: /api/v1/namespaces (Caused by NameResolutionError(\"HTTPSConnection(host='a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com', port=6443): Failed to resolve 'a3400844a39984b4baf14ef0992f5cf9-5d09deedd00666d5.elb.us-east-1.amazonaws.com' ([Errno -2] Name or service not known)\"))\n\n../../python/kserve/.venv/lib64/python3.11/site-packages/urllib3/util/retry.py:543: MaxRetryError"}, "teardown": {"duration": 0.0024492229858879, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}], "warnings": [{"message": "The event_loop fixture provided by pytest-asyncio has been redefined in\n/workspace/source/test/e2e/conftest.py:43\nReplacing the event_loop fixture with a custom implementation is deprecated\nand will lead to errors in the future.\nIf you want to request an asyncio event loop with a scope other than function\nscope, use the \"scope\" argument to the asyncio mark when marking the tests.\nIf you want to return different types of event loops, use the event_loop_policy\nfixture.\n", "category": "DeprecationWarning", "when": "runtest", "filename": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/pytest_asyncio/plugin.py", "lineno": 761}, {"message": "The test <Function test_flow_control_smoke[flow-control-utilization-detector]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_flow_control.py", "lineno": 47}, {"message": "The test <Function test_flow_control_smoke[flow-control-concurrency-detector]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_flow_control.py", "lineno": 47}, {"message": "The event_loop fixture provided by pytest-asyncio has been redefined in\n/workspace/source/test/e2e/conftest.py:43\nReplacing the event_loop fixture with a custom implementation is deprecated\nand will lead to errors in the future.\nIf you want to request an asyncio event loop with a scope other than function\nscope, use the \"scope\" argument to the asyncio mark when marking the tests.\nIf you want to return different types of event loops, use the event_loop_policy\nfixture.\n", "category": "DeprecationWarning", "when": "runtest", "filename": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/pytest_asyncio/plugin.py", "lineno": 761}, {"message": "The test <Function test_llm_inference_service[router-no-scheduler-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-inline-config-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-replicas-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-custom-template-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-tokenizer-kvcache-workload-llmd-simulator-kvcache]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator0]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator1]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator2]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-pd-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_stop_feature[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service_stop.py", "lineno": 40}, {"message": "The test <Function test_llm_tls_resources[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_tls.py", "lineno": 92}, {"message": "The test <Function test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}, {"message": "The test <Function test_llm_inference_service[router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 243}]}