{"created": 1783624932.731711, "duration": 3925.233989238739, "exitcode": 1, "root": "/workspace/source/test/e2e", "environment": {}, "summary": {"passed": 41, "failed": 1, "total": 42, "collected": 42}, "collectors": [{"nodeid": "explainer/test_art_explainer.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/explainer/test_art_explainer.py', 38, 'Skipped: ODH does not support art explainer at the moment')"}, {"nodeid": "predictor/test_grpc.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/predictor/test_grpc.py', 35, 'Skipped: Not testable in ODH at the moment')"}, {"nodeid": "predictor/test_torchserve.py", "outcome": "skipped", "result": [], "longrepr": "('/workspace/source/test/e2e/predictor/test_torchserve.py', 34, 'Skipped: ODH does not support torchserve at the moment')"}], "tests": [{"nodeid": "llmisvc/test_gateway_section_name.py::test_gateway_section_name_propagation[cluster_single_node-cluster_cpu-with-section-name]", "lineno": 131, "outcome": "passed", "keywords": ["test_gateway_section_name_propagation[with-section-name]", "parametrize", "llmd_simulator", "cluster_single_node", "cluster_cpu", "llminferenceservice", "pytestmark", "with-section-name", "test_gateway_section_name.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.046604233008110896, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 12.950416720006615, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0009942420001607388, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 4.418872540991288, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 128.52762343300856, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0030053160153329372, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_gateway_section_name.py::test_gateway_section_name_propagation[cluster_single_node-cluster_cpu-without-section-name]", "lineno": 131, "outcome": "passed", "keywords": ["test_gateway_section_name_propagation[without-section-name]", "parametrize", "llmd_simulator", "cluster_single_node", "cluster_cpu", "llminferenceservice", "pytestmark", "without-section-name", "test_gateway_section_name.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.0389736509823706, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 41.02901752301841, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0006544650241266936, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_enabled_requires_token[cluster_cpu-cluster_single_node-auth-enabled-default]", "lineno": 221, "outcome": "passed", "keywords": ["test_llm_auth_enabled_requires_token[auth-enabled-default]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-enabled-default", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.5719519410049543, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 215.60038001602516, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.006143095990410075, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator0]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator0]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator0", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.14303713798290119, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 68.73542911800905, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0023217809794005007, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator1]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator1]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "llmd_simulator", "model_routing", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator1", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3436440459918231, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 170.4357471379917, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0036321200022939593, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_invalid_token_rejected[cluster_cpu-cluster_single_node-auth-invalid-token]", "lineno": 380, "outcome": "passed", "keywords": ["test_llm_auth_invalid_token_rejected[auth-invalid-token]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-invalid-token", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.44910331899882294, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 157.64888364900253, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0025168450083583593, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator2]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator2]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "llmd_simulator", "model_routing", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator2", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.25964829401345924, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 129.5833746099961, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.002940304984804243, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_auth.py::test_llm_auth_disabled_no_token_required[cluster_cpu-cluster_single_node-auth-disabled]", "lineno": 511, "outcome": "passed", "keywords": ["test_llm_auth_disabled_no_token_required[auth-disabled]", "parametrize", "auth", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "auth-disabled", "test_llm_auth.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 1.0378910560102668, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 164.47401893601636, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.002341131999855861, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "model_routing", "lora", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17301706300349906, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 154.8468182790093, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0023316020087804645, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "llmd_simulator", "custom_gateway", "__wrapped__", "pytestmark", "router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.9339437799935695, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 106.83448144802242, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0014950229960959405, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "model_routing", "lora", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.15972561700618826, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 126.2265959700162, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0035795699805021286, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.21960201897309162, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 177.82260165401385, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.003711922006914392, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-pvc]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-single-cpu-model-pvc]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-pvc", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 20.376356355001917, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 161.07142019300954, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.004538701003184542, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1805677700031083, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 151.7562438599998, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.002263021015096456, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-pd-cpu-model-pvc]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-pd-cpu-model-pvc]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-pd-cpu-model-pvc", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2594012870104052, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 204.36024944999372, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.003297563991509378, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "custom_gateway", "__wrapped__", "pytestmark", "router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 2.254884744004812, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 203.21143113900325, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.003150258999085054, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_multi_node-router-managed-workload-simulated-dp-ep-cpu-model-pvc]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-pvc]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_multi_node", "pvc_storage", "__wrapped__", "pytestmark", "router-managed-workload-simulated-dp-ep-cpu-model-pvc", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.39077664900105447, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 288.25914554999326, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0027964420150965452, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-pd-cpu-model-fb-opt-125m", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.18897471102536656, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 200.78230701800203, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.002998824988026172, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.23991063999710605, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 228.69703814099194, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.002958034980110824, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_v1alpha1_to_v1alpha2_conversion", "lineno": 212, "outcome": "passed", "keywords": ["test_v1alpha1_to_v1alpha2_conversion", "cluster_single_node", "cluster_cpu", "pytestmark", "TestLLMInferenceServiceConversion", "conversion", "llminferenceservice", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.05186273701838218, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.09068838699022308, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.032287674985127524, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_v1alpha2_to_v1alpha1_conversion", "lineno": 303, "outcome": "passed", "keywords": ["test_v1alpha2_to_v1alpha1_conversion", "cluster_single_node", "cluster_cpu", "pytestmark", "TestLLMInferenceServiceConversion", "conversion", "llminferenceservice", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.055127089988673106, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.09078727901214734, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.016643428010866046, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_criticality_preservation_via_annotations", "lineno": 394, "outcome": "passed", "keywords": ["test_criticality_preservation_via_annotations", "cluster_single_node", "cluster_cpu", "pytestmark", "TestLLMInferenceServiceConversion", "conversion", "llminferenceservice", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.04588286500074901, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.09670112002640963, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.040234379994217306, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_lora_criticality_preservation", "lineno": 531, "outcome": "passed", "keywords": ["test_lora_criticality_preservation", "cluster_single_node", "cluster_cpu", "pytestmark", "TestLLMInferenceServiceConversion", "conversion", "llminferenceservice", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.04594617601833306, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.13232568799867295, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0697451239975635, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_conversion.py::TestLLMInferenceServiceConversion::test_round_trip_conversion_preserves_fields", "lineno": 680, "outcome": "passed", "keywords": ["test_round_trip_conversion_preserves_fields", "cluster_single_node", "cluster_cpu", "pytestmark", "TestLLMInferenceServiceConversion", "conversion", "llminferenceservice", "test_llm_inference_service_conversion.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.04398012298042886, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 0.11597312599769793, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.06762334599625319, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service_stop.py::test_llm_stop_feature[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 39, "outcome": "passed", "keywords": ["test_llm_stop_feature[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "test_llm_inference_service_stop.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3314732639992144, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 415.9789444899943, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0023643219901714474, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "custom_gateway", "__wrapped__", "pytestmark", "router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 5.228392576012993, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 255.15415498398943, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.005550773988943547, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_lora_adapters.py::test_llm_with_lora_adapters[cluster_cpu-single-lora-adapter-hf]", "lineno": 203, "outcome": "passed", "keywords": ["test_llm_with_lora_adapters[single-lora-adapter-hf]", "parametrize", "llminferenceservice", "cluster_cpu", "lora", "__wrapped__", "pytestmark", "single-lora-adapter-hf", "test_llm_lora_adapters.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.04942052098340355, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 158.40683725601411, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.004504748998442665, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-no-scheduler-workload-single-cpu-model-fb-opt-125m]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-no-scheduler-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "no_scheduler", "__wrapped__", "pytestmark", "router-no-scheduler-workload-single-cpu-model-fb-opt-125m", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.18760232999920845, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 171.01337518298533, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0020254949922673404, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_lora_adapters.py::test_llm_with_lora_adapters[cluster_cpu-multiple-lora-adapters]", "lineno": 203, "outcome": "passed", "keywords": ["test_llm_with_lora_adapters[multiple-lora-adapters]", "parametrize", "llminferenceservice", "cluster_cpu", "lora", "__wrapped__", "pytestmark", "multiple-lora-adapters", "test_llm_lora_adapters.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.049179153982549906, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 175.87323449799442, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0020873169996775687, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_multi_node-router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_multi_node", "__wrapped__", "pytestmark", "router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3297755010135006, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 287.2824359530059, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.003249764005886391, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_tls.py::test_llm_tls_resources[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 92, "outcome": "passed", "keywords": ["test_llm_tls_resources[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "test_llm_tls.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16414272101246752, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 154.01153587701265, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0032591730123385787, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_prestop_hook.py::test_prestop_hook[cluster_cpu-cluster_single_node-router-managed-workload-single-cpu-model-fb-opt-125m]", "lineno": 40, "outcome": "passed", "keywords": ["test_prestop_hook[router-managed-workload-single-cpu-model-fb-opt-125m]", "parametrize", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-single-cpu-model-fb-opt-125m", "test_prestop_hook.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.19007677299669012, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 219.94305144401733, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.006779250019462779, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-inline-config-workload-llmd-simulator]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-inline-config-workload-llmd-simulator]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-inline-config-workload-llmd-simulator", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.3815212319896091, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 73.83612521699979, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0018357910157646984, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-model-qwen2.5-0.5b", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.16194226502557285, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 71.79853036699933, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0022767100017517805, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-configmap-ref-workload-llmd-simulator", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2103816670132801, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 58.392576565995114, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.04395561499404721, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_rolling_upgrade.py::test_rolling_upgrade_coordination[cluster_cpu-cluster_single_node-router-managed-workload-llmd-simulator-model-fb-opt-125m]", "lineno": 40, "outcome": "passed", "keywords": ["test_rolling_upgrade_coordination[router-managed-workload-llmd-simulator-model-fb-opt-125m]", "parametrize", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-workload-llmd-simulator-model-fb-opt-125m", "llmd_simulator", "test_rolling_upgrade.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.17432644497603178, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 97.54727951899986, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.002572186989709735, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-replicas-workload-llmd-simulator]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-replicas-workload-llmd-simulator]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-replicas-workload-llmd-simulator", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.33462497699656524, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 90.74421570400591, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0017733699933160096, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-with-custom-template-workload-llmd-simulator]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-with-custom-template-workload-llmd-simulator]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "__wrapped__", "pytestmark", "router-managed-scheduler-with-custom-template-workload-llmd-simulator", "llmd_simulator", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.1926969519990962, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 76.0017564209993, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.0018525010091252625, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_storage_version_migration.py::TestStorageVersionMigration::test_storage_version_migration_after_simulated_upgrade", "lineno": 113, "outcome": "passed", "keywords": ["test_storage_version_migration_after_simulated_upgrade", "cluster_single_node", "cluster_cpu", "pytestmark", "TestStorageVersionMigration", "conversion", "llminferenceservice", "test_storage_version_migration.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.049224779999349266, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 63.13743918700493, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 2.3057140450109728, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]", "lineno": 244, "outcome": "passed", "keywords": ["test_llm_inference_service[router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.2051012009906117, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 76.32731458899798, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "teardown": {"duration": 0.004176482005277649, "outcome": "passed", "longrepr": "[gw1] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}, {"nodeid": "llmisvc/test_llm_inference_service.py::test_llm_inference_service[cluster_cpu-cluster_single_node-router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]", "lineno": 244, "outcome": "failed", "keywords": ["test_llm_inference_service[router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]", "parametrize", "asyncio", "llminferenceservice", "cluster_cpu", "cluster_single_node", "llmd_simulator", "__wrapped__", "pytestmark", "router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd", "test_llm_inference_service.py", "llmisvc/__init__.py", "e2e"], "setup": {"duration": 0.703145781008061, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}, "call": {"duration": 1163.5221081400232, "outcome": "failed", "crash": {"path": "/workspace/source/test/e2e/llmisvc/test_llm_inference_service.py", "lineno": 1115, "message": "AssertionError: Service returned 502:"}, "traceback": [{"path": "llmisvc/test_llm_inference_service.py", "lineno": 816, "message": ""}, {"path": "llmisvc/logging.py", "lineno": 40, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1119, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1215, "message": ""}, {"path": "llmisvc/test_llm_inference_service.py", "lineno": 1115, "message": "AssertionError"}], "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python\n\ntest_case = TestCase(base_refs=['router-managed', 'scheduler-v06-pd-config-migration', 'workload-llmd-simulator-pd'], prompt='KSer...              {'name': 'workload-llmd-simulator-pd-sche-64d0fc8b'}]},\n 'status': None}, model_name='facebook/opt-125m')\n\n    @pytest.mark.llminferenceservice\n    @pytest.mark.asyncio(loop_scope=\"session\")\n    @pytest.mark.parametrize(\n        \"test_case\",\n        [\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-with-gateway-ref\",\n                        \"router-with-managed-route\",\n                        \"model-fb-opt-125m\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/completions\",\n                    prompt=\"KServe is a\",\n                    payload_formatter=completions_payload,\n                    response_assertion=create_response_assertion(with_field=\"choices\"),\n                    expected_gateway=ROUTER_GATEWAYS[0],\n                    before_test=[\n                        lambda: create_router_resources(\n                            gateways=[ROUTER_GATEWAYS[0]],\n                        )\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                    pytest.mark.custom_gateway,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    payload_formatter=completions_payload,\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-custom-route-timeout\",\n                        \"scheduler-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"custom-route-timeout-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-with-refs\",\n                        \"scheduler-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"router-with-refs-test\",\n                    expected_gateway=ROUTER_GATEWAYS[0],\n                    before_test=[\n                        lambda: create_router_resources(\n                            gateways=[ROUTER_GATEWAYS[0]],\n                            routes=[ROUTER_ROUTES[0], ROUTER_ROUTES[1]],\n                        )\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.custom_gateway,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\"router-managed\", \"workload-pd-cpu\", \"model-fb-opt-125m\"],\n                    prompt=\"You are an expert in Kubernetes-native machine learning serving platforms, with deep knowledge of the KServe project. \"\n                    \"Explain the challenges of serving large-scale models, GPU scheduling, and how KServe integrates with capabilities like multi-model serving. \"\n                    \"Provide a detailed comparison with open source alternatives, focusing on operational trade-offs.\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-custom-route-timeout-pd\",\n                        \"scheduler-managed\",\n                        \"workload-pd-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"You are an expert in Kubernetes-native machine learning serving platforms, with deep knowledge of the KServe project. \"\n                    \"Explain the challenges of serving large-scale models, GPU scheduling, and how KServe integrates with capabilities like multi-model serving. \"\n                    \"Provide a detailed comparison with open source alternatives, focusing on operational trade-offs.\",\n                    service_name=\"custom-route-timeout-pd-test\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-with-refs-pd\",\n                        \"scheduler-managed\",\n                        \"workload-pd-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"You are an expert in Kubernetes-native machine learning serving platforms, with deep knowledge of the KServe project. \"\n                    \"Explain the challenges of serving large-scale models, GPU scheduling, and how KServe integrates with capabilities like multi-model serving. \"\n                    \"Provide a detailed comparison with open source alternatives, focusing on operational trade-offs.\",\n                    service_name=\"router-with-refs-pd-test\",\n                    response_assertion=assert_200_with_choices,\n                    expected_gateway=ROUTER_GATEWAYS[1],\n                    before_test=[\n                        lambda: create_router_resources(\n                            gateways=[ROUTER_GATEWAYS[1]],\n                            routes=[ROUTER_ROUTES[2], ROUTER_ROUTES[3]],\n                        )\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.custom_gateway,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-dp-ep-gpu\",\n                        \"workload-dp-ep-prefill-gpu\",\n                        \"model-deepseek-v2-lite\",\n                    ],\n                    prompt=\"Delve into the multifaceted implications of a fully disaggregated cloud architecture, specifically \"\n                    \"where the compute plane (P) and the data plane (D) are independently deployed and managed for a \"\n                    \"geographically distributed, high-throughput, low-latency microservices ecosystem. Beyond the \"\n                    \"fundamental challenges of network latency and data consistency, elaborate on the advanced \"\n                    \"considerations and trade-offs inherent in such a setup: 1. Network Architecture and Protocols: \"\n                    \"How would the network fabric and underlying protocols (e.g., RDMA, custom transport layers) need to \"\n                    \"evolve to support optimal performance and minimize inter-plane communication overhead, especially for \"\n                    \"synchronous operations? Discuss the role of network programmability (e.g., SDN, P4) in dynamically \"\n                    \"optimizing routing and traffic flow between P and D. 2. Advanced Data Consistency and Durability: \"\n                    \"Explore sophisticated data consistency models (e.g., causal consistency, strong eventual consistency) \"\n                    \"and their applicability in balancing performance and data integrity across a globally distributed data plane. \"\n                    \"Detail strategies for ensuring data durability and fault tolerance, including multi-region replication, \"\n                    \"intelligent partitioning, and recovery mechanisms in the event of partial or full plane failures. \"\n                    \"3. Dynamic Resource Orchestration and Cost Optimization: Analyze how an orchestration layer would intelligently \"\n                    \"manage the independent scaling of compute (P) and data (D) resources, considering fluctuating workloads, \"\n                    \"cost efficiency, and performance targets (e.g., using predictive analytics for resource provisioning). \"\n                    \"Discuss mechanisms for dynamically reallocating compute nodes to different data partitions based on \"\n                    \"workload patterns and data locality, potentially involving live migration strategies. \"\n                    \"4. Security and Compliance in a Distributed Landscape: Address the enhanced security perimeter \"\n                    \"challenges, including securing communication channels between P and D (encryption in transit, mutual TLS), \"\n                    \"fine-grained access control to data at rest and in motion, and identity management across disaggregated \"\n                    \"components. Discuss how such an architecture impacts compliance with regulatory frameworks (e.g., GDPR, HIPAA) \"\n                    \"concerning data sovereignty, privacy, and auditability. 5. Operational Complexity and Observability: \"\n                    \"Examine the increased complexity in monitoring, logging, and tracing across highly decoupled compute and \"\n                    \"data planes. What specialized tooling and practices (e.g., distributed tracing with OpenTelemetry, advanced AIOps) \"\n                    \"would be essential? How would incident response and troubleshooting differ in this disaggregated environment \"\n                    \"compared to traditional integrated systems? Consider the challenges of pinpointing root causes across \"\n                    \"independent failures. 6. Real-world Applicability and Future Trends: Identify specific industries \"\n                    \"or use cases (e.g., high-frequency trading, IoT edge processing, large language model inference) \"\n                    \"where the benefits of P/D disaggregation would strongly outweigh its complexities. \"\n                    \"Conclude by speculating on emerging technologies or paradigms (e.g., serverless compute functions \"\n                    \"directly interacting with object storage, in-memory disaggregation) that could further drive or \"\n                    \"transform P/D disaggregation in cloud computing.\",\n                    max_tokens=2000,\n                ),\n                marks=[\n                    pytest.mark.cluster_gpu,\n                    pytest.mark.cluster_nvidia,\n                    pytest.mark.cluster_nvidia_roce,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-no-scheduler\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"What is KServe?\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.no_scheduler,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-simulated-dp-ep-cpu\",\n                        \"model-fb-opt-125m\",\n                    ],\n                    prompt=\"This test simulates DP+EP that can run on CPU, the idea is to test the LWS-based deployment, \"\n                    \"but without the resources requirements for DP+EP (GPUs and ROCe/IB).\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_multi_node],\n            ),\n            # Scheduler config tests\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-inline-config\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-inline-config-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            # Chat completions endpoint coverage\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                        \"model-qwen2.5-0.5b\",\n                    ],\n                    model_name=\"Qwen/Qwen2.5-0.5B-Instruct\",\n                    endpoint=\"/v1/chat/completions\",\n                    prompt=\"What is KServe?\",\n                    payload_formatter=chat_completions_payload,\n                    response_assertion=create_response_assertion(with_field=\"choices\"),\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-configmap-ref\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-configmap-ref-test\",\n                    before_test=[create_scheduler_configmap],\n                    after_test=[delete_scheduler_configmap],\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-replicas\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-ha-replicas-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-custom-template\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-custom-template-test\",\n                ),\n                marks=[pytest.mark.cluster_cpu, pytest.mark.cluster_single_node],\n            ),\n            # Scheduler v0.6 \u2192 v0.7 migration tests.\n            # Deploy v0.6-style configs and verify the controller migrates them\n            # so the v0.7 scheduler boots successfully.\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-v06-pd-config-migration\",\n                        \"workload-llmd-simulator-pd\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-v06-pd-migration-test\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-v06-nonzero-threshold-migration\",\n                        \"workload-llmd-simulator-pd\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"scheduler-v06-threshold-migration-test\",\n                    response_assertion=assert_200_with_choices,\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Precise prefix KV cache routing test\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"scheduler-with-precise-prefix-cache-inline-config\",\n                        \"workload-llmd-simulator-kvcache\",\n                    ],\n                    prompt=\"KServe is a\",\n                    service_name=\"precise-prefix-cache-test\",\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Models endpoint coverage\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/models\",\n                    response_assertion=create_response_assertion(with_field=\"data\"),\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 /v1/completions\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/completions\",\n                    prompt=\"KServe is a\",\n                    payload_formatter=completions_payload,\n                    response_assertion=assert_model_field_matches(\"facebook/opt-125m\"),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: f\"publishers/{KSERVE_TEST_NAMESPACE}/models/facebook/opt-125m\",\n                    },\n                    peers=[\n                        TestCase(\n                            base_refs=[\n                                \"router-managed\",\n                                \"workload-llmd-simulator\",\n                                \"model-qwen2.5-0.5b\",\n                            ],\n                            endpoint=\"/v1/completions\",\n                            prompt=\"KServe is a\",\n                            payload_formatter=completions_payload,\n                            response_assertion=assert_model_field_matches(\n                                \"Qwen/Qwen2.5-0.5B-Instruct\"\n                            ),\n                            url_getter=get_model_routing_url,\n                            extra_headers={\n                                MODEL_ROUTING_HEADER: f\"publishers/{KSERVE_TEST_NAMESPACE}/models/Qwen/Qwen2.5-0.5B-Instruct\",\n                            },\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                    pytest.mark.model_routing,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 /v1/chat/completions\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-llmd-simulator\",\n                    ],\n                    endpoint=\"/v1/chat/completions\",\n                    prompt=\"What is KServe?\",\n                    payload_formatter=chat_completions_payload,\n                    response_assertion=assert_model_field_matches(\"facebook/opt-125m\"),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: f\"publishers/{KSERVE_TEST_NAMESPACE}/models/facebook/opt-125m\",\n                    },\n                    peers=[\n                        TestCase(\n                            base_refs=[\n                                \"router-managed\",\n                                \"workload-llmd-simulator\",\n                                \"model-qwen2.5-0.5b\",\n                            ],\n                            endpoint=\"/v1/chat/completions\",\n                            prompt=\"What is KServe?\",\n                            payload_formatter=chat_completions_payload,\n                            response_assertion=assert_model_field_matches(\n                                \"Qwen/Qwen2.5-0.5B-Instruct\"\n                            ),\n                            url_getter=get_model_routing_url,\n                            extra_headers={\n                                MODEL_ROUTING_HEADER: f\"publishers/{KSERVE_TEST_NAMESPACE}/models/Qwen/Qwen2.5-0.5B-Instruct\",\n                            },\n                        ),\n                    ],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.llmd_simulator,\n                    pytest.mark.model_routing,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 LoRA adapter\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m-with-lora-hf\",\n                    ],\n                    endpoint=\"/v1/completions\",\n                    prompt=\"KServe is a\",\n                    model_name=f\"publishers/{KSERVE_TEST_NAMESPACE}/models/lora-adapter-1\",\n                    payload_formatter=completions_payload,\n                    response_assertion=assert_model_field_matches(\n                        f\"publishers/{KSERVE_TEST_NAMESPACE}/models/lora-adapter-1\"\n                    ),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: f\"publishers/{KSERVE_TEST_NAMESPACE}/models/lora-adapter-1\",\n                    },\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.model_routing,\n                    pytest.mark.lora,\n                ],\n            ),\n            # Model-based routing via X-Gateway-Model-Name header \u2014 /v1/models (base + LoRA)\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-fb-opt-125m-with-lora-hf\",\n                    ],\n                    endpoint=\"/v1/models\",\n                    response_assertion=assert_models_contains(\n                        \"facebook/opt-125m\",\n                        f\"publishers/{KSERVE_TEST_NAMESPACE}/models/facebook/opt-125m\",\n                        \"lora-adapter-1\",\n                        f\"publishers/{KSERVE_TEST_NAMESPACE}/models/lora-adapter-1\",\n                    ),\n                    url_getter=get_model_routing_url,\n                    extra_headers={\n                        MODEL_ROUTING_HEADER: f\"publishers/{KSERVE_TEST_NAMESPACE}/models/facebook/opt-125m\",\n                    },\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.model_routing,\n                    pytest.mark.lora,\n                ],\n            ),\n            # PVC storage tests -- validate direct PVC volume mount with real vLLM serving\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-single-cpu\",\n                        \"model-pvc\",\n                    ],\n                    prompt=\"KServe is a\",\n                    response_assertion=assert_200_with_choices,\n                    before_test=[ensure_pvc_with_model],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.pvc_storage,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-pd-cpu\",\n                        \"model-pvc\",\n                    ],\n                    prompt=\"KServe is a\",\n                    response_assertion=assert_200_with_choices,\n                    before_test=[ensure_pvc_with_model],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_single_node,\n                    pytest.mark.pvc_storage,\n                ],\n            ),\n            pytest.param(\n                TestCase(\n                    base_refs=[\n                        \"router-managed\",\n                        \"workload-simulated-dp-ep-cpu\",\n                        \"model-pvc\",\n                    ],\n                    prompt=\"KServe is a\",\n                    before_test=[ensure_pvc_with_model],\n                ),\n                marks=[\n                    pytest.mark.cluster_cpu,\n                    pytest.mark.cluster_multi_node,\n                    pytest.mark.pvc_storage,\n                ],\n            ),\n        ],\n        indirect=[\"test_case\"],\n        ids=generate_test_id,\n    )\n    @log_execution\n    def test_llm_inference_service(test_case: TestCase):  # noqa: F811\n        inject_k8s_proxy()\n    \n        kserve_client = KServeClient(\n            config_file=os.environ.get(\"KUBECONFIG\", \"~/.kube/config\"),\n            client_configuration=client.Configuration(),\n        )\n    \n        service_name = test_case.llm_service.metadata.name\n        if not test_case.llm_service.metadata.annotations:\n            test_case.llm_service.metadata.annotations = {}\n    \n        test_case.llm_service.metadata.annotations[\n            \"security.opendatahub.io/enable-auth\"\n        ] = \"false\"\n        prefix = test_case.log_prefix\n    \n        test_failed = False\n        try:\n            print(f\"{prefix} Creating LLMInferenceService {service_name}\")\n            create_llmisvc(kserve_client, test_case.llm_service)\n            print(f\"{prefix} Waiting for LLMInferenceService {service_name} to be ready\")\n            wait_for_llm_isvc_ready(\n                kserve_client, test_case.llm_service, test_case.wait_timeout\n            )\n            print(f\"{prefix} Waiting for model response from {service_name}\")\n>           wait_for_model_response(\n                kserve_client,\n                test_case,\n                test_case.wait_timeout,\n                extra_headers=test_case.extra_headers,\n            )\n\nllmisvc/test_llm_inference_service.py:816: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nargs = (<kserve.api.kserve_client.KServeClient object at 0x7fba4fa693d0>, TestCase(base_refs=['router-managed', 'scheduler-v0...        {'name': 'workload-llmd-simulator-pd-sche-64d0fc8b'}]},\n 'status': None}, model_name='facebook/opt-125m'), 900)\nkwargs = {'extra_headers': None}, func_name = 'wait_for_model_response'\ntimestamp_start = '2026-07-09T19:03:46.429233', start_time = 1783623826.430062\nduration = 1103.1077423095703, timestamp_end = '2026-07-09T19:22:09.537807'\n\n    @functools.wraps(func)\n    def wrapper(*args, **kwargs):\n        func_name = func.__name__\n    \n        timestamp_start = datetime.now().isoformat()\n        logger.info(\n            f\"[{func_name}] [{timestamp_start}] start - args={args}, kwargs={kwargs}\"\n        )\n        start_time = time.time()\n    \n        try:\n>           result = func(*args, **kwargs)\n\nllmisvc/logging.py:40: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nkserve_client = <kserve.api.kserve_client.KServeClient object at 0x7fba4fa693d0>\ntest_case = TestCase(base_refs=['router-managed', 'scheduler-v06-pd-config-migration', 'workload-llmd-simulator-pd'], prompt='KSer...              {'name': 'workload-llmd-simulator-pd-sche-64d0fc8b'}]},\n 'status': None}, model_name='facebook/opt-125m')\ntimeout_seconds = 900, extra_headers = None\n\n    @log_execution\n    def wait_for_model_response(\n        kserve_client: KServeClient,\n        test_case: TestCase,  # noqa: F811\n        timeout_seconds: int = 900,\n        extra_headers: Optional[Dict[str, str]] = None,\n    ) -> str:\n        def get_successful_response():\n            try:\n                if test_case.url_getter:\n                    service_url = test_case.url_getter(kserve_client, test_case.llm_service)\n                else:\n                    service_url = get_llm_service_url(kserve_client, test_case.llm_service)\n            except Exception as e:\n                raise AssertionError(f\"\u274c Failed to get service URL: {e}\") from e\n    \n            model_url = service_url + test_case.endpoint\n    \n            headers = {\"Content-Type\": \"application/json\"}\n            if extra_headers:\n                headers.update(extra_headers)\n    \n            if test_case.payload_formatter is not None:\n                test_payload = test_case.payload_formatter(test_case)\n            elif test_case.prompt is not None:\n                test_payload = {\n                    \"model\": test_case.model_name\n                    if not extra_headers or MODEL_ROUTING_HEADER not in extra_headers\n                    else extra_headers[MODEL_ROUTING_HEADER],\n                    \"prompt\": test_case.prompt,\n                    \"max_tokens\": test_case.max_tokens,\n                }\n            else:\n                test_payload = None\n    \n            logger.info(f\"Calling LLM service at {model_url} with payload {test_payload}\")\n            try:\n                if test_payload is not None:\n                    response = post_with_retry(\n                        model_url,\n                        headers=headers,\n                        json_data=test_payload,\n                        timeout=test_case.response_timeout,\n                    )\n                else:\n                    response = get_with_retry(\n                        model_url,\n                        headers=headers,\n                        timeout=test_case.response_timeout,\n                    )\n            except Exception as e:\n                logger.error(f\"\u274c Failed to call model: {e}\")\n                raise AssertionError(f\"\u274c Failed to call model: {e}\") from e\n    \n            logger.info(f\"Model response is {response.status_code}: {response.text[:500]}\")\n    \n            if 200 <= response.status_code < 300:\n                return response\n            raise AssertionError(\n                f\"Service returned {response.status_code}: {response.text}\"\n            )\n    \n>       response = wait_for(get_successful_response, timeout=timeout_seconds, interval=5.0)\n\nllmisvc/test_llm_inference_service.py:1119: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\nassertion_fn = <function wait_for_model_response.<locals>.get_successful_response at 0x7fba4fc02660>\ntimeout = 900, interval = 5.0\n\n    def wait_for(\n        assertion_fn: Callable[[], Any], timeout: float = 5.0, interval: float = 0.1\n    ) -> Any:\n        \"\"\"Wait for the assertion to succeed within timeout.\"\"\"\n        deadline = time.time() + timeout\n        last_msg = None\n        while True:\n            try:\n>               return assertion_fn()\n\nllmisvc/test_llm_inference_service.py:1215: \n_ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ _ \n\n    def get_successful_response():\n        try:\n            if test_case.url_getter:\n                service_url = test_case.url_getter(kserve_client, test_case.llm_service)\n            else:\n                service_url = get_llm_service_url(kserve_client, test_case.llm_service)\n        except Exception as e:\n            raise AssertionError(f\"\u274c Failed to get service URL: {e}\") from e\n    \n        model_url = service_url + test_case.endpoint\n    \n        headers = {\"Content-Type\": \"application/json\"}\n        if extra_headers:\n            headers.update(extra_headers)\n    \n        if test_case.payload_formatter is not None:\n            test_payload = test_case.payload_formatter(test_case)\n        elif test_case.prompt is not None:\n            test_payload = {\n                \"model\": test_case.model_name\n                if not extra_headers or MODEL_ROUTING_HEADER not in extra_headers\n                else extra_headers[MODEL_ROUTING_HEADER],\n                \"prompt\": test_case.prompt,\n                \"max_tokens\": test_case.max_tokens,\n            }\n        else:\n            test_payload = None\n    \n        logger.info(f\"Calling LLM service at {model_url} with payload {test_payload}\")\n        try:\n            if test_payload is not None:\n                response = post_with_retry(\n                    model_url,\n                    headers=headers,\n                    json_data=test_payload,\n                    timeout=test_case.response_timeout,\n                )\n            else:\n                response = get_with_retry(\n                    model_url,\n                    headers=headers,\n                    timeout=test_case.response_timeout,\n                )\n        except Exception as e:\n            logger.error(f\"\u274c Failed to call model: {e}\")\n            raise AssertionError(f\"\u274c Failed to call model: {e}\") from e\n    \n        logger.info(f\"Model response is {response.status_code}: {response.text[:500]}\")\n    \n        if 200 <= response.status_code < 300:\n            return response\n>       raise AssertionError(\n            f\"Service returned {response.status_code}: {response.text}\"\n        )\nE       AssertionError: Service returned 502:\n\nllmisvc/test_llm_inference_service.py:1115: AssertionError"}, "teardown": {"duration": 0.003165920003084466, "outcome": "passed", "longrepr": "[gw0] linux -- Python 3.11.13 /workspace/source/python/kserve/.venv/bin/python"}}], "warnings": [{"message": "The event_loop fixture provided by pytest-asyncio has been redefined in\n/workspace/source/test/e2e/conftest.py:43\nReplacing the event_loop fixture with a custom implementation is deprecated\nand will lead to errors in the future.\nIf you want to request an asyncio event loop with a scope other than function\nscope, use the \"scope\" argument to the asyncio mark when marking the tests.\nIf you want to return different types of event loops, use the event_loop_policy\nfixture.\n", "category": "DeprecationWarning", "when": "runtest", "filename": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/pytest_asyncio/plugin.py", "lineno": 761}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-precise-prefix-cache-inline-config-workload-llmd-simulator-kvcache]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator0]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator1]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator2]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf0]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The event_loop fixture provided by pytest-asyncio has been redefined in\n/workspace/source/test/e2e/conftest.py:43\nReplacing the event_loop fixture with a custom implementation is deprecated\nand will lead to errors in the future.\nIf you want to request an asyncio event loop with a scope other than function\nscope, use the \"scope\" argument to the asyncio mark when marking the tests.\nIf you want to return different types of event loops, use the event_loop_policy\nfixture.\n", "category": "DeprecationWarning", "when": "runtest", "filename": "/workspace/source/python/kserve/.venv/lib64/python3.11/site-packages/pytest_asyncio/plugin.py", "lineno": 761}, {"message": "The test <Function test_llm_inference_service[router-with-gateway-ref-router-with-managed-route-model-fb-opt-125m-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m-with-lora-hf1]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-single-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-custom-route-timeout-scheduler-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-pd-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-with-refs-scheduler-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-pd-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-pvc]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-custom-route-timeout-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_stop_feature[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service_stop.py", "lineno": 40}, {"message": "The test <Function test_llm_inference_service[router-with-refs-pd-scheduler-managed-workload-pd-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-no-scheduler-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_tls_resources[router-managed-workload-single-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_tls.py", "lineno": 93}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-simulated-dp-ep-cpu-model-fb-opt-125m]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-inline-config-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-workload-llmd-simulator-model-qwen2.5-0.5b]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-configmap-ref-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-replicas-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-with-custom-template-workload-llmd-simulator]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-v06-nonzero-threshold-migration-workload-llmd-simulator-pd]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}, {"message": "The test <Function test_llm_inference_service[router-managed-scheduler-v06-pd-config-migration-workload-llmd-simulator-pd]> is marked with '@pytest.mark.asyncio' but it is not an async function. Please remove the asyncio mark. If the test is not marked explicitly, check for global marks applied via 'pytestmark'.", "category": "PytestWarning", "when": "runtest", "filename": "llmisvc/test_llm_inference_service.py", "lineno": 245}]}