diff --git a/docs/node_scenarios.md b/docs/node_scenarios.md index 14311387..d774dc29 100644 --- a/docs/node_scenarios.md +++ b/docs/node_scenarios.md @@ -68,25 +68,26 @@ node_scenarios: - node_crash_scenario node_name: # node on which scenario has to be injected label_selector: node-role.kubernetes.io/worker # when node_name is not specified, a node with matching label_selector is selected for node chaos scenario injection - instance_kill_count: 1 # number of times to inject each scenario under actions + instance_count: 1 # Number of nodes to perform action/select that match the label selector + runs: 1 # number of times to inject each scenario under actions (will perform on same node each time) timeout: 120 # duration to wait for completion of node scenario injection cloud_type: aws # cloud type on which Kubernetes/OpenShift runs - actions: - node_reboot_scenario node_name: label_selector: node-role.kubernetes.io/infra - instance_kill_count: 1 + instance_count: 1 timeout: 120 cloud_type: azure - actions: - node_crash_scenario node_name: label_selector: node-role.kubernetes.io/infra - instance_kill_count: 1 + instance_count: 1 timeout: 120 - actions: - stop_start_helper_node_scenario # node chaos scenario for helper node - instance_kill_count: 1 + instance_count: 1 timeout: 120 helper_node_ip: # ip address of the helper node service: # check status of the services on the helper node @@ -99,7 +100,7 @@ node_scenarios: - node_stop_start_scenario node_name: label_selector: node-role.kubernetes.io/worker - instance_kill_count: 1 + instance_count: 1 timeout: 120 cloud_type: bm bmc_user: defaultuser # For baremetal (bm) cloud type. The default IPMI username. Optional if specified for all machines. diff --git a/kraken/node_actions/common_node_functions.py b/kraken/node_actions/common_node_functions.py index bf997faf..5be0fefa 100644 --- a/kraken/node_actions/common_node_functions.py +++ b/kraken/node_actions/common_node_functions.py @@ -10,9 +10,9 @@ node_general = False # Pick a random node with specified label selector -def get_node(node_name, label_selector): +def get_node(node_name, label_selector, instance_kill_count): if node_name in kubecli.list_killable_nodes(): - return node_name + return [node_name] elif node_name: logging.info("Node with provided node_name does not exist or the node might " "be in NotReady state.") nodes = kubecli.list_killable_nodes(label_selector) @@ -20,8 +20,14 @@ def get_node(node_name, label_selector): raise Exception("Ready nodes with the provided label selector do not exist") logging.info("Ready nodes with the label selector %s: %s" % (label_selector, nodes)) number_of_nodes = len(nodes) - node = nodes[random.randint(0, number_of_nodes - 1)] - return node + if instance_kill_count == number_of_nodes: + return nodes + nodes_to_return = [] + for i in range(instance_kill_count): + node_to_add = nodes[random.randint(0, len(nodes) - 1)] + nodes_to_return.append(node_to_add) + nodes.remove(node_to_add) + return nodes_to_return # Wait till node status becomes Ready diff --git a/kraken/node_actions/run.py b/kraken/node_actions/run.py index ba33b880..e7e77ec3 100644 --- a/kraken/node_actions/run.py +++ b/kraken/node_actions/run.py @@ -64,49 +64,55 @@ def run(scenarios_list, config, wait_duration): def inject_node_scenario(action, node_scenario, node_scenario_object): generic_cloud_scenarios = ("stop_kubelet_scenario", "node_crash_scenario") # Get the node scenario configurations - instance_kill_count = node_scenario.get("instance_kill_count", 1) + run_kill_count = node_scenario.get("runs", 1) + instance_kill_count = node_scenario.get("instance_count", 1) node_name = node_scenario.get("node_name", "") label_selector = node_scenario.get("label_selector", "") timeout = node_scenario.get("timeout", 120) service = node_scenario.get("service", "") ssh_private_key = node_scenario.get("ssh_private_key", "~/.ssh/id_rsa") # Get the node to apply the scenario - node = common_node_functions.get_node(node_name, label_selector) - - if node_general and action not in generic_cloud_scenarios: - logging.info("Scenario: " + action + " is not set up for generic cloud type, skipping action") + if node_name: + node_name_list = node_name.split(",") else: - if action == "node_start_scenario": - node_scenario_object.node_start_scenario(instance_kill_count, node, timeout) - elif action == "node_stop_scenario": - node_scenario_object.node_stop_scenario(instance_kill_count, node, timeout) - elif action == "node_stop_start_scenario": - node_scenario_object.node_stop_start_scenario(instance_kill_count, node, timeout) - elif action == "node_termination_scenario": - node_scenario_object.node_termination_scenario(instance_kill_count, node, timeout) - elif action == "node_reboot_scenario": - node_scenario_object.node_reboot_scenario(instance_kill_count, node, timeout) - elif action == "stop_start_kubelet_scenario": - node_scenario_object.stop_start_kubelet_scenario(instance_kill_count, node, timeout) - elif action == "stop_kubelet_scenario": - node_scenario_object.stop_kubelet_scenario(instance_kill_count, node, timeout) - elif action == "node_crash_scenario": - node_scenario_object.node_crash_scenario(instance_kill_count, node, timeout) - elif action == "stop_start_helper_node_scenario": - if node_scenario["cloud_type"] != "openstack": - logging.error( - "Scenario: " + action + " is not supported for " - "cloud type " + node_scenario["cloud_type"] + ", skipping action" - ) + node_name_list = [node_name] + for single_node_name in node_name_list: + nodes = common_node_functions.get_node(single_node_name, label_selector, instance_kill_count) + for single_node in nodes: + if node_general and action not in generic_cloud_scenarios: + logging.info("Scenario: " + action + " is not set up for generic cloud type, skipping action") else: - if not node_scenario["helper_node_ip"]: - logging.error("Helper node IP address is not provided") - sys.exit(1) - node_scenario_object.helper_node_stop_start_scenario( - instance_kill_count, node_scenario["helper_node_ip"], timeout - ) - node_scenario_object.helper_node_service_status( - node_scenario["helper_node_ip"], service, ssh_private_key, timeout - ) - else: - logging.info("There is no node action that matches %s, skipping scenario" % action) + if action == "node_start_scenario": + node_scenario_object.node_start_scenario(run_kill_count, single_node, timeout) + elif action == "node_stop_scenario": + node_scenario_object.node_stop_scenario(run_kill_count, single_node, timeout) + elif action == "node_stop_start_scenario": + node_scenario_object.node_stop_start_scenario(run_kill_count, single_node, timeout) + elif action == "node_termination_scenario": + node_scenario_object.node_termination_scenario(run_kill_count, single_node, timeout) + elif action == "node_reboot_scenario": + node_scenario_object.node_reboot_scenario(run_kill_count, single_node, timeout) + elif action == "stop_start_kubelet_scenario": + node_scenario_object.stop_start_kubelet_scenario(run_kill_count, single_node, timeout) + elif action == "stop_kubelet_scenario": + node_scenario_object.stop_kubelet_scenario(run_kill_count, single_node, timeout) + elif action == "node_crash_scenario": + node_scenario_object.node_crash_scenario(run_kill_count, single_node, timeout) + elif action == "stop_start_helper_node_scenario": + if node_scenario["cloud_type"] != "openstack": + logging.error( + "Scenario: " + action + " is not supported for " + "cloud type " + node_scenario["cloud_type"] + ", skipping action" + ) + else: + if not node_scenario["helper_node_ip"]: + logging.error("Helper node IP address is not provided") + sys.exit(1) + node_scenario_object.helper_node_stop_start_scenario( + run_kill_count, node_scenario["helper_node_ip"], timeout + ) + node_scenario_object.helper_node_service_status( + node_scenario["helper_node_ip"], service, ssh_private_key, timeout + ) + else: + logging.info("There is no node action that matches %s, skipping scenario" % action) diff --git a/scenarios/node_scenarios_example.yml b/scenarios/node_scenarios_example.yml index 50c0ba35..e8b55fc9 100644 --- a/scenarios/node_scenarios_example.yml +++ b/scenarios/node_scenarios_example.yml @@ -3,15 +3,16 @@ node_scenarios: - node_stop_start_scenario - stop_start_kubelet_scenario - node_crash_scenario - node_name: # node on which scenario has to be injected + node_name: # node on which scenario has to be injected; can set multiple names separated by comma label_selector: node-role.kubernetes.io/worker # when node_name is not specified, a node with matching label_selector is selected for node chaos scenario injection - instance_kill_count: 1 # number of times to inject each scenario under actions + instance_count: 1 # Number of nodes to perform action/select that match the label selector + runs: 1 # number of times to inject each scenario under actions (will perform on same node each time) timeout: 120 # duration to wait for completion of node scenario injection cloud_type: aws # cloud type on which Kubernetes/OpenShift runs - actions: - node_reboot_scenario node_name: label_selector: node-role.kubernetes.io/infra - instance_kill_count: 1 + instance_count: 1 timeout: 120 cloud_type: aws