mirror of
https://github.com/clearlinux/cloud-native-setup.git
synced 2026-08-18 21:16:16 +00:00
Compare commits
51 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 52d1a8406b | |||
| 61b8702472 | |||
| 00c1d60470 | |||
| 07c2231e62 | |||
| 4ef8d34671 | |||
| bc0f257176 | |||
| e70e32d36e | |||
| 518fa87f27 | |||
| 6cd87d74be | |||
| 927ceddc9c | |||
| 6298cf2054 | |||
| acf5a95177 | |||
| 68a62f50bc | |||
| 9431dd9f38 | |||
| 9b8c7c093f | |||
| bdcb4fb5b7 | |||
| a0ca2a2017 | |||
| 46b3f230ee | |||
| b6c7cf1b8e | |||
| f146c771cc | |||
| 9510b068e0 | |||
| 952e037420 | |||
| c846e9753d | |||
| 39c7cc643a | |||
| 54adf53cdd | |||
| 00df885b45 | |||
| a76cc3437e | |||
| e09285f1e1 | |||
| b4e6813ed6 | |||
| 7efe99f139 | |||
| 103bfcc681 | |||
| 9574f44b20 | |||
| f7254e2b30 | |||
| df0af2ab2c | |||
| e985ffd6e0 | |||
| 7840d720b0 | |||
| 43fa8bad8f | |||
| da6087762a | |||
| 5b03651467 | |||
| 598289573b | |||
| 6972963363 | |||
| d271a73fe3 | |||
| 674ab84d0d | |||
| 3984a18b18 | |||
| 02880b82ee | |||
| 88b899d7e3 | |||
| 2ae8360760 | |||
| aa86c554e8 | |||
| ad2fc108d3 | |||
| 4edbaebf87 | |||
| e7b7d33be0 |
@@ -0,0 +1,4 @@
|
||||
resources:
|
||||
- canal/canal.yaml
|
||||
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
resources:
|
||||
cilium/cilium.yaml
|
||||
@@ -0,0 +1,2 @@
|
||||
resources:
|
||||
cilium/cilium.yaml
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
resources:
|
||||
- flannel/Documentation/kube-flannel.yml
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
resources:
|
||||
- flannel/Documentation/kube-flannel.yml
|
||||
@@ -0,0 +1,8 @@
|
||||
resources:
|
||||
- metrics-server/deploy/1.8+/aggregated-metrics-reader.yaml
|
||||
- metrics-server/deploy/1.8+/auth-delegator.yaml
|
||||
- metrics-server/deploy/1.8+/auth-reader.yaml
|
||||
- metrics-server/deploy/1.8+/metrics-apiservice.yaml
|
||||
- metrics-server/deploy/1.8+/metrics-server-deployment.yaml
|
||||
- metrics-server/deploy/1.8+/metrics-server-service.yaml
|
||||
- metrics-server/deploy/1.8+/resource-reader.yaml
|
||||
@@ -6,7 +6,7 @@ patchesJson6902:
|
||||
# adds "networking.k8s.io" to ClusterRole's apiGroups
|
||||
- target:
|
||||
group: rbac.authorization.k8s.io
|
||||
version: v1beta1
|
||||
version: v1
|
||||
kind: ClusterRole
|
||||
name: nginx-ingress-clusterrole
|
||||
path: patch_clusterrole.yaml
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
resources:
|
||||
- ingress-nginx/deploy/static/mandatory.yaml
|
||||
- ingress-nginx/deploy/static/provider/baremetal/service-nodeport.yaml
|
||||
@@ -0,0 +1,6 @@
|
||||
resources:
|
||||
- metallb/manifests/example-layer2-config.yaml
|
||||
- metallb/manifests/metallb.yaml
|
||||
|
||||
patchesStrategicMerge:
|
||||
- patch_configmap.yaml
|
||||
@@ -0,0 +1,11 @@
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
name: config
|
||||
data:
|
||||
config: |
|
||||
address-pools:
|
||||
- name: my-ip-space
|
||||
protocol: layer2
|
||||
addresses:
|
||||
- 10.0.0.240/28
|
||||
@@ -1,5 +1,5 @@
|
||||
# operator
|
||||
apiVersion: apps/v1beta1
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: rook-ceph-operator
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
resources:
|
||||
- rook/cluster/examples/kubernetes/ceph/common.yaml
|
||||
- rook/cluster/examples/kubernetes/ceph/operator.yaml
|
||||
- rook/cluster/examples/kubernetes/ceph/cluster.yaml
|
||||
- rook/cluster/examples/kubernetes/ceph/csi/rbd/storageclass.yaml
|
||||
|
||||
patchesStrategicMerge:
|
||||
# patches rook to use 'directories' instead of partitions.
|
||||
# comment out to use partitions
|
||||
- patch_cephcluster.yaml
|
||||
@@ -0,0 +1,9 @@
|
||||
apiVersion: ceph.rook.io/v1
|
||||
kind: CephCluster
|
||||
metadata:
|
||||
name: rook-ceph
|
||||
namespace: rook-ceph
|
||||
spec:
|
||||
storage:
|
||||
directories:
|
||||
- path: /var/lib/rook
|
||||
@@ -4,3 +4,8 @@ resources:
|
||||
- packaging/kata-deploy/k8s-1.14/kata-fc-runtimeClass.yaml
|
||||
- packaging/kata-deploy/k8s-1.14/kata-qemu-runtimeClass.yaml
|
||||
|
||||
images:
|
||||
# change 'latest' to specified version
|
||||
- name: katadocker/kata-deploy
|
||||
newName: katadocker/kata-deploy
|
||||
newTag: 1.8.2
|
||||
@@ -0,0 +1,11 @@
|
||||
resources:
|
||||
- packaging/kata-deploy/kata-deploy.yaml
|
||||
- packaging/kata-deploy/kata-rbac.yaml
|
||||
- packaging/kata-deploy/k8s-1.14/kata-fc-runtimeClass.yaml
|
||||
- packaging/kata-deploy/k8s-1.14/kata-qemu-runtimeClass.yaml
|
||||
|
||||
images:
|
||||
# change 'latest' to specified version
|
||||
- name: katadocker/kata-deploy
|
||||
newName: katadocker/kata-deploy
|
||||
newTag: 1.9.1
|
||||
@@ -1,5 +1,5 @@
|
||||
---
|
||||
apiVersion: apiextensions.k8s.io/v1beta1
|
||||
apiVersion: apiextensions.k8s.io/v1
|
||||
kind: CustomResourceDefinition
|
||||
metadata:
|
||||
# name must match the spec fields below, and be in the form: <plural>.<group>
|
||||
|
||||
@@ -119,7 +119,7 @@ patchesJson6902:
|
||||
# adds "networking.k8s.io" to ClusterRole's apiGroups
|
||||
- target:
|
||||
group: rbac.authorization.k8s.io
|
||||
version: v1beta1
|
||||
version: v1
|
||||
kind: ClusterRole
|
||||
name: nginx-ingress-clusterrole
|
||||
path: patch_clusterrole.yaml
|
||||
@@ -137,7 +137,7 @@ itself contains the operation to perform, target path and value. The `rules/3/ap
|
||||
operation (in this case "add") at the `apiGroups:` list found under the 4th list item of `rules:`.
|
||||
|
||||
```yaml
|
||||
apiVersion: rbac.authorization.k8s.io/v1beta1
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRole
|
||||
metadata:
|
||||
name: nginx-ingress-clusterrole
|
||||
|
||||
@@ -13,8 +13,8 @@ Follow instructions in the [Vagrant docs](https://www.vagrantup.com/intro/gettin
|
||||
|
||||
Or, follow our [detailed steps](vagrant.md)
|
||||
|
||||
Now you have a 3 node cluster up and running. Each of them have 2 vCPU, 4GB Memory, 2x10GB disks, 1 additional private network.
|
||||
Customize the setup using environment variables. E.g., `NODES=1 MEMORY=8192 CPUS=8 vagrant up --provider=libvirt`
|
||||
Now you have a 3 node cluster up and running. Each of them have 4 vCPU, 8GB Memory, 2x10GB disks, 1 additional private network.
|
||||
Customize the setup using environment variables. E.g., `NODES=2 MEMORY=16384 CPUS=8 vagrant up --provider=libvirt`
|
||||
|
||||
To login to the master node and change to this directory
|
||||
|
||||
@@ -38,6 +38,14 @@ This script ensures the following
|
||||
script uses the runtime specified in the `RUNNER` environment variable and defaults to `crio`. To use the
|
||||
`containerd` runtime, set the `RUNNER` environment variable to `containerd`.
|
||||
|
||||
In case of vagrant, if you want to spin up VM's using different environment variable than declared in [`setup_system.sh`],
|
||||
specify when performing vagrant up. E.g., `RUNNER=containerd vagrant up`
|
||||
|
||||
### Specify a version of Clear Linux
|
||||
|
||||
To specify a particular version of Clear Linux to use, set the CLRK8S_CLR_VER environment variable to the desired
|
||||
version before starting setup_system.sh (e.g. `CLRK8S_CLR_VER=31400 ./setup_system.sh`)
|
||||
|
||||
### Configuration for high numbers of pods per node
|
||||
|
||||
In order to enable running greater than 110 pods per node, set the environment
|
||||
@@ -65,6 +73,11 @@ you need to setup other cluster wide properties.
|
||||
There are different flavors to install, run `./create_stack.sh help` to get
|
||||
more information.
|
||||
|
||||
> NOTE: Before running [`create_stack.sh`](create_stack.sh) script, make sure to export
|
||||
the necessary environment variables if needed to be changed. By default it will use
|
||||
`CLRK8S_CNI` to be canal, and `CLRK8S_RUNNER` to be crio. Cilium is tested only in the
|
||||
Vagrant.
|
||||
|
||||
```bash
|
||||
# default shows help
|
||||
./create_stack.sh <subcommand>
|
||||
|
||||
Vendored
+4
-3
@@ -6,8 +6,8 @@ require 'ipaddr'
|
||||
require 'securerandom'
|
||||
|
||||
$num_instances = (ENV['NODES'] || 3).to_i
|
||||
$cpus = (ENV['CPUS'] || 2).to_i
|
||||
$memory = (ENV['MEMORY'] || 4096).to_i
|
||||
$cpus = (ENV['CPUS'] || 4).to_i
|
||||
$memory = (ENV['MEMORY'] || 8192).to_i
|
||||
$disks = 2
|
||||
# Using folder prefix instead of uuid until vagrant-libvirt fixes disk cleanup
|
||||
$disk_prefix = File.basename(File.dirname(__FILE__), "/")
|
||||
@@ -23,6 +23,7 @@ $driveletters = ('a'..'z').to_a
|
||||
$setup_fc = true ? (['true', '1'].include? ENV['SETUP_FC'].to_s) : false
|
||||
$runner = ENV.has_key?('RUNNER') ? ENV['RUNNER'].to_s : "crio".to_s
|
||||
$high_pod_count = ENV.has_key?('HIGH_POD_COUNT') ? ENV['HIGH_POD_COUNT'].to_s : ""
|
||||
$CLRK8S_CLR_VER = ENV.has_key?('CLRK8S_CLR_VER') ? ENV['CLRK8S_CLR_VER'].to_s : ""
|
||||
if !(["crio","containerd"].include? $runner)
|
||||
abort("it's either crio or containerd. Cannot do anything else")
|
||||
end
|
||||
@@ -97,7 +98,7 @@ Vagrant.configure("2") do |config|
|
||||
c.vm.provider :virtualbox do |_, override|
|
||||
override.vm.provision "shell", privileged: true, inline: "sudo mkdir -p /etc/profile.d; echo export MASTER_IP=#{$hosts["clr-01"]} > /etc/profile.d/cnsetup.sh"
|
||||
end
|
||||
c.vm.provision "shell", privileged: false, path: "setup_system.sh", env: {"RUNNER" => $runner, "HIGH_POD_COUNT" => $high_pod_count}
|
||||
c.vm.provision "shell", privileged: false, path: "setup_system.sh", env: {"RUNNER" => $runner, "HIGH_POD_COUNT" => $high_pod_count, "CLRK8S_CLR_VER" => $CLRK8S_CLR_VER}
|
||||
if $setup_fc
|
||||
if $runner == "crio".to_s
|
||||
c.vm.provision "shell", privileged: false, path: "setup_kata_firecracker.sh"
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
apiVersion: admissionregistration.k8s.io/v1beta1
|
||||
apiVersion: admissionregistration.k8s.io/v1
|
||||
kind: MutatingWebhookConfiguration
|
||||
metadata:
|
||||
name: pod-annotate-webhook
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
apiVersion: extensions/v1beta1
|
||||
apiVersion: extensions/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: pod-annotate-webhook
|
||||
|
||||
@@ -44,18 +44,21 @@ sudo systemctl enable --now containerd-devmapper
|
||||
# no. of feature arguments
|
||||
# Skip zeroing blocks for new volumes.
|
||||
sudo dmsetup create contd-thin-pool \
|
||||
--table "0 2097152 thin-pool /dev/loop21 /dev/loop20 512 32768 1 skip_block_zeroing"
|
||||
--table "0 20971520 thin-pool /dev/loop21 /dev/loop20 512 32768 1 skip_block_zeroing"
|
||||
|
||||
sudo mkdir -p /etc/containerd/
|
||||
if [ -f /etc/containerd/config.toml ]
|
||||
then
|
||||
sudo sed -i 's|^\(\[plugins\]\).*|\1\n \[plugins.devmapper\]\n pool_name = \"contd-thin-pool\"\n base_image_size = \"512MB\"|' /etc/containerd/config.toml
|
||||
sudo sed -i 's|^\(\[plugins\]\).*|\1\n \[plugins.devmapper\]\n pool_name = \"contd-thin-pool\"\n base_image_size = \"4096MB\"|' /etc/containerd/config.toml
|
||||
else
|
||||
cat<<EOT | sudo tee /etc/containerd/config.toml
|
||||
[plugins]
|
||||
[plugins.devmapper]
|
||||
pool_name = "contd-thin-pool"
|
||||
base_image_size = "512MB"
|
||||
base_image_size = "4096MB"
|
||||
[plugins.cri]
|
||||
[plugins.cri.containerd]
|
||||
snapshotter = "devmapper"
|
||||
EOT
|
||||
fi
|
||||
|
||||
|
||||
@@ -12,14 +12,27 @@ CUR_DIR=$(pwd)
|
||||
SCRIPT_DIR="$(dirname "${BASH_SOURCE[0]}")"
|
||||
: ${TOKEN:=}
|
||||
: ${MASTER_IP:=}
|
||||
: ${CERT_SANS:=}
|
||||
HIGH_POD_COUNT=${HIGH_POD_COUNT:-""}
|
||||
|
||||
# versions
|
||||
CANAL_VER="${CLRK8S_CANAL_VER:-v3.9}"
|
||||
CANAL_VER="${CLRK8S_CANAL_VER:-v3.10}"
|
||||
CILIUM_VER="${CLRK8S_CILIUM_VER:-v1.6.4}"
|
||||
FLANNEL_VER="${CLRK8S_FLANNEL_VER:-960b3243b9a7faccdfe7b3c09097105e68030ea7}"
|
||||
K8S_VER="${CLRK8S_K8S_VER:-}"
|
||||
ROOK_VER="${CLRK8S_ROOK_VER:-v1.1.1}"
|
||||
METRICS_VER="${CLRK8S_METRICS_VER:-v0.3.5}"
|
||||
KATA_VER="${CLRK8S_KATA_VER:-1.9.1-kernel-config}"
|
||||
ROOK_VER="${CLRK8S_ROOK_VER:-v1.1.7}"
|
||||
METRICS_VER="${CLRK8S_METRICS_VER:-v0.3.6}"
|
||||
DASHBOARD_VER="${CLRK8S_DASHBOARD_VER:-v2.0.0-beta2}"
|
||||
INGRES_VER="${CLRK8S_INGRES_VER:-nginx-0.26.1}"
|
||||
EFK_VER="${CLRK8S_EFK_VER:-v1.15.1}"
|
||||
METALLB_VER="${CLRK8S_METALLB_VER:-v0.8.3}"
|
||||
NPD_VER="${CLRK8S_NPD_VER:-v0.6.6}"
|
||||
PROMETHEUS_VER="${CLRK8S_PROMETHEUS_VER:-f458e85e5d7675f7bc253072e1b4c8892b51af0f}"
|
||||
CNI=${CLRK8S_CNI:-"canal"}
|
||||
if [[ -z "${RUNNER+x}" ]]; then RUNNER="${CLRK8S_RUNNER:-"crio"}"; fi
|
||||
|
||||
NFD_VER="${CLRK8S_NFD_VER:-v0.4.0}"
|
||||
|
||||
function print_usage_exit() {
|
||||
exit_code=${1:-0}
|
||||
@@ -45,17 +58,30 @@ trap finish EXIT
|
||||
function cluster_init() {
|
||||
# Config replacements
|
||||
if ! [ -z ${TOKEN} ]; then
|
||||
sed -i "/InitConfiguration/a bootstrapTokens:\\n- token: ${TOKEN}" ./kubeadm/base/kubeadm.yaml
|
||||
sed -i "/InitConfiguration/a bootstrapTokens:\\n- token: ${TOKEN}" ./kubeadm.yaml
|
||||
fi
|
||||
if ! [ -z ${MASTER_IP} ]; then
|
||||
sed -i "/InitConfiguration/a localAPIEndpoint:\\n advertiseAddress: ${MASTER_IP}" ./kubeadm/base/kubeadm.yaml
|
||||
sed -i "/InitConfiguration/a localAPIEndpoint:\\n advertiseAddress: ${MASTER_IP}" ./kubeadm.yaml
|
||||
fi
|
||||
if [[ -n "$K8S_VER" && $(grep -c kubernetesVersion ./kubeadm/base/kubeadm.yaml) -eq 0 ]]; then
|
||||
sed -i "s/ClusterConfiguration/ClusterConfiguration\nkubernetesVersion: ${K8S_VER}/g" ./kubeadm/base/kubeadm.yaml
|
||||
if [[ -n "$K8S_VER" && $(grep -c kubernetesVersion ./kubeadm.yaml) -eq 0 ]]; then
|
||||
sed -i "s/ClusterConfiguration/ClusterConfiguration\nkubernetesVersion: ${K8S_VER}/g" ./kubeadm.yaml
|
||||
fi
|
||||
if [[ -n "$CERT_SANS" ]]; then
|
||||
if [[ $(grep -c certSANs ./kubeadm.yaml) -gt 0 ]]; then
|
||||
sed -i '/certSANs/,/[a-zA-Z]*:/{//!d}' ./kubeadm.yaml
|
||||
else
|
||||
sed -i "/ClusterConfiguration/a apiServer:\\n certSANs:" ./kubeadm.yaml
|
||||
fi
|
||||
for CERT_SAN in ${CERT_SANS[@]}; do
|
||||
sed -i "/certSANs/a \ \ - ${CERT_SAN}" ./kubeadm.yaml
|
||||
done
|
||||
fi
|
||||
# Config patches
|
||||
if [[ -n "${HIGH_POD_COUNT}" ]]; then
|
||||
echo "$(kubectl kustomize ./kubeadm/high-pod-count/)" > ./kubeadm/base/kubeadm.yaml
|
||||
# increase limits in kubelet
|
||||
sed -i "/KubeletConfiguration/a maxOpenFiles\: 1048576" ./kubeadm.yaml
|
||||
sed -i "/KubeletConfiguration/a maxPods\: 5000" ./kubeadm.yaml
|
||||
# increase the address range per node
|
||||
sed -i "/ClusterConfiguration/a controllerManager:\\n extraArgs:\\n node-cidr-mask-size: \"20\"" ./kubeadm.yaml
|
||||
fi
|
||||
#This only works with kubernetes 1.12+. The kubeadm.yaml is setup
|
||||
#to enable the RuntimeClass featuregate
|
||||
@@ -63,7 +89,7 @@ function cluster_init() {
|
||||
echo "/var/lib/etcd exists! skipping init."
|
||||
return
|
||||
fi
|
||||
sudo -E kubeadm init --config=./kubeadm/base/kubeadm.yaml
|
||||
sudo -E kubeadm init --config=./kubeadm.yaml
|
||||
|
||||
rm -rf "${HOME}/.kube"
|
||||
mkdir -p "${HOME}/.kube"
|
||||
@@ -86,7 +112,7 @@ function cluster_init() {
|
||||
}
|
||||
|
||||
function kata() {
|
||||
KATA_VER=${1:-1.8.2-kernel-config}
|
||||
KATA_VER=${1:-$KATA_VER}
|
||||
KATA_URL="https://github.com/kata-containers/packaging.git"
|
||||
KATA_DIR="8-kata"
|
||||
get_repo "${KATA_URL}" "${KATA_DIR}/overlays/${KATA_VER}"
|
||||
@@ -96,23 +122,48 @@ function kata() {
|
||||
}
|
||||
|
||||
function cni() {
|
||||
# note version is not semver
|
||||
CANAL_VER=${1:-$CANAL_VER}
|
||||
CANAL_URL="https://docs.projectcalico.org/${CANAL_VER}/manifests"
|
||||
if [[ "$CANAL_VER" == "v3.3" ]]; then
|
||||
CANAL_URL="https://docs.projectcalico.org/v3.3/getting-started/kubernetes/installation/hosted/canal"
|
||||
fi
|
||||
CANAL_DIR="0-canal"
|
||||
case "$CNI" in
|
||||
canal)
|
||||
# note version is not semver
|
||||
CANAL_VER=${1:-$CANAL_VER}
|
||||
CANAL_URL="https://docs.projectcalico.org/${CANAL_VER}/manifests"
|
||||
if [[ "$CANAL_VER" == "v3.3" ]]; then
|
||||
CANAL_URL="https://docs.projectcalico.org/v3.3/getting-started/kubernetes/installation/hosted/canal"
|
||||
fi
|
||||
CANAL_DIR="0-canal"
|
||||
|
||||
# canal manifests are not kept in repo but in docs site so use curl
|
||||
mkdir -p "${CANAL_DIR}/overlays/${CANAL_VER}/canal"
|
||||
curl -o "${CANAL_DIR}/overlays/${CANAL_VER}/canal/canal.yaml" "$CANAL_URL/canal.yaml"
|
||||
if [[ "$CANAL_VER" == "v3.3" ]]; then
|
||||
curl -o "${CANAL_DIR}/overlays/${CANAL_VER}/canal/rbac.yaml" "$CANAL_URL/rbac.yaml"
|
||||
fi
|
||||
# canal doesnt pass kustomize validation
|
||||
kubectl apply -k "${CANAL_DIR}/overlays/${CANAL_VER}" --validate=false
|
||||
# canal manifests are not kept in repo but in docs site so use curl
|
||||
mkdir -p "${CANAL_DIR}/overlays/${CANAL_VER}/canal"
|
||||
curl -o "${CANAL_DIR}/overlays/${CANAL_VER}/canal/canal.yaml" "$CANAL_URL/canal.yaml"
|
||||
if [[ "$CANAL_VER" == "v3.3" ]]; then
|
||||
curl -o "${CANAL_DIR}/overlays/${CANAL_VER}/canal/rbac.yaml" "$CANAL_URL/rbac.yaml"
|
||||
fi
|
||||
# canal doesnt pass kustomize validation
|
||||
kubectl apply -k "${CANAL_DIR}/overlays/${CANAL_VER}" --validate=false
|
||||
;;
|
||||
flannel)
|
||||
FLANNEL_VER=${1:-$FLANNEL_VER}
|
||||
FLANNEL_URL="https://github.com/coreos/flannel"
|
||||
FLANNEL_DIR="0-flannel"
|
||||
|
||||
get_repo "${FLANNEL_URL}" "${FLANNEL_DIR}/overlays/${FLANNEL_VER}"
|
||||
set_repo_version "${FLANNEL_VER}" "${FLANNEL_DIR}/overlays/${FLANNEL_VER}/flannel"
|
||||
kubectl apply -k "${FLANNEL_DIR}/overlays/${FLANNEL_VER}"
|
||||
;;
|
||||
cilium)
|
||||
CILIUM_VER=${1:-$CILIUM_VER}
|
||||
CILIUM_URL="https://github.com/cilium/cilium.git"
|
||||
CILIUM_DIR="0-cilium"
|
||||
|
||||
get_repo "${CILIUM_URL}" "${CILIUM_DIR}/overlays/${CILIUM_VER}"
|
||||
set_repo_version "${CILIUM_VER}" "${CILIUM_DIR}/overlays/${CILIUM_VER}/cilium/"
|
||||
helm template "${CILIUM_DIR}/overlays/${CILIUM_VER}/cilium/install/kubernetes/cilium" --namespace kube-system --set global.containerRuntime.integration="$RUNNER" | kubectl apply -f -
|
||||
;;
|
||||
*)
|
||||
echo"Unknown cni $CNI"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
function metrics() {
|
||||
@@ -194,7 +245,7 @@ function monitoring() {
|
||||
}
|
||||
|
||||
function dashboard() {
|
||||
DASHBOARD_VER=${1:-v2.0.0-beta2}
|
||||
DASHBOARD_VER=${1:-$DASHBOARD_VER}
|
||||
DASHBOARD_URL="https://github.com/kubernetes/dashboard.git"
|
||||
DASHBOARD_DIR="2-dashboard"
|
||||
get_repo "${DASHBOARD_URL}" "${DASHBOARD_DIR}/overlays/${DASHBOARD_VER}"
|
||||
@@ -203,7 +254,7 @@ function dashboard() {
|
||||
}
|
||||
|
||||
function ingres() {
|
||||
INGRES_VER=${1:-nginx-0.25.1}
|
||||
INGRES_VER=${1:-$INGRES_VER}
|
||||
INGRES_URL="https://github.com/kubernetes/ingress-nginx.git"
|
||||
INGRES_DIR="5-ingres-lb"
|
||||
get_repo "${INGRES_URL}" "${INGRES_DIR}/overlays/${INGRES_VER}"
|
||||
@@ -212,7 +263,7 @@ function ingres() {
|
||||
}
|
||||
|
||||
function efk() {
|
||||
EFK_VER=${1:-v1.15.1}
|
||||
EFK_VER=${1:-$EFK_VER}
|
||||
EFK_URL="https://github.com/kubernetes/kubernetes.git"
|
||||
EFK_DIR="3-efk"
|
||||
get_repo "${EFK_URL}" "${EFK_DIR}/overlays/${EFK_VER}"
|
||||
@@ -222,7 +273,7 @@ function efk() {
|
||||
}
|
||||
|
||||
function metallb() {
|
||||
METALLB_VER=${1:-v0.8.1}
|
||||
METALLB_VER=${1:-$METALLB_VER}
|
||||
METALLB_URL="https://github.com/danderson/metallb.git"
|
||||
METALLB_DIR="6-metal-lb"
|
||||
get_repo "${METALLB_URL}" "${METALLB_DIR}/overlays/${METALLB_VER}"
|
||||
@@ -230,9 +281,8 @@ function metallb() {
|
||||
kubectl apply -k "${METALLB_DIR}/overlays/${METALLB_VER}"
|
||||
|
||||
}
|
||||
|
||||
function npd() {
|
||||
NPD_VER=${1:-v0.6.6}
|
||||
NPD_VER=${1:-$NPD_VER}
|
||||
NPD_URL="https://github.com/kubernetes/node-problem-detector.git"
|
||||
NPD_DIR="node-problem-detector"
|
||||
get_repo "${NPD_URL}" "${NPD_DIR}/overlays/${NPD_VER}"
|
||||
@@ -240,6 +290,16 @@ function npd() {
|
||||
kubectl apply -k "${NPD_DIR}/overlays/${NPD_VER}"
|
||||
}
|
||||
|
||||
# node feature discovery
|
||||
function nfd() {
|
||||
NFD_VER=${1:-$NFD_VER}
|
||||
NFD_URL="https://github.com/kubernetes-sigs/node-feature-discovery.git"
|
||||
NFD_DIR="node-feature-discovery"
|
||||
get_repo "${NFD_URL}" "${NFD_DIR}/overlays/${NFD_VER}"
|
||||
set_repo_version "${NFD_VER}" "${NFD_DIR}/overlays/${NFD_VER}/node-feature-discovery"
|
||||
kubectl apply -k "${NFD_DIR}/overlays/${NFD_VER}"
|
||||
}
|
||||
|
||||
function miscellaneous() {
|
||||
|
||||
# dashboard
|
||||
@@ -287,7 +347,7 @@ function set_repo_version() {
|
||||
pushd "$(pwd)"
|
||||
cd "${path}"
|
||||
git fetch origin "${ver}"
|
||||
git checkout "${ver}"
|
||||
git -c advice.detachedHead=false checkout "${ver}"
|
||||
popd
|
||||
|
||||
}
|
||||
@@ -306,6 +366,7 @@ command_handlers[storage]=storage
|
||||
command_handlers[monitoring]=monitoring
|
||||
command_handlers[metallb]=metallb
|
||||
command_handlers[npd]=npd
|
||||
command_handlers[nfd]=nfd
|
||||
|
||||
declare -A command_help
|
||||
command_help[init]="Only inits a cluster using kubeadm"
|
||||
@@ -313,6 +374,7 @@ command_help[cni]="Setup network for running cluster"
|
||||
command_help[minimal]="init + cni + kata + metrics"
|
||||
command_help[all]="minimal + storage + monitoring + miscellaneous"
|
||||
command_help[help]="show this message"
|
||||
command_help[nfd]="node feature discovery"
|
||||
|
||||
cd "${SCRIPT_DIR}"
|
||||
|
||||
|
||||
Executable
+166
@@ -0,0 +1,166 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
###
|
||||
# update_checker.sh
|
||||
# Parses create_stack.sh for urls and versions
|
||||
# Curls urls for latest version and reports the comparison
|
||||
##
|
||||
|
||||
# first argument is path to create_stach.sh
|
||||
COMPONENT_FILE="${1:-$create_stack.sh}"
|
||||
# set CLRK8S_DEBUG=1 for debug output
|
||||
DEBUG=${CLRK8S_DEBUG:-0}
|
||||
# set CLRK8S_NO_COLOR=1 for no colors
|
||||
NO_COLOR=${CLRK8S_NO_COLOR:-""}
|
||||
# set CLRK8S_ALL=1 for all results, not just changed
|
||||
ALL=${CLRK8S_ALL:-""}
|
||||
# add components to skip (not check)
|
||||
# - canal doesn't use git repo tags for revisions
|
||||
declare -a COMPONENT_SKIP=( CANAL )
|
||||
|
||||
# internal vars
|
||||
declare -A COMPONENT_VER
|
||||
declare -A COMPONENT_URL
|
||||
LATEST_URL=""
|
||||
|
||||
# usage prints help and exit
|
||||
function usage(){
|
||||
echo "Compare default component versions to latest release"
|
||||
echo "usage: update_checker.sh <path to create_stack.sh>"
|
||||
exit 0
|
||||
}
|
||||
# log echoes to stdout
|
||||
function log(){
|
||||
echo "$1"
|
||||
}
|
||||
# debug echoes to stdout if debug is enabled
|
||||
function debug(){
|
||||
if [[ "${DEBUG}" -ne 0 ]]; then
|
||||
echo "$1"
|
||||
fi
|
||||
}
|
||||
# extract_component_data scans file for component versions and urls and add them to maps
|
||||
function extract_component_data(){
|
||||
file=${1:-$COMPONENT_FILE}
|
||||
name=""
|
||||
version=""
|
||||
url=""
|
||||
while read -r line
|
||||
do
|
||||
# versions
|
||||
if [[ $line =~ "_VER=" ]]; then
|
||||
debug "Found component version $line"
|
||||
name=${line%%_*}
|
||||
if [[ "$COMPONENT_SKIP" =~ (^|[[:space:]])"$name"($|[[:space:]]) ]]; then
|
||||
debug "Skipping component $name"
|
||||
continue
|
||||
fi
|
||||
versions=${line#*=}
|
||||
if [[ $versions =~ ":-" ]]; then
|
||||
version=${line#*:-}
|
||||
fi
|
||||
# cleanup value
|
||||
version=${version%\}}
|
||||
version=${version%\}\"}
|
||||
if [[ -n "$name" && ${COMPONENT_VER[$name]} == "" ]]; then
|
||||
debug "Adding component $name=$version to COMPONENT_VER"
|
||||
COMPONENT_VER[$name]=$version
|
||||
fi
|
||||
fi
|
||||
|
||||
# urls
|
||||
if [[ $line =~ "_URL=" ]]; then
|
||||
debug "Found component URL $line"
|
||||
name=${line%%_*}
|
||||
if [[ $COMPONENT_SKIP =~ (^|[[:space:]])"$name"($|[[:space:]]) ]]; then
|
||||
debug "Skipping component $name"
|
||||
continue
|
||||
fi
|
||||
urls=${line#*=}
|
||||
|
||||
if [[ $urls =~ ":-" ]]; then
|
||||
urls=${line#*:-}
|
||||
fi
|
||||
# cleanup value
|
||||
url=${urls%\"}
|
||||
url=${url#\"}
|
||||
if [[ -n "$name" && ${COMPONENT_URL[$name]} == "" ]]; then
|
||||
debug "Adding component $name=$url to COMPONENT_URL"
|
||||
COMPONENT_URL[$name]=$url
|
||||
fi
|
||||
fi
|
||||
|
||||
done < $file
|
||||
|
||||
}
|
||||
# resolve_latest_url extracts the real release/latest url from a repo url
|
||||
function resolve_latest_url(){
|
||||
repo=$1
|
||||
url=${repo%.git*}/releases/latest
|
||||
LATEST_URL=$(curl -Ls -o /dev/null -w %{url_effective} $url)
|
||||
if [[ "$?" -gt 0 ]]; then
|
||||
echo "curl error, exiting."
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
# function_exists checks if a function exists
|
||||
function function_exists() {
|
||||
declare -f -F "$1" > /dev/null
|
||||
return $?
|
||||
}
|
||||
function report(){
|
||||
if [[ -z $NO_COLOR ]]; then
|
||||
BOLD="\e[1m\e[33m"
|
||||
BOLD_OFF="\e[0m"
|
||||
fi
|
||||
mode="changed"
|
||||
out=""
|
||||
if [[ -n "$1" ]]; then
|
||||
mode="all"
|
||||
fi
|
||||
out+="\n"
|
||||
out+="Components ($mode)\n"
|
||||
out+="--------------------------\n"
|
||||
echo -e $out
|
||||
out="NAME CURRENT LATEST\n"
|
||||
# loop thru each url, get latest version and report
|
||||
for k in "${!COMPONENT_URL[@]}";
|
||||
do
|
||||
name=$k
|
||||
resolve_latest_url "${COMPONENT_URL[$k]}"
|
||||
latest_url="$LATEST_URL"
|
||||
latest_ver="${latest_url#*tag/}"
|
||||
current_ver="${COMPONENT_VER[$name]}"
|
||||
if [[ "${current_ver}" != "${latest_ver}" ]]; then
|
||||
out+="$name $current_ver $BOLD $latest_ver $BOLD_OFF\n";
|
||||
fi
|
||||
if [[ "${current_ver}" == "${latest_ver}" && -n $ALL ]]; then
|
||||
out+="$name $current_ver $latest_ver\n";
|
||||
fi
|
||||
done;
|
||||
echo -e "${out}" | column -t
|
||||
if [[ "${#COMPONENT_SKIP[@]}" -gt 0 ]]; then
|
||||
echo "---"
|
||||
echo "WARNING: Skipped comparisions for the following components:"
|
||||
for s in "${!COMPONENT_SKIP[@]}"
|
||||
do
|
||||
echo "${COMPONENT_SKIP[$s]}"
|
||||
done
|
||||
fi
|
||||
}
|
||||
###
|
||||
# Main
|
||||
##
|
||||
|
||||
# print help if no args
|
||||
if [[ "$#" -eq 0 ]]; then
|
||||
usage
|
||||
fi
|
||||
|
||||
# get the versions
|
||||
extract_component_data "$1"
|
||||
# output
|
||||
report "${ALL}"
|
||||
|
||||
|
||||
|
||||
@@ -1,12 +1,8 @@
|
||||
apiVersion: kubeadm.k8s.io/v1beta1
|
||||
apiVersion: kubeadm.k8s.io/v1beta2
|
||||
kind: InitConfiguration
|
||||
metadata:
|
||||
name: init-config
|
||||
---
|
||||
apiVersion: kubelet.config.k8s.io/v1beta1
|
||||
kind: KubeletConfiguration
|
||||
metadata:
|
||||
name: kubelet-config
|
||||
cgroupDriver: systemd
|
||||
# Allowing for CPU pinning and isolation in case of guaranteed QoS class
|
||||
cpuManagerPolicy: static
|
||||
@@ -17,11 +13,10 @@ kubeReserved:
|
||||
cpu: 500m
|
||||
memory: 256M
|
||||
---
|
||||
apiVersion: kubeadm.k8s.io/v1beta1
|
||||
apiVersion: kubeadm.k8s.io/v1beta2
|
||||
kind: ClusterConfiguration
|
||||
metadata:
|
||||
name: cluster-config
|
||||
networking:
|
||||
dnsDomain: cluster.local
|
||||
podSubnet: 10.244.0.0/16
|
||||
serviceSubnet: 10.96.0.0/12
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
resources:
|
||||
- kubeadm.yaml
|
||||
@@ -1,3 +0,0 @@
|
||||
[
|
||||
{"op": "add", "path": "/controllerManager", "value": {"extraArgs": {"node-cidr-mask-size": "20"}}}
|
||||
]
|
||||
@@ -1,4 +0,0 @@
|
||||
[
|
||||
{"op": "add", "path": "/maxOpenFiles", "value": 1048576},
|
||||
{"op": "add", "path": "/maxPods", "value": 5000}
|
||||
]
|
||||
@@ -1,17 +0,0 @@
|
||||
bases:
|
||||
- ../base/
|
||||
patchesJson6902:
|
||||
# increase limits in kubelet
|
||||
- target:
|
||||
group: kubelet.config.k8s.io
|
||||
version: v1beta1
|
||||
kind: KubeletConfiguration
|
||||
name: kubelet-config
|
||||
path: kubelet-config.json
|
||||
# increase the address range per node
|
||||
- target:
|
||||
group: kubeadm.k8s.io
|
||||
version: v1beta1
|
||||
kind: ClusterConfiguration
|
||||
name: cluster-config
|
||||
path: cluster-config.json
|
||||
@@ -0,0 +1,5 @@
|
||||
resources:
|
||||
- node-feature-discovery/nfd-daemonset-combined.yaml.template
|
||||
- node-feature-discovery/nfd-worker-daemonset.yaml.template
|
||||
|
||||
|
||||
@@ -52,3 +52,4 @@ sudo systemctl is-enabled containerd && sudo systemctl restart containerd
|
||||
sudo systemctl restart kubelet
|
||||
|
||||
reset_cluster
|
||||
sudo -E bash -c "rm -rf /var/lib/containerd/*"
|
||||
|
||||
@@ -5,15 +5,20 @@ set -o nounset
|
||||
|
||||
# global vars
|
||||
CLRK8S_OS=${CLRK8S_OS:-""}
|
||||
CLR_VER=${CLRK8S_CLR_VER:-""}
|
||||
HIGH_POD_COUNT=${HIGH_POD_COUNT:-""}
|
||||
|
||||
# set no proxy
|
||||
ADD_NO_PROXY="10.244.0.0/16,10.96.0.0/12"
|
||||
ADD_NO_PROXY=".svc,10.0.0.0/8,192.168.0.0/16"
|
||||
ADD_NO_PROXY+=",$(hostname -I | sed 's/[[:space:]]/,/g')"
|
||||
: "${RUNNER:=crio}"
|
||||
if [[ -z "${RUNNER+x}" ]]; then RUNNER="${CLRK8S_RUNNER:-crio}"; fi
|
||||
|
||||
# update os version
|
||||
function upate_os_version() {
|
||||
if [[ -n "${CLR_VER}" ]]; then
|
||||
sudo swupd repair -m "${CLR_VER}" --picky --force
|
||||
return
|
||||
fi
|
||||
sudo swupd update
|
||||
}
|
||||
|
||||
@@ -145,7 +150,7 @@ function setup_proxy() {
|
||||
sed_val=${ADD_NO_PROXY//\//\\/}
|
||||
[ -f /etc/environment ] && sudo sed -i "/no_proxy/I s/$/,${sed_val}/g" /etc/environment
|
||||
if [ -f /etc/profile.d/proxy.sh ]; then
|
||||
sudo sed -i "/no_proxy/I s/\"$/,${sed_val}\"/g" /etc/profile.d/proxy.sh
|
||||
sudo sed -i "/no_proxy/I s/$/,${sed_val}/g" /etc/profile.d/proxy.sh
|
||||
else
|
||||
echo "Warning, failed to find /etc/profile.d/proxy.sh to edit no_proxy line"
|
||||
fi
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
apiVersion: extensions/v1beta1
|
||||
apiVersion: extensions/v1
|
||||
kind: Ingress
|
||||
metadata:
|
||||
name: php-apache-kata
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
apiVersion: extensions/v1beta1
|
||||
apiVersion: extensions/v1
|
||||
kind: Ingress
|
||||
metadata:
|
||||
name: php-apache-runc
|
||||
|
||||
+30
-2
@@ -13,10 +13,11 @@ is below:
|
||||
|
||||
| Tool | Description |
|
||||
| ---- | ----------- |
|
||||
| collectd | `collectd` based statistics/metrics gathering daemonset code |
|
||||
| lib | General library helper functions for forming and launching workloads, and storing results in a uniform manner to aid later analysis |
|
||||
| scaling | Tests to measure scaling, such as linear or parallel launching of pods |
|
||||
| lib/cpu-load* | Routines to enable CPU load generation on a cluster |
|
||||
| report | Rmarkdown based report generator, used to produce a PDF comparison report of 1 or more sets of results |
|
||||
|
||||
| scaling | Tests to measure scaling, such as linear or parallel launching of pods |
|
||||
|
||||
## Results storage and analysis
|
||||
|
||||
@@ -130,3 +131,30 @@ If k8s_parallel.sh was run, the results file is named `k8s-parallel.json` rather
|
||||
└── scaling-4.png
|
||||
```
|
||||
More details about result reporting can be reviewed at [`report`](./report) directory.
|
||||
|
||||
# Developers
|
||||
|
||||
This section provides some details of how the code is structured and configured. This may be of use whilst modifying
|
||||
existing or creating new tests.
|
||||
|
||||
## Metrics gathering
|
||||
|
||||
Metrics can be gathered using either a daemonset deployment of privileged pods used to gather statistics directly from the nodes using a combination of `mpstat`, `free` and `df`, or a daemonset deployment based around `collectd`.
|
||||
|
||||
### `collectd` statistics
|
||||
|
||||
The `collected` based code can be found in the `collectd` subdirectory. It uses the `collected` configuration found in the `collectd.conf` file to gather statistics, and store the results on the nodes themselves whilst tests are running. At the end of the test, the results are copied from the nodes and stored in the results directory for later processing.
|
||||
|
||||
The `collectd` statistics are only configured and gathered if the environment variable `SMF_USE_COLLECTD` is set to non-empty by the test code (that is, only enabled upon request).
|
||||
|
||||
### privileged statistics pods
|
||||
|
||||
The privileged statistics pods `YAML` can be found in the `scaling/stats.yaml` file. An example of how to invoke and use this daemonset to extract statistics can be found in the `scaling/k8s_scale.sh` file.
|
||||
|
||||
## Configuring constant 'loads'
|
||||
|
||||
The framework includes some tooling to assist in setting up constant pre-defined 'loads' across the cluster to aid evaluation of their impacts on the scaling metrics.
|
||||
|
||||
### CPU load generator
|
||||
|
||||
Details of how to configure a constant CPU load are detailed in the [cpu-load documentation](lib/cpu-load.md).
|
||||
|
||||
@@ -25,8 +25,6 @@ init_stats() {
|
||||
}
|
||||
|
||||
cleanup_stats() {
|
||||
local delete_wait_time=$1
|
||||
|
||||
# attempting to provide buffer for collectd CPU collection to record adequate history
|
||||
sleep 6
|
||||
|
||||
|
||||
@@ -9,6 +9,8 @@ LoadPlugin memory
|
||||
LoadPlugin cpufreq
|
||||
LoadPlugin df
|
||||
|
||||
Hostname localhost
|
||||
|
||||
<Plugin "cpu">
|
||||
ReportByCpu true
|
||||
ReportByState true
|
||||
@@ -23,6 +25,7 @@ LoadPlugin df
|
||||
Interface "/^ens/"
|
||||
Interface "/^enp/"
|
||||
Interface "/^em/"
|
||||
Interface "/^eth/"
|
||||
IgnoreSelected false
|
||||
</Plugin>
|
||||
<Plugin "aggregation">
|
||||
|
||||
@@ -10,6 +10,7 @@ RESULT_DIR="${LIB_DIR}/../results"
|
||||
|
||||
source ${LIB_DIR}/json.bash
|
||||
source ${LIB_DIR}/k8s-api.bash
|
||||
source ${LIB_DIR}/cpu-load.bash
|
||||
source /etc/os-release || source /usr/lib/os-release
|
||||
|
||||
die() {
|
||||
@@ -67,6 +68,48 @@ init_env()
|
||||
# been deliberately injected into the cluster under test.
|
||||
}
|
||||
|
||||
framework_init() {
|
||||
info "Initialising"
|
||||
|
||||
check_cmds "${cmds[@]}"
|
||||
|
||||
info "Checking k8s accessible"
|
||||
local worked=$( kubectl get nodes > /dev/null 2>&1 && echo $? || echo $? )
|
||||
if [ "$worked" != 0 ]; then
|
||||
die "kubectl failed to get nodes"
|
||||
fi
|
||||
|
||||
info $(get_num_nodes) "k8s nodes in 'Ready' state found"
|
||||
|
||||
k8s_api_init
|
||||
|
||||
# Launch our stats gathering pod
|
||||
if [ -n "$SMF_USE_COLLECTD" ]; then
|
||||
info "Setting up collectd"
|
||||
init_stats $wait_time
|
||||
fi
|
||||
|
||||
# And now we can set up our results storage then...
|
||||
metrics_json_init "k8s"
|
||||
save_config
|
||||
|
||||
# Initialise the cpu load generators now - after json init, as they may
|
||||
# produce some json results (config) data.
|
||||
cpu_load_init
|
||||
|
||||
}
|
||||
|
||||
framework_shutdown() {
|
||||
metrics_json_save
|
||||
k8s_api_shutdown
|
||||
cpu_load_shutdown
|
||||
|
||||
if [ -n "$SMF_USE_COLLECTD" ]; then
|
||||
cleanup_stats
|
||||
fi
|
||||
|
||||
}
|
||||
|
||||
# finds elements in $1 that are not in $2
|
||||
find_unique_pods() {
|
||||
local list_a=$1
|
||||
@@ -86,3 +129,22 @@ find_unique_pods() {
|
||||
fi
|
||||
done
|
||||
}
|
||||
|
||||
# waits for process to complete within a given time range
|
||||
waitForProcess(){
|
||||
wait_time="$1"
|
||||
sleep_time="$2"
|
||||
cmd="$3"
|
||||
proc_info_msg="$4"
|
||||
|
||||
while [ "$wait_time" -gt 0 ]; do
|
||||
if eval "$cmd"; then
|
||||
return 0
|
||||
else
|
||||
info "$proc_info_msg"
|
||||
sleep "$sleep_time"
|
||||
wait_time=$((wait_time-sleep_time))
|
||||
fi
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# Copyright (c) 2019 Intel Corporation
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
# Helper routines for setting up a constant CPU load on the cluster/nodes
|
||||
|
||||
CPULOAD_DIR=${THIS_FILE%/*}
|
||||
|
||||
# Default to testing all cores
|
||||
SMF_CPU_LOAD_NODES_NCPU=${SMF_CPU_LOAD_NODES_NCPU:-0}
|
||||
# Default to 100% load (yes, this might kill your node)
|
||||
SMF_CPU_LOAD_NODES_PERCENT=${SMF_CPU_LOAD_NODES_PERCENT:-}
|
||||
# Default to not setting any limits or requests, so no cpuset limiting and
|
||||
# no cpu core pinning
|
||||
SMF_CPU_LOAD_NODES_LIMIT=${SMF_CPU_LOAD_NODES_LIMIT:-}
|
||||
SMF_CPU_LOAD_NODES_REQUEST=${SMF_CPU_LOAD_NODES_REQUEST:-}
|
||||
|
||||
cpu_load_post_deploy_sleep=${cpu_load_post_deploy_sleep:-30}
|
||||
|
||||
cpu_per_node_daemonset=cpu-load
|
||||
clean_up_cpu_per_node=false
|
||||
|
||||
# Use a DaemonSet to place one cpu stressor on each node.
|
||||
cpu_per_node_init() {
|
||||
info "Generating per-node CPU load daemonset"
|
||||
|
||||
local ds_template=${CPULOAD_DIR}/cpu_load_daemonset.yaml.in
|
||||
local ds_yaml=${ds_template%\.in}
|
||||
|
||||
# Grab a copy of the template
|
||||
cp -f ${ds_template} ${ds_yaml}
|
||||
|
||||
# If a setting is not used (defined), then delete its relevant
|
||||
# lines from the YAML. Note, the YAML is constructed when necessary
|
||||
# with comments on the correct lines to ensure all necessary lines are
|
||||
# deleted
|
||||
if [ -z "$SMF_CPU_LOAD_NODES_NCPU" ]; then
|
||||
sed -i '/CPU_NCPU/d' ${ds_yaml}
|
||||
fi
|
||||
|
||||
if [ -z "${SMF_CPU_LOAD_NODES_PERCENT}" ]; then
|
||||
sed -i '/CPU_PERCENT/d' ${ds_yaml}
|
||||
fi
|
||||
|
||||
if [ -z "${SMF_CPU_LOAD_NODES_LIMIT}" ]; then
|
||||
sed -i '/CPU_LIMIT/d' ${ds_yaml}
|
||||
fi
|
||||
|
||||
if [ -z "${SMF_CPU_LOAD_NODES_REQUEST}" ]; then
|
||||
sed -i '/CPU_REQUEST/d' ${ds_yaml}
|
||||
fi
|
||||
|
||||
# And then finally replace all the remaining defined parts with the
|
||||
# real values.
|
||||
sed -i \
|
||||
-e "s|@CPU_NCPU@|${SMF_CPU_LOAD_NODES_NCPU}|g" \
|
||||
-e "s|@CPU_PERCENT@|${SMF_CPU_LOAD_NODES_PERCENT}|g" \
|
||||
-e "s|@CPU_LIMIT@|${SMF_CPU_LOAD_NODES_LIMIT}|g" \
|
||||
-e "s|@CPU_REQUEST@|${SMF_CPU_LOAD_NODES_REQUEST}|g" \
|
||||
${ds_yaml}
|
||||
|
||||
# Launch the daemonset...
|
||||
info "Deploying cpu-load-per-node daemonset"
|
||||
kubectl apply -f ${ds_yaml}
|
||||
kubectl rollout status --timeout=${wait_time}s daemonset/${cpu_per_node_daemonset}
|
||||
clean_up_cpu_per_node=yes
|
||||
info "cpu-load-per-node daemonset Deployed"
|
||||
if [ -n "$cpu_load_post_deploy_sleep" ]; then
|
||||
info "Sleeping ${cpu_load_post_deploy_sleep}s for cpu-load to settle"
|
||||
sleep ${cpu_load_post_deploy_sleep}
|
||||
fi
|
||||
|
||||
# And store off our config into the JSON results
|
||||
metrics_json_start_array
|
||||
local json="$(cat << EOF
|
||||
{
|
||||
"LOAD_NODES_NCPU": "${SMF_CPU_LOAD_NODES_NCPU}",
|
||||
"LOAD_NODES_PERCENT": "${SMF_CPU_LOAD_NODES_PERCENT}",
|
||||
"LOAD_NODES_LIMIT": "${SMF_CPU_LOAD_NODES_LIMIT}",
|
||||
"LOAD_NODES_REQUEST": "${SMF_CPU_LOAD_NODES_REQUEST}"
|
||||
}
|
||||
EOF
|
||||
)"
|
||||
metrics_json_add_array_element "$json"
|
||||
metrics_json_end_array "cpu-load"
|
||||
}
|
||||
|
||||
cpu_load_init() {
|
||||
info "Check if we need CPU load generators..."
|
||||
# This is defaulted of off (not defined), unless the high level test requests it.
|
||||
if [ -n "$SMF_CPU_LOAD_NODES" ]; then
|
||||
info "Initialising per-node CPU load"
|
||||
cpu_per_node_init
|
||||
fi
|
||||
}
|
||||
|
||||
cpu_load_shutdown() {
|
||||
if [ "$clean_up_cpu_per_node" = "yes" ]; then
|
||||
info "Cleaning up cpu per node load daemonset"
|
||||
kubectl delete daemonset --wait=true --timeout=${delete_wait_time}s "${cpu_per_node_daemonset}" || true
|
||||
fi
|
||||
}
|
||||
@@ -0,0 +1,81 @@
|
||||
# `cpu-load` stack stresser
|
||||
|
||||
The `cpu-load` stress functionality of the scaling framework allows you to optionally add a constant CPU stress
|
||||
load to cluster under test whilst the tests are running. This aids impact analysis of CPU load.
|
||||
|
||||
The `cpu-load` functionality utilises the [`stress-ng`](https://kernel.ubuntu.com/git/cking/stress-ng.git/) tool
|
||||
to generate the CPU load. Some of the configuration parameters are taken directoy from the `stress-ng` command line.
|
||||
|
||||
## Configuration
|
||||
|
||||
`cpu-load` is configured via a number of environment variables:
|
||||
|
||||
| Tool | Description |
|
||||
| ---- | ----------- |
|
||||
| collectd | `collectd` based statistics/metrics gathering daemonset code |
|
||||
| lib | General library helper functions for forming and launching workloads, and storing results in a uniform manner to aid later analysis |
|
||||
| report | Rmarkdown based report generator, used to produce a PDF comparison report of 1 or more sets of results |
|
||||
| scaling | Tests to measure scaling, such as linear or parallel launching of pods |
|
||||
|
||||
| Variable | Description | Default |
|
||||
| -------- | ----------- | ------- |
|
||||
| `SMF_CPU_LOAD_NODES` | Set to non-empty to deploy `cpu-load` stressor | unset (off) |
|
||||
| `SMF_CPU_LOAD_NODES_NCPU` | Number of stressor threads to launch per node | 0 (one per cpu) |
|
||||
| `SMF_CPU_LOAD_NODES_PERCENT` | Percentage of CPU to load | unset (100%) |
|
||||
| `SMF_CPU_LOAD_NODES_LIMIT` | k8s cpu resource limit to set | unset (none) |
|
||||
| `SMF_CPU_LOAD_NODES_REQUEST` | k8s cpu resource request to set | unset (none) |
|
||||
| `cpu_load_post_deploy_sleep` | Seconds to sleep for `cpu-load` deployment to settle | 30 |
|
||||
|
||||
`SMF_CPU_LOAD_NODES` must be set to a non-empty string to enable the `cpu-load` functionality. `cpu-load` uses
|
||||
a daemonSet to deploy one `stress-ng` single container pod to each active node in the cluster.
|
||||
|
||||
|
||||
Any of the `SMF_CPU_LOAD_NODES_*` variables can be set, or unset, and the daemonSet pods will be configured
|
||||
appropriately.
|
||||
|
||||
## Examples
|
||||
|
||||
The combinations of settings available allow a lot of flexibility. Below are some common example setups:
|
||||
|
||||
### 50% CPU load on all cores of all nodes (`stress-ng`)
|
||||
|
||||
Here we allow `stress-ng` to spawn workers to cover all the CPUs on each node, but ask it to restrict its
|
||||
bandwidth use to 50% of the CPU. We do not use the k8s limits.
|
||||
|
||||
```bash
|
||||
export SMF_CPU_LOAD_NODES=true
|
||||
#export SMF_CPU_LOAD_NODES_NCPU=
|
||||
export SMF_CPU_LOAD_NODES_PERCENT=50
|
||||
#export SMF_CPU_LOAD_NODES_LIMIT=999m
|
||||
#export SMF_CPU_LOAD_NODES_REQUEST=999m
|
||||
```
|
||||
|
||||
### 50% CPU load on 1 un-pinned core of all nodes (k8s `limits`)
|
||||
|
||||
Here we set `stress-ng` to run a single worker thread at 100% CPU, but use the k8s resource limits to restrict
|
||||
actual CPU usage to 50%. Because the k8s limit and request are not whole interger units, if the static policy is
|
||||
in place on the k8s cluster, the pods will be classified as Guaranteed QoS, but will *not* get pinned to a specific
|
||||
cpuset.
|
||||
|
||||
```bash
|
||||
export SMF_CPU_LOAD_NODES=true
|
||||
export SMF_CPU_LOAD_NODES_NCPU=1
|
||||
export SMF_CPU_LOAD_NODES_PERCENT=100
|
||||
export SMF_CPU_LOAD_NODES_LIMIT=500m
|
||||
export SMF_CPU_LOAD_NODES_REQUEST=500m
|
||||
```
|
||||
|
||||
### 50% CPU load pinned to 1 core, on all nodes
|
||||
|
||||
Here we set `stress-ng` to run a single worker thread at 50% CPU, and use the k8s resource limits to classify the
|
||||
pod as Guaranteed, and as we are using whole integer units of CPU resource requests, if the static policy manager is
|
||||
in play, the thread will be pinned to a single cpu cpuset.
|
||||
|
||||
```bash
|
||||
export SMF_CPU_LOAD_NODES=true
|
||||
export SMF_CPU_LOAD_NODES_NCPU=1
|
||||
export SMF_CPU_LOAD_NODES_PERCENT=50
|
||||
export SMF_CPU_LOAD_NODES_LIMIT=1
|
||||
export SMF_CPU_LOAD_NODES_REQUEST=1
|
||||
```
|
||||
|
||||
@@ -0,0 +1,30 @@
|
||||
apiVersion: apps/v1
|
||||
kind: DaemonSet
|
||||
metadata:
|
||||
name: cpu-load
|
||||
spec:
|
||||
selector:
|
||||
matchLabels:
|
||||
name: cpu-load-pods
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
name: cpu-load-pods
|
||||
spec:
|
||||
hostNetwork: true
|
||||
terminationGracePeriodSeconds: 0
|
||||
containers:
|
||||
- name: cpu-load
|
||||
imagePullPolicy: IfNotPresent
|
||||
image: polinux/stress-ng
|
||||
command: ["stress-ng"]
|
||||
args: # comment fields here so we can *delete* sections on demand
|
||||
- "--cpu"
|
||||
- "@CPU_NCPU@"
|
||||
- "-l" #CPU_PERCENT
|
||||
- "@CPU_PERCENT@" #CPU_PERCENT
|
||||
resources:
|
||||
limits:
|
||||
cpu: @CPU_LIMIT@
|
||||
requests:
|
||||
cpu: @CPU_REQUEST@
|
||||
@@ -100,7 +100,7 @@ setup() {
|
||||
}
|
||||
|
||||
run() {
|
||||
docker run -ti --rm -v ${HOSTINPUTDIR}:${GUESTINPUTDIR} -v ${HOSTOUTPUTDIR}:${GUESTOUTPUTDIR} ${extra_volumes} ${IMAGE} ${extra_command}
|
||||
docker run ${extra_opts} --rm -v ${HOSTINPUTDIR}:${GUESTINPUTDIR} -v ${HOSTOUTPUTDIR}:${GUESTOUTPUTDIR} ${extra_volumes} ${IMAGE} ${extra_command}
|
||||
ls -la ${HOSTOUTPUTDIR}/*
|
||||
}
|
||||
|
||||
@@ -113,6 +113,7 @@ main() {
|
||||
# In debug mode, run a shell instead of the default report generation
|
||||
extra_command="bash"
|
||||
extra_volumes="-v ${HOSTSCRIPTDIR}:${GUESTSCRIPTDIR}"
|
||||
extra_opts="-ti"
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
@@ -13,7 +13,7 @@
|
||||
# We would have used the 'verse' base, that already has some of the docs processing
|
||||
# installed, but I could not figure out how to add in the extra bits we needed to
|
||||
# the lite tex version is uses.
|
||||
FROM rocker/tidyverse
|
||||
FROM rocker/tidyverse:latest
|
||||
|
||||
# Version of the Dockerfile
|
||||
LABEL DOCKERFILE_VERSION="1.0"
|
||||
|
||||
@@ -30,15 +30,6 @@ cpustats=c() # Statistics for cpu usage
|
||||
bootstats=c() # Statistics for boot (launch) times
|
||||
inodestats=c() # Statistics for inode usage
|
||||
|
||||
# values for scaling the secondary y axes on some graphs
|
||||
mem_scale=1
|
||||
cpu_scale=1
|
||||
inode_scale=1
|
||||
ip_scale=1
|
||||
oct_scale=1
|
||||
drop_scale=1
|
||||
error_scale=1
|
||||
|
||||
# iterate over every set of results (test run)
|
||||
for (currentdir in resultdirs) {
|
||||
# For every results file we are interested in evaluating
|
||||
@@ -140,47 +131,52 @@ for (currentdir in resultdirs) {
|
||||
# filename has date on the end, so look for the right file name
|
||||
freemem_pattern='^memory\\-free'
|
||||
files=list.files(memory_dir, pattern=freemem_pattern)
|
||||
# collectd csv plugin starts a new file for each day of data collected
|
||||
for(file in files) {
|
||||
mem_free_csv=paste(memory_dir, file, sep="/")
|
||||
node_mem_free_data=read.csv(mem_free_csv, header=TRUE, sep=",")
|
||||
node_mem_free_data=cbind(node_mem_free_data,
|
||||
node=rep(n, length(node_mem_free_data$value)))
|
||||
node_mem_free_data=cbind(node_mem_free_data,
|
||||
testname=rep(testname, length(node_mem_free_data$value)))
|
||||
node_mem_free_data$s_offset = node_mem_free_data$epoch - local_bootdata[1,]$epoch
|
||||
|
||||
mem_free_csv=paste(memory_dir, files[1], sep="/")
|
||||
node_mem_free_data=read.csv(mem_free_csv, header=TRUE, sep=",")
|
||||
node_mem_free_data=cbind(node_mem_free_data,
|
||||
node=rep(n, length(node_mem_free_data$value)))
|
||||
node_mem_free_data=cbind(node_mem_free_data,
|
||||
testname=rep(testname, length(node_mem_free_data$value)))
|
||||
node_mem_free_data$s_offset = node_mem_free_data$epoch - local_bootdata[1,]$epoch
|
||||
|
||||
mem_free_data=rbind(mem_free_data, node_mem_free_data)
|
||||
mem_free_data=rbind(mem_free_data, node_mem_free_data)
|
||||
}
|
||||
|
||||
# grab CPU data
|
||||
cpu_dir=paste(localhost_dir, "aggregation-cpu-average", sep="/")
|
||||
# filename has date on the end, so look for the right file name
|
||||
percent_idle_pattern='^percent\\-idle'
|
||||
files=list.files(cpu_dir, pattern=percent_idle_pattern)
|
||||
for(file in files) {
|
||||
cpu_idle_csv=paste(cpu_dir, file, sep="/")
|
||||
node_cpu_idle_data=read.csv(cpu_idle_csv, header=TRUE, sep=",")
|
||||
node_cpu_idle_data=cbind(node_cpu_idle_data,
|
||||
node=rep(n, length(node_cpu_idle_data$value)))
|
||||
node_cpu_idle_data=cbind(node_cpu_idle_data,
|
||||
testname=rep(testname, length(node_cpu_idle_data$value)))
|
||||
node_cpu_idle_data$s_offset = node_cpu_idle_data$epoch - local_bootdata[1,]$epoch
|
||||
|
||||
cpu_idle_csv=paste(cpu_dir, files[1], sep="/")
|
||||
node_cpu_idle_data=read.csv(cpu_idle_csv, header=TRUE, sep=",")
|
||||
node_cpu_idle_data=cbind(node_cpu_idle_data,
|
||||
node=rep(n, length(node_cpu_idle_data$value)))
|
||||
node_cpu_idle_data=cbind(node_cpu_idle_data,
|
||||
testname=rep(testname, length(node_cpu_idle_data$value)))
|
||||
node_cpu_idle_data$s_offset = node_cpu_idle_data$epoch - local_bootdata[1,]$epoch
|
||||
|
||||
cpu_idle_data=rbind(cpu_idle_data, node_cpu_idle_data)
|
||||
cpu_idle_data=rbind(cpu_idle_data, node_cpu_idle_data)
|
||||
}
|
||||
|
||||
# grab inode data
|
||||
inode_dir=paste(localhost_dir, "df-root", sep="/")
|
||||
# filename has date on the end, so look for the right file name
|
||||
inode_free_pattern='^df_inodes\\-free'
|
||||
files=list.files(inode_dir, pattern=inode_free_pattern)
|
||||
inode_free_csv=paste(inode_dir, files[1], sep="/")
|
||||
node_inode_free_data=read.csv(inode_free_csv, header=TRUE, sep=",")
|
||||
node_inode_free_data=cbind(node_inode_free_data,
|
||||
node=rep(n, length(node_inode_free_data$value)))
|
||||
node_inode_free_data=cbind(node_inode_free_data,
|
||||
testname=rep(testname, length(node_inode_free_data$value)))
|
||||
node_inode_free_data$s_offset = node_inode_free_data$epoch - local_bootdata[1,]$epoch
|
||||
for(file in files) {
|
||||
inode_free_csv=paste(inode_dir, file, sep="/")
|
||||
node_inode_free_data=read.csv(inode_free_csv, header=TRUE, sep=",")
|
||||
node_inode_free_data=cbind(node_inode_free_data,
|
||||
node=rep(n, length(node_inode_free_data$value)))
|
||||
node_inode_free_data=cbind(node_inode_free_data,
|
||||
testname=rep(testname, length(node_inode_free_data$value)))
|
||||
node_inode_free_data$s_offset = node_inode_free_data$epoch - local_bootdata[1,]$epoch
|
||||
|
||||
inode_free_data=rbind(inode_free_data, node_inode_free_data)
|
||||
inode_free_data=rbind(inode_free_data, node_inode_free_data)
|
||||
}
|
||||
|
||||
# grab interface data
|
||||
interface_dir_pattern='^interface\\-'
|
||||
@@ -191,71 +187,79 @@ for (currentdir in resultdirs) {
|
||||
|
||||
# filename has date on the end, so look for the right file name
|
||||
interface_packets_pattern='^if_packets'
|
||||
files=list.files(interface_dir, pattern=interface_packets_pattern)
|
||||
interface_packets_csv=paste(interface_dir, files[1], sep="/")
|
||||
node_interface_packets_data=read.csv(interface_packets_csv, header=TRUE, sep=",")
|
||||
node_interface_packets_data=cbind(node_interface_packets_data,
|
||||
node=rep(n, length(node_interface_packets_data$epoch)))
|
||||
node_interface_packets_data=cbind(node_interface_packets_data,
|
||||
testname=rep(testname,
|
||||
int_files=list.files(interface_dir, pattern=interface_packets_pattern)
|
||||
for (int_file in int_files) {
|
||||
interface_packets_csv=paste(interface_dir, int_file, sep="/")
|
||||
node_interface_packets_data=read.csv(interface_packets_csv, header=TRUE, sep=",")
|
||||
node_interface_packets_data=cbind(node_interface_packets_data,
|
||||
node=rep(n, length(node_interface_packets_data$epoch)))
|
||||
node_interface_packets_data=cbind(node_interface_packets_data,
|
||||
testname=rep(testname,
|
||||
length(node_interface_packets_data$epoch)))
|
||||
node_interface_packets_data=cbind(node_interface_packets_data,
|
||||
name=rep(interface_name,
|
||||
length(node_interface_packets_data$epoch)))
|
||||
node_interface_packets_data=cbind(node_interface_packets_data,
|
||||
name=rep(interface_name,
|
||||
length(node_interface_packets_data$epoch)))
|
||||
node_interface_packets_data$s_offset = node_interface_packets_data$epoch - local_bootdata[1,]$epoch
|
||||
node_interface_packets_data$s_offset = node_interface_packets_data$epoch - local_bootdata[1,]$epoch
|
||||
|
||||
interface_packets_data=rbind(interface_packets_data, node_interface_packets_data)
|
||||
interface_packets_data=rbind(interface_packets_data, node_interface_packets_data)
|
||||
}
|
||||
|
||||
# filename has date on the end, so look for the right file name
|
||||
interface_octets_pattern='^if_octets'
|
||||
files=list.files(interface_dir, pattern=interface_octets_pattern)
|
||||
interface_octets_csv=paste(interface_dir, files[1], sep="/")
|
||||
node_interface_octets_data=read.csv(interface_octets_csv, header=TRUE, sep=",")
|
||||
node_interface_octets_data=cbind(node_interface_octets_data,
|
||||
node=rep(n, length(node_interface_octets_data$epoch)))
|
||||
node_interface_octets_data=cbind(node_interface_octets_data,
|
||||
testname=rep(testname,
|
||||
int_files=list.files(interface_dir, pattern=interface_octets_pattern)
|
||||
for (int_file in int_files) {
|
||||
interface_octets_csv=paste(interface_dir, int_file, sep="/")
|
||||
node_interface_octets_data=read.csv(interface_octets_csv, header=TRUE, sep=",")
|
||||
node_interface_octets_data=cbind(node_interface_octets_data,
|
||||
node=rep(n, length(node_interface_octets_data$epoch)))
|
||||
node_interface_octets_data=cbind(node_interface_octets_data,
|
||||
testname=rep(testname,
|
||||
length(node_interface_octets_data$epoch)))
|
||||
node_interface_octets_data=cbind(node_interface_octets_data,
|
||||
name=rep(interface_name,
|
||||
length(node_interface_octets_data$epoch)))
|
||||
node_interface_octets_data=cbind(node_interface_octets_data,
|
||||
name=rep(interface_name,
|
||||
length(node_interface_octets_data$epoch)))
|
||||
node_interface_octets_data$s_offset = node_interface_octets_data$epoch - local_bootdata[1,]$epoch
|
||||
node_interface_octets_data$s_offset = node_interface_octets_data$epoch - local_bootdata[1,]$epoch
|
||||
|
||||
interface_octets_data=rbind(interface_octets_data, node_interface_octets_data)
|
||||
interface_octets_data=rbind(interface_octets_data, node_interface_octets_data)
|
||||
}
|
||||
|
||||
# filename has date on the end, so look for the right file name
|
||||
interface_dropped_pattern='^if_dropped'
|
||||
files=list.files(interface_dir, pattern=interface_dropped_pattern)
|
||||
interface_dropped_csv=paste(interface_dir, files[1], sep="/")
|
||||
node_interface_dropped_data=read.csv(interface_dropped_csv, header=TRUE, sep=",")
|
||||
node_interface_dropped_data=cbind(node_interface_dropped_data,
|
||||
node=rep(n, length(node_interface_dropped_data$epoch)))
|
||||
node_interface_dropped_data=cbind(node_interface_dropped_data,
|
||||
testname=rep(testname,
|
||||
int_files=list.files(interface_dir, pattern=interface_dropped_pattern)
|
||||
for (int_file in int_files) {
|
||||
interface_dropped_csv=paste(interface_dir, int_file, sep="/")
|
||||
node_interface_dropped_data=read.csv(interface_dropped_csv, header=TRUE, sep=",")
|
||||
node_interface_dropped_data=cbind(node_interface_dropped_data,
|
||||
node=rep(n, length(node_interface_dropped_data$epoch)))
|
||||
node_interface_dropped_data=cbind(node_interface_dropped_data,
|
||||
testname=rep(testname,
|
||||
length(node_interface_dropped_data$epoch)))
|
||||
node_interface_dropped_data=cbind(node_interface_dropped_data,
|
||||
name=rep(interface_name,
|
||||
length(node_interface_dropped_data$epoch)))
|
||||
node_interface_dropped_data=cbind(node_interface_dropped_data,
|
||||
name=rep(interface_name,
|
||||
length(node_interface_dropped_data$epoch)))
|
||||
node_interface_dropped_data$s_offset = node_interface_dropped_data$epoch - local_bootdata[1,]$epoch
|
||||
node_interface_dropped_data$s_offset = node_interface_dropped_data$epoch - local_bootdata[1,]$epoch
|
||||
|
||||
interface_dropped_data=rbind(interface_dropped_data, node_interface_dropped_data)
|
||||
interface_dropped_data=rbind(interface_dropped_data, node_interface_dropped_data)
|
||||
}
|
||||
|
||||
# filename has date on the end, so look for the right file name
|
||||
interface_errors_pattern='^if_errors'
|
||||
files=list.files(interface_dir, pattern=interface_errors_pattern)
|
||||
interface_errors_csv=paste(interface_dir, files[1], sep="/")
|
||||
node_interface_errors_data=read.csv(interface_errors_csv, header=TRUE, sep=",")
|
||||
node_interface_errors_data=cbind(node_interface_errors_data,
|
||||
node=rep(n, length(node_interface_errors_data$epoch)))
|
||||
node_interface_errors_data=cbind(node_interface_errors_data,
|
||||
testname=rep(testname,
|
||||
int_files=list.files(interface_dir, pattern=interface_errors_pattern)
|
||||
for (int_file in int_files) {
|
||||
interface_errors_csv=paste(interface_dir, int_file, sep="/")
|
||||
node_interface_errors_data=read.csv(interface_errors_csv, header=TRUE, sep=",")
|
||||
node_interface_errors_data=cbind(node_interface_errors_data,
|
||||
node=rep(n, length(node_interface_errors_data$epoch)))
|
||||
node_interface_errors_data=cbind(node_interface_errors_data,
|
||||
testname=rep(testname,
|
||||
length(node_interface_errors_data$epoch)))
|
||||
node_interface_errors_data=cbind(node_interface_errors_data,
|
||||
name=rep(interface_name,
|
||||
length(node_interface_errors_data$epoch)))
|
||||
node_interface_errors_data=cbind(node_interface_errors_data,
|
||||
name=rep(interface_name,
|
||||
length(node_interface_errors_data$epoch)))
|
||||
node_interface_errors_data$s_offset = node_interface_errors_data$epoch - local_bootdata[1,]$epoch
|
||||
node_interface_errors_data$s_offset = node_interface_errors_data$epoch - local_bootdata[1,]$epoch
|
||||
|
||||
interface_errors_data=rbind(interface_errors_data, node_interface_errors_data)
|
||||
interface_errors_data=rbind(interface_errors_data, node_interface_errors_data)
|
||||
}
|
||||
}
|
||||
|
||||
# Do not use the master (non-schedulable) nodes to calculate
|
||||
@@ -264,43 +268,79 @@ for (currentdir in resultdirs) {
|
||||
next
|
||||
}
|
||||
|
||||
max_free_mem=max(node_mem_free_data$value)
|
||||
min_free_mem=min(node_mem_free_data$value)
|
||||
# get the epoch time of first and last pod launch
|
||||
start_time=local_bootdata$epoch[1]
|
||||
end_time=local_bootdata$epoch[length(local_bootdata$epoch)]
|
||||
|
||||
# get value closest to first pod launch
|
||||
mem_start_index=Position(function(x) x > start_time, node_mem_free_data$epoch)
|
||||
# take the reading previous to the index as long as a valid index
|
||||
if (is.na(mem_start_index)) {
|
||||
mem_start_index = 1
|
||||
} else if (mem_start_index > 1) {
|
||||
mem_start_index = mem_start_index - 1
|
||||
}
|
||||
max_free_mem=node_mem_free_data$value[mem_start_index]
|
||||
|
||||
# get value closest to last pod launch
|
||||
mem_end_index=Position(function(x) x > end_time, node_mem_free_data$epoch)
|
||||
# take the reading previous to the index as long as a valid index
|
||||
if (is.na(mem_end_index)) {
|
||||
mem_end_index = length(node_mem_free_data$epoch)
|
||||
} else if (mem_end_index > 1) {
|
||||
mem_end_index = mem_end_index - 1
|
||||
}
|
||||
min_free_mem=node_mem_free_data$value[mem_end_index]
|
||||
|
||||
memtotal = memtotal + (max_free_mem - min_free_mem)
|
||||
max_idle_cpu=max(node_cpu_idle_data$value)
|
||||
min_idle_cpu=min(node_cpu_idle_data$value)
|
||||
|
||||
# get value closest to first pod launch
|
||||
cpu_start_index=Position(function(x) x > start_time, node_cpu_idle_data$epoch)
|
||||
# take the reading previous to the index as long as a valid index
|
||||
if (is.na(cpu_start_index)) {
|
||||
cpu_start_index = 1
|
||||
} else if (cpu_start_index > 1) {
|
||||
cpu_start_index = cpu_start_index - 1
|
||||
}
|
||||
max_idle_cpu=node_cpu_idle_data$value[cpu_start_index]
|
||||
|
||||
# get value closest to last pod launch
|
||||
cpu_end_index=Position(function(x) x > end_time, node_cpu_idle_data$epoch)
|
||||
# take the reading previous to the index as long as a valid index
|
||||
if (is.na(cpu_end_index)) {
|
||||
cpu_end_index = length(node_cpu_idle_data$epoch)
|
||||
} else if (cpu_end_index > 1) {
|
||||
cpu_end_index = cpu_end_index - 1
|
||||
}
|
||||
min_idle_cpu=node_cpu_idle_data$value[cpu_end_index]
|
||||
|
||||
cputotal = cputotal + (max_idle_cpu - min_idle_cpu)
|
||||
max_free_inode=max(node_inode_free_data$value)
|
||||
min_free_inode=min(node_inode_free_data$value)
|
||||
|
||||
# get value closest to first pod launch
|
||||
inode_start_index=Position(function(x) x > start_time, node_inode_free_data$epoch)
|
||||
# take the reading previous to the index as long as a valid index
|
||||
if (is.na(inode_start_index)) {
|
||||
inode_start_index = 1
|
||||
} else if (inode_start_index > 1) {
|
||||
inode_start_index = inode_start_index - 1
|
||||
}
|
||||
max_free_inode=node_inode_free_data$value[inode_start_index]
|
||||
|
||||
# get value closest to last pod launch
|
||||
inode_end_index=Position(function(x) x > end_time, node_inode_free_data$epoch)
|
||||
# take the reading previous to the index as long as a valid index
|
||||
if (is.na(inode_end_index)) {
|
||||
inode_end_index = length(node_cpu_idle_data$epoch)
|
||||
} else if (inode_end_index > 1) {
|
||||
inode_end_index = inode_end_index - 1
|
||||
}
|
||||
min_free_inode=node_inode_free_data$value[inode_end_index]
|
||||
|
||||
inodetotal = inodetotal + (max_free_inode - min_free_inode)
|
||||
}
|
||||
|
||||
num_pods = local_bootdata$n_pods[length(local_bootdata$n_pods)]
|
||||
|
||||
# calculate scaling for secondary y axis
|
||||
# the two y scales, in R, must be mathematically related
|
||||
mem_scale = max(c(mem_scale,
|
||||
(max(mem_free_data$value) / (1024*1024*1024)) / num_pods))
|
||||
cpu_scale = max(c(cpu_scale,
|
||||
max(cpu_idle_data$value) / num_pods))
|
||||
inode_scale = max(c(inode_scale,
|
||||
max(inode_free_data$value) / num_pods))
|
||||
ip_scale = max(c(ip_scale,
|
||||
max(c(max(interface_packets_data$tx, na.rm=TRUE),
|
||||
max(interface_packets_data$rx, na.rm=TRUE))) / num_pods))
|
||||
oct_scale = max(c(oct_scale,
|
||||
max(c(max(interface_octets_data$tx, na.rm=TRUE),
|
||||
max(interface_octets_data$rx, na.rm=TRUE))) / num_pods))
|
||||
# drops and scale are often 0, so providing 1 so we won't scale by infinity
|
||||
drop_scale = max(c(drop_scale,
|
||||
max(c(1,
|
||||
max(interface_dropped_data$tx, na.rm=TRUE),
|
||||
max(interface_dropped_data$rx, na.rm=TRUE))) / num_pods))
|
||||
error_scale = max(c(error_scale,
|
||||
max(c(1,
|
||||
max(interface_errors_data$tx, na.rm=TRUE),
|
||||
max(interface_errors_data$rx, na.rm=TRUE))) / num_pods))
|
||||
|
||||
# We get data in b, but want the graphs in Gb.
|
||||
memtotal = memtotal / (1024*1024*1024)
|
||||
gb_per_pod = memtotal/num_pods
|
||||
@@ -367,13 +407,14 @@ memfreedata$mem_free_gb = memfreedata$value/(1024*1024*1024)
|
||||
# And show the boot times in seconds, not ms
|
||||
podbootdata$launch_time_s = podbootdata$launch_time/1000.0
|
||||
|
||||
|
||||
########### Output memory page ##############
|
||||
mem_stats_plot = suppressWarnings(ggtexttable(data.frame(memstats),
|
||||
theme=ttheme(base_size=10),
|
||||
rows=NULL
|
||||
))
|
||||
|
||||
#mem_sec_axis_scale=
|
||||
mem_scale = (max(memfreedata$value) / (1024*1024*1024)) / max(podbootdata$n_pods)
|
||||
mem_line_plot <- ggplot() +
|
||||
geom_line(data=memfreedata,
|
||||
aes(s_offset, mem_free_gb, colour=interaction(testname, node),
|
||||
@@ -394,6 +435,7 @@ mem_line_plot <- ggplot() +
|
||||
ylab("System Avail (Gb)") +
|
||||
scale_y_continuous(labels=comma, sec.axis=sec_axis(~ ./mem_scale, name="pods")) +
|
||||
ggtitle("System Memory free") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
page1 = grid.arrange(
|
||||
@@ -411,6 +453,7 @@ cpu_stats_plot = suppressWarnings(ggtexttable(data.frame(cpustats),
|
||||
rows=NULL
|
||||
))
|
||||
|
||||
cpu_scale = max(cpuidledata$value) / max(podbootdata$n_pods)
|
||||
cpu_line_plot <- ggplot() +
|
||||
geom_line(data=cpuidledata,
|
||||
aes(x=s_offset, y=value, colour=interaction(testname, node),
|
||||
@@ -431,6 +474,7 @@ cpu_line_plot <- ggplot() +
|
||||
xlab("seconds") +
|
||||
ylab("System CPU Idle (%)") +
|
||||
ggtitle("System CPU usage") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
page2 = grid.arrange(
|
||||
@@ -455,6 +499,7 @@ boot_line_plot <- ggplot() +
|
||||
xlab("pods") +
|
||||
ylab("Boot time (s)") +
|
||||
ggtitle("Pod boot time") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
page3 = grid.arrange(
|
||||
@@ -472,6 +517,7 @@ inode_stats_plot = suppressWarnings(ggtexttable(data.frame(inodestats),
|
||||
rows=NULL
|
||||
))
|
||||
|
||||
inode_scale = max(inodefreedata$value) / max(podbootdata$n_pods)
|
||||
inode_line_plot <- ggplot() +
|
||||
geom_line(data=inodefreedata,
|
||||
aes(x=s_offset, y=value, colour=interaction(testname, node),
|
||||
@@ -492,6 +538,7 @@ inode_line_plot <- ggplot() +
|
||||
ylab("inodes free") +
|
||||
scale_y_continuous(labels=comma, sec.axis=sec_axis(~ ./inode_scale, name="pods")) +
|
||||
ggtitle("inodes free") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
page4 = grid.arrange(
|
||||
@@ -504,6 +551,8 @@ page4 = grid.arrange(
|
||||
cat("\n\n\\pagebreak\n")
|
||||
|
||||
########## Output interface page packets and octets ##############
|
||||
ip_scale = max(c(max(ifpacketdata$tx, na.rm=TRUE),
|
||||
max(ifpacketdata$rx, na.rm=TRUE))) / max(podbootdata$n_pods)
|
||||
interface_packet_line_plot <- ggplot() +
|
||||
geom_line(data=ifpacketdata,
|
||||
aes(x=s_offset, y=tx, colour=interaction(testname, node, name, "tx"),
|
||||
@@ -532,8 +581,11 @@ interface_packet_line_plot <- ggplot() +
|
||||
ylab("packets") +
|
||||
scale_y_continuous(labels=comma, sec.axis=sec_axis(~ ./ip_scale, name="pods")) +
|
||||
ggtitle("interface packets") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
oct_scale = max(c(max(ifoctetdata$tx, na.rm=TRUE),
|
||||
max(ifoctetdata$rx, na.rm=TRUE))) / max(podbootdata$n_pods)
|
||||
interface_octet_line_plot <- ggplot() +
|
||||
geom_line(data=ifoctetdata,
|
||||
aes(x=s_offset, y=tx, colour=interaction(testname, node, name, "tx"),
|
||||
@@ -562,9 +614,9 @@ interface_octet_line_plot <- ggplot() +
|
||||
ylab("octets") +
|
||||
scale_y_continuous(labels=comma, sec.axis=sec_axis(~ ./oct_scale, name="pods")) +
|
||||
ggtitle("interface octets") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
|
||||
page5 = grid.arrange(
|
||||
interface_packet_line_plot,
|
||||
interface_octet_line_plot,
|
||||
@@ -575,6 +627,10 @@ page5 = grid.arrange(
|
||||
cat("\n\n\\pagebreak\n")
|
||||
|
||||
########## Output interface page drops and errors ##############
|
||||
# drops are often 0, so providing 1 so we won't scale by infinity
|
||||
drop_scale = max(c(1,
|
||||
max(ifdropdata$tx, na.rm=TRUE),
|
||||
max(ifdropdata$rx, na.rm=TRUE))) / max(podbootdata$n_pods)
|
||||
interface_drop_line_plot <- ggplot() +
|
||||
geom_line(data=ifdropdata,
|
||||
aes(x=s_offset, y=tx, colour=interaction(testname, node, name, "tx"),
|
||||
@@ -601,10 +657,15 @@ interface_drop_line_plot <- ggplot() +
|
||||
labs(colour="") +
|
||||
xlab("seconds") +
|
||||
ylab("drops") +
|
||||
scale_y_continuous(labels=comma, sec.axis=sec_axis(~ ./drop_scale, name="pods")) +
|
||||
scale_y_continuous(breaks=pretty_breaks(), sec.axis=sec_axis(~ ./drop_scale, name="pods", labels=comma)) +
|
||||
ggtitle("interface drops") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
# errors are often 0, so providing 1 so we won't scale by infinity
|
||||
error_scale = max(c(1,
|
||||
max(iferrordata$tx, na.rm=TRUE),
|
||||
max(iferrordata$rx, na.rm=TRUE))) / max(podbootdata$n_pods)
|
||||
interface_error_line_plot <- ggplot() +
|
||||
geom_line(data=iferrordata,
|
||||
aes(x=s_offset, y=tx, colour=interaction(testname, node, name, "tx"),
|
||||
@@ -631,8 +692,9 @@ interface_error_line_plot <- ggplot() +
|
||||
labs(colour="") +
|
||||
xlab("seconds") +
|
||||
ylab("errors") +
|
||||
scale_y_continuous(labels=comma, sec.axis=sec_axis(~ ./error_scale, name="pods")) +
|
||||
scale_y_continuous(breaks=pretty_breaks(), sec.axis=sec_axis(~ ./error_scale, name="pods", labels=comma)) +
|
||||
ggtitle("interface errors") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
page6 = grid.arrange(
|
||||
|
||||
@@ -8,6 +8,10 @@ author: "Auto generated"
|
||||
date: "`r format(Sys.time(), '%d %B, %Y')`"
|
||||
output:
|
||||
pdf_document:
|
||||
# Shrink the page margins so we get bigger/better resolution on the graphs
|
||||
# Keep the top and bottom margins reasonable, as we are really interested in
|
||||
# gaining 'width', and if we trim the bottom too much, we lose the page numbers.
|
||||
geometry: "left=1cm, right=1cm, top=2cm, bottom=2cm"
|
||||
urlcolor: blue
|
||||
---
|
||||
|
||||
|
||||
@@ -190,13 +190,13 @@ for (currentdir in resultdirs) {
|
||||
"avg_inode"=round(inodetotal/num_pods, 4)
|
||||
)
|
||||
inodestats=rbind(inodestats, local_inodes)
|
||||
}
|
||||
|
||||
# And collect up our rows into our global table of all results
|
||||
# These two tables *should* be the source of all the data we need to
|
||||
# process and plot (apart from the stats....)
|
||||
bootdata=rbind(bootdata, local_bootdata, make.row.names=FALSE)
|
||||
nodedata=rbind(nodedata, local_nodedata, make.row.names=FALSE)
|
||||
# And collect up our rows into our global table of all results
|
||||
# These two tables *should* be the source of all the data we need to
|
||||
# process and plot (apart from the stats....)
|
||||
bootdata=rbind(bootdata, local_bootdata, make.row.names=FALSE)
|
||||
nodedata=rbind(nodedata, local_nodedata, make.row.names=FALSE)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -229,6 +229,7 @@ mem_line_plot <- ggplot(data=nodedata, aes(n_pods,
|
||||
ylab("System Avail (Gb)") +
|
||||
scale_y_continuous(labels=comma) +
|
||||
ggtitle("System Memory free") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
page1 = grid.arrange(
|
||||
@@ -256,6 +257,7 @@ cpu_line_plot <- ggplot(data=nodedata, aes(n_pods,
|
||||
xlab("pods") +
|
||||
ylab("System CPU Idle (%)") +
|
||||
ggtitle("System CPU usage") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
page2 = grid.arrange(
|
||||
@@ -279,6 +281,7 @@ boot_line_plot <- ggplot() +
|
||||
xlab("pods") +
|
||||
ylab("Boot time (s)") +
|
||||
ggtitle("Pod boot time") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
page3 = grid.arrange(
|
||||
@@ -307,6 +310,7 @@ inode_line_plot <- ggplot(data=nodedata, aes(n_pods,
|
||||
ylab("inodes free") +
|
||||
scale_y_continuous(labels=comma) +
|
||||
ggtitle("inodes free") +
|
||||
theme(legend.position="bottom") +
|
||||
theme(axis.text.x=element_text(angle=90))
|
||||
|
||||
page4 = grid.arrange(
|
||||
|
||||
@@ -13,6 +13,7 @@ deployment="busybox"
|
||||
stats_pod="stats"
|
||||
|
||||
NUM_PODS=${NUM_PODS:-20}
|
||||
NUM_DEPLOYMENTS=${NUM_DEPLOYMENTS:-20}
|
||||
STEP=${STEP:-1}
|
||||
|
||||
LABEL=${LABEL:-magiclabel}
|
||||
@@ -24,6 +25,8 @@ delete_wait_time=${delete_wait_time:-600}
|
||||
settle_time=${settle_time:-5}
|
||||
use_api=${use_api:-yes}
|
||||
grace=${grace:-30}
|
||||
proc_wait_time=${proc_wait_time:-20}
|
||||
proc_sleep_time=2
|
||||
|
||||
declare -a new_pods
|
||||
declare -A node_basemem
|
||||
|
||||
@@ -126,14 +126,10 @@ init() {
|
||||
# a nice way to do it (unless you want to parse 'descibe nodes')
|
||||
# Have a read of https://github.com/kubernetes/kubernetes/issues/25353
|
||||
|
||||
k8s_api_init
|
||||
framework_init
|
||||
|
||||
# Ensure we pre-cache the container image etc.
|
||||
warmup
|
||||
|
||||
# And now we can set up our results storage then...
|
||||
metrics_json_init "k8s"
|
||||
save_config
|
||||
}
|
||||
|
||||
save_config(){
|
||||
@@ -218,11 +214,8 @@ cleanup() {
|
||||
|
||||
# First try to save any results we got
|
||||
metrics_json_end_array "BootResults"
|
||||
metrics_json_save
|
||||
|
||||
kill_deployment "${deployment}" "${LABEL}" "${LABELVALUE}" ${delete_wait_time}
|
||||
|
||||
k8s_api_shutdown
|
||||
framework_shutdown
|
||||
}
|
||||
|
||||
show_vars()
|
||||
@@ -253,8 +246,8 @@ help()
|
||||
usage=$(cat << EOF
|
||||
Usage: $0 [-h] [options]
|
||||
Description:
|
||||
Launch a series of workloads and take memory metric measurements after
|
||||
each launch.
|
||||
Launch a series of workloads in a parallel manner and take memory metric measurements
|
||||
after each launch.
|
||||
Options:
|
||||
-h, Help page.
|
||||
EOF
|
||||
|
||||
@@ -169,7 +169,7 @@ EOF
|
||||
if [ $n_pods -eq 0 ]; then
|
||||
local pods_per_gb=0
|
||||
else
|
||||
local pods_per_gb=$(bc -l <<< "scale=2; ($total_mem_used/1024) / $n_pods")
|
||||
local pods_per_gb=$(printf "%0f" $(bc -l <<< "scale=2; ($total_mem_used/1024) / $n_pods"))
|
||||
fi
|
||||
local mem_json="$(cat << EOF
|
||||
"memory": {
|
||||
@@ -209,7 +209,7 @@ init() {
|
||||
# FIXME - check the node(s) can run enough pods - check 'max-pods' in the
|
||||
# kubelet config - from 'kubectl describe node -o json' ?
|
||||
|
||||
k8s_api_init
|
||||
framework_init
|
||||
|
||||
# Launch our stats gathering pod
|
||||
kubectl apply -f ${SCRIPT_PATH}/${stats_pod}.yaml
|
||||
@@ -218,10 +218,6 @@ init() {
|
||||
# FIXME - we should probably 'warm up' the cluster with the container image(s) we will
|
||||
# use for testing, otherwise the download time will likely be included in the first pod
|
||||
# boot time.
|
||||
|
||||
# And now we can set up our results storage then...
|
||||
metrics_json_init "k8s"
|
||||
save_config
|
||||
}
|
||||
|
||||
save_config(){
|
||||
@@ -347,9 +343,7 @@ EOF
|
||||
)"
|
||||
|
||||
metrics_json_add_fragment "$json"
|
||||
metrics_json_save
|
||||
|
||||
k8s_api_shutdown
|
||||
framework_shutdown
|
||||
}
|
||||
|
||||
show_vars()
|
||||
@@ -380,8 +374,8 @@ help()
|
||||
usage=$(cat << EOF
|
||||
Usage: $0 [-h] [options]
|
||||
Description:
|
||||
Launch a series of workloads and take memory metric measurements after
|
||||
each launch.
|
||||
Launch a series of workloads in a linear manner and take memory metric measurements
|
||||
after each launch.
|
||||
Options:
|
||||
-h, Help page.
|
||||
EOF
|
||||
|
||||
@@ -195,7 +195,7 @@ EOF
|
||||
if [ $n_pods -eq 0 ]; then
|
||||
local pods_per_gb=0
|
||||
else
|
||||
local pods_per_gb=$(bc -l <<< "scale=2; ($total_mem_used/1024) / $n_pods")
|
||||
local pods_per_gb=$(printf "%0f" $(bc -l <<< "scale=2; ($total_mem_used/1024) / $n_pods"))
|
||||
fi
|
||||
local mem_json="$(cat << EOF
|
||||
"memory": {
|
||||
@@ -235,7 +235,7 @@ init() {
|
||||
# FIXME - check the node(s) can run enough pods - check 'max-pods' in the
|
||||
# kubelet config - from 'kubectl describe node -o json' ?
|
||||
|
||||
k8s_api_init
|
||||
framework_init
|
||||
|
||||
# Launch our stats gathering pod
|
||||
kubectl apply -f ${SCRIPT_PATH}/${stats_pod}.yaml
|
||||
@@ -244,10 +244,6 @@ init() {
|
||||
# FIXME - we should probably 'warm up' the cluster with the container image(s) we will
|
||||
# use for testing, otherwise the download time will likely be included in the first pod
|
||||
# boot time.
|
||||
|
||||
# And now we can set up our results storage then...
|
||||
metrics_json_init "k8s"
|
||||
save_config
|
||||
}
|
||||
|
||||
save_config(){
|
||||
@@ -410,9 +406,7 @@ EOF
|
||||
)"
|
||||
|
||||
metrics_json_add_fragment "$json"
|
||||
metrics_json_save
|
||||
|
||||
k8s_api_shutdown
|
||||
framework_shutdown
|
||||
}
|
||||
|
||||
show_vars()
|
||||
|
||||
Executable
+285
@@ -0,0 +1,285 @@
|
||||
#!/bin/bash
|
||||
# Copyright (c) 2019 Intel Corporation
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
#
|
||||
|
||||
set -e
|
||||
|
||||
# Pull in some common, useful, items
|
||||
SCRIPT_PATH=$(dirname "$(readlink -f "$0")")
|
||||
source "${SCRIPT_PATH}/../lib/common.bash"
|
||||
source "${SCRIPT_PATH}/common.bash"
|
||||
|
||||
LABELVALUE=${LABELVALUE:-scale_net}
|
||||
|
||||
# Set some default metrics env vars
|
||||
TEST_ARGS="runtime=${RUNTIME}"
|
||||
TEST_NAME="k8s scaling net"
|
||||
input_yaml="${SCRIPT_PATH}/net-serve.yaml.in"
|
||||
input_json="${SCRIPT_PATH}/net-serve.json.in"
|
||||
name_base_depl="net-serve"
|
||||
|
||||
# $1 is the launch time in seconds this pod/container took to start up.
|
||||
# $2 is the number of pod/containers under test
|
||||
# $3 is the time to pod network measure
|
||||
grab_stats(){
|
||||
local launch_time_ms=$1
|
||||
local n_pods=$2
|
||||
local net_time=$3
|
||||
|
||||
info "And grab some stats"
|
||||
|
||||
local date_json="$(cat << EOF
|
||||
"date": {
|
||||
"ns": $(date +%s%N),
|
||||
"Date": "$(date -u +"%Y-%m-%dT%T.%3N")"
|
||||
}
|
||||
EOF
|
||||
)"
|
||||
metrics_json_add_array_fragment "$date_json"
|
||||
|
||||
local pods_json="$(cat << EOF
|
||||
"n_pods": {
|
||||
"Result": ${n_pods},
|
||||
"Units" : "int"
|
||||
}
|
||||
EOF
|
||||
)"
|
||||
metrics_json_add_array_fragment "$pods_json"
|
||||
|
||||
local time_to_pod_net_json="$(cat << EOF
|
||||
"time_to_pod_net": {
|
||||
"Result": ${net_time},
|
||||
"Units" : "ms"
|
||||
}
|
||||
EOF
|
||||
)"
|
||||
metrics_json_add_array_fragment "$time_to_pod_net_json"
|
||||
|
||||
local launch_json="$(cat << EOF
|
||||
"launch_time": {
|
||||
"Result": $launch_time_ms,
|
||||
"Units" : "ms"
|
||||
}
|
||||
EOF
|
||||
)"
|
||||
metrics_json_add_array_fragment "$launch_json"
|
||||
|
||||
info "launch [$launch_time_ms]"
|
||||
|
||||
metrics_json_close_array_element
|
||||
}
|
||||
|
||||
init() {
|
||||
info "Initialising"
|
||||
|
||||
local cmds=("bc" "jq")
|
||||
check_cmds "${cmds[@]}"
|
||||
|
||||
info "Checking Kubernetes accessible"
|
||||
local worked=$( kubectl get nodes > /dev/null 2>&1 && echo $? || echo $? )
|
||||
if [ "$worked" != 0 ]; then
|
||||
die "kubectl failed to get nodes"
|
||||
fi
|
||||
|
||||
info $(get_num_nodes) "Kubernetes nodes in 'Ready' state found"
|
||||
|
||||
framework_init
|
||||
}
|
||||
|
||||
save_config() {
|
||||
metrics_json_start_array
|
||||
|
||||
local json="$(cat << EOF
|
||||
{
|
||||
"testname": "${TEST_NAME}",
|
||||
"NUM_DEPLOYMENTS": ${NUM_DEPLOYMENTS},
|
||||
"STEP": ${STEP},
|
||||
"wait_time": ${wait_time},
|
||||
"delete_wait_time": ${delete_wait_time},
|
||||
"settle_time": ${settle_time}
|
||||
}
|
||||
EOF
|
||||
)"
|
||||
metrics_json_add_array_element "$json"
|
||||
metrics_json_end_array "Config"
|
||||
}
|
||||
|
||||
run() {
|
||||
info "Running test"
|
||||
local header_post="Content-Type: application/json"
|
||||
local base_curl=${API_ADDRESS}:${API_PORT}/apis/apps/v1/namespaces/default/deployments
|
||||
|
||||
trap cleanup EXIT QUIT KILL
|
||||
|
||||
metrics_json_start_array
|
||||
|
||||
for reqs in $(seq ${STEP} ${STEP} ${NUM_DEPLOYMENTS}); do
|
||||
local deployment="${name_base_depl}${reqs}"
|
||||
info "Testing replicas ${reqs} of ${NUM_DEPLOYMENTS}"
|
||||
# Generate the next yaml file
|
||||
|
||||
local runtime_command
|
||||
if [ -n "$RUNTIME" ]; then
|
||||
runtime_command="s|@RUNTIMECLASS@|${RUNTIME}|g"
|
||||
else
|
||||
runtime_command="/@RUNTIMECLASS@/d"
|
||||
fi
|
||||
|
||||
local input_template
|
||||
local generated_file
|
||||
if [ "$use_api" != "no" ]; then
|
||||
input_template=$input_json
|
||||
generated_file=$generated_json
|
||||
else
|
||||
input_template=$input_yaml
|
||||
generated_file=$generated_yaml
|
||||
fi
|
||||
|
||||
sed -e $runtime_command \
|
||||
-e "s|@DEPLOYMENT@|${deployment}|g" \
|
||||
-e "s|@LABEL@|${LABEL}|g" \
|
||||
-e "s|@LABELVALUE@|${LABELVALUE}|g" \
|
||||
-e "s|@GRACE@|${grace}|g" \
|
||||
< ${input_template} > ${generated_file}
|
||||
|
||||
info "Applying changes"
|
||||
local start_time=$(date +%s%N)
|
||||
|
||||
if [ "$use_api" != "no" ]; then
|
||||
curl -s ${base_curl} -XPOST -H "${header_post}" -d@${generated_file} > /dev/null
|
||||
else
|
||||
kubectl apply -f ${generated_file}
|
||||
fi
|
||||
|
||||
kubectl rollout status --timeout=${wait_time}s deployment/${deployment}
|
||||
kubectl expose --port=8080 deployment $deployment
|
||||
|
||||
# Check service exposed
|
||||
cmd="kubectl get services $deployment -n default --no-headers=true"
|
||||
waitForProcess "$proc_wait_time" "$proc_sleep_time" "$cmd" "Waiting for service"
|
||||
|
||||
IP=$(kubectl get services $deployment -n default --no-headers=true | awk '{printf $3}')
|
||||
end_net=$(date +%s%N)
|
||||
info "IP: $IP"
|
||||
|
||||
# service health check
|
||||
cmd="curl --noproxy \"*\" http://$IP:8080/healthz"
|
||||
waitForProcess "$proc_wait_time" "$proc_sleep_time" "$cmd" "http server is not ready yet!!"
|
||||
|
||||
RESP=$(curl -s --noproxy "*" http://$IP:8080/echo?msg=curl%20request%20to%20$deployment)
|
||||
local end_time=$(date +%s%N)
|
||||
info "http reply: $RESP"
|
||||
|
||||
local total_milliseconds=$(( (end_time - start_time) / 1000000 ))
|
||||
local net_diff=$(( (end_net - start_time) / 1000000 ))
|
||||
info "Took $total_milliseconds ms ($end_time - $start_time)"
|
||||
info "Net took $net_diff ms"
|
||||
|
||||
kubectl delete service $deployment
|
||||
if [ $? -ne 0 ]; then
|
||||
echo "kubectl delete service failed"
|
||||
exit
|
||||
fi
|
||||
|
||||
sleep ${settle_time}
|
||||
grab_stats $total_milliseconds $reqs $net_diff
|
||||
done
|
||||
}
|
||||
|
||||
cleanup() {
|
||||
info "Cleaning up"
|
||||
|
||||
# First try to save any results we got
|
||||
metrics_json_end_array "BootResults"
|
||||
|
||||
local start_time=$(date +%s%N)
|
||||
|
||||
for reqs in $(seq ${STEP} ${STEP} ${NUM_DEPLOYMENTS}); do
|
||||
local deployment="${name_base_depl}${reqs}"
|
||||
kubectl delete deployment --wait=true --timeout=${delete_wait_time}s ${deployment} || true
|
||||
done
|
||||
|
||||
for x in $(seq 1 ${delete_wait_time}); do
|
||||
local npods=$(kubectl get pods -l=${LABEL}=${LABELVALUE} -o=name | wc -l)
|
||||
if [ $npods -eq 0 ]; then
|
||||
echo "All pods have terminated at cycle $x"
|
||||
local alldied=true
|
||||
break;
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
|
||||
if [ -z "$alldied" ]; then
|
||||
echo "ERROR: Not all pods died!"
|
||||
fi
|
||||
|
||||
local end_time=$(date +%s%N)
|
||||
local total_milliseconds=$(( (end_time - start_time) / 1000000 ))
|
||||
info "Delete Took $total_milliseconds ms ($end_time - $start_time)"
|
||||
|
||||
local json="$(cat << EOF
|
||||
"Delete": {
|
||||
"Result": ${total_milliseconds},
|
||||
"Units" : "ms"
|
||||
}
|
||||
EOF
|
||||
)"
|
||||
|
||||
metrics_json_add_fragment "$json"
|
||||
framework_shutdown
|
||||
}
|
||||
|
||||
show_vars() {
|
||||
echo -e "\nEnvironment variables:"
|
||||
echo -e "\tName (default)"
|
||||
echo -e "\t\tDescription"
|
||||
echo -e "\tNUM_DEPLOYMENTS (${NUM_DEPLOYMENTS})"
|
||||
echo -e "\t\tNumber of deployments to launch"
|
||||
echo -e "\tSTEP (${STEP})"
|
||||
echo -e "\t\tNumber of pods to launch per cycle"
|
||||
echo -e "\twait_time (${wait_time})"
|
||||
echo -e "\t\tSeconds to wait for pods to become ready"
|
||||
echo -e "\tproc_wait_time (${proc_wait_time})"
|
||||
echo -e "\t\tSeconds to wait for net server process to become ready"
|
||||
echo -e "\tdelete_wait_time (${delete_wait_time})"
|
||||
echo -e "\t\tSeconds to wait for all pods to be deleted"
|
||||
echo -e "\tsettle_time (${settle_time})"
|
||||
echo -e "\t\tSeconds to wait after pods ready before taking measurements"
|
||||
echo -e "\tuse_api (${use_api})"
|
||||
echo -e "\t\tspecify yes or no to use the API to launch pods"
|
||||
echo -e "\tgrace (${grace})"
|
||||
echo -e "\t\tspecify the grace period in seconds for workload pod termination"
|
||||
}
|
||||
|
||||
help() {
|
||||
usage=$(cat << EOF
|
||||
Usage: $0 [-h] [options]
|
||||
Description:
|
||||
Launch a series of workloads and take time to pod network metric measurements after
|
||||
each launch.
|
||||
Options:
|
||||
-h, Help page.
|
||||
EOF
|
||||
)
|
||||
echo "$usage"
|
||||
show_vars
|
||||
}
|
||||
|
||||
main() {
|
||||
local OPTIND
|
||||
while getopts "h" opt;do
|
||||
case ${opt} in
|
||||
h)
|
||||
help
|
||||
exit 0;
|
||||
;;
|
||||
esac
|
||||
done
|
||||
shift $((OPTIND-1))
|
||||
init
|
||||
run
|
||||
}
|
||||
|
||||
main "$@"
|
||||
@@ -15,6 +15,8 @@ source "${SCRIPT_PATH}/../collectd/collectd.bash"
|
||||
NUM_PODS=${NUM_PODS:-20}
|
||||
STEP=${STEP:-1}
|
||||
|
||||
SMF_USE_COLLECTD=true
|
||||
|
||||
LABELVALUE=${LABELVALUE:-gandalf}
|
||||
|
||||
pod_command="[\"tail\", \"-f\", \"/dev/null\"]"
|
||||
@@ -64,27 +66,7 @@ EOF
|
||||
}
|
||||
|
||||
init() {
|
||||
info "Initialising"
|
||||
|
||||
local cmds=("bc" "jq")
|
||||
check_cmds "${cmds[@]}"
|
||||
|
||||
info "Checking k8s accessible"
|
||||
local worked=$( kubectl get nodes > /dev/null 2>&1 && echo $? || echo $? )
|
||||
if [ "$worked" != 0 ]; then
|
||||
die "kubectl failed to get nodes"
|
||||
fi
|
||||
|
||||
info $(get_num_nodes) "k8s nodes in 'Ready' state found"
|
||||
|
||||
k8s_api_init
|
||||
|
||||
# Launch our stats gathering pod
|
||||
init_stats $wait_time
|
||||
|
||||
# And now we can set up our results storage then...
|
||||
metrics_json_init "k8s"
|
||||
save_config
|
||||
framework_init
|
||||
}
|
||||
|
||||
save_config(){
|
||||
@@ -198,11 +180,7 @@ EOF
|
||||
)"
|
||||
|
||||
metrics_json_add_fragment "$json"
|
||||
metrics_json_save
|
||||
|
||||
cleanup_stats $delete_wait_time
|
||||
|
||||
k8s_api_shutdown
|
||||
framework_shutdown
|
||||
}
|
||||
|
||||
show_vars()
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
{
|
||||
"apiVersion": "apps/v1",
|
||||
"kind": "Deployment",
|
||||
"metadata": {
|
||||
"labels": {
|
||||
"run": "net-serve"
|
||||
},
|
||||
"name": "@DEPLOYMENT@"
|
||||
},
|
||||
"spec": {
|
||||
"replicas": 1,
|
||||
"selector": {
|
||||
"matchLabels": {
|
||||
"run": "net-serve"
|
||||
}
|
||||
},
|
||||
"template": {
|
||||
"metadata": {
|
||||
"labels": {
|
||||
"run": "net-serve",
|
||||
"@LABEL@": "@LABELVALUE@"
|
||||
}
|
||||
},
|
||||
"spec": {
|
||||
"terminationGracePeriodSeconds": @GRACE@,
|
||||
"runtimeClassName": "@RUNTIMECLASS@",
|
||||
"automountServiceAccountToken": false,
|
||||
"containers": [{
|
||||
"name": "net-serve",
|
||||
"image": "gcr.io/kubernetes-e2e-test-images/agnhost:2.8",
|
||||
"imagePullPolicy": "IfNotPresent",
|
||||
"args": [
|
||||
"netexec"
|
||||
]
|
||||
}],
|
||||
"restartPolicy": "Always"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
labels:
|
||||
run: net-serve
|
||||
name: @DEPLOYMENT@
|
||||
spec:
|
||||
replicas: 1
|
||||
selector:
|
||||
matchLabels:
|
||||
run: net-serve
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
run: net-serve
|
||||
@LABEL@: @LABELVALUE@
|
||||
spec:
|
||||
terminationGracePeriodSeconds: @GRACE@
|
||||
runtimeClassName: @RUNTIMECLASS@
|
||||
automountServiceAccountToken: false
|
||||
containers:
|
||||
- name: net-serve
|
||||
image: gcr.io/kubernetes-e2e-test-images/agnhost:2.8
|
||||
imagePullPolicy: IfNotPresent
|
||||
args:
|
||||
- netexec
|
||||
restartPolicy: Always
|
||||
Reference in New Issue
Block a user