diff --git a/assets/loadtest/.gitignore b/assets/loadtest/.gitignore index 1057e07b48f..249797b7fd9 100644 --- a/assets/loadtest/.gitignore +++ b/assets/loadtest/.gitignore @@ -1,15 +1,17 @@ etcd/certs/*.pem teleport/oidc.yaml -k8s/certificate.yaml +k8s/*certificate*.yaml k8s/secrets/** !k8s/secrets/Makefile -**/*-gen.yaml +**/*-gen*.yaml **/profiles/** **/.terraform **/.terraform.lock* **/terraform.tfstate* -**/terraform.tfvars \ No newline at end of file +**/terraform.tfvars + +.secrets \ No newline at end of file diff --git a/assets/loadtest/Makefile b/assets/loadtest/Makefile index ff923787afa..244612c1560 100644 --- a/assets/loadtest/Makefile +++ b/assets/loadtest/Makefile @@ -1,6 +1,7 @@ SOAK_TEST_DURATION ?= 30m USE_CERT_MANAGER ?= yes -TELEPORT_IMAGE ?= quay.io/gravitational/teleport-ent:8.0.0 +TELEPORT_IMAGE ?= quay.io/gravitational/teleport-ent:9.0.0 +NAMESPACE ?= loadtest .PHONY: reserve-ips reserve-ips: @@ -20,12 +21,17 @@ delete-cluster: # deploy teleport with etcd backend to loadtest namespace .PHONY: deploy-etcd-cluster deploy-etcd-cluster: - @make -C k8s apply BACKEND=etcd USE_CERT_MANAGER=$(USE_CERT_MANAGER) TELEPORT_IMAGE=$(TELEPORT_IMAGE) + @make -C k8s apply BACKEND=etcd USE_CERT_MANAGER=$(USE_CERT_MANAGER) TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) -# deploy teleport with etcd backend to loadtest namespace +# deploy teleport with firestore backend to loadtest namespace .PHONY: deploy-firestore-cluster deploy-firestore-cluster: - @make -C k8s apply BACKEND=firestore USE_CERT_MANAGER=$(USE_CERT_MANAGER) TELEPORT_IMAGE=$(TELEPORT_IMAGE) + @make -C k8s apply BACKEND=firestore USE_CERT_MANAGER=$(USE_CERT_MANAGER) TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) + +# deploy teleport with dynamo backend to loadtest namespace +.PHONY: deploy-dynamo-cluster +deploy-dynamo-cluster: + @make -C k8s apply BACKEND=dynamo USE_CERT_MANAGER=$(USE_CERT_MANAGER) TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) # delete the loadtest namespace .PHONY: delete-deploy @@ -35,7 +41,7 @@ delete-deploy: # run soak tests .PHONY: run-soak-tests run-soak-tests: - @make -C k8s run-soak-tests SOAK_TEST_DURATION=$(SOAK_TEST_DURATION) TELEPORT_IMAGE=$(TELEPORT_IMAGE) + @make -C k8s run-soak-tests SOAK_TEST_DURATION=$(SOAK_TEST_DURATION) TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) # run 500 node trusted cluster scaling test # This installs 500 trusted clusters, waits for a period of time and then @@ -43,7 +49,7 @@ run-soak-tests: # The test helps identify any potential memory leaks caused by remote clusters .PHONY: run-tc-scaling-test run-tc-scaling-test: - @make -C k8s run-tc-scaling-test TELEPORT_IMAGE=$(TELEPORT_IMAGE) + @make -C k8s run-tc-scaling-test TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) # run the 10k scaling test # This tests installs 10,000 IoT nodes and then removes all 10,000 nodes after a fixed period of time. @@ -51,12 +57,12 @@ run-tc-scaling-test: # potential memory leaks caused by large clusters for both non-IoT and IoT nodes. .PHONY: run-scaling-test run-scaling-test: - @make -C k8s run-scaling-test TELEPORT_IMAGE=$(TELEPORT_IMAGE) + @make -C k8s run-scaling-test TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) # list pods in loadtest namespace .PHONY: pods pods: - @make -C k8s pods + @make -C k8s pods NAMESPACE=$(NAMESPACE) # get cluster credentials .PHONY: get-creds @@ -66,4 +72,4 @@ get-creds: # collect heap and goroutine profiles .PHONY: collect-profiles collect-profiles: - @make -C k8s collect-profiles PROFILE_LOCATION=$(shell pwd) \ No newline at end of file + @make -C k8s collect-profiles PROFILE_LOCATION=$(shell pwd) NAMESPACE=$(NAMESPACE) \ No newline at end of file diff --git a/assets/loadtest/README.md b/assets/loadtest/README.md index fba50c9b3f3..56c9cdd2149 100644 --- a/assets/loadtest/README.md +++ b/assets/loadtest/README.md @@ -18,26 +18,25 @@ Teleports manual release test plan. - Make sure that you have a GCP service account key with `Compute Admin`, `Compute Network Admin`, `Kubernetes Engine Admin`, `Kubernetes Engine Cluster Admin`, and `Service Account User` - To authenticate as the service account follow these [instructions](https://cloud.google.com/docs/authentication/production) -- Make sure you have reserved static ip addresses for the proxy and grafana +- Make sure you have reserved static ip addresses for the proxy - This only needs to be done once per GCP project, see the [network docs](./network/README.md) for details ### Creating the Cluster First create a cluster, if you are running this automation for the first time, you may be asked to run -`terraform init` from the cluster directory before continuing. To resize the cluster, edit [`terraform.tfvars`](./cluster/terraform.tfvars) as needed. +`terraform init` from the cluster directory before continuing. To resize the cluster, edit [`terraform.tfvars`](cluster/terraform.tfvars) as needed. ```bash $ make create-cluster ``` ### DNS Entries -Before deploying anything to the cluster you first need to set `PROXY_HOST` and `GRAFANA_HOST`. These variables should -be the DNS names to be used for the [`proxy`](./k8s/proxy.yaml) and [`grafana`](./k8s/grafana.yaml) services. When everything is successfully deployed you should be able -to navigate to `https://PROXY_HOST:3080` and `https://GRAFANA_HOST:8443` in your browser. +Before deploying anything to the cluster you first need to set `PROXY_HOST`. These variables should +be the DNS names to be used for the [`proxy`](./k8s/proxy.yaml). When everything is successfully deployed you should be able +to navigate to `https://PROXY_HOST:3080` in your browser. ```bash $ export PROXY_HOST=proxy.loadtest.com -$ export GRAFANA_HOST=grafana.loadtest.com ``` ### TLS Certificates @@ -93,7 +92,7 @@ $ make run-soak-tests ``` -**Note:** You must have enough nodes in the cluster to run the following tests. Ensure your `node_count` in [`terraform.tfvars`](./cluster/terraform.tfvars) is correctly set. +**Note:** You must have enough nodes in the cluster to run the following tests. Ensure your `node_count` in [`terraform.tfvars`](cluster/terraform.tfvars) is correctly set. To run the 10k node scaling tests: diff --git a/assets/loadtest/etcd/telegraf.conf b/assets/loadtest/etcd/telegraf.conf deleted file mode 100644 index e85560f79ee..00000000000 --- a/assets/loadtest/etcd/telegraf.conf +++ /dev/null @@ -1,137 +0,0 @@ -# Configuration for telegraf agent -[agent] - ## Default data collection interval for all inputs - interval = "10s" - ## Rounds collection interval to 'interval' - ## ie, if interval="10s" then always collect on :00, :10, :20, etc. - round_interval = true - - ## Telegraf will send metrics to outputs in batches of at - ## most metric_batch_size metrics. - metric_batch_size = 1000 - ## For failed writes, telegraf will cache metric_buffer_limit metrics for each - ## output, and will flush this buffer on a successful write. Oldest metrics - ## are dropped first when this buffer fills. - metric_buffer_limit = 10000 - - ## Collection jitter is used to jitter the collection by a random amount. - ## Each plugin will sleep for a random time within jitter before collecting. - ## This can be used to avoid many plugins querying things like sysfs at the - ## same time, which can have a measurable effect on the system. - collection_jitter = "0s" - - ## Default flushing interval for all outputs. You shouldn't set this below - ## interval. Maximum flush_interval will be flush_interval + flush_jitter - flush_interval = "10s" - ## Jitter the flush interval by a random amount. This is primarily to avoid - ## large write spikes for users running a large number of telegraf instances. - ## ie, a jitter of 5s and interval 10s means flushes will happen every 10-15s - flush_jitter = "0s" - - ## By default, precision will be set to the same timestamp order as the - ## collection interval, with the maximum being 1s. - ## Precision will NOT be used for service inputs, such as logparser and statsd. - precision = "" - ## Run telegraf in debug mode - debug = false - ## Run telegraf in quiet mode - quiet = false - ## Override default hostname, if empty use os.Hostname() - hostname = "" - ## If set to true, do no set the "host" tag in the telegraf agent. - omit_hostname = false - - -############################################################################### -# INPUT PLUGINS # -############################################################################### - -[[inputs.procstat]] - exe = "etcd" - prefix = "etcd" - -[[inputs.prometheus]] - # An array of urls to scrape metrics from. - urls = ["https://127.0.0.1:2379/metrics"] - tls_ca = "/etc/etcd/certs/ca-cert.pem" - tls_cert = "/etc/etcd/certs/client-cert.pem" - tls_key = "/etc/etcd/certs/client-key.pem" - name_prefix = "etcd_" - - # Add tags to be able to make beautiful dashboards - [inputs.prometheus.tags] - teleservice = "etcd" - -# Read metrics about cpu usage -[[inputs.cpu]] - ## Whether to report per-cpu stats or not - percpu = true - ## Whether to report total system cpu stats or not - totalcpu = true - ## If true, collect raw CPU time metrics. - collect_cpu_time = false - ## If true, compute and report the sum of all non-idle CPU states. - report_active = false - -# Read metrics about disk usage by mount point -[[inputs.disk]] - ## By default, telegraf gather stats for all mountpoints. - ## Setting mountpoints will restrict the stats to the specified mountpoints. - # mount_points = ["/"] - - ## Ignore some mountpoints by filesystem type. For example (dev)tmpfs (usually - ## present on /run, /var/run, /dev/shm or /dev). - ignore_fs = ["tmpfs", "devtmpfs", "devfs"] - -# Read metrics about disk IO by device -[[inputs.diskio]] - -# Get kernel statistics from /proc/stat -[[inputs.kernel]] - # no configuration - -# Read metrics about memory usage -[[inputs.mem]] - # no configuration - -# Get the number of processes and group them by status -[[inputs.processes]] - # no configuration - -# Read metrics about swap memory usage -[[inputs.swap]] - # no configuration - -# Read metrics about system load & uptime -[[inputs.system]] - # no configuration - -# Read netstat info -[[inputs.netstat]] - -# Read net info -[[inputs.net]] - -############################################################################### -# OUTPUT PLUGINS # -############################################################################### - -# Configuration for influxdb server to send metrics to -[[outputs.influxdb_v2]] - ## The full HTTP or UDP endpoint URL for your InfluxDB instance. - ## Multiple urls can be specified as part of the same cluster, - ## this means that only ONE of the urls will be written to each interval. - urls = ["http://influxdb:8086"] # required - - ## Token for authentication. - token = "${INFLUXDB_TOKEN}" - - ## Organization is the name of the organization you wish to write to; must exist. - organization = "teleport" - - ## Destination bucket to write into. - bucket = "telegraf" - - ## Write timeout (for the InfluxDB client), formatted as a string. - ## If not provided, will default to 5s. 0s means no timeout (not recommended). - timeout = "5s" \ No newline at end of file diff --git a/assets/loadtest/grafana/dashboard.yaml b/assets/loadtest/grafana/dashboard.yaml deleted file mode 100644 index 012d60cc4b6..00000000000 --- a/assets/loadtest/grafana/dashboard.yaml +++ /dev/null @@ -1,7 +0,0 @@ -apiVersion: 1 - -providers: - - name: Default - type: file - options: - path: /var/lib/grafana/dashboards \ No newline at end of file diff --git a/assets/loadtest/grafana/health-dashboard.json b/assets/loadtest/grafana/health-dashboard.json deleted file mode 100644 index f141f0b4cc2..00000000000 --- a/assets/loadtest/grafana/health-dashboard.json +++ /dev/null @@ -1,781 +0,0 @@ -{ - "annotations": { - "list": [ - { - "builtIn": 1, - "datasource": "-- Grafana --", - "enable": true, - "hide": true, - "iconColor": "rgba(0, 211, 255, 1)", - "name": "Annotations & Alerts", - "target": { - "limit": 100, - "matchAny": false, - "tags": [], - "type": "dashboard" - }, - "type": "dashboard" - } - ] - }, - "editable": true, - "fiscalYearStartMonth": 0, - "gnetId": null, - "graphTooltip": 0, - "id": 1, - "links": [], - "liveNow": false, - "panels": [ - { - "collapsed": false, - "datasource": null, - "gridPos": { - "h": 1, - "w": 24, - "x": 0, - "y": 0 - }, - "id": 8, - "panels": [], - "title": "Teleport", - "type": "row" - }, - { - "datasource": null, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisLabel": "Goroutine Count", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "auto", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "red", - "value": 80 - } - ] - } - }, - "overrides": [] - }, - "gridPos": { - "h": 10, - "w": 12, - "x": 0, - "y": 1 - }, - "id": 10, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom" - }, - "tooltip": { - "mode": "single" - } - }, - "targets": [ - { - "hide": false, - "query": " from(bucket: v.bucket)\n |> range(start: v.timeRangeStart, stop: v.timeRangeStop)\n |> filter(fn: (r) => r._measurement == \"teleport_go_goroutines\" and r._field == \"gauge\")", - "refId": "A" - } - ], - "title": "Goroutines (Max Per Interval)", - "transformations": [ - { - "id": "labelsToFields", - "options": { - "valueLabel": "host" - } - } - ], - "type": "timeseries" - }, - { - "datasource": null, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisLabel": "File Descriptors (Max)", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "auto", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "red", - "value": 80 - } - ] - } - }, - "overrides": [] - }, - "gridPos": { - "h": 10, - "w": 12, - "x": 12, - "y": 1 - }, - "id": 12, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom" - }, - "tooltip": { - "mode": "single" - } - }, - "targets": [ - { - "query": " from(bucket: v.bucket)\n |> range(start: v.timeRangeStart, stop: v.timeRangeStop)\n |> filter(fn: (r) =>r._measurement == \"teleport_process_open_fds\" and r._field == \"gauge\")", - "refId": "A" - } - ], - "title": "Open File Descriptors (Max)", - "transformations": [ - { - "id": "labelsToFields", - "options": { - "valueLabel": "host" - } - } - ], - "type": "timeseries" - }, - { - "datasource": null, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisLabel": "Megabytes", - "axisPlacement": "auto", - "barAlignment": -1, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "lineInterpolation": "linear", - "lineStyle": { - "fill": "solid" - }, - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "auto", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "red", - "value": 80 - } - ] - }, - "unit": "bytes" - }, - "overrides": [] - }, - "gridPos": { - "h": 11, - "w": 12, - "x": 0, - "y": 11 - }, - "id": 14, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom" - }, - "tooltip": { - "mode": "single" - } - }, - "targets": [ - { - "query": " from(bucket: v.bucket)\n |> range(start: v.timeRangeStart, stop: v.timeRangeStop)\n |> filter(fn: (r) =>r._measurement == \"teleport_go_memstats_heap_inuse_bytes\" and r._field == \"gauge\")", - "refId": "A" - } - ], - "title": "Heap In Use Bytes", - "transformations": [ - { - "id": "labelsToFields", - "options": { - "valueLabel": "host" - } - } - ], - "type": "timeseries" - }, - { - "datasource": null, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisLabel": "Number of Cores", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "auto", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "red", - "value": 80 - } - ] - } - }, - "overrides": [] - }, - "gridPos": { - "h": 11, - "w": 12, - "x": 12, - "y": 11 - }, - "id": 16, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom" - }, - "tooltip": { - "mode": "single" - } - }, - "targets": [ - { - "query": " from(bucket: v.bucket)\n |> range(start: v.timeRangeStart, stop: v.timeRangeStop)\n |> filter(fn: (r) =>r._measurement == \"teleport_process_cpu_seconds_total\" and r._field == \"counter\")\n |> derivative(nonNegative: true, unit: 10s)\n |> map(fn: (r) => ({ r with _value: r._value / float(v: 10) }))", - "refId": "A" - } - ], - "title": "CPU Cores", - "transformations": [ - { - "id": "labelsToFields", - "options": { - "valueLabel": "host" - } - } - ], - "type": "timeseries" - }, - { - "datasource": null, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisLabel": "Event Size", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "auto", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "red", - "value": 80 - } - ] - }, - "unit": "bytes" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 0, - "y": 22 - }, - "id": 18, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom" - }, - "tooltip": { - "mode": "single" - } - }, - "pluginVersion": "8.2.3", - "targets": [ - { - "query": "from(bucket: v.bucket)\n |> range(start: v.timeRangeStart, stop: v.timeRangeStop)\n |> filter(fn: (r) =>r._measurement == \"teleport_watcher_events\" and r._field == \"sum\")\n |> group(columns: [\"host\"])\n |> sort(columns: [\"_time\"], desc: true)\n |> aggregateWindow(every: 10s, fn: sum, createEmpty: false)\n |> derivative(nonNegative: true, unit: 10s)", - "refId": "A" - } - ], - "title": "Events Per/Sec By Host", - "transformations": [ - { - "id": "labelsToFields", - "options": { - "valueLabel": "host" - } - } - ], - "type": "timeseries" - }, - { - "datasource": null, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisLabel": "Number of Events", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "auto", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "red", - "value": 80 - } - ] - }, - "unit": "none" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 12, - "y": 22 - }, - "id": 19, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom" - }, - "tooltip": { - "mode": "single" - } - }, - "targets": [ - { - "query": " from(bucket: v.bucket)\n |> range(start: v.timeRangeStart, stop: v.timeRangeStop)\n |> filter(fn: (r) =>r._measurement == \"teleport_watcher_events\" and r._field == \"count\")\n |> group(columns: [\"host\"])\n |> sort(columns: [\"_time\"], desc: true)\n |> aggregateWindow(every: 10s, fn: sum, createEmpty: false)\n |> derivative(nonNegative: true, unit: 10s)\n", - "refId": "A" - } - ], - "title": "Events Emitted", - "transformations": [ - { - "id": "labelsToFields", - "options": { - "valueLabel": "host" - } - } - ], - "type": "timeseries" - }, - { - "datasource": null, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "axisLabel": "Event Size", - "axisPlacement": "auto", - "barAlignment": 0, - "drawStyle": "line", - "fillOpacity": 10, - "gradientMode": "none", - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - }, - "lineInterpolation": "linear", - "lineWidth": 1, - "pointSize": 5, - "scaleDistribution": { - "type": "linear" - }, - "showPoints": "auto", - "spanNulls": false, - "stacking": { - "group": "A", - "mode": "none" - }, - "thresholdsStyle": { - "mode": "off" - } - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "red", - "value": 80 - } - ] - }, - "unit": "bytes" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 0, - "y": 30 - }, - "id": 20, - "options": { - "legend": { - "calcs": [], - "displayMode": "list", - "placement": "bottom" - }, - "tooltip": { - "mode": "single" - } - }, - "targets": [ - { - "hide": false, - "query": "from(bucket: v.bucket)\n |> range(start: v.timeRangeStart, stop: v.timeRangeStop)\n |> filter(fn: (r) => r._measurement == \"teleport_watcher_events\" and r._field == \"sum\")\n |> map(fn: (r) => ({r with resource: if r.resource =~ /tunnel_connection\\/.*$/ then \"/tunnel_connection\" else r.resource}))\n |> group(columns: [\"resource\"])\n |> sort(columns: [\"_time\"], desc: true)\n |> aggregateWindow(every: 10s, fn: sum, createEmpty: false)\n |> derivative(nonNegative: true, unit: 10s)\n\n", - "refId": "A" - } - ], - "title": "Events Per/Sec By Resource", - "transformations": [], - "type": "timeseries" - }, - { - "datasource": null, - "fieldConfig": { - "defaults": { - "color": { - "mode": "palette-classic" - }, - "custom": { - "hideFrom": { - "legend": false, - "tooltip": false, - "viz": false - } - }, - "mappings": [], - "unit": "none" - }, - "overrides": [] - }, - "gridPos": { - "h": 8, - "w": 12, - "x": 12, - "y": 30 - }, - "id": 22, - "options": { - "legend": { - "displayMode": "list", - "placement": "bottom" - }, - "pieType": "pie", - "reduceOptions": { - "calcs": [ - "lastNotNull" - ], - "fields": "", - "values": false - }, - "tooltip": { - "mode": "single" - } - }, - "pluginVersion": "8.2.3", - "targets": [ - { - "query": "from(bucket: v.bucket)\n |> range(start: v.timeRangeStart, stop: v.timeRangeStop)\n |> filter(fn: (r) => r._measurement == \"teleport_watcher_events\" and r._field == \"count\")\n |> map(fn: (r) => ({r with resource: if r.resource =~ /tunnel_connection\\/.*$/ then \"/tunnel_connection\" else r.resource}))\n |> group(columns: [\"resource\"])\n |> sort(columns: [\"_time\"], desc: true) \n |> aggregateWindow(every: v.windowPeriod, fn: sum, createEmpty: false)", - "refId": "A" - } - ], - "title": "Resources Emitted", - "transformations": [ - { - "id": "labelsToFields", - "options": { - "valueLabel": "resource" - } - } - ], - "type": "piechart" - }, - { - "datasource": null, - "fieldConfig": { - "defaults": { - "color": { - "mode": "thresholds" - }, - "mappings": [], - "thresholds": { - "mode": "absolute", - "steps": [ - { - "color": "green", - "value": null - }, - { - "color": "red", - "value": 80 - } - ] - }, - "unit": "bytes" - }, - "overrides": [] - }, - "gridPos": { - "h": 7, - "w": 24, - "x": 0, - "y": 38 - }, - "id": 24, - "options": { - "displayMode": "gradient", - "orientation": "auto", - "reduceOptions": { - "calcs": [ - "lastNotNull" - ], - "fields": "", - "values": false - }, - "showUnfilled": true, - "text": {} - }, - "pluginVersion": "8.2.3", - "targets": [ - { - "query": "from(bucket: v.bucket)\n |> range(start: v.timeRangeStart, stop: v.timeRangeStop)\n |> filter(fn: (r) => r._measurement == \"teleport_watcher_events\" and r._field == \"sum\")\n |> map(fn: (r) => ({r with resource: if r.resource =~ /tunnel_connection\\/.*$/ then \"/tunnel_connection\" else r.resource}))\n |> group(columns: [\"resource\"])\n |> sort(columns: [\"_time\"], desc: true) \n |> aggregateWindow(every: v.windowPeriod, fn: sum, createEmpty: false)", - "refId": "A" - } - ], - "title": "Resource Sizes", - "type": "bargauge" - } - ], - "refresh": "10s", - "schemaVersion": 31, - "style": "dark", - "tags": [], - "templating": { - "list": [] - }, - "time": { - "from": "now-30m", - "to": "now" - }, - "timepicker": {}, - "timezone": "", - "title": "Teleport Health Stats", - "uid": "anXQA8F7z", - "version": 1 -} \ No newline at end of file diff --git a/assets/loadtest/grafana/influxdb-datasource.yaml b/assets/loadtest/grafana/influxdb-datasource.yaml deleted file mode 100644 index 0bc3355f83e..00000000000 --- a/assets/loadtest/grafana/influxdb-datasource.yaml +++ /dev/null @@ -1,17 +0,0 @@ -apiVersion: 1 -datasources: - - name: InfluxDB-Flux - type: influxdb - access: proxy - url: http://influxdb:8086 - user: admin - password: ${INFLUXDB_PASS} - isDefault: true - editable: true - basicAuth: true - jsonData: - version: Flux - defaultBucket: telegraf - organization: teleport - secureJsonData: - token: ${INFLUXDB_TOKEN} \ No newline at end of file diff --git a/assets/loadtest/grafana/nginx.conf b/assets/loadtest/grafana/nginx.conf deleted file mode 100644 index b231350870e..00000000000 --- a/assets/loadtest/grafana/nginx.conf +++ /dev/null @@ -1,65 +0,0 @@ -worker_processes auto; -user www-data; -pid /run/nginx.pid; - -events { - worker_connections 2048; -} - -http { - sendfile on; - tcp_nopush on; - tcp_nodelay on; - keepalive_timeout 65; - types_hash_max_size 2048; - # server_tokens off; - - # server_names_hash_bucket_size 64; - # server_name_in_redirect off; - - include /etc/nginx/mime.types; - default_type application/octet-stream; - - ## - # TLS settings - we are pretty strict here - # but well, it's a dev service, why not? - ## - ssl_protocols TLSv1.2; - ssl_prefer_server_ciphers on; - - ## - # Logging Settings - ## - error_log stderr debug; - access_log /dev/stdout; - - - ## - # Gzip Settings - ## - gzip on; - - # - # Frontend grafana with TLS - # - server { - listen 8443 default_server ssl; - ssl_certificate_key /etc/tls/certs/tls.key; - ssl_certificate /etc/tls/certs/tls.crt; - ssl_ciphers AES256+EECDH:AES256+EDH:!aNULL; - - location / { - proxy_pass http://grafana:3000; - } - } - - server { - listen 8888; - - location /healthz { - return 200 'alive and kicking!'; - } - } -} - - diff --git a/assets/loadtest/k8s/Makefile b/assets/loadtest/k8s/Makefile index 71f785b0b32..68355e2db63 100644 --- a/assets/loadtest/k8s/Makefile +++ b/assets/loadtest/k8s/Makefile @@ -1,13 +1,13 @@ LICENSE_PATH ?= /var/lib/teleport/license.pem -CERT_MANAGER_VERSION ?= 1.6.0 +CERT_MANAGER_VERSION ?= v1.7.1 SOAK_TEST_DURATION ?= 30m BACKEND ?= etcd USE_CERT_MANAGER ?= yes -TELEPORT_IMAGE ?= quay.io/gravitational/teleport-ent:8.0.0 +TELEPORT_IMAGE ?= quay.io/gravitational/teleport-ent:9.0.0 +NAMESPACE ?= loadtest # performs initialization needed for cluster # 1) generates etcd certs -# 2) generates credentials for grafana and influx # 2) creates loadtest namespace # 3) installs cert-manager # 4) creates and applies secrets @@ -23,10 +23,10 @@ setup: ifeq ($(BACKEND), etcd) make -C ../etcd/certs all endif - kubectl create namespace loadtest --dry-run=client -o yaml | kubectl apply -f - + kubectl create namespace $(NAMESPACE) --dry-run=client -o yaml | kubectl apply -f - make -C ./secrets all ifeq ($(USE_CERT_MANAGER), yes) - kubectl apply -f https://github.com/jetstack/cert-manager/releases/download/v$(CERT_MANAGER_VERSION)/cert-manager.yaml + kubectl apply -f https://github.com/jetstack/cert-manager/releases/download/$(CERT_MANAGER_VERSION)/cert-manager.yaml endif make generate-secrets @@ -34,13 +34,13 @@ endif .PHONY: generate-secrets generate-secrets: ifeq ($(BACKEND), etcd) - kubectl create secret generic etcd-client-certs -n loadtest \ + kubectl create secret generic etcd-client-certs -n $(NAMESPACE) \ --from-file=client-cert.pem=../etcd/certs/client-cert.pem \ --from-file=client-key.pem=../etcd/certs/client-key.pem \ --from-file=ca-cert.pem=../etcd/certs/ca-cert.pem \ --dry-run=client -o yaml | kubectl apply -f - - kubectl create secret generic etcd-server-certs -n loadtest \ + kubectl create secret generic etcd-server-certs -n $(NAMESPACE) \ --from-file=server-cert.pem=../etcd/certs/server-cert.pem \ --from-file=server-key.pem=../etcd/certs/server-key.pem \ --from-file=ca-cert.pem=../etcd/certs/ca-cert.pem \ @@ -48,21 +48,12 @@ ifeq ($(BACKEND), etcd) endif ifeq ($(BACKEND), firestore) - kubectl create secret generic gcp-creds -n loadtest \ + kubectl create secret generic gcp-creds -n $(NAMESPACE) \ --from-file=gcp_creds.json=${GCP_CREDS_LOCATION} \ --dry-run=client -o yaml | kubectl apply -f - endif - kubectl create secret generic influxdb-creds -n loadtest \ - --from-file=INFLUXDB_PASS=./secrets/influx-pass \ - --from-file=INFLUXDB_TOKEN=./secrets/influx-token \ - --dry-run=client -o yaml | kubectl apply -f - - - kubectl create secret generic grafana-creds -n loadtest \ - --from-file=GF_SECURITY_ADMIN_PASSWORD=./secrets/grafana-pass \ - --dry-run=client -o yaml | kubectl apply -f - - - kubectl create secret generic license -n loadtest \ + kubectl create secret generic license -n $(NAMESPACE) \ --from-file=license.pem=$(LICENSE_PATH) \ --dry-run=client -o yaml | kubectl apply -f - @@ -71,20 +62,18 @@ endif clean: make -C secrets clean make -C ../etcd/certs clean - kubectl delete namespace loadtest --ignore-not-found - kubectl delete -f https://github.com/jetstack/cert-manager/releases/download/v$(CERT_MANAGER_VERSION)/cert-manager.yaml --ignore-not-found + kubectl delete namespace $(NAMESPACE) --ignore-not-found + kubectl delete -f https://github.com/jetstack/cert-manager/releases/download/$(CERT_MANAGER_VERSION)/cert-manager.yaml --ignore-not-found ifeq ($(BACKEND), etcd) -# deploys etcd, grafana, influxdb, and teleport to the loadtest namespace +# deploys etcd and teleport to the loadtest namespace .PHONY: apply apply: setup install-etcd generate-certificates install-monitor install-teleport - -else ifeq ($(BACKEND), firestore) -# deploys grafana, influxdb, and teleport to the loadtest namespace +else +# deploys and teleport to the loadtest namespace .PHONY: apply apply: setup generate-certificates install-monitor install-teleport - endif ifeq ($(USE_CERT_MANAGER), yes) @@ -115,56 +104,38 @@ install-teleport: install-auth install-proxy install-node install-iot-node .PHONY: delete-teleport delete-teleport: delete-tc delete-nodes delete-proxy delete-auth -# installs grafana, influxdb, and prometheus +# installs prometheus exporter .PHONY: install-monitor install-monitor: - kubectl create configmap grafana-config -n loadtest \ - --from-file=influxdb-datasource.yaml=../grafana/influxdb-datasource.yaml \ - --from-file=health-dashboard.json=../grafana/health-dashboard.json \ - --from-file=default.yaml=../grafana/dashboard.yaml \ - --from-file=nginx.conf=../grafana/nginx.conf \ - --dry-run=client -o yaml | kubectl apply -f - - - kubectl apply -f influxdb.yaml - @make expand-yaml FILENAME=grafana - kubectl apply -f grafana-gen.yaml - @make expand-yaml FILENAME=prometheus + @make expand-yaml FILENAME=prometheus NAMESPACE=$(NAMESPACE) kubectl apply -f prometheus-gen.yaml -# deletes grafana, influxdb, and prometheus deployments, services and configmaps +# deletes prometheus exporter .PHONY: delete-monitor delete-monitor: - kubectl delete -f influxdb.yaml --ignore-not-found - kubectl delete -f grafana-gen.yaml --ignore-not-found - kubectl delete configmap grafana-config -n loadtest --ignore-not-found kubectl delete -f prometheus-gen.yaml --ignore-not-found # installs an etcd cluster .PHONY: install-etcd install-etcd: - kubectl create configmap etcd-telegraf-config -n loadtest \ - --from-file=telegraf.conf=../etcd/telegraf.conf \ - --dry-run=client -o yaml | kubectl apply -f - - kubectl apply -f etcd.yaml # deletes etcd deployment, services, and configmaps .PHONY: delete-etcd delete-etcd: kubectl delete -f etcd.yaml --ignore-not-found - kubectl delete configmap etcd-telegraf-config -n loadtest --ignore-not-found # install auth and applies required teleport resources for loadtests .PHONY: install-auth install-auth: setup-auth - kubectl wait --for=condition=ready pod -l teleport-role=auth -n loadtest --timeout=120s + kubectl wait --for=condition=ready pod -l teleport-role=auth -n $(NAMESPACE) --timeout=120s - kubectl -n loadtest exec deploy/auth -c teleport -it \ + kubectl -n $(NAMESPACE) exec deploy/auth -c teleport -it \ -- tctl --config /etc/teleport/teleport.yaml create -f /etc/teleport/admin.yaml - kubectl -n loadtest exec deploy/auth -c teleport -it \ + kubectl -n $(NAMESPACE) exec deploy/auth -c teleport -it \ -- tctl --config /etc/teleport/teleport.yaml create -f /etc/teleport/oidc.yaml - kubectl -n loadtest exec deploy/auth -c teleport -it \ + kubectl -n $(NAMESPACE) exec deploy/auth -c teleport -it \ -- tctl --config /etc/teleport/teleport.yaml create -f /etc/teleport/user.yaml @@ -173,31 +144,43 @@ ifeq ($(BACKEND), etcd) setup-auth: @make expand-yaml FILENAME=../teleport/teleport-auth-etcd - kubectl create configmap auth-config -n loadtest \ + kubectl create configmap auth-config -n $(NAMESPACE) \ --from-file=teleport.yaml=../teleport/teleport-auth-etcd-gen.yaml \ - --from-file=telegraf.conf=../teleport/telegraf.conf \ --from-file=oidc.yaml=../teleport/oidc.yaml \ --from-file=admin.yaml=../teleport/admin.yaml \ --from-file=user.yaml=../teleport/soaktest-user.yaml \ --dry-run=client -o yaml | kubectl apply -f - - @make expand-yaml FILENAME=auth-etcd + @make expand-yaml FILENAME=auth-etcd TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) kubectl apply -f auth-etcd-gen.yaml else ifeq ($(BACKEND), firestore) .PHONY: setup-auth setup-auth: @make expand-yaml FILENAME=../teleport/teleport-auth-firestore - kubectl create configmap auth-config -n loadtest \ + kubectl create configmap auth-config -n $(NAMESPACE) \ --from-file=teleport.yaml=../teleport/teleport-auth-firestore-gen.yaml \ - --from-file=telegraf.conf=../teleport/telegraf.conf \ --from-file=oidc.yaml=../teleport/oidc.yaml \ --from-file=admin.yaml=../teleport/admin.yaml \ --from-file=user.yaml=../teleport/soaktest-user.yaml \ --dry-run=client -o yaml | kubectl apply -f - - @make expand-yaml FILENAME=auth-firestore + @make expand-yaml FILENAME=auth-firestore TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) kubectl apply -f auth-firestore-gen.yaml +else ifeq ($(BACKEND), dynamo) +.PHONY: setup-auth +setup-auth: + @make expand-yaml DYNAMO_TABLE=${DYNAMO_TABLE} DYNAMO_REGION=${DYNAMO_REGION} FILENAME=../teleport/teleport-auth-dynamo + + kubectl create configmap auth-config -n $(NAMESPACE) \ + --from-file=teleport.yaml=../teleport/teleport-auth-dynamo-gen.yaml \ + --from-file=oidc.yaml=../teleport/oidc.yaml \ + --from-file=admin.yaml=../teleport/admin.yaml \ + --from-file=user.yaml=../teleport/soaktest-user.yaml \ + --dry-run=client -o yaml | kubectl apply -f - + + @make expand-yaml FILENAME=auth-dynamo TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) + kubectl apply -f auth-dynamo-gen.yaml else .PHONY: setup-auth setup-auth: @@ -209,27 +192,27 @@ endif .PHONY: delete-auth delete-auth: kubectl delete -f auth-etcd-gen.yaml --ignore-not-found + kubectl delete -f auth-dynamo-gen.yaml --ignore-not-found kubectl delete -f auth-firestore-gen.yaml --ignore-not-found - kubectl delete configmap auth-config -n loadtest --ignore-not-found + kubectl delete configmap auth-config -n $(NAMESPACE) --ignore-not-found # install proxy .PHONY: install-proxy install-proxy: @make expand-yaml FILENAME=../teleport/teleport-proxy - kubectl create configmap proxy-config -n loadtest \ + kubectl create configmap proxy-config -n $(NAMESPACE) \ --from-file=teleport.yaml=../teleport/teleport-proxy-gen.yaml \ - --from-file=telegraf.conf=../teleport/telegraf.conf \ --dry-run=client -o yaml | kubectl apply -f - - @make expand-yaml FILENAME=proxy + @make expand-yaml FILENAME=proxy TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) kubectl apply -f proxy-gen.yaml # deletes proxy deployment, services and configmaps .PHONY: delete-proxy delete-proxy: kubectl delete -f proxy-gen.yaml --ignore-not-found - kubectl delete configmap proxy-config -n loadtest --ignore-not-found + kubectl delete configmap proxy-config -n $(NAMESPACE) --ignore-not-found # deletes all node deployment and configmaps .PHONY: delete-nodes @@ -239,13 +222,13 @@ delete-nodes: delete-node delete-iot-node .PHONY: delete-node delete-node: kubectl delete -f node-gen.yaml --ignore-not-found - kubectl delete configmap node-config -n loadtest --ignore-not-found + kubectl delete configmap node-config -n $(NAMESPACE) --ignore-not-found # deletes all IoT nodes .PHONY: delete-iot-node delete-iot-node: kubectl delete -f iot-node-gen.yaml --ignore-not-found - kubectl delete configmap iot-node-config -n loadtest --ignore-not-found + kubectl delete configmap iot-node-config -n $(NAMESPACE) --ignore-not-found # install one IoT node and one non-IoT node .PHONY: install-nodes @@ -255,133 +238,137 @@ install-nodes: install-iot-node install-node .PHONY: install-iot-node install-iot-node: @make expand-yaml FILENAME=../teleport/teleport-iot-node - kubectl create configmap iot-node-config -n loadtest \ + kubectl create configmap iot-node-config -n $(NAMESPACE) \ --from-file=teleport.yaml=../teleport/teleport-iot-node-gen.yaml \ --dry-run=client -o yaml | kubectl apply -f - - @make expand-yaml FILENAME=iot-node + @make expand-yaml FILENAME=iot-node TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) kubectl apply -f iot-node-gen.yaml # install a non-IoT mode node .PHONY: install-node install-node: @make expand-yaml FILENAME=../teleport/teleport-node - kubectl create configmap node-config -n loadtest \ + kubectl create configmap node-config -n $(NAMESPACE) \ --from-file=teleport.yaml=../teleport/teleport-node-gen.yaml \ --dry-run=client -o yaml | kubectl apply -f - - @make expand-yaml FILENAME=node + @make expand-yaml FILENAME=node TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) kubectl apply -f node-gen.yaml # installs a trusted cluster .PHONY: install-tc install-tc: @make expand-yaml FILENAME=../teleport/tc - kubectl create configmap tc-config -n loadtest \ + kubectl create configmap tc-config -n $(NAMESPACE) \ --from-file=teleport.yaml=../teleport/teleport-tc.yaml \ --from-file=cluster.yaml=../teleport/tc-gen.yaml \ --dry-run=client -o yaml | kubectl apply -f - - @make expand-yaml FILENAME=tc + @make expand-yaml FILENAME=tc TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) NAMESPACE=$(NAMESPACE) kubectl apply -f tc-gen.yaml # deletes all rc resources from teleport and deletes trusted cluster deployments and configmaps .PHONY: delete-tc delete-tc: kubectl delete -f tc-gen.yaml --ignore-not-found - kubectl delete configmap tc-config -n loadtest --ignore-not-found + kubectl delete configmap tc-config -n $(NAMESPACE) --ignore-not-found - kubectl -n loadtest exec deploy/auth -c teleport -it \ + kubectl -n $(NAMESPACE) exec deploy/auth -c teleport -it \ -- /bin/bash -c "tctl --config /etc/teleport/teleport.yaml get rc | grep ' name:' | cut -d ':' -f2- | xargs -P 20 -n 1 -I {} tctl --config /etc/teleport/teleport.yaml rm rc/{}" # joins all trusted clusters to root cluster .PHONY: setup-tc setup-tc: - kubectl get pod -n loadtest -l app="tc" -o custom-columns=name:metadata.name --no-headers \ - | xargs -P 20 -n 1 -I {} kubectl -n loadtest exec {} -- tctl --config /etc/teleport/teleport.yaml create -f /etc/teleport/cluster.yaml + kubectl get pod -n $(NAMESPACE) -l app="tc" -o custom-columns=name:metadata.name --no-headers \ + | xargs -P 20 -n 1 -I {} kubectl -n $(NAMESPACE) exec {} -- tctl --config /etc/teleport/teleport.yaml create -f /etc/teleport/cluster.yaml # scales trusted clusters to 500 .PHONY: scale-tc-500 scale-tc-500: - kubectl scale --replicas=50 deploy tc -n loadtest + kubectl scale --replicas=500 deploy tc -n $(NAMESPACE) -# scales trusted cluters to 1 +# scales trusted clusters to 1 .PHONY: scale-tc-1 scale-tc-1: - kubectl scale --replicas=1 deploy tc -n loadtest + kubectl scale --replicas=1 deploy tc -n $(NAMESPACE) # scales nodes to 1 .PHONY: scale-1-non-iot scale-1-non-iot: - kubectl scale --replicas=1 deploy node -n loadtest + kubectl scale --replicas=1 deploy node -n $(NAMESPACE) # scales nodes to 1000 .PHONY: scale-1k-non-iot scale-1k-non-iot: - kubectl scale --replicas=100 deploy node -n loadtest + kubectl scale --replicas=1000 deploy node -n $(NAMESPACE) # scales nodes to 10000 .PHONY: scale-10k-non-iot scale-10k-non-iot: - kubectl scale --replicas=10000 deploy node -n loadtest + kubectl scale --replicas=10000 deploy node -n $(NAMESPACE) # scales nodes to 1 .PHONY: scale-1-iot scale-1-iot: - kubectl scale --replicas=1 deploy iot-node -n loadtest + kubectl scale --replicas=1 deploy iot-node -n $(NAMESPACE) # scales nodes to 1000 .PHONY: scale-1k-iot scale-1k-iot: - kubectl scale --replicas=100 deploy iot-node -n loadtest + kubectl scale --replicas=1000 deploy iot-node -n $(NAMESPACE) # scales nodes to 10000 .PHONY: scale-10k-iot scale-10k-iot: - kubectl scale --replicas=10000 deploy iot-node -n loadtest + kubectl scale --replicas=10000 deploy iot-node -n $(NAMESPACE) # gets pods in loadtest namespace .PHONY: pods pods: - kubectl get pods -n loadtest + kubectl get pods -n $(NAMESPACE) # removes all soak test jobs and configmaps .PHONY: delete-soaktest .PHONY: delete-soaktest delete-soaktest: - kubectl delete job -l app=soaktest -n loadtest --ignore-not-found + kubectl delete job -l app=soaktest -n $(NAMESPACE) --ignore-not-found - kubectl delete configmap soaktest-config -n loadtest --ignore-not-found + kubectl delete configmap soaktest-config -n $(NAMESPACE) --ignore-not-found # creates the soak test job .PHONY: install-soaktest install-soaktest: - kubectl create configmap soaktest-config -n loadtest \ + kubectl create configmap soaktest-config -n $(NAMESPACE) \ + --from-file=soaktest.sh=../teleport/soaktest.sh \ --from-literal=DURATION=$(SOAK_TEST_DURATION) \ - --from-file=auth=./secrets/soaktest-auth \ --dry-run=client -o yaml | kubectl apply -f - + kubectl create secret generic soaktest -n $(NAMESPACE) \ + --from-file=auth=./secrets/soaktest-auth \ + --from-literal=PROXY_HOST=$$(cat ./secrets/secrets.env | grep PROXY_HOST | cut -d '=' -f2- ) \ + --dry-run=client -o yaml | kubectl apply -f - - @make expand-yaml FILENAME=soaktest + @make expand-yaml FILENAME=soaktest TELEPORT_IMAGE=$(TELEPORT_IMAGE) NAMESPACE=$(NAMESPACE) kubectl create -f soaktest-gen.yaml # deploys a job to run the soak tests .PHONY: run-soak-tests run-soak-tests: - kubectl -n loadtest exec $$(kubectl get pod -n loadtest -l teleport-role="auth" -o jsonpath="{.items[0].metadata.name}") -c teleport -it \ + kubectl -n $(NAMESPACE) exec $$(kubectl get pod -n $(NAMESPACE) -l teleport-role="auth" -o jsonpath="{.items[0].metadata.name}") -c teleport -it \ -- tctl auth sign --overwrite --user=soaktest-runner --out=/data/soaktest-auth --ttl=8760h --config /etc/teleport/teleport.yaml - kubectl cp -c teleport loadtest/$$(kubectl get pod -n loadtest -l teleport-role="auth" -o jsonpath="{.items[0].metadata.name}"):/data/soaktest-auth ./secrets/soaktest-auth + kubectl cp -c teleport loadtest/$$(kubectl get pod -n $(NAMESPACE) -l teleport-role="auth" -o jsonpath="{.items[0].metadata.name}"):/data/soaktest-auth ./secrets/soaktest-auth - kubectl wait --for=condition=available --timeout=600s deploy/node -n loadtest - kubectl wait --for=condition=available --timeout=600s deploy/iot-node -n loadtest + kubectl wait --for=condition=available --timeout=600s deploy/node -n $(NAMESPACE) + kubectl wait --for=condition=available --timeout=600s deploy/iot-node -n $(NAMESPACE) @make install-soaktest @sleep 1 - kubectl wait --for=condition=ready pod $$(kubectl get pods --sort-by=.metadata.creationTimestamp -o jsonpath="{.items[-1:].metadata.name}" -l app=soaktest -n loadtest) -n loadtest --timeout=120s - kubectl logs $$(kubectl get pods --sort-by=.metadata.creationTimestamp -o jsonpath="{.items[-1:].metadata.name}" -l app=soaktest -n loadtest) -n loadtest --tail -1 -f + kubectl wait --for=condition=ready pod $$(kubectl get pods --sort-by=.metadata.creationTimestamp -o jsonpath="{.items[-1:].metadata.name}" -l app=soaktest -n $(NAMESPACE)) -n $(NAMESPACE) --timeout=120s + kubectl logs $$(kubectl get pods --sort-by=.metadata.creationTimestamp -o jsonpath="{.items[-1:].metadata.name}" -l app=soaktest -n $(NAMESPACE)) -n $(NAMESPACE) --tail -1 -f # runs the node scaling tests .PHONY: run-scaling-test @@ -389,12 +376,12 @@ run-scaling-test: @make delete-nodes @make install-node @make scale-10k-non-iot - @kubectl wait --for=condition=available deploy/node -n loadtest --timeout=60m + @kubectl wait --for=condition=available deploy/node -n $(NAMESPACE) --timeout=60m @sleep 30 @make scale-1-non-iot @sleep 15 @make scale-10k-non-iot - @kubectl wait --for=condition=available deploy/node -n loadtest --timeout=60m + @kubectl wait --for=condition=available deploy/node -n $(NAMESPACE) --timeout=60m @sleep 15 @make scale-1-non-iot @@ -403,12 +390,12 @@ run-scaling-test: @make delete-nodes @make install-iot-node @make scale-10k-iot - @kubectl wait --for=condition=available deploy/iot-node -n loadtest --timeout=60m + @kubectl wait --for=condition=available deploy/iot-node -n $(NAMESPACE) --timeout=60m @sleep 30 @make scale-1-iot @sleep 15 @make scale-10k-iot - @kubectl wait --for=condition=available deploy/iot-node -n loadtest --timeout=60m + @kubectl wait --for=condition=available deploy/iot-node -n $(NAMESPACE) --timeout=60m @sleep 15 @make scale-1 @@ -420,35 +407,46 @@ run-scaling-test: run-tc-scaling-test: @make install-tc @make scale-tc-500 - kubectl wait --for=condition=available deploy/tc -n loadtest --timeout=60m - @sleep 60 + kubectl wait --for=condition=available deploy/tc -n $(NAMESPACE) --timeout=60m + @sleep 120 @make setup-tc - @sleep 180 + @sleep 1200 @make delete-tc - @sleep 60 + @sleep 180 + @make install-tc @make scale-tc-500 - kubectl wait --for=condition=available deploy/tc -n loadtest --timeout=60m - @sleep 60 + kubectl wait --for=condition=available deploy/tc -n $(NAMESPACE) --timeout=60m + @sleep 120 @make setup-tc - @sleep 180 + @sleep 1200 @make delete-tc # collect goroutine and heap go profiles from the auth deployment .PHONY: collect-profiles collect-profiles: - kubectl port-forward service/auth 3434:3434 -n loadtest > /dev/null 2>&1 & + kubectl port-forward service/auth 3434:3434 -n $(NAMESPACE) > /dev/null 2>&1 & @echo "waiting for auth to be available..." @timeout 30 sh -c 'until nc -z localhost 3434; do sleep 0.5; done' - @make fetch-profiles LOCATION=$(shell date +%s) + @make fetch-profiles LOCATION=auth-$(shell date +%s) + + kill -s kill $$(pgrep -f 3434:3434) + + kubectl port-forward service/proxy 3434:3434 -n $(NAMESPACE) > /dev/null 2>&1 & + + @echo "waiting for proxy to be available..." + + @timeout 30 sh -c 'until nc -z localhost 3434; do sleep 0.5; done' + + @make fetch-profiles LOCATION=proxy-$(shell date +%s) kill -s kill $$(pgrep -f 3434:3434) @@ -463,4 +461,4 @@ fetch-profiles: # output file will be named the same with a -gen suffix, i.e input = test then output will be test-gen.yaml .PHONY: expand-yaml expand-yaml: - @bash -c "set -a && source ./secrets/secrets.env && set +a && envsubst < $(FILENAME).yaml > $(FILENAME)-gen.yaml" + @bash -c "set -a && source ./secrets/secrets.env && set +a && envsubst < $(FILENAME).yaml > $(FILENAME)-gen.yaml" \ No newline at end of file diff --git a/assets/loadtest/k8s/auth-dynamo.yaml b/assets/loadtest/k8s/auth-dynamo.yaml new file mode 100644 index 00000000000..1d4fa0a5421 --- /dev/null +++ b/assets/loadtest/k8s/auth-dynamo.yaml @@ -0,0 +1,83 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: auth + namespace: ${NAMESPACE} + labels: + teleport-role: auth +spec: + replicas: 3 + selector: + matchLabels: + teleport-role: auth + template: + metadata: + annotations: + prometheus.io/scrape: "true" + prometheus.io/port: "3434" + labels: + backend: dynamo + teleport-role: auth + spec: + volumes: + - name: config + configMap: + name: auth-config + - name: license + secret: + secretName: license + - name: storage + emptyDir: {} + containers: + - name: teleport + image: ${TELEPORT_IMAGE} + args: ["-d", "--insecure", "--diag-addr=0.0.0.0:3434"] + ports: + - name: diag + containerPort: 3434 + protocol: TCP + readinessProbe: + failureThreshold: 3 + httpGet: + path: /healthz + port: 3434 + scheme: HTTP + initialDelaySeconds: 10 + periodSeconds: 30 + successThreshold: 1 + timeoutSeconds: 2 + livenessProbe: + failureThreshold: 3 + initialDelaySeconds: 30 + periodSeconds: 10 + successThreshold: 1 + tcpSocket: + port: 3434 + timeoutSeconds: 1 + volumeMounts: + - name: config + mountPath: /etc/teleport/ + readOnly: true + - name: license + mountPath: /var/lib/teleport/license.pem + subPath: license.pem + readOnly: true + - mountPath: /data + name: storage +--- +apiVersion: v1 +kind: Service +metadata: + name: auth + namespace: ${NAMESPACE} +spec: + ports: + - name: auth + port: 3025 + targetPort: 3025 + - name: diag + port: 3434 + targetPort: 3434 + selector: + teleport-role: auth + type: ClusterIP diff --git a/assets/loadtest/k8s/auth-etcd.yaml b/assets/loadtest/k8s/auth-etcd.yaml index 049df023b70..33cb2887ac6 100644 --- a/assets/loadtest/k8s/auth-etcd.yaml +++ b/assets/loadtest/k8s/auth-etcd.yaml @@ -2,16 +2,19 @@ apiVersion: apps/v1 kind: Deployment metadata: name: auth - namespace: loadtest + namespace: ${NAMESPACE} labels: teleport-role: auth spec: - replicas: 1 + replicas: 3 selector: matchLabels: teleport-role: auth template: metadata: + annotations: + prometheus.io/scrape: "true" + prometheus.io/port: "3434" labels: teleport-role: auth backend: etcd @@ -31,16 +34,6 @@ spec: - name: storage emptyDir: {} containers: - - name: telegraf - image: telegraf:1.20.3 - envFrom: - - secretRef: - name: influxdb-creds - volumeMounts: - - name: config - mountPath: /etc/telegraf/telegraf.conf - subPath: telegraf.conf - readOnly: true - name: teleport image: ${TELEPORT_IMAGE} args: ["-d", "--insecure", "--diag-addr=0.0.0.0:3434"] @@ -84,7 +77,7 @@ apiVersion: v1 kind: Service metadata: name: auth - namespace: loadtest + namespace: ${NAMESPACE} spec: ports: - name: auth diff --git a/assets/loadtest/k8s/auth-firestore.yaml b/assets/loadtest/k8s/auth-firestore.yaml index 3b5c3c3affb..e0a8afc7b86 100644 --- a/assets/loadtest/k8s/auth-firestore.yaml +++ b/assets/loadtest/k8s/auth-firestore.yaml @@ -2,7 +2,7 @@ apiVersion: apps/v1 kind: Deployment metadata: name: auth - namespace: loadtest + namespace: ${NAMESPACE} labels: teleport-role: auth spec: @@ -11,12 +11,15 @@ spec: matchLabels: teleport-role: auth template: - metadata: - labels: - teleport-role: auth - backend: firestore - prometheus.io/scrape: "true" - prometheus.io/port: "3434" + metadata: + annotations: + prometheus.io/scrape: "true" + prometheus.io/port: "3434" + labels: + teleport-role: auth + backend: firestore + prometheus.io/scrape: "true" + prometheus.io/port: "3434" spec: volumes: - name: config @@ -31,16 +34,6 @@ spec: - name: storage emptyDir: {} containers: - - name: telegraf - image: telegraf:1.20.3 - envFrom: - - secretRef: - name: influxdb-creds - volumeMounts: - - name: config - mountPath: /etc/telegraf/telegraf.conf - subPath: telegraf.conf - readOnly: true - name: teleport image: ${TELEPORT_IMAGE} args: ["-d", "--insecure", "--diag-addr=0.0.0.0:3434"] @@ -85,7 +78,7 @@ apiVersion: v1 kind: Service metadata: name: auth - namespace: loadtest + namespace: ${NAMESPACE} spec: ports: - name: auth diff --git a/assets/loadtest/k8s/etcd.yaml b/assets/loadtest/k8s/etcd.yaml index 1f31832a61a..3a35bc04ca4 100644 --- a/assets/loadtest/k8s/etcd.yaml +++ b/assets/loadtest/k8s/etcd.yaml @@ -18,9 +18,6 @@ spec: app: etcd spec: volumes: - - name: telegraf-config - configMap: - name: etcd-telegraf-config - name: server-certs secret: secretName: etcd-server-certs @@ -28,18 +25,6 @@ spec: secret: secretName: etcd-client-certs containers: - - name: telegraf - image: telegraf:1.20.3 - envFrom: - - secretRef: - name: influxdb-creds - volumeMounts: - - name: telegraf-config - mountPath: /etc/telegraf - readOnly: true - - name: client-certs - mountPath: /etc/etcd/certs/ - readOnly: true - name: etcd image: quay.io/coreos/etcd:v3.3.25 ports: diff --git a/assets/loadtest/k8s/grafana.yaml b/assets/loadtest/k8s/grafana.yaml deleted file mode 100644 index ca07afdb67c..00000000000 --- a/assets/loadtest/k8s/grafana.yaml +++ /dev/null @@ -1,140 +0,0 @@ -apiVersion: apps/v1 -kind: Deployment -metadata: - labels: - app: grafana - name: grafana - namespace: loadtest -spec: - selector: - matchLabels: - app: grafana - template: - metadata: - labels: - app: grafana - spec: - volumes: - - name: grafana-config - configMap: - name: grafana-config - - name: teleport-tls - secret: - secretName: teleport-tls - securityContext: - fsGroup: 472 - supplementalGroups: - - 0 - containers: - - name: nginx - image: nginx:1.14.2 - ports: - - containerPort: 8443 - name: https-grafana - protocol: TCP - - containerPort: 8888 - name: health - protocol: TCP - volumeMounts: - - name: teleport-tls - mountPath: /etc/tls/certs/ - readOnly: true - - name: grafana-config - mountPath: /etc/nginx/nginx.conf - subPath: nginx.conf - readOnly: true - readinessProbe: - failureThreshold: 3 - httpGet: - path: /healthz - port: 8888 - scheme: HTTP - initialDelaySeconds: 10 - periodSeconds: 30 - successThreshold: 1 - timeoutSeconds: 2 - livenessProbe: - failureThreshold: 3 - initialDelaySeconds: 30 - periodSeconds: 10 - successThreshold: 1 - tcpSocket: - port: 8888 - - name: grafana - image: grafana/grafana:8.2.3 - imagePullPolicy: IfNotPresent - envFrom: - - secretRef: - name: grafana-creds - - secretRef: - name: influxdb-creds - ports: - - containerPort: 3000 - name: http-grafana - protocol: TCP - readinessProbe: - failureThreshold: 3 - httpGet: - path: /robots.txt - port: 3000 - scheme: HTTP - initialDelaySeconds: 10 - periodSeconds: 30 - successThreshold: 1 - timeoutSeconds: 2 - livenessProbe: - failureThreshold: 3 - initialDelaySeconds: 30 - periodSeconds: 10 - successThreshold: 1 - tcpSocket: - port: 3000 - timeoutSeconds: 1 - resources: - requests: - cpu: 250m - memory: 750Mi - volumeMounts: - - mountPath: /etc/grafana/provisioning/datasources/influxdb-datasource.yaml - name: grafana-config - readOnly: true - subPath: influxdb-datasource.yaml - - mountPath: /etc/grafana/provisioning/dashboards/default.yaml - name: grafana-config - readOnly: true - subPath: default.yaml - - mountPath: /var/lib/grafana/dashboards/health-dashboard.json - name: grafana-config - subPath: health-dashboard.json - readOnly: true ---- -apiVersion: v1 -kind: Service -metadata: - name: grafana-https - namespace: loadtest -spec: - type: LoadBalancer - loadBalancerIP: ${GRAFANA_IP} - selector: - app: grafana - ports: - - name: web - protocol: TCP - port: 8443 - targetPort: 8443 ---- -apiVersion: v1 -kind: Service -metadata: - name: grafana - namespace: loadtest -spec: - selector: - app: grafana - ports: - - name: web - protocol: TCP - port: 3000 - targetPort: 3000 - diff --git a/assets/loadtest/k8s/influxdb.yaml b/assets/loadtest/k8s/influxdb.yaml deleted file mode 100644 index dda506468be..00000000000 --- a/assets/loadtest/k8s/influxdb.yaml +++ /dev/null @@ -1,63 +0,0 @@ -apiVersion: v1 -kind: ConfigMap -metadata: - name: influxdb-config - namespace: loadtest -data: - DOCKER_INFLUXDB_INIT_MODE: setup - DOCKER_INFLUXDB_INIT_USERNAME: admin - DOCKER_INFLUXDB_INIT_ORG: teleport - DOCKER_INFLUXDB_INIT_BUCKET: telegraf - DOCKER_INFLUXDB_INIT_RETENTION: 1w ---- -apiVersion: apps/v1 -kind: Deployment -metadata: - labels: - app: influxdb - name: influxdb - namespace: loadtest -spec: - replicas: 1 - selector: - matchLabels: - app: influxdb - template: - metadata: - labels: - app: influxdb - spec: - containers: - - image: influxdb:2.0.9 - name: influxdb - ports: - - containerPort: 8086 - name: influxdb - env: - - name: DOCKER_INFLUXDB_INIT_PASSWORD - valueFrom: - secretKeyRef: - name: influxdb-creds - key: INFLUXDB_PASS - - name: DOCKER_INFLUXDB_INIT_ADMIN_TOKEN - valueFrom: - secretKeyRef: - name: influxdb-creds - key: INFLUXDB_TOKEN - envFrom: - - configMapRef: - name: influxdb-config ---- -apiVersion: v1 -kind: Service -metadata: - name: influxdb - namespace: loadtest -spec: - ports: - - name: influxdb - port: 8086 - targetPort: 8086 - selector: - app: influxdb - type: ClusterIP diff --git a/assets/loadtest/k8s/iot-node.yaml b/assets/loadtest/k8s/iot-node.yaml index fec6f2b337d..0a65a092f8d 100644 --- a/assets/loadtest/k8s/iot-node.yaml +++ b/assets/loadtest/k8s/iot-node.yaml @@ -4,7 +4,7 @@ metadata: labels: teleport-role: node name: iot-node - namespace: loadtest + namespace: ${NAMESPACE} spec: replicas: 1 selector: @@ -20,11 +20,32 @@ spec: containers: - image: ${TELEPORT_IMAGE} name: teleport - args: ["-d", "--insecure"] + args: ["-d", "--insecure", "--diag-addr=0.0.0.0:3434"] ports: - name: nodessh containerPort: 3022 protocol: TCP + - name: diag + containerPort: 3434 + protocol: TCP + readinessProbe: + failureThreshold: 3 + httpGet: + path: /healthz + port: 3434 + scheme: HTTP + initialDelaySeconds: 10 + periodSeconds: 30 + successThreshold: 1 + timeoutSeconds: 2 + livenessProbe: + failureThreshold: 3 + initialDelaySeconds: 30 + periodSeconds: 10 + successThreshold: 1 + tcpSocket: + port: 3434 + timeoutSeconds: 1 volumeMounts: - mountPath: /etc/teleport name: config diff --git a/assets/loadtest/k8s/node.yaml b/assets/loadtest/k8s/node.yaml index 8b468d481ca..b1ecc77234d 100644 --- a/assets/loadtest/k8s/node.yaml +++ b/assets/loadtest/k8s/node.yaml @@ -4,7 +4,7 @@ metadata: labels: teleport-role: node name: node - namespace: loadtest + namespace: ${NAMESPACE} spec: replicas: 1 selector: @@ -20,11 +20,32 @@ spec: containers: - image: ${TELEPORT_IMAGE} name: teleport - args: ["-d", "--insecure"] + args: ["-d", "--insecure", "--diag-addr=0.0.0.0:3434"] ports: - name: nodessh containerPort: 3022 protocol: TCP + - name: diag + containerPort: 3434 + protocol: TCP + readinessProbe: + failureThreshold: 3 + httpGet: + path: /healthz + port: 3434 + scheme: HTTP + initialDelaySeconds: 10 + periodSeconds: 30 + successThreshold: 1 + timeoutSeconds: 2 + livenessProbe: + failureThreshold: 3 + initialDelaySeconds: 30 + periodSeconds: 10 + successThreshold: 1 + tcpSocket: + port: 3434 + timeoutSeconds: 1 volumeMounts: - mountPath: /etc/teleport name: config diff --git a/assets/loadtest/k8s/proxy.yaml b/assets/loadtest/k8s/proxy.yaml index 4fd0dac3381..c377949554d 100644 --- a/assets/loadtest/k8s/proxy.yaml +++ b/assets/loadtest/k8s/proxy.yaml @@ -2,7 +2,7 @@ apiVersion: apps/v1 kind: Deployment metadata: name: proxy - namespace: loadtest + namespace: ${NAMESPACE} labels: teleport-role: proxy spec: @@ -12,6 +12,9 @@ spec: teleport-role: proxy template: metadata: + annotations: + prometheus.io/scrape: "true" + prometheus.io/port: "3434" labels: teleport-role: proxy prometheus.io/scrape: "true" @@ -25,15 +28,6 @@ spec: secret: secretName: teleport-tls containers: - - name: telegraf - image: telegraf:1.20.3 - envFrom: - - secretRef: - name: influxdb-creds - volumeMounts: - - name: config - mountPath: /etc/telegraf/telegraf.conf - subPath: telegraf.conf - name: teleport image: ${TELEPORT_IMAGE} args: ["-d", "--insecure", "--diag-addr=0.0.0.0:3434"] @@ -74,11 +68,11 @@ apiVersion: v1 kind: Service metadata: name: proxy - namespace: loadtest + namespace: ${NAMESPACE} labels: teleport-role: proxy spec: - type: LoadBalancer + type: LoadBalancer loadBalancerIP: ${PROXY_IP} ports: - name: https @@ -101,5 +95,8 @@ spec: port: 3036 targetPort: 3036 protocol: TCP + - name: diag + port: 3434 + targetPort: 3434 selector: teleport-role: proxy diff --git a/assets/loadtest/k8s/secrets/Makefile b/assets/loadtest/k8s/secrets/Makefile index 6151d28f801..9e3ad14ff26 100644 --- a/assets/loadtest/k8s/secrets/Makefile +++ b/assets/loadtest/k8s/secrets/Makefile @@ -1,14 +1,11 @@ # output: -# grafana-pass : admin password for grafana -# influx-pass : admin password for influxdb -# influx-token : admin token for influxdb # node-token : static join token for teleport nodes # proxy-token : static join token for teleport proxies # tc-token : static join token for teleport trusted clusters # secrets.env : env file to be used to replace placeholder values in yaml files .PHONY:all -all: grafana-pass influx-pass influx-token join-tokens env +all: join-tokens env @rm -rf *csr .PHONY: env @@ -18,11 +15,6 @@ env: exit 1; \ fi - @if [ -z ${GRAFANA_HOST} ]; then \ - echo "GRAFANA_IP is not set, cannot apply cluster."; \ - exit 1; \ - fi - @if [ -z ${PROM_REMOTE_URL} ]; then \ echo "PROM_REMOTE_URL is not set, cannot apply cluster."; \ exit 1; \ @@ -40,8 +32,6 @@ env: @echo PROXY_IP=$(shell make -C ../../network get-proxy-ip) > secrets.env @echo PROXY_HOST=${PROXY_HOST} >> secrets.env - @echo GRAFANA_IP=$(shell make -C ../../network get-grafana-ip) >> secrets.env - @echo GRAFANA_HOST=${GRAFANA_HOST} >> secrets.env @echo NODE_TOKEN=$(shell cat node-token) >> secrets.env @echo PROXY_TOKEN=$(shell cat proxy-token) >> secrets.env @echo TC_TOKEN=$(shell cat tc-token) >> secrets.env @@ -50,15 +40,6 @@ env: @echo PROM_USER=${PROM_USER} >> secrets.env @echo PROM_PASSWORD=${PROM_PASSWORD} >> secrets.env -grafana-pass: - openssl rand -base64 32 | tr -d '\n' > grafana-pass - -influx-pass: - openssl rand -base64 32 | tr -d '\n' > influx-pass - -influx-token: - openssl rand -base64 32 | tr -d '\n' > influx-token - node-token: openssl rand -base64 32 | tr -d '\n' > node-token @@ -74,4 +55,4 @@ join-tokens: node-token proxy-token tc-token # removes everything .PHONY:clean clean: - rm -rf *-pass *-token *-auth *.env + rm -rf *-pass *-token *-auth *.env \ No newline at end of file diff --git a/assets/loadtest/k8s/soaktest.yaml b/assets/loadtest/k8s/soaktest.yaml index dd86d9b0780..98cc56f6f8c 100644 --- a/assets/loadtest/k8s/soaktest.yaml +++ b/assets/loadtest/k8s/soaktest.yaml @@ -2,50 +2,46 @@ apiVersion: batch/v1 kind: Job metadata: generateName: soaktest- - namespace: loadtest + namespace: ${NAMESPACE} labels: app: soaktest spec: completions: 1 parallelism: 1 + backoffLimit: 2 template: metadata: labels: app: soaktest spec: + restartPolicy: Never volumes: - name: config configMap: name: soaktest-config + defaultMode: 0777 + - name: soaktest-auth + secret: + secretName: soaktest containers: - image: ${TELEPORT_IMAGE} name: teleport envFrom: - configMapRef: name: soaktest-config + - secretRef: + name: soaktest command: - /bin/sh - -c - | - node=$(tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth -l root ls -f names | grep -v iot) - iot_node=$(tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth -l root ls -f names | grep iot) - - echo "----Non-IoT Mode Test via ${node}----" - echo "tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth bench --duration=${DURATION} root@${node} ls" - tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth bench --duration=${DURATION} root@${node} ls - - echo "tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth bench --duration=${DURATION} --interactive root@${node} ps aux" - tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth bench --duration=${DURATION} --interactive root@${node} ps aux - - echo "----IoT Mode Test via ${iot_node}----" - echo "tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth bench --duration=${DURATION} root@${iot_node} ls" - tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth bench --duration=${DURATION} root@${iot_node} ls - - echo "tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth bench --duration=${DURATION} --interactive root@${iot_node} ps aux" - tsh --insecure --proxy=monster.gravitational.co:3080 -i /etc/teleport/auth bench --duration=${DURATION} --interactive root@${iot_node} ps aux + cp /scripts/soaktest.sh /tmp + chmod +x /tmp/soaktest.sh + /tmp/soaktest.sh volumeMounts: - - mountPath: /etc/teleport + - mountPath: /scripts name: config readOnly: true - - restartPolicy: Never \ No newline at end of file + - mountPath: /etc/teleport + name: soaktest-auth + readOnly: true diff --git a/assets/loadtest/k8s/tc.yaml b/assets/loadtest/k8s/tc.yaml index 587218c51cc..606ae866314 100644 --- a/assets/loadtest/k8s/tc.yaml +++ b/assets/loadtest/k8s/tc.yaml @@ -4,7 +4,7 @@ metadata: labels: app: tc name: tc - namespace: loadtest + namespace: ${NAMESPACE} spec: replicas: 1 selector: @@ -24,12 +24,33 @@ spec: secretName: license containers: - image: ${TELEPORT_IMAGE} - args: ["-d", "--insecure"] + args: ["-d", "--insecure", "--diag-addr=0.0.0.0:3434"] name: tc ports: - containerPort: 3022 name: nodessh protocol: TCP + - name: diag + containerPort: 3434 + protocol: TCP + readinessProbe: + failureThreshold: 3 + httpGet: + path: /healthz + port: 3434 + scheme: HTTP + initialDelaySeconds: 10 + periodSeconds: 30 + successThreshold: 1 + timeoutSeconds: 2 + livenessProbe: + failureThreshold: 3 + initialDelaySeconds: 30 + periodSeconds: 10 + successThreshold: 1 + tcpSocket: + port: 3434 + timeoutSeconds: 1 volumeMounts: - name: config mountPath: /etc/teleport/ diff --git a/assets/loadtest/teleport/soaktest.sh b/assets/loadtest/teleport/soaktest.sh new file mode 100755 index 00000000000..0753f330d6e --- /dev/null +++ b/assets/loadtest/teleport/soaktest.sh @@ -0,0 +1,30 @@ +#!/bin/bash +# This script runs the teleport load test soak tests. +set -e +set -x + +node=$(tsh --insecure --proxy="${PROXY_HOST}":3080 -i /etc/teleport/auth -l root ls -f names | grep -v iot) +iot_node=$(tsh --insecure --proxy="${PROXY_HOST}":3080 -i /etc/teleport/auth -l root ls -f names | grep iot) + +echo "${node}" +echo "${iot_node}" + +if [ -z "${node}" ]; then + echo "no regular nodes found to run soak test on."; + exit 1; +fi + +if [ -z "${iot_node}" ]; then + echo "no IoT nodes found to run soak test on."; + exit 1; +fi + +echo "----Non-IoT Node Test----" +tsh --insecure --proxy="${PROXY_HOST}":3080 -i /etc/teleport/auth bench --duration="${DURATION}" root@"${node}" ls + +tsh --insecure --proxy="${PROXY_HOST}":3080 -i /etc/teleport/auth bench --duration="${DURATION}" --interactive root@"${node}" ps aux + +echo "----IoT Node Test----" +tsh --insecure --proxy="${PROXY_HOST}":3080 -i /etc/teleport/auth bench --duration="${DURATION}" root@"${iot_node}" ls + +tsh --insecure --proxy="${PROXY_HOST}":3080 -i /etc/teleport/auth bench --duration="${DURATION}" --interactive root@"${iot_node}" ps aux \ No newline at end of file diff --git a/assets/loadtest/teleport/telegraf.conf b/assets/loadtest/teleport/telegraf.conf deleted file mode 100644 index 1c8caf6fa52..00000000000 --- a/assets/loadtest/teleport/telegraf.conf +++ /dev/null @@ -1,134 +0,0 @@ -# Configuration for telegraf agent -[agent] - ## Default data collection interval for all inputs - interval = "10s" - ## Rounds collection interval to 'interval' - ## ie, if interval="10s" then always collect on :00, :10, :20, etc. - round_interval = true - - ## Telegraf will send metrics to outputs in batches of at - ## most metric_batch_size metrics. - metric_batch_size = 1000 - ## For failed writes, telegraf will cache metric_buffer_limit metrics for each - ## output, and will flush this buffer on a successful write. Oldest metrics - ## are dropped first when this buffer fills. - metric_buffer_limit = 10000 - - ## Collection jitter is used to jitter the collection by a random amount. - ## Each plugin will sleep for a random time within jitter before collecting. - ## This can be used to avoid many plugins querying things like sysfs at the - ## same time, which can have a measurable effect on the system. - collection_jitter = "0s" - - ## Default flushing interval for all outputs. You shouldn't set this below - ## interval. Maximum flush_interval will be flush_interval + flush_jitter - flush_interval = "10s" - ## Jitter the flush interval by a random amount. This is primarily to avoid - ## large write spikes for users running a large number of telegraf instances. - ## ie, a jitter of 5s and interval 10s means flushes will happen every 10-15s - flush_jitter = "0s" - - ## By default, precision will be set to the same timestamp order as the - ## collection interval, with the maximum being 1s. - ## Precision will NOT be used for service inputs, such as logparser and statsd. - precision = "" - ## Run telegraf in debug mode - debug = false - ## Run telegraf in quiet mode - quiet = false - ## Override default hostname, if empty use os.Hostname() - hostname = "" - ## If set to true, do no set the "host" tag in the telegraf agent. - omit_hostname = false - - -############################################################################### -# INPUT PLUGINS # -############################################################################### - -[[inputs.procstat]] - exe = "teleport" - prefix = "teleport" - -[[inputs.prometheus]] - # An array of urls to scrape metrics from. - urls = ["http://127.0.0.1:3434/metrics"] - name_prefix = "teleport_" - - # Add tags to be able to make beautiful dashboards - [inputs.prometheus.tags] - teleservice = "teleport" - -# Read metrics about cpu usage -[[inputs.cpu]] - ## Whether to report per-cpu stats or not - percpu = true - ## Whether to report total system cpu stats or not - totalcpu = true - ## If true, collect raw CPU time metrics. - collect_cpu_time = false - ## If true, compute and report the sum of all non-idle CPU states. - report_active = false - -# Read metrics about disk usage by mount point -[[inputs.disk]] - ## By default, telegraf gather stats for all mountpoints. - ## Setting mountpoints will restrict the stats to the specified mountpoints. - # mount_points = ["/"] - - ## Ignore some mountpoints by filesystem type. For example (dev)tmpfs (usually - ## present on /run, /var/run, /dev/shm or /dev). - ignore_fs = ["tmpfs", "devtmpfs", "devfs"] - -# Read metrics about disk IO by device -[[inputs.diskio]] - -# Get kernel statistics from /proc/stat -[[inputs.kernel]] - # no configuration - -# Read metrics about memory usage -[[inputs.mem]] - # no configuration - -# Get the number of processes and group them by status -[[inputs.processes]] - # no configuration - -# Read metrics about swap memory usage -[[inputs.swap]] - # no configuration - -# Read metrics about system load & uptime -[[inputs.system]] - # no configuration - -# Read netstat info -[[inputs.netstat]] - -# Read net info -[[inputs.net]] - -############################################################################### -# OUTPUT PLUGINS # -############################################################################### - -# Configuration for influxdb server to send metrics to -[[outputs.influxdb_v2]] - ## The full HTTP or UDP endpoint URL for your InfluxDB instance. - ## Multiple urls can be specified as part of the same cluster, - ## this means that only ONE of the urls will be written to each interval. - urls = ["http://influxdb:8086"] # required - - ## Token for authentication. - token = "${INFLUXDB_TOKEN}" - - ## Organization is the name of the organization you wish to write to; must exist. - organization = "teleport" - - ## Destination bucket to write into. - bucket = "telegraf" - - ## Write timeout (for the InfluxDB client), formatted as a string. - ## If not provided, will default to 5s. 0s means no timeout (not recommended). - timeout = "5s" \ No newline at end of file diff --git a/assets/loadtest/teleport/teleport-auth-dynamo.yaml b/assets/loadtest/teleport/teleport-auth-dynamo.yaml new file mode 100644 index 00000000000..3094c70a0bf --- /dev/null +++ b/assets/loadtest/teleport/teleport-auth-dynamo.yaml @@ -0,0 +1,36 @@ +teleport: + log: + severity: DEBUG + + data_dir: /var/lib/teleport + + advertise_ip: auth + + storage: + type: dynamodb + table_name: ${DYNAMO_TABLE} + region: ${DYNAMO_REGION} + + connection_limits: + max_connections: 65000 + max_users: 10000 + +auth_service: + enabled: yes + + listen_addr: 0.0.0.0:3025 + + authentication: + type: oidc + + cluster_name: one + tokens: + - "node:node-${NODE_TOKEN}" + - "proxy:proxy-${PROXY_TOKEN}" + - "trusted_cluster:cluster-${TC_TOKEN}" + +ssh_service: + enabled: no + +proxy_service: + enabled: no \ No newline at end of file diff --git a/trace.out b/trace.out deleted file mode 100644 index bbb077e130c..00000000000 Binary files a/trace.out and /dev/null differ